summaryrefslogtreecommitdiff
path: root/snippets/hyperstack
diff options
context:
space:
mode:
authorPaul Buetow <paul@buetow.org>2026-03-21 09:46:58 +0200
committerPaul Buetow <paul@buetow.org>2026-03-21 09:46:58 +0200
commitc693f37a6115f3567cd4fcff4c256a6d20dd6fac (patch)
tree04e18f502616535013bab0c7c513a1aabdb9c2f2 /snippets/hyperstack
parent3f6ef419f52c3361c8914a27c7949c2c8f2be1c8 (diff)
moved
Diffstat (limited to 'snippets/hyperstack')
-rw-r--r--snippets/hyperstack/.crush/logs/crush.log4
-rw-r--r--snippets/hyperstack/.gitignore4
-rw-r--r--snippets/hyperstack/.pi/settings.json8
-rw-r--r--snippets/hyperstack/Gemfile3
-rw-r--r--snippets/hyperstack/Gemfile.lock16
-rw-r--r--snippets/hyperstack/README.md186
-rw-r--r--snippets/hyperstack/hyperstack-vm.toml204
-rw-r--r--snippets/hyperstack/hyperstack-vm1.toml185
-rw-r--r--snippets/hyperstack/hyperstack-vm2.toml182
-rwxr-xr-xsnippets/hyperstack/hyperstack.rb2731
-rwxr-xr-xsnippets/hyperstack/pi-vm17
-rwxr-xr-xsnippets/hyperstack/pi-vm27
-rw-r--r--snippets/hyperstack/vllm-setup.txt487
-rwxr-xr-xsnippets/hyperstack/wg1-setup.sh414
14 files changed, 0 insertions, 4438 deletions
diff --git a/snippets/hyperstack/.crush/logs/crush.log b/snippets/hyperstack/.crush/logs/crush.log
deleted file mode 100644
index 7745db8..0000000
--- a/snippets/hyperstack/.crush/logs/crush.log
+++ /dev/null
@@ -1,4 +0,0 @@
-{"time":"2026-01-29T21:33:05.561515639+02:00","level":"INFO","source":{"function":"github.com/charmbracelet/crush/internal/config.(*catwalkSync).Get.func1","file":"/home/paul/go/pkg/mod/github.com/charmbracelet/crush@v0.36.0/internal/config/catwalk.go","line":55},"msg":"Fetching providers from Catwalk"}
-{"time":"2026-01-29T21:33:05.920268417+02:00","level":"INFO","source":{"function":"github.com/charmbracelet/crush/internal/config.cache[...].Store","file":"/home/paul/go/pkg/mod/github.com/charmbracelet/crush@v0.36.0/internal/config/provider.go","line":213},"msg":"Saving provider data to disk","path":"/home/paul/.local/share/crush/providers.json"}
-{"time":"2026-01-29T21:33:05.923610816+02:00","level":"WARN","source":{"function":"github.com/charmbracelet/crush/internal/config.(*Config).configureProviders-range1","file":"/home/paul/go/pkg/mod/github.com/charmbracelet/crush@v0.36.0/internal/config/load.go","line":295},"msg":"Provider is missing API key, this might be OK for local providers","provider":"ollama"}
-{"time":"2026-01-29T21:33:05.923686216+02:00","level":"WARN","source":{"function":"github.com/charmbracelet/crush/internal/config.(*Config).configureProviders-range1","file":"/home/paul/go/pkg/mod/github.com/charmbracelet/crush@v0.36.0/internal/config/load.go","line":309},"msg":"Provider is missing API key, this might be OK for local providers","provider":"ollama"}
diff --git a/snippets/hyperstack/.gitignore b/snippets/hyperstack/.gitignore
deleted file mode 100644
index 132d791..0000000
--- a/snippets/hyperstack/.gitignore
+++ /dev/null
@@ -1,4 +0,0 @@
-.bundle/
-vendor/bundle/
-.hyperstack-vm-state.json
-.hyperstack-vm*-state.json*
diff --git a/snippets/hyperstack/.pi/settings.json b/snippets/hyperstack/.pi/settings.json
deleted file mode 100644
index 23f5df6..0000000
--- a/snippets/hyperstack/.pi/settings.json
+++ /dev/null
@@ -1,8 +0,0 @@
-{
- "defaultProvider": "hyperstack1",
- "defaultModel": "cyankiwi/NVIDIA-Nemotron-3-Super-120B-A12B-AWQ-4bit",
- "enabledModels": [
- "hyperstack1/cyankiwi/NVIDIA-Nemotron-3-Super-120B-A12B-AWQ-4bit",
- "hyperstack2/bullpoint/Qwen3-Coder-Next-AWQ-4bit"
- ]
-}
diff --git a/snippets/hyperstack/Gemfile b/snippets/hyperstack/Gemfile
deleted file mode 100644
index a1bbd94..0000000
--- a/snippets/hyperstack/Gemfile
+++ /dev/null
@@ -1,3 +0,0 @@
-source "https://rubygems.org"
-
-gem "toml-rb", "~> 2.2"
diff --git a/snippets/hyperstack/Gemfile.lock b/snippets/hyperstack/Gemfile.lock
deleted file mode 100644
index 80e05d4..0000000
--- a/snippets/hyperstack/Gemfile.lock
+++ /dev/null
@@ -1,16 +0,0 @@
-GEM
- remote: https://rubygems.org/
- specs:
- citrus (3.0.2)
- toml-rb (2.2.0)
- citrus (~> 3.0, > 3.0)
-
-PLATFORMS
- ruby
- x86_64-linux
-
-DEPENDENCIES
- toml-rb (~> 2.2)
-
-BUNDLED WITH
- 2.6.9
diff --git a/snippets/hyperstack/README.md b/snippets/hyperstack/README.md
deleted file mode 100644
index 730b310..0000000
--- a/snippets/hyperstack/README.md
+++ /dev/null
@@ -1,186 +0,0 @@
-# hyperstack
-
-Automates Hyperstack GPU VM lifecycle: create, bootstrap, WireGuard tunnel, vLLM inference, LiteLLM proxy.
-
-## Architecture
-
-```
-Claude Code (local) Hyperstack VM (A100 80GB)
-┌─────────────────┐ ┌──────────────────────────────────┐
-│ claude CLI │── Anthropic API ─▶│ LiteLLM proxy (:4000) │
-│ │ /v1/messages │ Anthropic → OpenAI translation │
-│ │ via WireGuard │ │ │
-└─────────────────┘ │ ▼ │
- │ vLLM engine (:11434) │
-OpenCode (local) │ bullpoint/Qwen3-Coder-Next- │
-┌─────────────────┐ │ AWQ-4bit (45 GB, MoE 80B) │
-│ opencode │── OpenAI API ────▶│ FlashAttention v2 │
-│ │ /v1/chat/... │ prefix caching │
-└─────────────────┘ └──────────────────────────────────┘
-```
-
-Both local clients connect over a WireGuard tunnel (`wg1`, subnet `192.168.3.0/24`).
-The VM gets `192.168.3.1`; your local machine gets `192.168.3.2`.
-
-## Prerequisites
-
-- Hyperstack account with API key in `~/.hyperstack`
-- SSH key registered in Hyperstack as `earth` (or change `ssh.hyperstack_key_name` in the TOML)
-- Review `[network].allowed_ssh_cidrs` and `[network].allowed_wireguard_cidrs` in your TOML.
- The secure default is `["auto"]`, which resolves your current public egress IP to `/32`.
- Set explicit CIDRs or `HYPERSTACK_OPERATOR_CIDR` if you deploy from a different network.
-- WireGuard setup script: `wg1-setup.sh` (present in this directory)
-- Ruby with `toml-rb` gem: `bundle install`
-
-## Quickstart
-
-```bash
-# Deploy VM, set up WireGuard + vLLM + LiteLLM (~10 min on first run)
-ruby hyperstack.rb create
-
-# Verify everything is working
-ruby hyperstack.rb test
-
-# Use Claude Code against the local vLLM
-ANTHROPIC_BASE_URL=http://hyperstack.wg1:4000 \
-ANTHROPIC_API_KEY=sk-litellm-master \
-claude --model claude-opus-4-6-20260604 --dangerously-skip-permissions
-
-# Tear down
-# Also removes the tracked local wg1 peer, hostname alias, and pinned SSH host key.
-ruby hyperstack.rb delete
-```
-
-## Using Pi
-
-Bring both VMs up first:
-
-```bash
-ruby hyperstack.rb create-both
-```
-
-Then start one Pi session per terminal:
-
-```bash
-./pi-vm1
-./pi-vm2
-```
-
-These wrappers `cd` into this repo before launching Pi, so the project-local
-settings in `.pi/settings.json` still apply.
-
-## Using Claude Code with vLLM
-
-WireGuard (`wg1`) must be active before connecting.
-
-```bash
-ANTHROPIC_BASE_URL=http://hyperstack.wg1:4000 \
-ANTHROPIC_API_KEY=sk-litellm-master \
-claude --model claude-opus-4-6-20260604 --dangerously-skip-permissions
-```
-
-If you see an **"Auth conflict"** warning, clear the saved claude.ai session first:
-
-```bash
-claude /logout
-```
-
-**Fish shell alias** (add to `~/.config/fish/config.fish`):
-
-```fish
-alias claude-local='ANTHROPIC_BASE_URL=http://hyperstack.wg1:4000 \
- ANTHROPIC_API_KEY=sk-litellm-master \
- claude --model claude-opus-4-6-20260604 --dangerously-skip-permissions'
-```
-
-**Available model aliases** — all map to the same vLLM model:
-
-| Alias | Use case |
-|-------|----------|
-| `claude-opus-4-6-20260604` | Recommended (most future-proof) |
-| `claude-opus-4-20250514` | |
-| `claude-sonnet-4-20250514` | |
-| `claude-haiku-3-5-20241022` | |
-
-Add new Anthropic model IDs to `vllm.litellm_claude_model_names` in `hyperstack-vm.toml` as they are released.
-
-## Using OpenCode with vLLM
-
-OpenCode speaks OpenAI natively — connect directly to vLLM, no LiteLLM needed:
-
-```bash
-OPENAI_BASE_URL=http://hyperstack.wg1:11434/v1 \
-OPENAI_API_KEY=EMPTY \
-opencode
-```
-
-Set the model name to `bullpoint/Qwen3-Coder-Next-AWQ-4bit` in your OpenCode config.
-
-## CLI reference
-
-```
-ruby hyperstack.rb [--config path] <command> [options]
-
-Commands:
- create Deploy a new VM and run full provisioning
- delete Destroy the tracked VM
- status Show VM and WireGuard status
- test Run end-to-end inference tests (vLLM + LiteLLM)
-
-create options:
- --replace Delete existing tracked VM before creating
- --dry-run Print the plan without making changes
- --vllm / --no-vllm Override config: enable/disable vLLM+LiteLLM setup
- --ollama / --no-ollama Override config: enable/disable Ollama setup
-```
-
-## Configuration
-
-Edit `hyperstack-vm.toml` to change defaults. Key sections:
-
-| Section | Purpose |
-|---------|---------|
-| `[vm]` | Flavor, image, environment name |
-| `[vllm]` | Model, container settings, LiteLLM key and Claude aliases |
-| `[ollama]` | Ollama settings (disabled by default; set `install = true` to use instead) |
-| `[network]` | Ports, WireGuard subnet, allowed CIDRs |
-| `[wireguard]` | Auto-setup script path |
-
-`allowed_ssh_cidrs` and `allowed_wireguard_cidrs` accept either explicit CIDRs such as
-`["203.0.113.4/32"]` or `["auto"]`. `auto` resolves the current public operator IP at runtime;
-set `HYPERSTACK_OPERATOR_CIDR` to override that detection when needed.
-
-SSH host keys are pinned per state file in `<state>.known_hosts`. `delete` and `--replace`
-clear that trust file for intentional reprovisioning; unexpected host key changes now fail closed.
-
-## Monitoring vLLM
-
-```bash
-# Live engine stats (throughput, KV cache, prefix cache hit rate)
-ssh ubuntu@<vm-ip> 'docker logs -f vllm_qwen3 2>&1 | grep "Engine 000"'
-
-# Last 1 minute of stats
-ssh ubuntu@<vm-ip> 'docker logs --since 1m vllm_qwen3 2>&1 | grep "Engine 000"'
-
-# GPU stats (every 5 s)
-ssh ubuntu@<vm-ip> 'nvidia-smi --query-gpu=temperature.gpu,utilization.gpu,power.draw,memory.used --format=csv -l 5'
-
-# LiteLLM proxy log
-ssh ubuntu@<vm-ip> 'sudo journalctl -fu litellm'
-```
-
-Healthy baseline (A100 80GB PCIe, qwen3-coder-next AWQ 4-bit):
-
-| Metric | Expected |
-|--------|----------|
-| Prefill throughput | 5,000–11,000 tok/s |
-| Decode throughput | 40–99 tok/s |
-| KV cache usage | 2–5% for typical sessions |
-| Prefix cache hit (Claude Code) | 0% (expected — prompt prefix mutates each turn) |
-| Prefix cache hit (OpenCode) | >50% after warm-up |
-
-## Switching models
-
-Stop the current container, start a new one with a different `--model`, then update `vllm.model` in `hyperstack-vm.toml` and re-run `ruby hyperstack.rb create` to reinstall LiteLLM with the updated config.
-
-See `vllm-setup.txt` for detailed vLLM and LiteLLM setup notes, VRAM sizing guide, and troubleshooting.
diff --git a/snippets/hyperstack/hyperstack-vm.toml b/snippets/hyperstack/hyperstack-vm.toml
deleted file mode 100644
index e82c97f..0000000
--- a/snippets/hyperstack/hyperstack-vm.toml
+++ /dev/null
@@ -1,204 +0,0 @@
-[auth]
-api_key_file = "~/.hyperstack"
-
-[hyperstack]
-base_url = "https://infrahub-api.nexgencloud.com/v1"
-
-[state]
-file = ".hyperstack-vm-state.json"
-
-[vm]
-name_prefix = "hyperstack"
-hostname = "hyperstack"
-environment_name = "snonux-ollama"
-
-# A100-80GB is the cost-first default for gpt-oss-120b inference.
-# Switch this to n3-H100x1 if you want safer throughput and compatibility headroom.
-flavor_name = "n3-A100x1"
-image_name = "Ubuntu Server 24.04 LTS R570 CUDA 12.8 with Docker"
-assign_floating_ip = true
-create_bootable_volume = false
-enable_port_randomization = false
-labels = ["gpt-oss-120b", "wireguard"]
-
-[ssh]
-username = "ubuntu"
-private_key_path = "~/.ssh/id_rsa"
-hyperstack_key_name = "earth"
-port = 22
-connect_timeout_sec = 10
-
-[network]
-wireguard_udp_port = 56710
-wireguard_subnet = "192.168.3.0/24"
-# Secure default: "auto" resolves your current public egress IP to /32 at runtime.
-# Override with explicit CIDRs if you deploy from multiple networks or want broader access.
-allowed_ssh_cidrs = ["auto"]
-allowed_wireguard_cidrs = ["auto"]
-# Port 11434 is shared by both Ollama and vLLM for firewall compatibility.
-ollama_port = 11434
-# Port 4000: LiteLLM Anthropic-API proxy (used with vLLM).
-litellm_port = 4000
-
-[bootstrap]
-enable_guest_bootstrap = true
-install_wireguard = true
-configure_ufw = true
-configure_ollama_host = false
-
-[ollama]
-# Disabled in favour of vLLM; set install = true to switch back to Ollama.
-install = false
-models_dir = "/ephemeral/ollama/models"
-listen_host = "0.0.0.0:11434"
-gpu_overhead_mb = 2000
-num_parallel = 1
-context_length = 32768
-pull_models = ["qwen3-coder-next", "qwen3-coder:30b", "gpt-oss:20b", "gpt-oss:120b", "nemotron-3-super"]
-
-# vLLM serves one model via Docker; LiteLLM translates Anthropic API → OpenAI.
-# Use --vllm / --no-vllm CLI flags to override install at runtime.
-[vllm]
-install = true
-model = "bullpoint/Qwen3-Coder-Next-AWQ-4bit"
-# HuggingFace model cache on ephemeral NVMe (fast; survives reboots on most providers).
-hug_cache_dir = "/ephemeral/hug"
-container_name = "vllm_qwen3"
-max_model_len = 262144
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "qwen3_coder"
-# LiteLLM maps each entry to the vLLM model; add new Anthropic model IDs here.
-litellm_master_key = "sk-litellm-master"
-litellm_claude_model_names = [
- "claude-sonnet-4-20250514",
- "claude-opus-4-20250514",
- "claude-opus-4-6-20260604",
- "claude-haiku-3-5-20241022"
-]
-
-# Named model presets for 'ruby hyperstack.rb model switch <name>'.
-# Each preset overrides the matching [vllm] field; unset fields fall back to [vllm] defaults.
-# Switch examples:
-# ruby hyperstack.rb model switch qwen3-coder-next # fast coding, 256k context
-# ruby hyperstack.rb model switch nemotron-super # extended analysis, 131k context
-
-[vllm.presets.qwen3-coder-next]
-model = "bullpoint/Qwen3-Coder-Next-AWQ-4bit"
-container_name = "vllm_qwen3"
-max_model_len = 262144
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "qwen3_coder"
-
-# NVIDIA Nemotron-3-Super-120B-A12B AWQ 4-bit — hybrid Mamba+MoE (12B active / 120B total).
-# ~60 GB weights on A100 80GB. Uses NoPE (no positional embeddings) so context can be set to
-# 1M by just raising max_model_len; no YaRN needed. May OOM above 256K on A100 80GB.
-# Requires trust_remote_code=true for the nemotron_h architecture.
-# Note: cyankiwi AWQ has model_type="nemotron_nas" (underscore); vLLM keys on "nemotron-nas"
-# (hyphen), so vLLM may not recognise it without trust_remote_code and latest vLLM.
-# NVIDIA Nemotron-3-Super uses the same XML tool call format as Qwen3 XML:
-# <tool_call><function=name><parameter=p>value</parameter></function></tool_call>
-# qwen3_xml handles this format and is compatible with Nemotron's chat template.
-[vllm.presets.nemotron-super]
-model = "cyankiwi/NVIDIA-Nemotron-3-Super-120B-A12B-AWQ-4bit"
-container_name = "vllm_nemotron_super"
-max_model_len = 262144
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "qwen3_xml"
-trust_remote_code = true
-# nemotron_v3 reasoning parser exposes <think> tokens as reasoning_content in the API.
-extra_vllm_args = ["--reasoning-parser", "nemotron_v3"]
-
-# OpenAI GPT-OSS 20B — ultra-fast MoE (3.6B active / 20B total, MXFP4), ~14 GB on A100.
-# Native MXFP4 quantization; vLLM auto-detects it (no --quantization flag needed).
-# With only 14 GB weights, most of the 80 GB is available for KV cache (64K+ context).
-# tool_call_parser = "" disables --enable-auto-tool-choice: the llama3_json parser crashes
-# on gpt-oss responses (vLLM 0.17.1 adds token_ids to responses, breaking the parser API).
-[vllm.presets.gpt-oss-20b]
-model = "openai/gpt-oss-20b"
-container_name = "vllm_gpt_oss_20b"
-max_model_len = 65536
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = ""
-
-# OpenAI GPT-OSS 120B — powerful MoE (5.1B active / 117B total, MXFP4), ~65 GB on A100.
-# Hard architecture limit: max_position_embeddings=131072 in model config.json.
-# 131072 is the absolute ceiling — exceeding it causes NaN or CUDA OOB errors.
-# For sessions approaching this limit, start a fresh opencode conversation.
-# tool_call_parser = "" disables --enable-auto-tool-choice (same reason as gpt-oss-20b).
-[vllm.presets.gpt-oss-120b]
-model = "openai/gpt-oss-120b"
-container_name = "vllm_gpt_oss_120b"
-max_model_len = 131072
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = ""
-
-# Qwen2.5-Coder-32B-Instruct AWQ — best-in-class open coding model at 32B, ~18 GB on A100.
-# Official Qwen AWQ release; max_position_embeddings=32768 per model config.json.
-[vllm.presets.qwen25-coder-32b]
-model = "Qwen/Qwen2.5-Coder-32B-Instruct-AWQ"
-container_name = "vllm_qwen25_coder32b"
-max_model_len = 32768
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "hermes"
-
-# Qwen3-Coder-30B-A3B AWQ — Qwen3 generation coding MoE (3B active / 30B total), ~18 GB.
-# Note: model card warns of significant quality loss at 4-bit for this MoE architecture.
-[vllm.presets.qwen3-coder-30b]
-model = "QuantTrio/Qwen3-Coder-30B-A3B-Instruct-AWQ"
-container_name = "vllm_qwen3_coder30b"
-max_model_len = 65536
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "qwen3_coder"
-
-# DeepSeek-R1-Distill-Qwen-32B AWQ — R1 reasoning distillation of Qwen 32B, ~18 GB on A100.
-# Generates <think> reasoning tokens; --reasoning-parser deepseek_r1 exposes them in the API.
-# tool_call_parser="" disables tool calling (reasoning models don't support it reliably).
-[vllm.presets.deepseek-r1-32b]
-model = "casperhansen/deepseek-r1-distill-qwen-32b-awq"
-container_name = "vllm_deepseek_r1_32b"
-max_model_len = 32768
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = ""
-extra_vllm_args = ["--reasoning-parser", "deepseek_r1"]
-
-# Qwen3-32B AWQ — dense 32B reasoning model with extended context, ~18 GB on A100.
-# Native thinking mode; --reasoning-parser deepseek_r1 is compatible with Qwen3 thinking format.
-# tool_call_parser="" disables tool calling (reasoning models don't support it reliably).
-[vllm.presets.qwen3-32b]
-model = "Qwen/Qwen3-32B-AWQ"
-container_name = "vllm_qwen3_32b"
-max_model_len = 32768
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = ""
-extra_vllm_args = ["--reasoning-parser", "deepseek_r1"]
-
-# Devstral-Small-2507 AWQ — Mistral's coding agent model (~15 GB on A100).
-# Uses HF safetensors weights but Mistral tokenizer (tekken.json) and config (params.json).
-# --load_format mistral is NOT used: AWQ weights are in standard HF safetensors format.
-# --tokenizer_mode mistral and --config_format mistral handle the Mistral-native files.
-[vllm.presets.devstral]
-model = "cyankiwi/Devstral-Small-2507-AWQ-4bit"
-container_name = "vllm_devstral"
-max_model_len = 32768
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "mistral"
-extra_vllm_args = ["--tokenizer_mode", "mistral", "--config_format", "mistral"]
-
-[wireguard]
-auto_setup = true
-setup_script = "./wg1-setup.sh"
-
-[local_client]
-check_wg1_service = true
-interface_name = "wg1"
-config_path = "/etc/wireguard/wg1.conf"
diff --git a/snippets/hyperstack/hyperstack-vm1.toml b/snippets/hyperstack/hyperstack-vm1.toml
deleted file mode 100644
index 1b116bd..0000000
--- a/snippets/hyperstack/hyperstack-vm1.toml
+++ /dev/null
@@ -1,185 +0,0 @@
-[auth]
-api_key_file = "~/.hyperstack"
-
-[hyperstack]
-base_url = "https://infrahub-api.nexgencloud.com/v1"
-
-[state]
-# Separate state file for VM1 so vm1 and vm2 can be managed independently.
-file = ".hyperstack-vm1-state.json"
-
-[vm]
-name_prefix = "hyperstack1"
-hostname = "hyperstack1"
-environment_name = "snonux-ollama"
-
-# A100-80GB is the cost-first default for nemotron-3-super inference.
-# Switch this to n3-H100x1 if you want safer throughput and compatibility headroom.
-flavor_name = "n3-A100x1"
-image_name = "Ubuntu Server 24.04 LTS R570 CUDA 12.8 with Docker"
-assign_floating_ip = true
-create_bootable_volume = false
-enable_port_randomization = false
-labels = ["nemotron-3-super", "wireguard"]
-
-[ssh]
-username = "ubuntu"
-private_key_path = "~/.ssh/id_rsa"
-hyperstack_key_name = "earth"
-port = 22
-connect_timeout_sec = 10
-
-[network]
-wireguard_udp_port = 56710
-wireguard_subnet = "192.168.3.0/24"
-# VM1 gets the first server-side WireGuard IP (gateway address + 0).
-# earth (client) is 192.168.3.2; VM1 is 192.168.3.1; VM2 is 192.168.3.3.
-wireguard_server_ip = "192.168.3.1"
-# Secure default: "auto" resolves your current public egress IP to /32 at runtime.
-# Override with explicit CIDRs if you deploy from multiple networks or want broader access.
-allowed_ssh_cidrs = ["auto"]
-allowed_wireguard_cidrs = ["auto"]
-# Port 11434 is shared by both Ollama and vLLM for firewall compatibility.
-ollama_port = 11434
-# Port 4000: LiteLLM Anthropic-API proxy (used with vLLM).
-litellm_port = 4000
-
-[bootstrap]
-enable_guest_bootstrap = true
-install_wireguard = true
-configure_ufw = true
-configure_ollama_host = false
-
-[ollama]
-# Disabled in favour of vLLM; set install = true to switch back to Ollama.
-install = false
-models_dir = "/ephemeral/ollama/models"
-listen_host = "0.0.0.0:11434"
-gpu_overhead_mb = 2000
-num_parallel = 1
-context_length = 32768
-pull_models = ["nemotron-3-super"]
-
-# vLLM serves one model via Docker; LiteLLM translates Anthropic API → OpenAI.
-# VM1 defaults to nemotron-3-super; use 'model switch' to load any other preset.
-[vllm]
-install = true
-model = "cyankiwi/NVIDIA-Nemotron-3-Super-120B-A12B-AWQ-4bit"
-# HuggingFace model cache on ephemeral NVMe (fast; survives reboots on most providers).
-hug_cache_dir = "/ephemeral/hug"
-container_name = "vllm_nemotron_super"
-max_model_len = 262144
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-# NVIDIA Nemotron-3-Super uses the same XML tool call format as Qwen3 XML.
-tool_call_parser = "qwen3_xml"
-trust_remote_code = true
-extra_vllm_args = ["--reasoning-parser", "nemotron_v3"]
-# LiteLLM maps each entry to the vLLM model; add new Anthropic model IDs here.
-litellm_master_key = "sk-litellm-master"
-litellm_claude_model_names = [
- "claude-sonnet-4-20250514",
- "claude-opus-4-20250514",
- "claude-opus-4-6-20260604",
- "claude-haiku-3-5-20241022"
-]
-
-# Named model presets for 'ruby hyperstack.rb --config hyperstack-vm1.toml model switch <name>'.
-# Each preset overrides the matching [vllm] field; unset fields fall back to [vllm] defaults.
-
-[vllm.presets.qwen3-coder-next]
-model = "bullpoint/Qwen3-Coder-Next-AWQ-4bit"
-container_name = "vllm_qwen3"
-max_model_len = 262144
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "qwen3_coder"
-
-# NVIDIA Nemotron-3-Super-120B-A12B AWQ 4-bit — hybrid Mamba+MoE (12B active / 120B total).
-# ~60 GB weights on A100 80GB. Uses NoPE so context can be set to 1M; no YaRN needed.
-# Requires trust_remote_code=true for the nemotron_h architecture.
-[vllm.presets.nemotron-super]
-model = "cyankiwi/NVIDIA-Nemotron-3-Super-120B-A12B-AWQ-4bit"
-container_name = "vllm_nemotron_super"
-max_model_len = 262144
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "qwen3_xml"
-trust_remote_code = true
-extra_vllm_args = ["--reasoning-parser", "nemotron_v3"]
-
-# OpenAI GPT-OSS 20B — ultra-fast MoE (3.6B active / 20B total, MXFP4), ~14 GB on A100.
-[vllm.presets.gpt-oss-20b]
-model = "openai/gpt-oss-20b"
-container_name = "vllm_gpt_oss_20b"
-max_model_len = 65536
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = ""
-
-# OpenAI GPT-OSS 120B — powerful MoE (5.1B active / 117B total, MXFP4), ~65 GB on A100.
-# Hard architecture limit: max_position_embeddings=131072 in model config.json.
-[vllm.presets.gpt-oss-120b]
-model = "openai/gpt-oss-120b"
-container_name = "vllm_gpt_oss_120b"
-max_model_len = 131072
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = ""
-
-# Qwen2.5-Coder-32B-Instruct AWQ — best-in-class open coding model at 32B, ~18 GB on A100.
-[vllm.presets.qwen25-coder-32b]
-model = "Qwen/Qwen2.5-Coder-32B-Instruct-AWQ"
-container_name = "vllm_qwen25_coder32b"
-max_model_len = 32768
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "hermes"
-
-# Qwen3-Coder-30B-A3B AWQ — Qwen3 generation coding MoE (3B active / 30B total), ~18 GB.
-[vllm.presets.qwen3-coder-30b]
-model = "QuantTrio/Qwen3-Coder-30B-A3B-Instruct-AWQ"
-container_name = "vllm_qwen3_coder30b"
-max_model_len = 65536
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "qwen3_coder"
-
-# DeepSeek-R1-Distill-Qwen-32B AWQ — R1 reasoning distillation of Qwen 32B, ~18 GB on A100.
-[vllm.presets.deepseek-r1-32b]
-model = "casperhansen/deepseek-r1-distill-qwen-32b-awq"
-container_name = "vllm_deepseek_r1_32b"
-max_model_len = 32768
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = ""
-extra_vllm_args = ["--reasoning-parser", "deepseek_r1"]
-
-# Qwen3-32B AWQ — dense 32B reasoning model with extended context, ~18 GB on A100.
-[vllm.presets.qwen3-32b]
-model = "Qwen/Qwen3-32B-AWQ"
-container_name = "vllm_qwen3_32b"
-max_model_len = 32768
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = ""
-extra_vllm_args = ["--reasoning-parser", "deepseek_r1"]
-
-# Devstral-Small-2507 AWQ — Mistral's coding agent model (~15 GB on A100).
-[vllm.presets.devstral]
-model = "cyankiwi/Devstral-Small-2507-AWQ-4bit"
-container_name = "vllm_devstral"
-max_model_len = 32768
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "mistral"
-extra_vllm_args = ["--tokenizer_mode", "mistral", "--config_format", "mistral"]
-
-[wireguard]
-auto_setup = true
-setup_script = "./wg1-setup.sh"
-
-[local_client]
-check_wg1_service = true
-interface_name = "wg1"
-config_path = "/etc/wireguard/wg1.conf"
diff --git a/snippets/hyperstack/hyperstack-vm2.toml b/snippets/hyperstack/hyperstack-vm2.toml
deleted file mode 100644
index e8e9b00..0000000
--- a/snippets/hyperstack/hyperstack-vm2.toml
+++ /dev/null
@@ -1,182 +0,0 @@
-[auth]
-api_key_file = "~/.hyperstack"
-
-[hyperstack]
-base_url = "https://infrahub-api.nexgencloud.com/v1"
-
-[state]
-# Separate state file for VM2 so vm1 and vm2 can be managed independently.
-file = ".hyperstack-vm2-state.json"
-
-[vm]
-name_prefix = "hyperstack2"
-hostname = "hyperstack2"
-environment_name = "snonux-ollama"
-
-# A100-80GB is the cost-first default for qwen3-coder-next inference.
-# Switch this to n3-H100x1 if you want safer throughput and compatibility headroom.
-flavor_name = "n3-A100x1"
-image_name = "Ubuntu Server 24.04 LTS R570 CUDA 12.8 with Docker"
-assign_floating_ip = true
-create_bootable_volume = false
-enable_port_randomization = false
-labels = ["qwen3-coder-next", "wireguard"]
-
-[ssh]
-username = "ubuntu"
-private_key_path = "~/.ssh/id_rsa"
-hyperstack_key_name = "earth"
-port = 22
-connect_timeout_sec = 10
-
-[network]
-wireguard_udp_port = 56710
-wireguard_subnet = "192.168.3.0/24"
-# VM2 gets the third server-side WireGuard IP (skipping .2 which is the earth client).
-# earth (client) is 192.168.3.2; VM1 is 192.168.3.1; VM2 is 192.168.3.3.
-wireguard_server_ip = "192.168.3.3"
-# Secure default: "auto" resolves your current public egress IP to /32 at runtime.
-# Override with explicit CIDRs if you deploy from multiple networks or want broader access.
-allowed_ssh_cidrs = ["auto"]
-allowed_wireguard_cidrs = ["auto"]
-# Port 11434 is shared by both Ollama and vLLM for firewall compatibility.
-ollama_port = 11434
-# Port 4000: LiteLLM Anthropic-API proxy (used with vLLM).
-litellm_port = 4000
-
-[bootstrap]
-enable_guest_bootstrap = true
-install_wireguard = true
-configure_ufw = true
-configure_ollama_host = false
-
-[ollama]
-# Disabled in favour of vLLM; set install = true to switch back to Ollama.
-install = false
-models_dir = "/ephemeral/ollama/models"
-listen_host = "0.0.0.0:11434"
-gpu_overhead_mb = 2000
-num_parallel = 1
-context_length = 32768
-pull_models = ["qwen3-coder-next"]
-
-# vLLM serves one model via Docker; LiteLLM translates Anthropic API → OpenAI.
-# VM2 defaults to qwen3-coder-next; use 'model switch' to load any other preset.
-[vllm]
-install = true
-model = "bullpoint/Qwen3-Coder-Next-AWQ-4bit"
-# HuggingFace model cache on ephemeral NVMe (fast; survives reboots on most providers).
-hug_cache_dir = "/ephemeral/hug"
-container_name = "vllm_qwen3"
-max_model_len = 262144
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "qwen3_coder"
-# LiteLLM maps each entry to the vLLM model; add new Anthropic model IDs here.
-litellm_master_key = "sk-litellm-master"
-litellm_claude_model_names = [
- "claude-sonnet-4-20250514",
- "claude-opus-4-20250514",
- "claude-opus-4-6-20260604",
- "claude-haiku-3-5-20241022"
-]
-
-# Named model presets for 'ruby hyperstack.rb --config hyperstack-vm2.toml model switch <name>'.
-# Each preset overrides the matching [vllm] field; unset fields fall back to [vllm] defaults.
-
-[vllm.presets.qwen3-coder-next]
-model = "bullpoint/Qwen3-Coder-Next-AWQ-4bit"
-container_name = "vllm_qwen3"
-max_model_len = 262144
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "qwen3_coder"
-
-# NVIDIA Nemotron-3-Super-120B-A12B AWQ 4-bit — hybrid Mamba+MoE (12B active / 120B total).
-# ~60 GB weights on A100 80GB. Uses NoPE so context can be set to 1M; no YaRN needed.
-# Requires trust_remote_code=true for the nemotron_h architecture.
-[vllm.presets.nemotron-super]
-model = "cyankiwi/NVIDIA-Nemotron-3-Super-120B-A12B-AWQ-4bit"
-container_name = "vllm_nemotron_super"
-max_model_len = 262144
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "qwen3_xml"
-trust_remote_code = true
-extra_vllm_args = ["--reasoning-parser", "nemotron_v3"]
-
-# OpenAI GPT-OSS 20B — ultra-fast MoE (3.6B active / 20B total, MXFP4), ~14 GB on A100.
-[vllm.presets.gpt-oss-20b]
-model = "openai/gpt-oss-20b"
-container_name = "vllm_gpt_oss_20b"
-max_model_len = 65536
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = ""
-
-# OpenAI GPT-OSS 120B — powerful MoE (5.1B active / 117B total, MXFP4), ~65 GB on A100.
-# Hard architecture limit: max_position_embeddings=131072 in model config.json.
-[vllm.presets.gpt-oss-120b]
-model = "openai/gpt-oss-120b"
-container_name = "vllm_gpt_oss_120b"
-max_model_len = 131072
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = ""
-
-# Qwen2.5-Coder-32B-Instruct AWQ — best-in-class open coding model at 32B, ~18 GB on A100.
-[vllm.presets.qwen25-coder-32b]
-model = "Qwen/Qwen2.5-Coder-32B-Instruct-AWQ"
-container_name = "vllm_qwen25_coder32b"
-max_model_len = 32768
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "hermes"
-
-# Qwen3-Coder-30B-A3B AWQ — Qwen3 generation coding MoE (3B active / 30B total), ~18 GB.
-[vllm.presets.qwen3-coder-30b]
-model = "QuantTrio/Qwen3-Coder-30B-A3B-Instruct-AWQ"
-container_name = "vllm_qwen3_coder30b"
-max_model_len = 65536
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "qwen3_coder"
-
-# DeepSeek-R1-Distill-Qwen-32B AWQ — R1 reasoning distillation of Qwen 32B, ~18 GB on A100.
-[vllm.presets.deepseek-r1-32b]
-model = "casperhansen/deepseek-r1-distill-qwen-32b-awq"
-container_name = "vllm_deepseek_r1_32b"
-max_model_len = 32768
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = ""
-extra_vllm_args = ["--reasoning-parser", "deepseek_r1"]
-
-# Qwen3-32B AWQ — dense 32B reasoning model with extended context, ~18 GB on A100.
-[vllm.presets.qwen3-32b]
-model = "Qwen/Qwen3-32B-AWQ"
-container_name = "vllm_qwen3_32b"
-max_model_len = 32768
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = ""
-extra_vllm_args = ["--reasoning-parser", "deepseek_r1"]
-
-# Devstral-Small-2507 AWQ — Mistral's coding agent model (~15 GB on A100).
-[vllm.presets.devstral]
-model = "cyankiwi/Devstral-Small-2507-AWQ-4bit"
-container_name = "vllm_devstral"
-max_model_len = 32768
-gpu_memory_utilization = 0.92
-tensor_parallel_size = 1
-tool_call_parser = "mistral"
-extra_vllm_args = ["--tokenizer_mode", "mistral", "--config_format", "mistral"]
-
-[wireguard]
-auto_setup = true
-setup_script = "./wg1-setup.sh"
-
-[local_client]
-check_wg1_service = true
-interface_name = "wg1"
-config_path = "/etc/wireguard/wg1.conf"
diff --git a/snippets/hyperstack/hyperstack.rb b/snippets/hyperstack/hyperstack.rb
deleted file mode 100755
index 7cd817d..0000000
--- a/snippets/hyperstack/hyperstack.rb
+++ /dev/null
@@ -1,2731 +0,0 @@
-#!/usr/bin/env ruby
-# frozen_string_literal: true
-
-begin
- require 'bundler/setup'
-rescue LoadError, Gem::GemNotFoundException, Gem::LoadError, Errno::ENOENT
- nil
-end
-
-require 'json'
-require 'fileutils'
-require 'net/http'
-require 'open3'
-require 'optparse'
-require 'ipaddr'
-require 'shellwords'
-require 'socket'
-require 'time'
-require 'timeout'
-
-begin
- require 'toml-rb'
-rescue LoadError
- warn "Missing dependency: toml-rb. Run `bundle install` in #{__dir__} first."
- exit 2
-end
-
-module HyperstackVM
- class Error < StandardError; end
-
- class ConfigLoader
- attr_reader :path
-
- def self.load(path)
- expanded = File.expand_path(path)
- raise Error, "Config file not found: #{expanded}" unless File.exist?(expanded)
-
- raw = TomlRB.load_file(expanded)
- new(raw, expanded)
- rescue TomlRB::ParseError => e
- raise Error, "Failed to parse TOML config #{expanded}: #{e.message}"
- end
-
- def initialize(raw, path)
- @path = path
- @data = deep_merge(DEFAULTS, raw || {})
- validate!
- end
-
- def config
- Config.new(@data, @path)
- end
-
- private
-
- DEFAULTS = {