From 14da24c02bc40456a0d23cdc3385872a491cb1e0 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 14:12:46 +0000 Subject: [PATCH 01/15] Add GPU setup research for rack server AI workloads Research comparing P40, A2000, T4, and M40 GPUs for LLM inference and image generation in Dell R720/R730 rack servers. Includes benchmarks, compatibility notes, pricing, and recommendations. https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- docs/gpu-setup-research.md | 117 +++++++++++++++++++++++++++++++++++++ 1 file changed, 117 insertions(+) create mode 100644 docs/gpu-setup-research.md diff --git a/docs/gpu-setup-research.md b/docs/gpu-setup-research.md new file mode 100644 index 0000000..c0a9483 --- /dev/null +++ b/docs/gpu-setup-research.md @@ -0,0 +1,117 @@ +# GPU Setup Research: Rack Server AI Workloads + +## Goal +Cost-efficient rack-mountable GPU setup for: +1. **LLM coding inference** (Ollama + qwen2.5-coder models) +2. **Image generation** (ComfyUI / InvokeAI with Stable Diffusion) + +Target servers: Dell R720/R730 or HP DL380 equivalent (2U rack) + +## GPU Candidates Compared + +| Feature | Tesla P40 | RTX A2000 12GB | Tesla T4 | Tesla M40 | +|---------|-----------|----------------|----------|-----------| +| Architecture | Pascal (2016) | Ampere (2020) | Turing (2018) | Maxwell (2015) | +| VRAM | 24GB GDDR5 | 12GB GDDR6 | 16GB GDDR6 | 24GB GDDR5 | +| Tensor Cores | No | Yes | Yes | No | +| TDP | 250W | 70W | 70W | 250W | +| Compute Capability | 6.1 | 8.6 | 7.5 | 5.2 | +| Cooling | Passive (needs fan) | Active (blower) | Passive (needs fan) | Passive (needs fan) | +| Aux Power Required | Yes (8-pin) | No (bus-powered) | No (bus-powered) | Yes (8-pin) | +| Used Price (2026) | ~$300 | ~$490 | ~$800+ | ~$100-150 | +| PCIe | Gen3 x16 | Gen4 x16 | Gen3 x16 | Gen3 x16 | + +## Performance Benchmarks + +### LLM Inference (tokens/sec via Ollama) + +| Model | A2000 12GB | P40 24GB | +|-------|------------|----------| +| qwen2.5:14b | ~21 t/s | ~17 t/s | +| llama3.2:3b-Q8 | ~50 t/s | ~40 t/s | +| llama3.2:3b-Q4 | ~60 t/s | ~48 t/s | + +### Image Generation (ComfyUI) + +| GPU | SDXL 20 steps | Notes | +|-----|---------------|-------| +| P40 | ~49 seconds | Requires `--force-fp32` flag | +| A2000 | ~16 seconds | ~3x faster than P40 | +| RTX 4090 | ~3 seconds | Reference (16x faster than P40) | + +## Rack Server Compatibility + +### RTX A2000 in R720/R730 +- **Physical fit**: Yes. Dual-slot, low-profile, 167mm length +- **Power**: 70W bus-powered, no aux cable needed. Must use 75W slots (slots 4-7 on R720) +- **Cooling**: Blower-style fan exhausts out bracket — ideal for rack airflow +- **Requirement**: Dual CPUs needed for GPU PCIe slots +- **Confirmed working** in Dell R740XD (similar architecture) +- Third-party single-slot cooler available from n3rdware for tighter fits + +### Tesla P40 in R720/R730 +- **Physical fit**: Yes. Full-length, single-slot, designed for rack servers +- **Power**: 250W, requires 8-pin aux power. Needs GPU enablement kit (power cables + riser) +- **Cooling**: Passive — relies on server chassis fans (which R720/R730 have) +- **Requirement**: Dual CPUs, redundant 1100W PSUs recommended +- **Natively supported** in these servers + +### R720 vs R730 +- R720: PCIe Gen2 (not a bottleneck for LLM inference, which is VRAM-bound) +- R730: PCIe Gen3, generally preferred +- Both support up to 2x double-wide or 4x single-wide GPUs + +## Setup Options Analysis + +### Option A: P40 + A2000 (~$800) +- P40 for LLM coding (24GB fits 14B models) +- A2000 for image gen (3x faster than P40, tensor cores) +- Total power: ~320W GPU +- **Best performance split** + +### Option B: Single P40 (~$300) +- Use for both LLM and image gen +- 24GB handles 14B models well +- Image gen slow but usable (~49s SDXL) +- **Best budget option**, upgrade later + +### Option C: Dual P40 (~$600) +- One dedicated to LLM, one to image gen +- 500W total GPU power draw +- Both need passive cooling (server fans handle this) +- Image gen still slow on P40 + +### Option D: P40 + P4 (~$350) +- P40 for LLM (24GB) +- P4 for light image gen (8GB, passive, 75W) +- P4 limited to SD 1.5 and smaller models +- Low total power: ~325W + +## Recommendations + +### Budget Priority (under $500): Single P40 +Start with one P40. Your local-ai stack auto-selects qwen2.5:14b at 24GB VRAM. +Image gen works with `--force-fp32` flag. Add a second GPU later. + +### Performance Priority (~$800): P40 + A2000 +Best of both worlds. P40 handles LLM inference with full 24GB VRAM. +A2000 handles image gen 3x faster with modern Ampere architecture. +Both fit in R720/R730. A2000 blower cooling works well in rack. + +### Configuration Notes for local-ai stack +- Set `INVOKEAI_PRECISION=float32` when using P40 for image gen +- Ollama `OLLAMA_NUM_GPU=999` works with both GPUs +- For dual GPU: assign specific GPUs via `CUDA_VISIBLE_DEVICES` +- P40 needs `--force-fp32` in ComfyUI launch args + +## Sources +- [NVIDIA RTX A2000 Datasheet](https://www.nvidia.com/content/dam/en-zz/Solutions/design-visualization/rtx-a2000/nvidia-rtx-a2000-datasheet-1987439-r5.pdf) +- [Lenovo ThinkSystem RTX A2000 Product Guide](https://lenovopress.lenovo.com/lp1919-thinksystem-nvidia-rtx-a2000-12gb-pcie-active-gpu) +- [n3rdware Single-Slot A2000 Cooler](https://n3rdware.com/gpu-coolers/single-slot-rtx-a2000-cooler) +- [Dell R730 Owner's Manual - Expansion Cards](https://www.dell.com/support/manuals/en-us/poweredge-r730/r730_ompublication/expansion-card-installation-guidelines) +- [Dell R720 Owner's Manual - Expansion Cards](https://www.dell.com/support/manuals/en-us/poweredge-r720/720720xdom/expansion-card-installation-guidelines) +- [ComfyUI GPU Benchmarks Discussion](https://github.com/Comfy-Org/ComfyUI/discussions/2970) +- [ComfyUI P40 FP32 Issue](https://github.com/Comfy-Org/ComfyUI/issues/4363) +- [How to Use Tesla P40 Guide](https://github.com/JingShing/How-to-use-tesla-p40) +- [Build a Local LLM Server Under $1000](https://sanj.dev/post/affordable-ai-hardware-local-llms) +- [Best Budget GPUs for AI 2026](https://techtactician.com/best-budget-gpus-for-local-ai-workflows/) From a4e0ebc1bdbb3f09e27d84d33899876551905692 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 14:21:02 +0000 Subject: [PATCH 02/15] Update GPU research with real March 2026 prices and 32B model analysis - Corrected all GPU prices to actual eBay/Newegg listings as of March 22, 2026 - Added analysis of 32B coding models (Qwen2.5-Coder 32B, Qwen3.5 27B) on 24GB VRAM - Added honest comparison of local LLM quality vs Claude Code - Revised recommendations: dual P40 ($400-500) or P40+T4 ($400-550) - Added configuration notes for 32B models, dual-GPU, and newer Qwen3 models - RTX A4000 at $700+ is too expensive for this build https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- docs/gpu-setup-research.md | 182 +++++++++++++++++++++++++++---------- 1 file changed, 133 insertions(+), 49 deletions(-) diff --git a/docs/gpu-setup-research.md b/docs/gpu-setup-research.md index c0a9483..cfcdd58 100644 --- a/docs/gpu-setup-research.md +++ b/docs/gpu-setup-research.md @@ -1,35 +1,69 @@ # GPU Setup Research: Rack Server AI Workloads +*Last updated: March 22, 2026* + ## Goal Cost-efficient rack-mountable GPU setup for: -1. **LLM coding inference** (Ollama + qwen2.5-coder models) -2. **Image generation** (ComfyUI / InvokeAI with Stable Diffusion) +1. **LLM coding inference** — Run 32B parameter coding models (Qwen2.5-Coder 32B, Qwen3 32B) for quality closest to Claude/GPT-4o +2. **Image generation** — ComfyUI / InvokeAI with Stable Diffusion SDXL / Flux Target servers: Dell R720/R730 or HP DL380 equivalent (2U rack) +## Why 24GB VRAM Matters for Coding + +32B parameter models are the sweet spot for local coding quality: +- **Qwen2.5-Coder 32B** scores 73.7 on Aider (between GPT-4o at 71% and Claude 3.5 Haiku at 75%) +- Competitive with GPT-4o on EvalPlus, LiveCodeBench, BigCodeBench +- At Q4_K_M quantization (~20GB), fits on 24GB with room for small context +- 14B models are noticeably worse for complex coding tasks + +Newer models also fit 24GB: +- **Qwen 3.5 27B** (dense): ties GPT-5 mini on SWE-bench (72.4%), ~16GB at Q4 +- **Qwen3-Coder 30B-A3B** (MoE): only 3.3B active params, fast inference, fits easily +- **Qwen2.5-Coder 32B** remains the FIM (autocomplete) king: 92.7% HumanEval + +### Honest Assessment: Local vs Claude Code +Nothing local approaches Claude Opus 4.6 quality for complex multi-file agentic coding. +These 32B models are competitive with **GPT-4o** — a tier below Claude Sonnet, two tiers below Opus. +Best strategy: use local models for routine tasks, save Claude credits for hard problems. + ## GPU Candidates Compared -| Feature | Tesla P40 | RTX A2000 12GB | Tesla T4 | Tesla M40 | +| Feature | Tesla P40 | RTX A2000 12GB | Tesla T4 | RTX A4000 | |---------|-----------|----------------|----------|-----------| -| Architecture | Pascal (2016) | Ampere (2020) | Turing (2018) | Maxwell (2015) | -| VRAM | 24GB GDDR5 | 12GB GDDR6 | 16GB GDDR6 | 24GB GDDR5 | -| Tensor Cores | No | Yes | Yes | No | -| TDP | 250W | 70W | 70W | 250W | -| Compute Capability | 6.1 | 8.6 | 7.5 | 5.2 | -| Cooling | Passive (needs fan) | Active (blower) | Passive (needs fan) | Passive (needs fan) | -| Aux Power Required | Yes (8-pin) | No (bus-powered) | No (bus-powered) | Yes (8-pin) | -| Used Price (2026) | ~$300 | ~$490 | ~$800+ | ~$100-150 | -| PCIe | Gen3 x16 | Gen4 x16 | Gen3 x16 | Gen3 x16 | +| Architecture | Pascal (2016) | Ampere (2020) | Turing (2018) | Ampere (2020) | +| VRAM | 24GB GDDR5 | 12GB GDDR6 | 16GB GDDR6 | 16GB GDDR6 | +| Tensor Cores | No | Yes | Yes | Yes | +| TDP | 250W | 70W | 70W | 140W | +| Compute Capability | 6.1 | 8.6 | 7.5 | 8.6 | +| Cooling | Passive (server fans) | Active (blower) | Passive (server fans) | Active (single-slot) | +| Aux Power Required | Yes (8-pin) | No (bus-powered) | No (bus-powered) | Yes (6-pin) | +| PCIe | Gen3 x16 | Gen4 x16 | Gen3 x16 | Gen4 x16 | +| Can run 32B Q4? | Yes (tight) | No (12GB) | No (16GB) | No (16GB) | + +## Current Prices (March 22, 2026) + +| GPU | VRAM | Price Range | Best Deals | Notes | +|-----|------|-------------|------------|-------| +| **Tesla P40** | 24GB | $150-320 | Newegg refurb $219-270; eBay used $150-200 | Best VRAM/$ ratio | +| **RTX A2000 12GB** | 12GB | $250-535 | eBay used ~$250-350; one listing at $490 | New retail ~$535 | +| **Tesla T4** | 16GB | $150-350 | eBay used $150-250 | Great power efficiency | +| **RTX A4000** | 16GB | $700-750+ | eBay used ~$700; new $720+ | Too expensive for this build | + +Sources: eBay active listings, Newegg, Lowpi.com, Pangoly price history (all checked March 2026) ## Performance Benchmarks ### LLM Inference (tokens/sec via Ollama) -| Model | A2000 12GB | P40 24GB | -|-------|------------|----------| -| qwen2.5:14b | ~21 t/s | ~17 t/s | -| llama3.2:3b-Q8 | ~50 t/s | ~40 t/s | -| llama3.2:3b-Q4 | ~60 t/s | ~48 t/s | +| Model | A2000 12GB | P40 24GB | RTX 3090 24GB (ref) | +|-------|------------|----------|---------------------| +| qwen2.5:14b | ~21 t/s | ~17 t/s | — | +| qwen2.5-coder:32b Q4_K_M | Won't fit | ~5-12 t/s (est.) | ~37-40 t/s | +| llama3.2:3b-Q4 | ~60 t/s | ~48 t/s | — | + +**P40 reality check for 32B**: The model barely fits (~22GB for weights), leaving only ~2GB for KV cache. +Context window will be severely limited. Expect 5-12 tok/s — usable for short prompts, painful for long sessions. ### Image Generation (ComfyUI) @@ -55,6 +89,7 @@ Target servers: Dell R720/R730 or HP DL380 equivalent (2U rack) - **Cooling**: Passive — relies on server chassis fans (which R720/R730 have) - **Requirement**: Dual CPUs, redundant 1100W PSUs recommended - **Natively supported** in these servers +- Users report power-limiting to 140W with little performance impact ### R720 vs R730 - R720: PCIe Gen2 (not a bottleneck for LLM inference, which is VRAM-bound) @@ -63,46 +98,88 @@ Target servers: Dell R720/R730 or HP DL380 equivalent (2U rack) ## Setup Options Analysis -### Option A: P40 + A2000 (~$800) -- P40 for LLM coding (24GB fits 14B models) -- A2000 for image gen (3x faster than P40, tensor cores) +### Option A: Dual P40 (~$400-500) +- **P40 #1**: Qwen2.5-Coder 32B (tight fit, 5-12 tok/s, limited context) +- **P40 #2**: Image gen with ComfyUI (`--force-fp32`, ~49s/image SDXL) +- Or: split 32B model across both P40s via tensor parallelism for better speed +- Both are native rack server GPUs (passive, designed for R720/R730) +- Total power: ~500W GPU (can power-limit to ~280W) +- **Best value for 32B + image gen** + +### Option B: P40 + A2000 (~$550-810) +- P40 for 32B coding model (24GB, tight but works) +- A2000 for image gen (3x faster than P40, tensor cores, 70W) - Total power: ~320W GPU -- **Best performance split** +- A2000 at $490 is overpriced — shop for $250-350 range +- **Better image gen speed vs Option A** -### Option B: Single P40 (~$300) -- Use for both LLM and image gen -- 24GB handles 14B models well -- Image gen slow but usable (~49s SDXL) -- **Best budget option**, upgrade later +### Option C: Single P40 (~$200-300) +- Run 32B coding model OR image gen (not both simultaneously) +- 32B model leaves no room for anything else in VRAM +- Swap between tasks by unloading/loading models +- **Cheapest entry point**, upgrade later -### Option C: Dual P40 (~$600) -- One dedicated to LLM, one to image gen -- 500W total GPU power draw -- Both need passive cooling (server fans handle this) -- Image gen still slow on P40 - -### Option D: P40 + P4 (~$350) -- P40 for LLM (24GB) -- P4 for light image gen (8GB, passive, 75W) -- P4 limited to SD 1.5 and smaller models -- Low total power: ~325W +### Option D: P40 (coding) + T4 (image gen) (~$400-550) +- P40: 32B coding model (24GB) +- T4: Image gen with tensor cores, 16GB, 70W, passive +- T4 is faster than P40 for image gen (Turing tensor cores, native FP16) +- Both passive-cooled = true rack-native +- **Good balance of price and image gen speed** ## Recommendations -### Budget Priority (under $500): Single P40 -Start with one P40. Your local-ai stack auto-selects qwen2.5:14b at 24GB VRAM. -Image gen works with `--force-fp32` flag. Add a second GPU later. +### If 32B coding quality is the priority: Dual P40 ($400-500) +Two P40s give you 48GB total. Run 32B coding model on one, image gen on the other. +Or split the model across both for faster inference. Both fit natively in R720/R730. -### Performance Priority (~$800): P40 + A2000 -Best of both worlds. P40 handles LLM inference with full 24GB VRAM. -A2000 handles image gen 3x faster with modern Ampere architecture. -Both fit in R720/R730. A2000 blower cooling works well in rack. +### If image gen speed matters equally: P40 + T4 ($400-550) +P40 for 32B coding, T4 for image gen. Both passive-cooled, both rack-native. +T4 has tensor cores + FP16 for much faster image gen than P40. -### Configuration Notes for local-ai stack -- Set `INVOKEAI_PRECISION=float32` when using P40 for image gen -- Ollama `OLLAMA_NUM_GPU=999` works with both GPUs -- For dual GPU: assign specific GPUs via `CUDA_VISIBLE_DEVICES` -- P40 needs `--force-fp32` in ComfyUI launch args +### Cheapest possible: Single P40 ($200-300) +Run 32B coding model with limited context. Swap to image gen when needed. +Upgrade to dual-GPU later. + +## Configuration Notes for local-ai stack + +### For 32B models on P40 +```bash +# In Ollama environment +OLLAMA_NUM_GPU=999 +OLLAMA_NUM_CTX=4096 # Keep context small to fit in remaining VRAM +OLLAMA_KEEP_ALIVE=24h + +# Pull the right quantization +ollama pull qwen2.5-coder:32b-instruct-q4_K_M +``` + +### For dual-GPU setup +```bash +# Assign GPU 0 to Ollama (coding), GPU 1 to InvokeAI (image gen) +# In docker-compose.yml for Ollama: +CUDA_VISIBLE_DEVICES=0 + +# In docker-compose.yml for InvokeAI: +CUDA_VISIBLE_DEVICES=1 +``` + +### For P40 image gen +```bash +# InvokeAI +INVOKEAI_PRECISION=float32 + +# ComfyUI launch args +--force-fp32 +``` + +### For newer coding models (2026) +```bash +# Qwen 3.5 27B — fits easily on 24GB at Q4, better quality than 2.5 +ollama pull qwen3.5:27b + +# Qwen3-Coder 30B-A3B MoE — fast inference, agentic coding +ollama pull qwen3-coder:30b +``` ## Sources - [NVIDIA RTX A2000 Datasheet](https://www.nvidia.com/content/dam/en-zz/Solutions/design-visualization/rtx-a2000/nvidia-rtx-a2000-datasheet-1987439-r5.pdf) @@ -115,3 +192,10 @@ Both fit in R720/R730. A2000 blower cooling works well in rack. - [How to Use Tesla P40 Guide](https://github.com/JingShing/How-to-use-tesla-p40) - [Build a Local LLM Server Under $1000](https://sanj.dev/post/affordable-ai-hardware-local-llms) - [Best Budget GPUs for AI 2026](https://techtactician.com/best-budget-gpus-for-local-ai-workflows/) +- [Tesla P40 for Local LLMs 2026](https://like2byte.com/tesla-p40-local-llm-guide/) +- [Best Local LLMs for 24GB VRAM 2026](https://localllm.in/blog/best-local-llms-24gb-vram) +- [Best Coding Models 2026](https://localvram.com/en/guides/best-coding-models/) +- [Ollama VRAM Requirements Guide](https://localllm.in/blog/ollama-vram-requirements-for-local-llms) +- [Local LLMs That Can Replace Claude Code](https://agentnativedev.medium.com/local-llms-that-can-replace-claude-code-6f5b6cac93bf) +- [Qwen2.5-Coder 32B on Ollama](https://ollama.com/library/qwen2.5-coder:32b-instruct-q4_K_M) +- [Qwen2.5-Coder 32B HuggingFace Discussion](https://huggingface.co/Qwen/Qwen2.5-Coder-32B-Instruct/discussions/28) From a14a45bce6dd8daf3fffbd37bc8e8ce9a3ec4c5d Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 14:29:58 +0000 Subject: [PATCH 03/15] Major update: 48GB GPU analysis with RTX 8000 as best value - Added full 48GB GPU market comparison (RTX 8000, A40, A6000, L40, RTX 6000 Ada) - Quadro RTX 8000 Passive at $750-1,400 is 4-5x cheaper than alternatives - Added RTX 8000 LLM benchmarks (34 t/s on 30B models at 8K context) - Explained why 48GB >> 24GB for coding: context window is the bottleneck - Added 2026 coding model landscape (Qwen3.5 27B, Qwen3-Coder, etc.) - Revised recommendations: RTX 8000 as primary, dual P40 as budget alt - Updated config notes for 48GB (32K context, higher quantization options) - All prices verified from real listings as of March 22, 2026 https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- docs/gpu-setup-research.md | 245 ++++++++++++++++++++----------------- 1 file changed, 133 insertions(+), 112 deletions(-) diff --git a/docs/gpu-setup-research.md b/docs/gpu-setup-research.md index cfcdd58..14ce3af 100644 --- a/docs/gpu-setup-research.md +++ b/docs/gpu-setup-research.md @@ -4,156 +4,176 @@ ## Goal Cost-efficient rack-mountable GPU setup for: -1. **LLM coding inference** — Run 32B parameter coding models (Qwen2.5-Coder 32B, Qwen3 32B) for quality closest to Claude/GPT-4o +1. **LLM coding inference** — Run 32B+ parameter coding models with maximum context windows 2. **Image generation** — ComfyUI / InvokeAI with Stable Diffusion SDXL / Flux Target servers: Dell R720/R730 or HP DL380 equivalent (2U rack) -## Why 24GB VRAM Matters for Coding +## Why 48GB VRAM is the Right Target -32B parameter models are the sweet spot for local coding quality: -- **Qwen2.5-Coder 32B** scores 73.7 on Aider (between GPT-4o at 71% and Claude 3.5 Haiku at 75%) -- Competitive with GPT-4o on EvalPlus, LiveCodeBench, BigCodeBench -- At Q4_K_M quantization (~20GB), fits on 24GB with room for small context -- 14B models are noticeably worse for complex coding tasks +### The Problem with 24GB +32B coding models at Q4_K_M quantization use ~20GB of weights, leaving only ~4GB for KV cache +on a 24GB card. This severely limits context window size — the key ingredient for complex +coding sessions where the model needs to understand your entire codebase. -Newer models also fit 24GB: -- **Qwen 3.5 27B** (dense): ties GPT-5 mini on SWE-bench (72.4%), ~16GB at Q4 -- **Qwen3-Coder 30B-A3B** (MoE): only 3.3B active params, fast inference, fits easily -- **Qwen2.5-Coder 32B** remains the FIM (autocomplete) king: 92.7% HumanEval +### What 48GB Unlocks +- **32B models at higher quantization** (Q6_K/Q8_0) = better output quality +- **28GB+ free for KV cache** = massive context windows (32K+ tokens) +- **70B models** in aggressive quantization (~12 t/s but functional) +- **Simultaneous model loading** — coding model + image gen model at once +- Room for future larger models without hardware changes + +## Best Local Coding Models (2026) + +| Model | Size at Q4_K_M | Quality | Notes | +|-------|---------------|---------|-------| +| **Qwen2.5-Coder 32B** | ~20GB | 73.7 Aider (≈ GPT-4o) | FIM king, 92.7% HumanEval | +| **Qwen 3.5 27B** | ~16GB | 72.4% SWE-bench (ties GPT-5 mini) | 262K context, multimodal | +| **Qwen3-Coder 30B-A3B** (MoE) | ~18GB | #1 SWE-rebench (64.6%) | Only 3.3B active, very fast | +| **Qwen3-Coder-Next 80B** (MoE) | needs 64GB+ RAM offload | Beats Claude Opus 4.6 on SWE-rebench | Hybrid attention, 256K context | ### Honest Assessment: Local vs Claude Code Nothing local approaches Claude Opus 4.6 quality for complex multi-file agentic coding. These 32B models are competitive with **GPT-4o** — a tier below Claude Sonnet, two tiers below Opus. Best strategy: use local models for routine tasks, save Claude credits for hard problems. -## GPU Candidates Compared +## 48GB GPU Market (March 22, 2026 — Real Prices) -| Feature | Tesla P40 | RTX A2000 12GB | Tesla T4 | RTX A4000 | -|---------|-----------|----------------|----------|-----------| -| Architecture | Pascal (2016) | Ampere (2020) | Turing (2018) | Ampere (2020) | -| VRAM | 24GB GDDR5 | 12GB GDDR6 | 16GB GDDR6 | 16GB GDDR6 | -| Tensor Cores | No | Yes | Yes | Yes | -| TDP | 250W | 70W | 70W | 140W | -| Compute Capability | 6.1 | 8.6 | 7.5 | 8.6 | -| Cooling | Passive (server fans) | Active (blower) | Passive (server fans) | Active (single-slot) | -| Aux Power Required | Yes (8-pin) | No (bus-powered) | No (bus-powered) | Yes (6-pin) | -| PCIe | Gen3 x16 | Gen4 x16 | Gen3 x16 | Gen4 x16 | -| Can run 32B Q4? | Yes (tight) | No (12GB) | No (16GB) | No (16GB) | +| GPU | Arch | Used Price | TDP | Cooling | Tensor Cores | Mem BW | +|-----|------|------------|-----|---------|--------------|--------| +| **Quadro RTX 8000** | Turing (2018) | **$750–1,400** | 260W | Passive variant | Yes (576) | 672 GB/s | +| **A40** | Ampere (2020) | **~$5,050+** | 300W | Passive | Yes (336 3rd-gen) | 696 GB/s | +| **RTX A6000** | Ampere (2020) | **~$5,400+** | 300W | Active (blower) | Yes (336 3rd-gen) | 768 GB/s | +| **L40** | Ada (2022) | **~$6,500+** | 300W | Passive | Yes (568 4th-gen) | 864 GB/s | +| **RTX 6000 Ada** | Ada (2022) | **~$6,500+** | 300W | Active | Yes (568 4th-gen) | 960 GB/s | -## Current Prices (March 22, 2026) +Sources: eBay active/sold listings, GPUPoet price tracking, Pangoly, CamelCamelCamel (all March 2026) + +### Winner: Quadro RTX 8000 Passive ($750–1,400) + +The RTX 8000 is **4–5x cheaper** than every other 48GB option. The passive variant is +purpose-built for rack servers — no fan, relies on chassis airflow, designed for 24/7 operation +in 2U/4U systems. + +Key advantages over the P40: +- **48GB vs 24GB** — room for models + massive context +- **Has Tensor Cores** (576 Turing) — native FP16, no `--force-fp32` hacks for image gen +- **NVLink support** — pair two for 96GB combined (100 GB/s bidirectional) +- 10W idle power draw + +## RTX 8000 Performance Benchmarks + +### LLM Inference (Exllama, 5.0 bpw quantization) + +| Model | Context | Prompt Processing | Generation | +|-------|---------|-------------------|------------| +| Qwen3 30B-A3B (MoE) | 8K | 950 t/s | **34 t/s** | +| Qwen3 30B-A3B (MoE) | 16K | 673 t/s | **21 t/s** | +| Qwen3 30B-A3B (MoE) | 32K | 345 t/s | **11 t/s** | +| Llama 3.3 70B | short | 36 t/s | **13 t/s** | +| Llama 3.1 8B | — | — | **72 t/s** | + +### Compared to P40 (24GB) + +| Metric | P40 (24GB) | RTX 8000 (48GB) | +|--------|-----------|-----------------| +| 32B model fit | Barely (~2GB free) | Comfortable (~28GB free) | +| 32B generation speed | ~5-12 t/s (est.) | ~20-34 t/s | +| Max practical context | ~4K tokens | **32K+ tokens** | +| Image gen (SDXL) | ~49s (`--force-fp32`) | Faster (native FP16) | +| Rack server ready | Yes (passive) | Yes (passive variant) | + +### Image Generation +The RTX 8000 has Turing Tensor Cores with native FP16 support. Unlike the P40, it does NOT +need `--force-fp32` workarounds. Image gen performance is significantly better than the P40, +though still behind Ampere/Ada cards. + +## 24GB GPU Options (Previous Research — Still Valid for Tighter Budgets) | GPU | VRAM | Price Range | Best Deals | Notes | |-----|------|-------------|------------|-------| -| **Tesla P40** | 24GB | $150-320 | Newegg refurb $219-270; eBay used $150-200 | Best VRAM/$ ratio | -| **RTX A2000 12GB** | 12GB | $250-535 | eBay used ~$250-350; one listing at $490 | New retail ~$535 | +| **Tesla P40** | 24GB | $150-320 | Newegg refurb $219-270; eBay used $150-200 | Best VRAM/$ at 24GB | +| **RTX A2000 12GB** | 12GB | $250-535 | eBay used ~$250-350; one listing at $490 | Can't run 32B models | | **Tesla T4** | 16GB | $150-350 | eBay used $150-250 | Great power efficiency | -| **RTX A4000** | 16GB | $700-750+ | eBay used ~$700; new $720+ | Too expensive for this build | - -Sources: eBay active listings, Newegg, Lowpi.com, Pangoly price history (all checked March 2026) - -## Performance Benchmarks - -### LLM Inference (tokens/sec via Ollama) - -| Model | A2000 12GB | P40 24GB | RTX 3090 24GB (ref) | -|-------|------------|----------|---------------------| -| qwen2.5:14b | ~21 t/s | ~17 t/s | — | -| qwen2.5-coder:32b Q4_K_M | Won't fit | ~5-12 t/s (est.) | ~37-40 t/s | -| llama3.2:3b-Q4 | ~60 t/s | ~48 t/s | — | - -**P40 reality check for 32B**: The model barely fits (~22GB for weights), leaving only ~2GB for KV cache. -Context window will be severely limited. Expect 5-12 tok/s — usable for short prompts, painful for long sessions. - -### Image Generation (ComfyUI) - -| GPU | SDXL 20 steps | Notes | -|-----|---------------|-------| -| P40 | ~49 seconds | Requires `--force-fp32` flag | -| A2000 | ~16 seconds | ~3x faster than P40 | -| RTX 4090 | ~3 seconds | Reference (16x faster than P40) | +| **RTX A4000** | 16GB | $700-750+ | eBay used ~$700; new $720+ | Too expensive for 16GB | ## Rack Server Compatibility +### Quadro RTX 8000 Passive in R720/R730 +- **Physical fit**: Full-length, dual-slot — fits in GPU riser slots +- **Power**: 260W, requires 8-pin aux power + GPU enablement kit +- **Cooling**: Passive — relies on server chassis fans (same as P40) +- **Requirement**: Dual CPUs, redundant 1100W PSUs recommended +- **NVLink**: Can pair two RTX 8000s for 96GB combined VRAM +- Very similar physical/power requirements to the Tesla P40 + ### RTX A2000 in R720/R730 - **Physical fit**: Yes. Dual-slot, low-profile, 167mm length - **Power**: 70W bus-powered, no aux cable needed. Must use 75W slots (slots 4-7 on R720) - **Cooling**: Blower-style fan exhausts out bracket — ideal for rack airflow - **Requirement**: Dual CPUs needed for GPU PCIe slots - **Confirmed working** in Dell R740XD (similar architecture) -- Third-party single-slot cooler available from n3rdware for tighter fits ### Tesla P40 in R720/R730 - **Physical fit**: Yes. Full-length, single-slot, designed for rack servers -- **Power**: 250W, requires 8-pin aux power. Needs GPU enablement kit (power cables + riser) -- **Cooling**: Passive — relies on server chassis fans (which R720/R730 have) +- **Power**: 250W, requires 8-pin aux power. Needs GPU enablement kit +- **Cooling**: Passive — relies on server chassis fans - **Requirement**: Dual CPUs, redundant 1100W PSUs recommended - **Natively supported** in these servers -- Users report power-limiting to 140W with little performance impact ### R720 vs R730 - R720: PCIe Gen2 (not a bottleneck for LLM inference, which is VRAM-bound) - R730: PCIe Gen3, generally preferred - Both support up to 2x double-wide or 4x single-wide GPUs -## Setup Options Analysis +## Recommended Setups -### Option A: Dual P40 (~$400-500) -- **P40 #1**: Qwen2.5-Coder 32B (tight fit, 5-12 tok/s, limited context) -- **P40 #2**: Image gen with ComfyUI (`--force-fp32`, ~49s/image SDXL) -- Or: split 32B model across both P40s via tensor parallelism for better speed -- Both are native rack server GPUs (passive, designed for R720/R730) -- Total power: ~500W GPU (can power-limit to ~280W) -- **Best value for 32B + image gen** +### Best Overall: RTX 8000 Passive ($750–1,400) +Single card handles both coding and image gen. 48GB VRAM fits 32B models with massive +context windows. Passive cooling is rack-native. Tensor cores handle FP16 image gen properly. +One card, one slot, simple setup. -### Option B: P40 + A2000 (~$550-810) -- P40 for 32B coding model (24GB, tight but works) -- A2000 for image gen (3x faster than P40, tensor cores, 70W) -- Total power: ~320W GPU -- A2000 at $490 is overpriced — shop for $250-350 range -- **Better image gen speed vs Option A** +### Best Overall + Dedicated Image Gen: RTX 8000 + A2000 ($1,000–1,750) +RTX 8000 for coding with full 48GB dedicated to LLM context. +A2000 for image gen (3x faster than Turing, 70W, bus-powered, blower cooled). +Best separation of concerns — no model swapping needed. -### Option C: Single P40 (~$200-300) -- Run 32B coding model OR image gen (not both simultaneously) -- 32B model leaves no room for anything else in VRAM -- Swap between tasks by unloading/loading models -- **Cheapest entry point**, upgrade later +### Budget Alternative: Dual P40 ($400–500) +Two P40s for 48GB total, but split across cards (can't combine for one model without +tensor parallelism). One for 32B coding (tight fit), one for image gen (slow, needs --force-fp32). -### Option D: P40 (coding) + T4 (image gen) (~$400-550) -- P40: 32B coding model (24GB) -- T4: Image gen with tensor cores, 16GB, 70W, passive -- T4 is faster than P40 for image gen (Turing tensor cores, native FP16) -- Both passive-cooled = true rack-native -- **Good balance of price and image gen speed** - -## Recommendations - -### If 32B coding quality is the priority: Dual P40 ($400-500) -Two P40s give you 48GB total. Run 32B coding model on one, image gen on the other. -Or split the model across both for faster inference. Both fit natively in R720/R730. - -### If image gen speed matters equally: P40 + T4 ($400-550) -P40 for 32B coding, T4 for image gen. Both passive-cooled, both rack-native. -T4 has tensor cores + FP16 for much faster image gen than P40. - -### Cheapest possible: Single P40 ($200-300) -Run 32B coding model with limited context. Swap to image gen when needed. -Upgrade to dual-GPU later. +### Cheapest Entry: Single P40 ($200–300) +Run 32B coding model with very limited context (~4K tokens). Swap to image gen when needed. +Good for testing whether local LLM coding works for your workflow before investing more. ## Configuration Notes for local-ai stack -### For 32B models on P40 +### For 48GB RTX 8000 +```bash +# Ollama — take advantage of the full 48GB +OLLAMA_NUM_GPU=999 +OLLAMA_NUM_CTX=32768 # Large context window — 48GB can handle it +OLLAMA_KEEP_ALIVE=24h + +# Pull best coding models +ollama pull qwen2.5-coder:32b-instruct-q4_K_M # ~20GB, leaves 28GB for context +ollama pull qwen3.5:27b # ~16GB at Q4, even more context room +ollama pull qwen3-coder:30b # MoE, very fast inference + +# Higher quantization for better quality (48GB allows this) +# Look for Q6_K or Q8_0 variants on Ollama for better output quality +``` + +### For 32B models on P40 (24GB — tight fit) ```bash -# In Ollama environment OLLAMA_NUM_GPU=999 OLLAMA_NUM_CTX=4096 # Keep context small to fit in remaining VRAM OLLAMA_KEEP_ALIVE=24h -# Pull the right quantization ollama pull qwen2.5-coder:32b-instruct-q4_K_M ``` -### For dual-GPU setup +### For dual-GPU setup (RTX 8000 + A2000 or P40 + anything) ```bash # Assign GPU 0 to Ollama (coding), GPU 1 to InvokeAI (image gen) # In docker-compose.yml for Ollama: @@ -163,7 +183,7 @@ CUDA_VISIBLE_DEVICES=0 CUDA_VISIBLE_DEVICES=1 ``` -### For P40 image gen +### For image gen on P40 (no tensor cores) ```bash # InvokeAI INVOKEAI_PRECISION=float32 @@ -172,30 +192,31 @@ INVOKEAI_PRECISION=float32 --force-fp32 ``` -### For newer coding models (2026) +### For image gen on RTX 8000 / A2000 / T4 (has tensor cores) ```bash -# Qwen 3.5 27B — fits easily on 24GB at Q4, better quality than 2.5 -ollama pull qwen3.5:27b +# InvokeAI — native FP16 works fine +INVOKEAI_PRECISION=float16 -# Qwen3-Coder 30B-A3B MoE — fast inference, agentic coding -ollama pull qwen3-coder:30b +# ComfyUI — no special flags needed ``` ## Sources +- [Quadro RTX 8000 for Local LLMs — Hardware Corner](https://www.hardware-corner.net/guides/quadro-rtx-8000-for-llm/) +- [RTX 8000 Passive — Network Outlet](https://networkoutlet.com/blogs/articles/nvidia-quadro-rtx-8000-48gb-passive-cooling-powering-ai-rendering-server-workloads) +- [LLM Benchmarks on Turing/Ampere GPUs — Stefandroid](https://blog.stefandroid.com/2025/06/02/benchmark-llm-performance-nvidia-gpus.html) +- [NVIDIA A40 Price Tracking — GPUPoet](https://gpupoet.com/gpu/learn/card/nvidia-a40) +- [NVIDIA L40 Price Tracking — GPUPoet](https://gpupoet.com/gpu/learn/card/nvidia-l40) +- [RTX A6000 Price History — CamelCamelCamel](https://camelcamelcamel.com/product/B09BDH8VZV) +- [RTX A6000 Price History — Pangoly](https://pangoly.com/en/price-history/pny-nvidia-quadro-rtx-a6000) - [NVIDIA RTX A2000 Datasheet](https://www.nvidia.com/content/dam/en-zz/Solutions/design-visualization/rtx-a2000/nvidia-rtx-a2000-datasheet-1987439-r5.pdf) -- [Lenovo ThinkSystem RTX A2000 Product Guide](https://lenovopress.lenovo.com/lp1919-thinksystem-nvidia-rtx-a2000-12gb-pcie-active-gpu) -- [n3rdware Single-Slot A2000 Cooler](https://n3rdware.com/gpu-coolers/single-slot-rtx-a2000-cooler) -- [Dell R730 Owner's Manual - Expansion Cards](https://www.dell.com/support/manuals/en-us/poweredge-r730/r730_ompublication/expansion-card-installation-guidelines) -- [Dell R720 Owner's Manual - Expansion Cards](https://www.dell.com/support/manuals/en-us/poweredge-r720/720720xdom/expansion-card-installation-guidelines) +- [Dell R730 Owner's Manual — Expansion Cards](https://www.dell.com/support/manuals/en-us/poweredge-r730/r730_ompublication/expansion-card-installation-guidelines) +- [Dell R720 Owner's Manual — Expansion Cards](https://www.dell.com/support/manuals/en-us/poweredge-r720/720720xdom/expansion-card-installation-guidelines) - [ComfyUI GPU Benchmarks Discussion](https://github.com/Comfy-Org/ComfyUI/discussions/2970) - [ComfyUI P40 FP32 Issue](https://github.com/Comfy-Org/ComfyUI/issues/4363) -- [How to Use Tesla P40 Guide](https://github.com/JingShing/How-to-use-tesla-p40) -- [Build a Local LLM Server Under $1000](https://sanj.dev/post/affordable-ai-hardware-local-llms) -- [Best Budget GPUs for AI 2026](https://techtactician.com/best-budget-gpus-for-local-ai-workflows/) -- [Tesla P40 for Local LLMs 2026](https://like2byte.com/tesla-p40-local-llm-guide/) - [Best Local LLMs for 24GB VRAM 2026](https://localllm.in/blog/best-local-llms-24gb-vram) - [Best Coding Models 2026](https://localvram.com/en/guides/best-coding-models/) - [Ollama VRAM Requirements Guide](https://localllm.in/blog/ollama-vram-requirements-for-local-llms) - [Local LLMs That Can Replace Claude Code](https://agentnativedev.medium.com/local-llms-that-can-replace-claude-code-6f5b6cac93bf) +- [7 Local LLM Families to Replace Claude/Codex](https://agentnativedev.medium.com/7-local-llm-families-to-replace-claude-codex-for-everyday-tasks-25ba74c3635d) - [Qwen2.5-Coder 32B on Ollama](https://ollama.com/library/qwen2.5-coder:32b-instruct-q4_K_M) -- [Qwen2.5-Coder 32B HuggingFace Discussion](https://huggingface.co/Qwen/Qwen2.5-Coder-32B-Instruct/discussions/28) +- [Qwen3-Coder — How to Run Locally](https://unsloth.ai/docs/models/qwen3-coder-how-to-run-locally) From 8646dca2bbfb34cdd7f5a91ba8fd2e291cc6a63b Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 14:32:28 +0000 Subject: [PATCH 04/15] Fix RTX 8000 pricing: $2,000-2,900 realistic, not $750 The $750 listing was a single outlier. Actual market price for Quadro RTX 8000 Passive 48GB is $2,000-2,900 on eBay. Updated all recommendations and cost comparisons accordingly. https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- docs/gpu-setup-research.md | 37 +++++++++++++++++++++++++------------ 1 file changed, 25 insertions(+), 12 deletions(-) diff --git a/docs/gpu-setup-research.md b/docs/gpu-setup-research.md index 14ce3af..e72d231 100644 --- a/docs/gpu-setup-research.md +++ b/docs/gpu-setup-research.md @@ -41,19 +41,20 @@ Best strategy: use local models for routine tasks, save Claude credits for hard | GPU | Arch | Used Price | TDP | Cooling | Tensor Cores | Mem BW | |-----|------|------------|-----|---------|--------------|--------| -| **Quadro RTX 8000** | Turing (2018) | **$750–1,400** | 260W | Passive variant | Yes (576) | 672 GB/s | +| **Quadro RTX 8000** | Turing (2018) | **$2,000–2,900** | 260W | Passive variant | Yes (576) | 672 GB/s | | **A40** | Ampere (2020) | **~$5,050+** | 300W | Passive | Yes (336 3rd-gen) | 696 GB/s | | **RTX A6000** | Ampere (2020) | **~$5,400+** | 300W | Active (blower) | Yes (336 3rd-gen) | 768 GB/s | | **L40** | Ada (2022) | **~$6,500+** | 300W | Passive | Yes (568 4th-gen) | 864 GB/s | | **RTX 6000 Ada** | Ada (2022) | **~$6,500+** | 300W | Active | Yes (568 4th-gen) | 960 GB/s | Sources: eBay active/sold listings, GPUPoet price tracking, Pangoly, CamelCamelCamel (all March 2026) +Note: One outlier RTX 8000 listing at ~$750 exists but is not representative of the market. -### Winner: Quadro RTX 8000 Passive ($750–1,400) +### Cheapest 48GB Option: Quadro RTX 8000 Passive ($2,000–2,900) -The RTX 8000 is **4–5x cheaper** than every other 48GB option. The passive variant is -purpose-built for rack servers — no fan, relies on chassis airflow, designed for 24/7 operation -in 2U/4U systems. +The RTX 8000 is still the cheapest 48GB card — roughly half the price of an A40 and a +third of an A6000. The passive variant is purpose-built for rack servers — no fan, relies +on chassis airflow, designed for 24/7 operation in 2U/4U systems. Key advantages over the P40: - **48GB vs 24GB** — room for models + massive context @@ -61,6 +62,16 @@ Key advantages over the P40: - **NVLink support** — pair two for 96GB combined (100 GB/s bidirectional) - 10W idle power draw +### Cost Reality Check +At $2,000–2,900 the RTX 8000 is a significant investment. The key question: is unified +48GB VRAM worth 4–6x the cost of dual P40s ($400–500)? + +**Yes, if** you need large context windows (32K+) for complex coding — KV cache can't +be split across two GPUs without NVLink (which P40s don't have). + +**No, if** you're mostly doing short-prompt coding tasks and image gen — dual P40s give +you 48GB total (split) at a fraction of the cost, and each card can handle its own workload. + ## RTX 8000 Performance Benchmarks ### LLM Inference (Exllama, 5.0 bpw quantization) @@ -77,6 +88,7 @@ Key advantages over the P40: | Metric | P40 (24GB) | RTX 8000 (48GB) | |--------|-----------|-----------------| +| **Used price** | **$150–320** | **$2,000–2,900** | | 32B model fit | Barely (~2GB free) | Comfortable (~28GB free) | | 32B generation speed | ~5-12 t/s (est.) | ~20-34 t/s | | Max practical context | ~4K tokens | **32K+ tokens** | @@ -128,21 +140,22 @@ though still behind Ampere/Ada cards. ## Recommended Setups -### Best Overall: RTX 8000 Passive ($750–1,400) +### If budget allows ($2,000–2,900): RTX 8000 Passive Single card handles both coding and image gen. 48GB VRAM fits 32B models with massive -context windows. Passive cooling is rack-native. Tensor cores handle FP16 image gen properly. -One card, one slot, simple setup. +context windows (32K+). Passive cooling is rack-native. Tensor cores handle FP16 image gen +properly. One card, one slot, simple setup. The premium buys you unified VRAM = big context. -### Best Overall + Dedicated Image Gen: RTX 8000 + A2000 ($1,000–1,750) +### If budget allows + dedicated image gen ($2,300–3,250): RTX 8000 + A2000 RTX 8000 for coding with full 48GB dedicated to LLM context. A2000 for image gen (3x faster than Turing, 70W, bus-powered, blower cooled). Best separation of concerns — no model swapping needed. -### Budget Alternative: Dual P40 ($400–500) +### Best value ($400–500): Dual P40 Two P40s for 48GB total, but split across cards (can't combine for one model without -tensor parallelism). One for 32B coding (tight fit), one for image gen (slow, needs --force-fp32). +NVLink, which P40s lack). One for 32B coding (tight fit, ~4K context), one for image gen +(slow, needs --force-fp32). **5x cheaper than RTX 8000** but with significant context limitations. -### Cheapest Entry: Single P40 ($200–300) +### Cheapest entry ($200–300): Single P40 Run 32B coding model with very limited context (~4K tokens). Swap to image gen when needed. Good for testing whether local LLM coding works for your workflow before investing more. From 7fa91e98162f560061dae37720b64b7447439a7d Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 15:57:24 +0000 Subject: [PATCH 05/15] Add Quadro RTX 5000 NVLink budget build guide and Qwen 3.5 models MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Research and document the $850 budget build: 2x Quadro RTX 5000 (16GB each) connected via NVLink for 32GB unified VRAM. With Qwen 3.5's Gated Delta Network architecture (Feb 2026), this setup runs the 35B-A3B MoE model with full 262K context in ~25GB — the best price-to-capability ratio for local AI coding available. Includes: - Exact NVLink bridge part numbers (RTX 5000 uses unique smaller connector) - Motherboard/PSU requirements and slot spacing guidance - llama.cpp and Ollama multi-GPU configuration - VRAM budget calculations for all Qwen 3.5 model sizes - Phased build plan (start with 1 card at $400, add second later) - Updated model table with full Qwen 3.5 family specs - Cost comparison vs RTX 8000, RTX 3090, Claude Max, and API pricing https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- docs/gpu-setup-research.md | 228 ++++++++++++++++++++++++++++++++++++- 1 file changed, 227 insertions(+), 1 deletion(-) diff --git a/docs/gpu-setup-research.md b/docs/gpu-setup-research.md index e72d231..acafb52 100644 --- a/docs/gpu-setup-research.md +++ b/docs/gpu-setup-research.md @@ -25,10 +25,25 @@ coding sessions where the model needs to understand your entire codebase. ## Best Local Coding Models (2026) +### Qwen 3.5 Family (February 2026 — Gated Delta Networks) + +Architecture breakthrough: 3 of every 4 layers use **linear attention** (O(n) scaling), +drastically reducing KV cache memory. These models need far less VRAM for long contexts +than traditional transformers. + +| Model | Type | Active Params | Size at Q4_K_M | Max Context | Quality | Notes | +|-------|------|---------------|---------------|-------------|---------|-------| +| **Qwen3.5-35B-A3B** | **MoE** | **3B** | **~12GB** | **262K** | **A-** | Best bang for buck — 35B model, 3B active, fits 262K ctx in 25GB | +| **Qwen3.5-27B** | Dense | 27B | ~17GB | 262K | A- | 72.4% SWE-bench, ties GPT-5 mini | +| **Qwen3.5-122B-A10B** | MoE | 10B | ~76GB | 262K | A | Matches GPT-5 mini across the board | +| **Qwen3.5-9B** | Dense | 9B | ~6GB | 262K | B+ | Fits on any modern GPU | +| **Qwen3.5-4B** | Dense | 4B | ~3GB | 262K | B | Tiny but capable | + +### Previous Generation (Still Relevant) + | Model | Size at Q4_K_M | Quality | Notes | |-------|---------------|---------|-------| | **Qwen2.5-Coder 32B** | ~20GB | 73.7 Aider (≈ GPT-4o) | FIM king, 92.7% HumanEval | -| **Qwen 3.5 27B** | ~16GB | 72.4% SWE-bench (ties GPT-5 mini) | 262K context, multimodal | | **Qwen3-Coder 30B-A3B** (MoE) | ~18GB | #1 SWE-rebench (64.6%) | Only 3.3B active, very fast | | **Qwen3-Coder-Next 80B** (MoE) | needs 64GB+ RAM offload | Beats Claude Opus 4.6 on SWE-rebench | Hybrid attention, 256K context | @@ -100,6 +115,217 @@ The RTX 8000 has Turing Tensor Cores with native FP16 support. Unlike the P40, i need `--force-fp32` workarounds. Image gen performance is significantly better than the P40, though still behind Ampere/Ada cards. +## Budget Build: 2x Quadro RTX 5000 + NVLink ($850 Total) + +*The best price-to-capability ratio for local AI coding in 2026.* + +### Why This Works Now + +Qwen 3.5 (February 2026) introduced **Gated Delta Networks** — 3 out of 4 layers use linear +attention (O(n) scaling) instead of quadratic. KV cache memory usage is dramatically lower +than traditional transformers. A 35B MoE model with 262K context now fits in ~25GB VRAM. + +### Hardware + +#### GPU: NVIDIA Quadro RTX 5000 (Turing, TU104) + +| Spec | Value | +|------|-------| +| VRAM | 16GB GDDR6 | +| CUDA Cores | 3072 | +| Tensor Cores | 384 (Gen 2, FP16) | +| TDP | ~230W | +| NVLink | **Yes — 50 GB/s bidirectional** | +| Form Factor | Dual-slot, blower cooler (rack-friendly) | +| PCIe | 3.0 x16 | +| Used Price | **~$400** | +| Part Number | VCQRTX5000-PB | + +#### NVLink Bridge (CRITICAL: RTX 5000 uses a unique smaller connector) + +The Quadro RTX 5000 has a **shorter NVLink connector** than all other Quadro RTX cards. +Bridges from the RTX 6000/8000 will NOT physically fit. You must buy the RTX 5000-specific bridge. + +| Detail | Value | +|--------|-------| +| Product | NVIDIA Quadro RTX 5000 NVLink HB Bridge 2-Slot | +| SKU | NVLINKX8-2SLOT-PB | +| Part Numbers | 1JF3K, 699-54934-0500-000, 900-54934-0100-000, P4934, 6FY12AA, L55997-001 | +| Price | **~$30-80** (eBay, Amazon) | +| Bandwidth | 50 GB/s total (25 GB/s per direction) | +| Sizing | 2-slot (cards adjacent) or 3-slot (one slot gap — better thermals) | + +**Where to buy:** +- eBay: search "Quadro RTX 5000 NVLink" or part numbers P4934 / L55997-001 / 1JF3K +- Amazon: search part number 6FY12AA or 1JF3K + +**WARNING:** The 3-slot bridge is recommended over 2-slot. With a 2-slot bridge the cards +sit directly adjacent — the top card's blower intake gets blocked by the bottom card. +A 3-slot bridge leaves an air gap for proper cooling. + +#### Motherboard Requirements + +| Requirement | Details | +|-------------|---------| +| PCIe slots | Two x16 slots (x8 electrical is fine — LLM inference is VRAM-bound, not PCIe-bound) | +| Slot spacing | Must match your NVLink bridge size (2-slot or 3-slot gap) | +| Power supply | 650W+ minimum (80 PLUS Gold recommended), 850W+ for headroom | +| Power connectors | 2x 8-pin PCIe power (one per card). Do NOT daisy-chain — use separate cables | +| CPU platform | Any modern platform works. Threadripper/Xeon not required | + +**Recommended motherboards (workstation/server):** +- Any board with 2x PCIe x16 slots spaced 2-3 slots apart +- Server: Dell R730/R740 with GPU riser (but verify 3-slot bridge clearance in 2U) +- Workstation: MSI X399 Creation, ASUS WS series, Supermicro X11/X12 boards +- Desktop: Most ATX boards with 2 full-length x16 slots work + +**Rack server note:** The Quadro RTX 5000's blower cooler exhausts out the bracket — +this works well in rack airflow. If using a 2U server, measure clearance for the NVLink +bridge sitting on top of the cards. A 4U chassis gives the most room. + +### What Runs on 32GB Unified (2x RTX 5000 + NVLink) + +| Model | Arch | Quant | Weights | Context | Total VRAM | Quality | +|-------|------|-------|---------|---------|------------|---------| +| **Qwen3.5-35B-A3B** | **MoE (3B active)** | **Q4_K_M** | **~12GB** | **262K** | **~25GB** | **A-** | +| Qwen3.5-27B | Dense | Q4_K_M | ~17GB | 128K+ | ~25GB | A- | +| Qwen3-Coder-Next (80B/3B active) | MoE | Q4 | ~20GB | 128K | ~28GB | A | +| Qwen2.5-Coder-14B | Dense | Q4_K_M | ~10GB | 128K | ~22GB | B+ | +| Qwen2.5-Coder-14B | Dense | Q8 | ~16GB | 64K | ~28GB | A- | +| Qwen2.5-Coder-32B | Dense | Q4_K_M | ~20GB | 16-24K | ~28GB | A- | + +**The sweet spot: Qwen3.5-35B-A3B at Q4_K_M with 262K context.** This is a 35B parameter +model with only 3B active at inference (MoE). The Gated Delta Network architecture slashes +KV cache memory. The entire model + full 262K context fits in ~25GB — well within 32GB. + +### What Runs on 16GB (Single RTX 5000 — Phase 1) + +| Model | Quant | Context | Quality | +|-------|-------|---------|---------| +| **Qwen3.5-35B-A3B** | Q4_K_L | ~64-128K | **A-** | +| Qwen3.5-9B | Q8 | 128K+ | B+ | +| Qwen3.5-4B | Q8 | 262K | B | +| Qwen2.5-Coder-7B | Q8 | 128K | B | +| Qwen2.5-Coder-14B | Q4_K_M | 16-32K | B+ | + +Even a single card can run the Qwen3.5-35B-A3B MoE model — just with a smaller context window. + +### Estimated Inference Speed + +| Model | 1x RTX 5000 | 2x RTX 5000 (NVLink) | +|-------|-------------|---------------------| +| Qwen3.5-35B-A3B Q4 (short ctx) | ~25-35 tok/s | ~25-35 tok/s | +| Qwen3.5-35B-A3B Q4 (128K ctx) | ~10-18 tok/s | ~15-25 tok/s | +| Qwen3.5-35B-A3B Q4 (262K ctx) | Won't fit | ~10-18 tok/s | +| Qwen2.5-Coder-14B Q4 | ~20-30 tok/s | ~25-35 tok/s | + +NVLink matters most at large context windows where KV cache spans both cards. +At short contexts that fit on one card, the second GPU adds less benefit. + +### Power Consumption & Cost + +| Config | Idle | Load | Monthly (8hr/day @ $0.09/kWh) | Annual | +|--------|------|------|-------------------------------|--------| +| 1x Quadro RTX 5000 | ~15W | ~210W | **~$4.50** | ~$54 | +| 2x Quadro RTX 5000 | ~30W | ~420W | **~$9.00** | ~$108 | + +### Software Setup + +#### llama.cpp (Recommended — Best Multi-GPU Support) + +```bash +# Build with CUDA support +git clone https://github.com/ggerganov/llama.cpp +cd llama.cpp +cmake -B build -DGGML_CUDA=ON +cmake --build build --config Release -j$(nproc) + +# Download Qwen3.5-35B-A3B GGUF (Q4_K_M) +# Get from: https://huggingface.co/unsloth/Qwen3.5-35B-A3B-GGUF + +# Run on dual GPU with NVLink +./build/bin/llama-server \ + -m Qwen3.5-35B-A3B-Q4_K_M.gguf \ + -ngl 999 \ + -c 262144 \ + --host 0.0.0.0 \ + --port 8080 + +# llama.cpp auto-detects NVLink and splits layers across both GPUs +# Use -ts 1,1 to manually set equal split if needed +``` + +#### Ollama + +```bash +# Requires Ollama v0.17+ for Qwen3.5 support +# NOTE: As of March 2026, some Qwen3.5 GGUFs have compatibility issues +# with Ollama due to mmproj vision files. llama.cpp may be more reliable. + +# Environment variables for multi-GPU +export OLLAMA_GPU_SPLIT=16,16 # Equal split across both 16GB cards +export OLLAMA_KV_CACHE_TYPE=q8_0 # Halves KV cache VRAM with minimal quality loss +export OLLAMA_KEEP_ALIVE=24h # Keep model loaded in VRAM +export OLLAMA_FLASH_ATTENTION=1 # Enable flash attention for VRAM savings + +# Pull and run +ollama pull qwen3.5:35b-a3b-q4_K_M +ollama run qwen3.5:35b-a3b-q4_K_M +``` + +#### Verify NVLink Is Working + +```bash +# Check NVLink status +nvidia-smi nvlink --status + +# Check NVLink bandwidth +nvidia-smi nvlink -gt d + +# Monitor both GPUs during inference +watch -n 0.5 nvidia-smi +``` + +### Total Cost Summary + +| Item | Cost | +|------|------| +| 1x Quadro RTX 5000 (Phase 1) | $400 | +| 1x Quadro RTX 5000 (Phase 2) | $400 | +| NVLink HB Bridge 3-slot | ~$50 | +| **Hardware total** | **$850** | +| Monthly power (2 cards, 8hr/day) | $9/mo | +| Claude Pro subscription | $20/mo | +| **Monthly operating cost** | **~$29/mo** | +| **3-year total cost of ownership** | **$850 + $1,044 = $1,894** | + +### Comparison: This Build vs Alternatives + +| Setup | Cost (3yr) | Best Model | Max Context | Quality | +|-------|-----------|------------|-------------|---------| +| **2x RTX 5000 + $20 Pro** | **$1,894** | Qwen3.5-35B-A3B + Opus | 262K local | **A- local, A+ cloud** | +| 1x RTX 3090 + $20 Pro | $1,420 | Qwen3.5-35B-A3B + Opus | ~128K local | A- local, A+ cloud | +| RTX 8000 (48GB) + $20 Pro | $3,220+ | Qwen3.5-35B-A3B + Opus | 262K+ local | A- local, A+ cloud | +| Claude Max only (no GPU) | $3,600 | Opus 4.6 | 200K | A+ cloud only | +| API-only (Opus heavy use) | $18,000+ | Opus 4.6 | 200K | A+ cloud only | + +### Phased Build Plan + +**Phase 1 — Start with one card ($400)** +1. Buy Quadro RTX 5000 (VCQRTX5000-PB) — ~$400 on eBay +2. Install in any PCIe x16 slot +3. Install llama.cpp or Ollama v0.17+ +4. Run Qwen3.5-35B-A3B at Q4_K_L with 64-128K context +5. Already A- quality for coding — test if local inference fits your workflow + +**Phase 2 — Add second card + NVLink ($450)** +1. Buy matching Quadro RTX 5000 — ~$400 +2. Buy NVLink HB Bridge 3-slot (part: P4934 / 1JF3K / 6FY12AA) — ~$50 +3. Install second card in adjacent/nearby x16 slot +4. Connect NVLink bridge +5. Verify with `nvidia-smi nvlink --status` +6. Now running 32GB unified — Qwen3.5-35B-A3B at Q4_K_M with full 262K context + ## 24GB GPU Options (Previous Research — Still Valid for Tighter Budgets) | GPU | VRAM | Price Range | Best Deals | Notes | From 1e7b087a5d6b024679029bf2773c88c2fab96df6 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 16:57:48 +0000 Subject: [PATCH 06/15] Add Dell R720/R730 installation guide for dual RTX 5000 + NVLink Complete shopping list with Dell-specific part numbers: - GPU power cables: 9H6FV / N08NH (~$10-15 each, need 2) - GPU Riser 3 required for second GPU slot - Low-profile heatsinks needed on R720 (usually pre-installed on R730) - 2x 1100W PSUs mandatory, non-redundant mode for full wattage Documents riser layout, NVLink bridge clearance in 2U, potential issues (CPU TDP limits, "unsupported" GPU warning, blower noise), and R720 vs R730 comparison. Total build cost ~$940-960 with Dell parts. https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- docs/gpu-setup-research.md | 100 ++++++++++++++++++++++++++++++++++++- 1 file changed, 99 insertions(+), 1 deletion(-) diff --git a/docs/gpu-setup-research.md b/docs/gpu-setup-research.md index acafb52..0be7d0d 100644 --- a/docs/gpu-setup-research.md +++ b/docs/gpu-setup-research.md @@ -320,12 +320,110 @@ watch -n 0.5 nvidia-smi **Phase 2 — Add second card + NVLink ($450)** 1. Buy matching Quadro RTX 5000 — ~$400 -2. Buy NVLink HB Bridge 3-slot (part: P4934 / 1JF3K / 6FY12AA) — ~$50 +2. Buy NVLink HB Bridge 2-slot (part: P4934 / 1JF3K / 6FY12AA) — ~$50 3. Install second card in adjacent/nearby x16 slot 4. Connect NVLink bridge 5. Verify with `nvidia-smi nvlink --status` 6. Now running 32GB unified — Qwen3.5-35B-A3B at Q4_K_M with full 262K context +### Dell R720/R730 Installation Guide + +#### Prerequisites (MUST HAVE before buying GPUs) + +| Requirement | R720 | R730 | Why | +|-------------|-------|-------|-----| +| **Dual CPUs** | Required | Required | GPU riser slots are wired to CPU2 — dead without it | +| **2x 1100W PSUs** | Required | Required | 2x 230W GPUs + system = ~600W+ under load | +| **GPU Riser 3** | Required for 2nd GPU | Required for GPUs | Provides the PCIe x16 slot + 8-pin power | +| **GPU Power Cable** | Required | Required | Riser-to-GPU power, not included by default | +| **Low-Profile Heatsinks** | Must swap (part of enablement kit) | Usually pre-installed | Standard heatsinks block GPU riser clearance | +| **Max ambient temp** | 30°C (not the usual 35°C) | 30°C | High GPU TDP restricts cooling headroom | + +#### Shopping List: Dell-Specific Parts + +| Part | Dell P/N | What It Is | Price | Where | +|------|----------|------------|-------|-------| +| **GPU Power Cable** | **9H6FV** (09H6FV) | 8-pin EPS (riser) → 6-pin + 6+2-pin PCIe. One cable powers one GPU | ~$10-15 | Amazon, eBay | +| **GPU Power Cable (alt)** | **N08NH** (0N08NH) | Same function, alternate Dell part number | ~$10-15 | Amazon, eBay | +| **GPU Riser 3** (R720) | Check eBay for "R720 riser 3" or "R720 GPU riser" | Second riser card that provides GPU-capable x16 slot | ~$15-30 | eBay | +| **GPU Riser 3** (R730) | Check eBay for "R730 riser 3" or "R730 GPU riser" | R730 version — NOT interchangeable with R720 | ~$15-30 | eBay | +| **Low-Profile Heatsinks** (R720 only) | Part of original GPU enablement kit | Shorter heatsinks that clear the GPU riser. Search "R720 low profile heatsink" | ~$10-20/pair | eBay | + +**You need 2x power cables** (one per GPU). Search Amazon for "Dell R720 R730 GPU power cable 9H6FV" — multiple sellers (COMeap, ZAHARA, BestParts) stock them for ~$10-15 each. + +#### How It Fits + +``` +Dell R720/R730 Riser Layout (rear view): +┌─────────────────────────────────────┐ +│ Riser 1 Riser 2 Riser 3│ +│ (network/ (GPU 1) (GPU 2)│ +│ storage) PCIe x16 PCIe x16│ +│ Gen2(720) Gen2(720)│ +│ Gen3(730) Gen3(730)│ +└─────────────────────────────────────┘ + ↑ RTX 5000 ↑ ↑ RTX 5000 ↑ + └── NVLink Bridge ──┘ +``` + +- Both GPUs sit on **adjacent risers** (Riser 2 + Riser 3) — this is 2-slot spacing +- The **2-slot NVLink bridge** (P4934) is the correct size for R720/R730 +- The cards mount **vertically** via risers, parallel to each other +- NVLink bridge connects across the top of both cards + +#### R720 vs R730 + +| Feature | R720 | R730 | +|---------|------|------| +| **PCIe** | Gen2 x16 | **Gen3 x16** | +| **Impact on LLM** | None — VRAM-bound | None — VRAM-bound | +| **Impact on NVLink** | None — NVLink bypasses PCIe | None — NVLink bypasses PCIe | +| **GPU power delivery** | Same 8-pin from riser | Same 8-pin from riser | +| **Heatsink swap** | Usually required | Usually already low-profile | +| **Used price** | ~$100-150 cheaper | Preferred if budget allows | +| **Recommendation** | Fine if you already have one | **Buy this one** if shopping new | + +#### Potential Issues + +1. **NVLink bridge clearance in 2U** — The bridge sits on top of both GPUs. In a 2U chassis + this is tight. The R720/R730 riser design mounts cards vertically which actually helps — + the bridge faces the chassis side panel, not the lid. Should fit, but measure before buying. + +2. **Blower fan noise** — The RTX 5000 has an active blower (unlike passive Tesla cards). + The server's own fans may spin higher to compensate. The blower exhausts out the bracket + which is correct for rack airflow. + +3. **"Unsupported" GPU warning** — Dell officially supports Tesla/Quadro cards from their era. + The Quadro RTX 5000 is a later generation than R720/R730 was designed for, but community + reports confirm Quadro RTX and even consumer RTX cards work fine. You won't get Dell support + if something goes wrong, but electrically it's standard PCIe. + +4. **CPU TDP limit** — Dell requires CPUs of 115W or less when GPUs are installed (R720). + Check your CPU model. Most common Xeon E5-2600 v1/v2 (R720) and E5-2600 v3/v4 (R730) + processors are within this range, but some high-core-count variants exceed it. + +5. **PSU mode** — With dual 300W GPUs, set PSU configuration to **non-redundant mode** + to use combined wattage from both PSUs. In redundant mode, you're limited to one PSU's + capacity (1100W) which may not be enough under full GPU + CPU load. + +#### Complete R720/R730 Shopping List + +``` +GPUS + NVLINK + 2x Quadro RTX 5000 ~$800 + 1x NVLink Bridge 2-slot (P4934 / L55997-001) ~$50 + +DELL-SPECIFIC PARTS + 2x GPU Power Cable (9H6FV or N08NH) ~$25 + 1x GPU Riser 3 (match your server model!) ~$20 + 2x Low-Profile Heatsinks (R720 only) ~$15 + +POWER (if not already installed) + 2x Dell 1100W PSU ~$30-50 ea + +TOTAL (assuming you have the server + dual CPUs) ~$940-960 +``` + ## 24GB GPU Options (Previous Research — Still Valid for Tighter Budgets) | GPU | VRAM | Price Range | Best Deals | Notes | From 5cfd8b269642731a979d64c66b17e5eaac92fe38 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 17:07:01 +0000 Subject: [PATCH 07/15] Reframe guide around Drop Max / Keep Pro strategy The real use case: drop Claude Max ($100/mo), keep Pro ($20/mo), offload bulk codebase work to local. Local handles the 80% (reading 10K-line projects, routine fixes, boilerplate) with no rate limits. Pro handles the hard 20% where Opus quality matters. GPU pays for itself in 13 months, saves $1,596 over 3 years vs Max. Also adds: NVLink speed reality check, image gen capabilities (SDXL/Flux included at no extra cost), hardware longevity estimate (3-5 years), and updated cost comparison tables. https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- docs/gpu-setup-research.md | 106 +++++++++++++++++++++++++++++++------ 1 file changed, 91 insertions(+), 15 deletions(-) diff --git a/docs/gpu-setup-research.md b/docs/gpu-setup-research.md index 0be7d0d..de120f5 100644 --- a/docs/gpu-setup-research.md +++ b/docs/gpu-setup-research.md @@ -48,9 +48,34 @@ than traditional transformers. | **Qwen3-Coder-Next 80B** (MoE) | needs 64GB+ RAM offload | Beats Claude Opus 4.6 on SWE-rebench | Hybrid attention, 256K context | ### Honest Assessment: Local vs Claude Code + Nothing local approaches Claude Opus 4.6 quality for complex multi-file agentic coding. These 32B models are competitive with **GPT-4o** — a tier below Claude Sonnet, two tiers below Opus. -Best strategy: use local models for routine tasks, save Claude credits for hard problems. + +**The real strategy: Drop Max ($100/mo), keep Pro ($20/mo), offload bulk work to local.** + +The problem with Pro for large projects: rate limits. A 10,000-line codebase needs the model +to read, understand, and hold context across many files. On Pro you'll hit usage caps mid-session +on complex multi-file work. Max ($100/mo) removes those limits — but that's $80/mo extra. + +Local AI eliminates this problem differently: +- **Local model (262K context)**: Reads your entire 10K-line project at once. No rate limits, + no usage caps, runs 24/7. Handles the bulk work — understanding codebase structure, routine + bug fixes, simple refactors, code explanation, test writing, boilerplate generation. +- **Claude Pro ($20/mo)**: Reserved for the hard problems — complex multi-file architectural + changes, subtle bugs that need Opus-level reasoning, code review on critical paths. + Pro limits are fine when you're only sending Claude the *hard* 20% instead of everything. + +This is the unlock: local doesn't replace Claude, it **reduces your Claude usage enough +that Pro limits stop being a problem.** The 80% of routine work that was burning through +your Max quota now runs locally with zero limits. + +| Plan | Monthly | What You Get | Limit Problem | +|------|---------|--------------|---------------| +| Max only | $100 | Opus unlimited | Paying $80/mo for unlimited when you don't need it | +| Pro only | $20 | Opus with rate limits | **Hits caps on 10K-line projects** | +| **Pro + Local GPU** | **$29** | Opus for hard stuff + unlimited local | **No caps — bulk work is local** | +| Local only (no Claude) | $9 | A- quality only | Stuck on hard problems with no escape hatch | ## 48GB GPU Market (March 22, 2026 — Real Prices) @@ -222,6 +247,44 @@ Even a single card can run the Qwen3.5-35B-A3B MoE model — just with a smaller NVLink matters most at large context windows where KV cache spans both cards. At short contexts that fit on one card, the second GPU adds less benefit. +**Speed reality check:** NVLink doesn't make it faster — it prevents the slowdown you'd get +from PCIe when the model spans both cards. The base speed is still Turing (2018 silicon). +10-35 tok/s is fast enough for coding (you read slower than that), but it's not instant. +The MoE architecture (only 3B active params at inference) is what makes it viable on older +hardware — NVLink just removes the inter-GPU bottleneck for 262K context. + +### Image Generation (Included — No Extra Cost) + +The RTX 5000 has **384 Tensor Cores with native FP16** — full SDXL/Flux support, no hacks. + +| Workload | VRAM Needed | Where It Runs | +|----------|-------------|---------------| +| SDXL (1024x1024) | ~8-10GB | Either card alone | +| Flux Dev | ~12-14GB | Single card (16GB) | +| Flux Dev (high-res / batched) | ~18-24GB | Both cards via NVLink (32GB) | +| ComfyUI / InvokeAI | Works natively | No `--force-fp32` needed | + +**Run both workloads simultaneously:** +```bash +# Option A: Dedicated cards (no model swapping) +CUDA_VISIBLE_DEVICES=0 # Ollama — coding LLM on GPU 0 +CUDA_VISIBLE_DEVICES=1 # ComfyUI/InvokeAI — image gen on GPU 1 + +# Option B: Both cards unified for whichever task you're doing +# Switch between LLM (32GB, 262K context) and image gen (32GB, high-res batches) +``` + +### Hardware Longevity: 3-5 Years Realistic + +- **2026-2027**: Sweet spot. MoE + linear attention models are getting smaller active params. + 32GB unified handles the best coding models at full context. Peak value. +- **2028-2029**: Still useful. The trend is more efficient models, not bigger ones. + 32GB likely still runs the best ~35-70B MoE coding models of that era. +- **2030+**: Questionable. New architectures may need FP8, newer tensor core ops that + Turing lacks. But VRAM is VRAM — something useful will always run on 32GB. +- **The cards themselves won't die** — Quadro-grade, designed for 24/7 data center use. + They'll be outclassed before they fail. + ### Power Consumption & Cost | Config | Idle | Load | Monthly (8hr/day @ $0.09/kWh) | Annual | @@ -290,24 +353,37 @@ watch -n 0.5 nvidia-smi | Item | Cost | |------|------| -| 1x Quadro RTX 5000 (Phase 1) | $400 | -| 1x Quadro RTX 5000 (Phase 2) | $400 | -| NVLink HB Bridge 3-slot | ~$50 | -| **Hardware total** | **$850** | +| 2x Quadro RTX 5000 | ~$800 | +| NVLink HB Bridge 2-slot (P4934) | ~$50 | +| Dell cables/riser (R720/R730) | ~$60 | +| Dell 1100W PSUs (if needed) | ~$60-100 | +| **Hardware total** | **~$960** | | Monthly power (2 cards, 8hr/day) | $9/mo | -| Claude Pro subscription | $20/mo | -| **Monthly operating cost** | **~$29/mo** | -| **3-year total cost of ownership** | **$850 + $1,044 = $1,894** | +| Claude Pro subscription (keep) | $20/mo | +| Claude Max subscription (drop) | -$100/mo saved | +| **Net monthly cost** | **$29/mo (was $100/mo)** | + +### The Math: Drop Max, Keep Pro, Add Local + +| | Year 1 | Year 2 | Year 3 | **3-Year Total** | +|---|--------|--------|--------|-----------------| +| **Claude Max (current)** | $1,200 | $1,200 | $1,200 | **$3,600** | +| **Pro + Local GPU** | $960 + $348 | $348 | $348 | **$2,004** | +| **Savings** | | | | **$1,596** | + +You save ~$71/mo after hardware payoff. The GPU pays for itself in **13 months**. +After that, you're saving $80/mo vs Max with no usage limits on bulk work. ### Comparison: This Build vs Alternatives -| Setup | Cost (3yr) | Best Model | Max Context | Quality | -|-------|-----------|------------|-------------|---------| -| **2x RTX 5000 + $20 Pro** | **$1,894** | Qwen3.5-35B-A3B + Opus | 262K local | **A- local, A+ cloud** | -| 1x RTX 3090 + $20 Pro | $1,420 | Qwen3.5-35B-A3B + Opus | ~128K local | A- local, A+ cloud | -| RTX 8000 (48GB) + $20 Pro | $3,220+ | Qwen3.5-35B-A3B + Opus | 262K+ local | A- local, A+ cloud | -| Claude Max only (no GPU) | $3,600 | Opus 4.6 | 200K | A+ cloud only | -| API-only (Opus heavy use) | $18,000+ | Opus 4.6 | 200K | A+ cloud only | +| Setup | Monthly | 3yr Total | Limits? | Quality | +|-------|---------|-----------|---------|---------| +| **Pro + 2x RTX 5000** | **$29** | **$2,004** | **Unlimited local, Pro limits for Opus** | **A- local, A+ cloud** | +| Pro + 1x RTX 3090 | $25 | $1,620 | Unlimited local (128K ctx), Pro limits | A- local, A+ cloud | +| Pro + RTX 8000 (48GB) | $29 | $3,040+ | Unlimited local, Pro limits | A- local, A+ cloud | +| **Claude Max (no GPU)** | **$100** | **$3,600** | **Unlimited Opus** | **A+ cloud only** | +| Claude Pro only (no GPU) | $20 | $720 | **Hits caps on large projects** | A+ cloud, limited | +| API-only (Opus heavy use) | $500+ | $18,000+ | Pay per token | A+ cloud only | ### Phased Build Plan From 209173e98831b71b170bac769b41a80065370375 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 17:09:20 +0000 Subject: [PATCH 08/15] Fix: 262K context requires both cards unified, not simultaneous Both GPUs must be dedicated to the LLM for 262K context window. Image gen and LLM run one at a time (swap takes seconds). Simultaneous only possible with smaller context (~64K on one card). https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- docs/gpu-setup-research.md | 22 +++++++++++++++------- 1 file changed, 15 insertions(+), 7 deletions(-) diff --git a/docs/gpu-setup-research.md b/docs/gpu-setup-research.md index de120f5..dd9e7c3 100644 --- a/docs/gpu-setup-research.md +++ b/docs/gpu-setup-research.md @@ -264,16 +264,24 @@ The RTX 5000 has **384 Tensor Cores with native FP16** — full SDXL/Flux suppor | Flux Dev (high-res / batched) | ~18-24GB | Both cards via NVLink (32GB) | | ComfyUI / InvokeAI | Works natively | No `--force-fp32` needed | -**Run both workloads simultaneously:** -```bash -# Option A: Dedicated cards (no model swapping) -CUDA_VISIBLE_DEVICES=0 # Ollama — coding LLM on GPU 0 -CUDA_VISIBLE_DEVICES=1 # ComfyUI/InvokeAI — image gen on GPU 1 +**Important: 262K context requires both cards unified.** You can't split one off for image +gen and keep 262K. It's one task at a time: -# Option B: Both cards unified for whichever task you're doing -# Switch between LLM (32GB, 262K context) and image gen (32GB, high-res batches) +```bash +# CODING SESSION: Both cards unified → 32GB → 262K context +ollama run qwen3.5:35b-a3b-q4_K_M # Uses both GPUs via NVLink + +# IMAGE GEN SESSION: Stop LLM, run image gen on one card (16GB is plenty) +ollama stop # Frees VRAM +comfyui --listen 0.0.0.0 # SDXL/Flux fits easily in 16GB + +# Swap takes a few seconds, not simultaneous but not painful ``` +If you want simultaneous coding + image gen, you'd run a smaller model at shorter context +on one card (e.g., Qwen3.5-35B-A3B at ~64K on 16GB) and image gen on the other. But for +full 262K context, both cards must be dedicated to the LLM. + ### Hardware Longevity: 3-5 Years Realistic - **2026-2027**: Sweet spot. MoE + linear attention models are getting smaller active params. From 6d6a47e02877414453ebc1bf46d76eda92f687e3 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 17:11:49 +0000 Subject: [PATCH 09/15] Honest quality ratings: 35B-A3B is B+ not A-, add reality checks The 35B-A3B MoE activates only 3B params per token - quality tracks active params, making it more like a smart 7B than a true 35B. Revised rating to B+ to A-. The 27B dense model is the real A- but needs 17GB weights. Also notes: 262K is a VRAM ceiling not a quality guarantee, Q4 quantization costs something, and context quality degrades at edges. Practical high-quality context is more like 64-128K. https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- docs/gpu-setup-research.md | 16 +++++++++++++--- 1 file changed, 13 insertions(+), 3 deletions(-) diff --git a/docs/gpu-setup-research.md b/docs/gpu-setup-research.md index dd9e7c3..c6a96fd 100644 --- a/docs/gpu-setup-research.md +++ b/docs/gpu-setup-research.md @@ -33,12 +33,22 @@ than traditional transformers. | Model | Type | Active Params | Size at Q4_K_M | Max Context | Quality | Notes | |-------|------|---------------|---------------|-------------|---------|-------| -| **Qwen3.5-35B-A3B** | **MoE** | **3B** | **~12GB** | **262K** | **A-** | Best bang for buck — 35B model, 3B active, fits 262K ctx in 25GB | -| **Qwen3.5-27B** | Dense | 27B | ~17GB | 262K | A- | 72.4% SWE-bench, ties GPT-5 mini | +| **Qwen3.5-35B-A3B** | **MoE** | **3B** | **~12GB** | **262K** | **B+ to A-** | 35B total but only 3B active — quality tracks active params | +| **Qwen3.5-27B** | Dense | 27B | ~17GB | 262K | **A-** | 72.4% SWE-bench, ties GPT-5 mini. The real A- option. | | **Qwen3.5-122B-A10B** | MoE | 10B | ~76GB | 262K | A | Matches GPT-5 mini across the board | | **Qwen3.5-9B** | Dense | 9B | ~6GB | 262K | B+ | Fits on any modern GPU | | **Qwen3.5-4B** | Dense | 4B | ~3GB | 262K | B | Tiny but capable | +**Quality reality check:** MoE models route tokens through only a subset of parameters. +The 35B-A3B activates **3B params per token** — think of it as a smart 7B model, not a 35B. +Quality is closer to B+ for complex coding. The 27B dense model is genuinely A- but needs +17GB weights (leaving less room for context on 32GB). At Q4 quantization there's a further +small quality loss. And 262K is a VRAM ceiling, not a quality guarantee — models degrade +at the edges of their context window. Practical high-quality context is more like 64-128K. + +**No local model approaches Claude Opus on hard problems.** The strategy isn't to replace +Opus — it's to offload the 80% of routine work so your Pro plan limits stop being an issue. + ### Previous Generation (Still Relevant) | Model | Size at Q4_K_M | Quality | Notes | @@ -212,7 +222,7 @@ bridge sitting on top of the cards. A 4U chassis gives the most room. | Model | Arch | Quant | Weights | Context | Total VRAM | Quality | |-------|------|-------|---------|---------|------------|---------| -| **Qwen3.5-35B-A3B** | **MoE (3B active)** | **Q4_K_M** | **~12GB** | **262K** | **~25GB** | **A-** | +| **Qwen3.5-35B-A3B** | **MoE (3B active)** | **Q4_K_M** | **~12GB** | **262K** | **~25GB** | **B+ to A-** | | Qwen3.5-27B | Dense | Q4_K_M | ~17GB | 128K+ | ~25GB | A- | | Qwen3-Coder-Next (80B/3B active) | MoE | Q4 | ~20GB | 128K | ~28GB | A | | Qwen2.5-Coder-14B | Dense | Q4_K_M | ~10GB | 128K | ~22GB | B+ | From f775b704ae4d2c187c54c08fb2b7ce41df59ac36 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 17:15:00 +0000 Subject: [PATCH 10/15] Fix misleading single vs dual card comparison Single card table was showing A- for 35B-A3B which undercut the case for a second card. Restructured to lead with 16GB (start here) at honest B+ ratings, then show what 32GB unlocks: dense 27B/32B models that physically can't fit on 16GB and are genuinely A- quality. The upgrade isn't about 262K context - it's about accessing better models (27B dense > 35B MoE with 3B active params). https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- docs/gpu-setup-research.md | 49 ++++++++++++++++++++++---------------- 1 file changed, 28 insertions(+), 21 deletions(-) diff --git a/docs/gpu-setup-research.md b/docs/gpu-setup-research.md index c6a96fd..e0d1368 100644 --- a/docs/gpu-setup-research.md +++ b/docs/gpu-setup-research.md @@ -218,32 +218,39 @@ A 3-slot bridge leaves an air gap for proper cooling. this works well in rack airflow. If using a 2U server, measure clearance for the NVLink bridge sitting on top of the cards. A 4U chassis gives the most room. -### What Runs on 32GB Unified (2x RTX 5000 + NVLink) +### What Runs on 16GB (Single RTX 5000 — Start Here) -| Model | Arch | Quant | Weights | Context | Total VRAM | Quality | -|-------|------|-------|---------|---------|------------|---------| -| **Qwen3.5-35B-A3B** | **MoE (3B active)** | **Q4_K_M** | **~12GB** | **262K** | **~25GB** | **B+ to A-** | -| Qwen3.5-27B | Dense | Q4_K_M | ~17GB | 128K+ | ~25GB | A- | -| Qwen3-Coder-Next (80B/3B active) | MoE | Q4 | ~20GB | 128K | ~28GB | A | -| Qwen2.5-Coder-14B | Dense | Q4_K_M | ~10GB | 128K | ~22GB | B+ | -| Qwen2.5-Coder-14B | Dense | Q8 | ~16GB | 64K | ~28GB | A- | -| Qwen2.5-Coder-32B | Dense | Q4_K_M | ~20GB | 16-24K | ~28GB | A- | +| Model | Quant | Context | Quality | Notes | +|-------|-------|---------|---------|-------| +| **Qwen3.5-35B-A3B** | Q4_K_L | ~64-128K | **B+** | MoE, 3B active. Good but VRAM is tight — context may be lower | +| Qwen3.5-9B | Q8 | 128K+ | B+ | Fits comfortably, high quant | +| Qwen3.5-4B | Q8 | 262K | B | Tiny model, long context | +| Qwen2.5-Coder-7B | Q8 | 128K | B | Solid for simple tasks | +| Qwen2.5-Coder-14B | Q4_K_M | 16-32K | B+ | Tight fit, limited context | -**The sweet spot: Qwen3.5-35B-A3B at Q4_K_M with 262K context.** This is a 35B parameter -model with only 3B active at inference (MoE). The Gated Delta Network architecture slashes -KV cache memory. The entire model + full 262K context fits in ~25GB — well within 32GB. +A single card is a solid start — B+ coding with decent context. But 16GB is the ceiling. +You can't run bigger dense models, can't use higher quantization, and context is squeezed. -### What Runs on 16GB (Single RTX 5000 — Phase 1) +### What the Second Card + NVLink Unlocks (32GB) -| Model | Quant | Context | Quality | -|-------|-------|---------|---------| -| **Qwen3.5-35B-A3B** | Q4_K_L | ~64-128K | **A-** | -| Qwen3.5-9B | Q8 | 128K+ | B+ | -| Qwen3.5-4B | Q8 | 262K | B | -| Qwen2.5-Coder-7B | Q8 | 128K | B | -| Qwen2.5-Coder-14B | Q4_K_M | 16-32K | B+ | +The second card doesn't just double context — it opens models that **don't fit on 16GB at all:** -Even a single card can run the Qwen3.5-35B-A3B MoE model — just with a smaller context window. +| Model | Arch | Quant | Weights | Context | Total VRAM | Quality | **Why it needs 32GB** | +|-------|------|-------|---------|---------|------------|---------|----------------------| +| **Qwen3.5-27B** | **Dense** | **Q4_K_M** | **~17GB** | **128K+** | **~25GB** | **A-** | **17GB weights won't fit on 16GB** | +| **Qwen2.5-Coder-32B** | **Dense** | **Q4_K_M** | **~20GB** | **16-24K** | **~28GB** | **A-** | **20GB weights won't fit on 16GB** | +| Qwen2.5-Coder-14B | Dense | **Q8** | ~16GB | 64K | ~28GB | A- | Q8 quant = better output, needs 16GB for weights alone | +| Qwen3-Coder-Next (80B) | MoE | Q4 | ~20GB | 128K | ~28GB | A | 20GB weights won't fit on 16GB | +| Qwen3.5-35B-A3B | MoE (3B active) | Q4_K_M | ~12GB | 262K | ~25GB | B+ | Fits on 1 card at reduced context, but 32GB = full 262K + headroom | + +**The real upgrade isn't 262K context — it's access to dense 27B/32B models that are +genuinely A- quality.** The 35B-A3B MoE runs on both setups, but its 3B active params +limit quality. The Qwen3.5-27B dense model uses all 27B params on every token — that's +the quality jump. And its 17GB of weights physically can't fit on a single 16GB card. + +Think of it this way: +- **1 card**: B+ coding (MoE or small dense models, squeezed context) +- **2 cards**: **A- coding** (full dense 27B/32B models, comfortable context, higher quant options) ### Estimated Inference Speed From 1d78957083aed6ab401bab9ff77491066c63a1d1 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 17:22:47 +0000 Subject: [PATCH 11/15] =?UTF-8?q?Add=20practical=20context=20windows=20sec?= =?UTF-8?q?tion=20=E2=80=94=20usability=20not=20ceilings?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit VRAM must hold weights AND KV cache. What's left after weights is your real context window. For a 10K-line project (~100-150K tokens): - 1 card: 32-100K usable, file-by-file workflow - 2 cards: 80-180K usable, whole-project-in-one-shot workflow - This Opus session: 1M tokens (neither setup comes close) Comparison table vs Claude Opus 1M session for perspective. Keep Pro for the hard stuff, use local for the daily 80%. https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- docs/gpu-setup-research.md | 44 ++++++++++++++++++++++++++++++++++++++ 1 file changed, 44 insertions(+) diff --git a/docs/gpu-setup-research.md b/docs/gpu-setup-research.md index e0d1368..714e6ba 100644 --- a/docs/gpu-setup-research.md +++ b/docs/gpu-setup-research.md @@ -252,6 +252,50 @@ Think of it this way: - **1 card**: B+ coding (MoE or small dense models, squeezed context) - **2 cards**: **A- coding** (full dense 27B/32B models, comfortable context, higher quant options) +### Practical Context Windows (Usability, Not Ceilings) + +Context window "support" is a ceiling, not what you actually get. VRAM must hold both the +model weights AND the KV cache. What's left after weights determines your real context. +Quality also degrades toward the edges of a model's context window. + +**Reference: A 10,000-line codebase ≈ 100-150K tokens** (varies by language/comments). +This Claude Opus session uses a **1 million token** context window for comparison. + +#### 1 Card (16GB) — Practical + +| Model | Weights | Free for KV | **Usable context** | 10K-line project? | +|-------|---------|-------------|-------------------|-------------------| +| Qwen3.5-35B-A3B (MoE) | ~12GB | ~3GB | **32-50K tokens** | **No — ~1/3 of it** | +| Qwen3.5-9B (dense) | ~6GB | ~9GB | **80-100K tokens** | **Mostly — but B+ quality** | +| Qwen2.5-Coder-14B | ~10GB | ~5GB | **16-24K tokens** | **No — a few files at a time** | + +**Workflow on 1 card:** You're feeding files in chunks. Good for "fix this function" or +"explain this file." Not for "read my whole project and refactor the auth system." + +#### 2 Cards (32GB via NVLink) — Practical + +| Model | Weights | Free for KV | **Usable context** | 10K-line project? | +|-------|---------|-------------|-------------------|-------------------| +| **Qwen3.5-27B (dense)** | ~17GB | ~14GB | **80-128K tokens** | **Yes — most/all of it at A-** | +| Qwen3.5-35B-A3B (MoE) | ~12GB | ~19GB | **128-180K tokens** | **Yes with room to spare (B+)** | +| Qwen2.5-Coder-32B | ~20GB | ~11GB | **32-48K tokens** | **Partial — but strong A- on what it sees** | + +**Workflow on 2 cards:** You can dump most/all of a 10K-line project in one shot with the +27B dense model. That's the real workflow change — "here's my whole project, find the bug" +becomes possible locally. + +#### vs This Claude Session + +| Setup | Usable context | vs Opus 1M | Whole-project workflow? | +|-------|---------------|------------|----------------------| +| 1x RTX 5000 (best) | ~50-100K | 5-10% | No — file by file | +| **2x RTX 5000 (best)** | **~128-180K** | **13-18%** | **Yes — for 10K-line projects** | +| Claude Opus (this session) | 1,000K | 100% | Yes — for anything | + +**Neither setup replaces this session** for complex multi-file work across a 50K+ line +codebase. That's why you keep Pro. But 2 cards handles the daily "read my project and +help me code" workflow locally with no rate limits — and that's 80% of the work. + ### Estimated Inference Speed | Model | 1x RTX 5000 | 2x RTX 5000 (NVLink) | From 0f57690692532074469b9078095e2c95f9c2485a Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 17:36:06 +0000 Subject: [PATCH 12/15] =?UTF-8?q?Add=20single-card=20optimization=20guide?= =?UTF-8?q?=20=E2=80=94=20squeeze=20every=20byte=20from=2016GB?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Six stackable techniques that compound: 1. KV cache quantization (Q8_0 = 2x context, asymmetric K=Q8/V=Q4 = 2.6x) 2. Flash attention (free VRAM + speed, zero quality loss) 3. Host-memory prompt caching (--cram, use the server's 128-384GB RAM) 4. KV to system RAM (-nkvo, last resort, 5-20x slower) 5. Architecture selection (GQA + MoE = tiny KV footprint) 6. NVMe mmap for model loading (fast cold starts, not inference) Stacked result: single card goes from ~50K to ~130K usable context with the MoE model. Updated llama.cpp config with all flags. https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- docs/gpu-setup-research.md | 166 ++++++++++++++++++++++++++++++++++++- 1 file changed, 165 insertions(+), 1 deletion(-) diff --git a/docs/gpu-setup-research.md b/docs/gpu-setup-research.md index 714e6ba..9519d87 100644 --- a/docs/gpu-setup-research.md +++ b/docs/gpu-setup-research.md @@ -296,6 +296,151 @@ becomes possible locally. codebase. That's why you keep Pro. But 2 cards handles the daily "read my project and help me code" workflow locally with no rate limits — and that's 80% of the work. +### Squeezing Every Byte: Single-Card Optimization (16GB) + +Before buying a second card, stack these techniques. They're cumulative — use all of them +together. The gains compound because they all free VRAM from the same bottleneck: KV cache. + +#### 1. Quantize the KV Cache (Biggest Single Win) + +By default, llama.cpp stores the KV cache in FP16. That's 2 bytes per value. You can +compress it with zero code changes — just flags: + +| Cache Type | Bytes/value | vs FP16 | Quality Impact | Verdict | +|-----------|-------------|---------|----------------|---------| +| FP16 (default) | 2.0 | baseline | none | wasteful on 16GB | +| **Q8_0** | **1.0** | **50% smaller** | **~0.002-0.05 perplexity** | **Always use this** | +| Q4_0 | 0.5 | 75% smaller | ~0.2 perplexity (noticeable) | Use if desperate | +| **Asymmetric: K=Q8_0, V=Q4_0** | **0.75 avg** | **62% smaller** | **Better than uniform Q4** | **Best bang/buck** | + +The K cache is more sensitive to quantization than V. Asymmetric (Q8 keys, Q4 values) gives +you ~62% savings with quality closer to Q8 than Q4. + +**Concrete example — Qwen3.5-35B-A3B on 1 card (16GB):** +- Weights: ~12GB → 4GB free for KV cache +- FP16 KV cache: 4GB → **~50K context** +- Q8_0 KV cache: 4GB buys 2x → **~100K context** +- K=Q8/V=Q4 KV cache: 4GB buys 2.6x → **~130K context** + +That's the difference between "a few files" and "a meaningful chunk of a project." + +```bash +# llama.cpp — always use these three flags together +llama-server \ + --cache-type-k q8_0 \ + --cache-type-v q4_0 \ + --flash-attn \ + -m model.gguf -ngl 99 -c 131072 + +# Ollama — set environment variable before starting +export OLLAMA_KV_CACHE_TYPE=q8_0 # or q4_0 for aggressive +export OLLAMA_FLASH_ATTENTION=1 +ollama serve +``` + +#### 2. Flash Attention (Free Speed + VRAM) + +Flash attention restructures how attention is computed — instead of materializing the full +attention matrix in VRAM, it computes it in tiles. Result: less VRAM used during inference, +slightly faster, **zero quality loss**. + +Always enable it. There's no downside on Turing GPUs with quantized KV cache. + +```bash +# llama.cpp +--flash-attn + +# Ollama +export OLLAMA_FLASH_ATTENTION=1 +``` + +#### 3. Host-Memory Prompt Caching (`--cram`) — System RAM as L2 Cache + +This is the smart use of system RAM. The `--cram` flag in llama-server stores pre-computed +prompt representations in host memory (system RAM). When you send the same system prompt +or reuse a conversation prefix, it skips reprocessing — hot-swaps the cached computation +back onto the GPU. + +This doesn't increase context window size, but it **dramatically reduces time-to-first-token** +for repeated workflows (which is most coding — same system prompt, same project context). + +```bash +# llama-server with 16GB RAM cache for prompts +llama-server \ + --cram 16384 \ + --cache-type-k q8_0 --cache-type-v q4_0 --flash-attn \ + -m model.gguf -ngl 99 -c 131072 +``` + +Your R720/R730 has 128-384GB of DDR3/DDR4 RAM. Use it. `--cram 65536` (64GB) is reasonable +for a dedicated inference server — it costs nothing, and repeat prompts become near-instant. + +#### 4. KV Cache to System RAM (`-nkvo`) — Last Resort for Context + +The `-nkvo` (no KV offload) flag moves the entire KV cache to system RAM, freeing all 16GB +of VRAM for model weights. This sounds great but comes with a brutal speed penalty: + +| Scenario | Speed Impact | +|----------|-------------| +| Full VRAM (normal) | Baseline (25-35 tok/s) | +| KV in system RAM via PCIe | **5-20x slower** (~2-7 tok/s) | +| KV on NVMe via mmap | **30x+ slower** (~1 tok/s) | + +**When it makes sense:** Loading a model that barely doesn't fit (e.g., Qwen3.5-27B dense +at 17GB weights on a 16GB card). You'd get ~2-5 tok/s with KV in RAM — painfully slow, but +it's the difference between "runs slowly" and "doesn't run at all." Fine for a batch job +where you walk away and come back. Not viable for interactive coding. + +**Don't do this routinely.** Quantized KV cache (technique #1) is 50-100x better because +the cache stays on the GPU. Only use `-nkvo` for models that literally can't fit otherwise. + +#### 5. Pick the Right Architecture (GQA + MoE = VRAM Efficient) + +Not all models consume KV cache equally. Modern architectures with **Grouped Query Attention +(GQA)** use far less KV cache than older Multi-Head Attention (MHA): + +| Architecture | KV cache at 64K context | Examples | +|-------------|------------------------|---------| +| MHA (old) | ~8-12GB | LLaMA-1, GPT-J | +| **GQA (modern)** | **~1-3GB** | **Qwen3.5 series, LLaMA-3** | +| **GQA + MoE** | **~1.2GB** | **Qwen3.5-35B-A3B** | + +The Qwen3.5-35B-A3B is almost purpose-built for your situation: 3B active params (fast on +Turing), MoE architecture (small memory footprint during inference), and GQA (tiny KV cache). +With quantized KV on top of that, 130K+ context on a single 16GB card is realistic. + +#### 6. NVMe as mmap Backing Store + +Your fast NVMe matters for **model loading**, not inference. llama.cpp uses mmap by default +to stream model weights from disk, so a fast NVMe means: +- Near-instant cold starts (weights stream in as needed) +- Graceful degradation if model slightly exceeds RAM (OS pages out unused layers) + +But NVMe is **not** a viable substitute for VRAM during inference. The bandwidth gap is too +large: VRAM runs at ~400 GB/s (RTX 5000), system RAM at ~50-100 GB/s (DDR4 quad-channel), +NVMe at ~3-7 GB/s. Three orders of magnitude difference from VRAM. + +**Practical use:** Keep all your GGUF model files on NVMe. Enable mmap (default). That's it. +Don't try to use NVMe as overflow for the KV cache — the latency kills interactive use. + +#### Stacking Everything: Revised Single-Card Numbers + +| Model | Optimization | Usable Context | Speed | Quality | +|-------|-------------|---------------|-------|---------| +| Qwen3.5-35B-A3B Q4 | None (defaults) | ~32-50K | 25-35 tok/s | B+ | +| Qwen3.5-35B-A3B Q4 | **KV Q8 + flash** | **~80-100K** | **25-35 tok/s** | **B+** | +| Qwen3.5-35B-A3B Q4 | **KV asym + flash** | **~100-130K** | **25-35 tok/s** | **B+ (tiny quality dip)** | +| Qwen3.5-9B Q8 | KV Q8 + flash | ~120-160K | 30-45 tok/s | B | +| Qwen2.5-Coder-14B Q4 | KV Q8 + flash | ~40-64K | 20-30 tok/s | B+ | + +Add `--cram` on top for instant repeated prompts. That's your real single-card ceiling. + +**The honest answer:** With all optimizations stacked, a single card goes from "a few files +at a time" to "maybe half a 10K-line project." That's a meaningful upgrade from the +unoptimized baseline, but it still doesn't match what 2 cards with a dense 27B model gives +you. The second card isn't about optimization tricks — it's about physics (more VRAM = more +data on the fast bus). + ### Estimated Inference Speed | Model | 1x RTX 5000 | 2x RTX 5000 (NVLink) | @@ -375,16 +520,35 @@ cmake --build build --config Release -j$(nproc) # Download Qwen3.5-35B-A3B GGUF (Q4_K_M) # Get from: https://huggingface.co/unsloth/Qwen3.5-35B-A3B-GGUF -# Run on dual GPU with NVLink +# Run on dual GPU with NVLink (all optimizations on) ./build/bin/llama-server \ -m Qwen3.5-35B-A3B-Q4_K_M.gguf \ -ngl 999 \ -c 262144 \ + --cache-type-k q8_0 \ + --cache-type-v q4_0 \ + --flash-attn \ + --cram 65536 \ --host 0.0.0.0 \ --port 8080 +# --cache-type-k q8_0 / --cache-type-v q4_0 = asymmetric KV quantization (62% smaller cache) +# --flash-attn = tiled attention (less VRAM, no quality loss) +# --cram 65536 = 64GB host RAM prompt cache (instant repeat prompts) # llama.cpp auto-detects NVLink and splits layers across both GPUs # Use -ts 1,1 to manually set equal split if needed + +# Single card variant (no NVLink) — same flags, smaller context +./build/bin/llama-server \ + -m Qwen3.5-35B-A3B-Q4_K_M.gguf \ + -ngl 999 \ + -c 131072 \ + --cache-type-k q8_0 \ + --cache-type-v q4_0 \ + --flash-attn \ + --cram 65536 \ + --host 0.0.0.0 \ + --port 8080 ``` #### Ollama From 8989815afa2df388fe8b3e5ffe2d7291f6d23a83 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 17:47:05 +0000 Subject: [PATCH 13/15] =?UTF-8?q?Wire=20in=20GPU=20optimizations=20and=20G?= =?UTF-8?q?itea=E2=86=94GitHub=20sync?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Ollama config (all 3 setup scripts): - Models updated from Qwen 2.5 → Qwen 3.5 series (Feb 2026) - Auto-detect multi-GPU: TOTAL_VRAM = per-card × count - KV cache quantization (q8_0) + flash attention enabled by default - Context windows scaled by total VRAM (4K→128K) - RAG server CHAT_MODEL now uses detected model variable Gitea↔GitHub sync (new): - gitea-github-sync.sh: bidirectional mirror with --init wizard - Modes: --pull-only, --push-only, --list (dry run), --repo single - Auto-discovers repos from both platforms via API - Systemd timer: --install-timer [interval] for scheduled sync - MCP tool: gitea_github_sync() for on-demand from Claude/WebUI - Sync script mounted read-only into mcp-server container - .env gets GITEA_URL variable for sync script - curl added to mcp-server container deps https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- gitea-github-sync.sh | 464 +++++++++++++++++++++++++++++++++++++++++ laptop_full_setup.sh | 114 ++++++---- local-ai-setup.sh | 44 ++-- mcp_server.py | 19 ++ ubuntu-post-install.sh | 38 ++-- 5 files changed, 609 insertions(+), 70 deletions(-) create mode 100755 gitea-github-sync.sh diff --git a/gitea-github-sync.sh b/gitea-github-sync.sh new file mode 100755 index 0000000..2624cdb --- /dev/null +++ b/gitea-github-sync.sh @@ -0,0 +1,464 @@ +#!/usr/bin/env bash +# ============================================================================= +# Gitea ↔ GitHub Mirror Sync +# +# Mirrors repos between your local Gitea and GitHub in both directions: +# GitHub → Gitea: Pulls repos you own on GitHub into Gitea (backup/offline use) +# Gitea → GitHub: Pushes Gitea repos to GitHub (remote backup) +# +# Usage: +# ./gitea-github-sync.sh — sync all configured repos +# ./gitea-github-sync.sh --pull-only — GitHub → Gitea only +# ./gitea-github-sync.sh --push-only — Gitea → GitHub only +# ./gitea-github-sync.sh --repo owner/name — sync one specific repo +# ./gitea-github-sync.sh --list — list what would sync (dry run) +# ./gitea-github-sync.sh --init — interactive first-time setup +# +# Config: ~/.config/gitea-github-sync/config +# Tokens: reads from .env in the same directory as this script (or $SYNC_ENV) +# +# Schedule: install the systemd timer with --install-timer +# ./gitea-github-sync.sh --install-timer — every 6 hours (default) +# ./gitea-github-sync.sh --install-timer 1h — custom interval +# ./gitea-github-sync.sh --remove-timer — remove the timer +# ============================================================================= +set -euo pipefail + +RED='\033[0;31m'; GREEN='\033[0;32m'; YELLOW='\033[1;33m' +CYAN='\033[0;36m'; BOLD='\033[1m'; NC='\033[0m' +info() { echo -e "${CYAN}[sync]${NC} $*"; } +ok() { echo -e "${GREEN}[ ok ]${NC} $*"; } +warn() { echo -e "${YELLOW}[warn]${NC} $*"; } +err() { echo -e "${RED}[err ]${NC} $*" >&2; } + +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +CONFIG_DIR="${XDG_CONFIG_HOME:-$HOME/.config}/gitea-github-sync" +CONFIG_FILE="$CONFIG_DIR/config" +WORK_DIR="$CONFIG_DIR/repos" +LOG_FILE="$CONFIG_DIR/sync.log" + +# ── load tokens from .env ─────────────────────────────────────────────────── +ENV_FILE="${SYNC_ENV:-$SCRIPT_DIR/.env}" +if [[ -f "$ENV_FILE" ]]; then + # shellcheck disable=SC1090 + set -a; source <(grep -E '^(GITEA_TOKEN|GITHUB_TOKEN|GITEA_URL)=' "$ENV_FILE" | sed 's/ *#.*//'); set +a +fi + +GITEA_URL="${GITEA_URL:-http://localhost:3001}" +GITEA_TOKEN="${GITEA_TOKEN:-}" +GITHUB_TOKEN="${GITHUB_TOKEN:-}" + +# ── parse args ────────────────────────────────────────────────────────────── +MODE="all" # all | pull | push | list | init | install-timer | remove-timer +SINGLE_REPO="" +TIMER_INTERVAL="6h" + +while [[ $# -gt 0 ]]; do + case "$1" in + --pull-only) MODE="pull"; shift ;; + --push-only) MODE="push"; shift ;; + --list) MODE="list"; shift ;; + --init) MODE="init"; shift ;; + --install-timer) MODE="install-timer"; shift; [[ "${1:-}" =~ ^[0-9]+[smhd]$ ]] && { TIMER_INTERVAL="$1"; shift; } ;; + --remove-timer) MODE="remove-timer"; shift ;; + --repo) shift; SINGLE_REPO="${1:-}"; shift ;; + -h|--help) + sed -n '2,/^# =====/{ /^# =====/d; s/^# \?//p; }' "$0"; exit 0 ;; + *) err "Unknown arg: $1"; exit 1 ;; + esac +done + +# ── helpers ───────────────────────────────────────────────────────────────── +_gitea_api() { + local method="$1" path="$2"; shift 2 + curl -sfL -X "$method" \ + -H "Authorization: token $GITEA_TOKEN" \ + -H "Content-Type: application/json" \ + "$GITEA_URL/api/v1$path" "$@" +} + +_github_api() { + local method="$1" path="$2"; shift 2 + curl -sfL -X "$method" \ + -H "Authorization: Bearer $GITHUB_TOKEN" \ + -H "Accept: application/vnd.github+json" \ + "https://api.github.com$path" "$@" +} + +_log() { echo "[$(date '+%Y-%m-%d %H:%M:%S')] $*" >> "$LOG_FILE"; } + +# ── config management ────────────────────────────────────────────────────── +load_config() { + mkdir -p "$CONFIG_DIR" "$WORK_DIR" + GITHUB_USER="" + GITEA_USER="" + SYNC_REPOS=() # explicit list (empty = auto-discover) + EXCLUDE_REPOS=() # repos to skip + PUSH_PRIVATE=false # push private Gitea repos to GitHub? + PULL_PRIVATE=true # pull private GitHub repos to Gitea? + PULL_FORKS=false # pull forked repos from GitHub? + + if [[ -f "$CONFIG_FILE" ]]; then + # shellcheck disable=SC1090 + source "$CONFIG_FILE" + fi +} + +save_config() { + mkdir -p "$CONFIG_DIR" + cat > "$CONFIG_FILE" << EOF +# Gitea-GitHub Sync — configuration +# Generated $(date '+%Y-%m-%d %H:%M:%S') + +# GitHub username (for discovering repos to pull) +GITHUB_USER="$GITHUB_USER" + +# Gitea username (for discovering repos to push) +GITEA_USER="$GITEA_USER" + +# Explicit repo list — if set, only these sync. Format: owner/repo +# Leave empty () to auto-discover from both platforms. +SYNC_REPOS=($(printf '"%s" ' "${SYNC_REPOS[@]}")) + +# Repos to skip (pattern matched against owner/repo) +EXCLUDE_REPOS=($(printf '"%s" ' "${EXCLUDE_REPOS[@]}")) + +# Push private Gitea repos to GitHub as private repos? +PUSH_PRIVATE=$PUSH_PRIVATE + +# Pull private GitHub repos to Gitea? +PULL_PRIVATE=$PULL_PRIVATE + +# Pull forked repos from GitHub? +PULL_FORKS=$PULL_FORKS +EOF + ok "Config saved: $CONFIG_FILE" +} + +# ── init (first-time setup) ──────────────────────────────────────────────── +do_init() { + echo -e "\n${BOLD}Gitea ↔ GitHub Sync — First-Time Setup${NC}\n" + + # Check tokens + if [[ -z "$GITEA_TOKEN" || "$GITEA_TOKEN" == "your-gitea-token-here" ]]; then + err "GITEA_TOKEN not set. Add it to $ENV_FILE first." + echo " Generate at: $GITEA_URL/user/settings/applications" + exit 1 + fi + if [[ -z "$GITHUB_TOKEN" || "$GITHUB_TOKEN" == "your-github-token-here" ]]; then + err "GITHUB_TOKEN not set. Add it to $ENV_FILE first." + echo " Generate at: https://github.com/settings/tokens" + echo " Scopes needed: repo (full control)" + exit 1 + fi + + # Discover usernames + info "Detecting GitHub user..." + GITHUB_USER=$(_github_api GET /user | python3 -c "import sys,json; print(json.load(sys.stdin)['login'])" 2>/dev/null) \ + || { err "Failed to reach GitHub API. Check GITHUB_TOKEN."; exit 1; } + ok "GitHub user: $GITHUB_USER" + + info "Detecting Gitea user..." + GITEA_USER=$(_gitea_api GET /user | python3 -c "import sys,json; print(json.load(sys.stdin)['login'])" 2>/dev/null) \ + || { err "Failed to reach Gitea API. Check GITEA_TOKEN and GITEA_URL ($GITEA_URL)."; exit 1; } + ok "Gitea user: $GITEA_USER" + + # Ask about sync scope + echo "" + read -rp "Pull private GitHub repos to Gitea? [Y/n] " ans + PULL_PRIVATE=true; [[ "${ans,,}" == "n" ]] && PULL_PRIVATE=false + + read -rp "Pull forked repos from GitHub? [y/N] " ans + PULL_FORKS=false; [[ "${ans,,}" == "y" ]] && PULL_FORKS=true + + read -rp "Push private Gitea repos to GitHub? [y/N] " ans + PUSH_PRIVATE=false; [[ "${ans,,}" == "y" ]] && PUSH_PRIVATE=true + + save_config + + echo "" + info "Run '$(basename "$0") --list' to preview what would sync." + info "Run '$(basename "$0")' to sync now." + info "Run '$(basename "$0") --install-timer' to sync automatically." +} + +# ── discover repos ───────────────────────────────────────────────────────── +get_github_repos() { + local page=1 repos=() + while true; do + local batch + batch=$(_github_api GET "/user/repos?per_page=100&page=$page&affiliation=owner" \ + | python3 -c " +import sys, json +for r in json.load(sys.stdin): + if r.get('fork') and not $($PULL_FORKS && echo True || echo False): + continue + if r.get('private') and not $($PULL_PRIVATE && echo True || echo False): + continue + print(r['full_name'] + '|' + r['clone_url'] + '|' + str(r.get('private',False)).lower()) +" 2>/dev/null) || break + [[ -z "$batch" ]] && break + while IFS= read -r line; do repos+=("$line"); done <<< "$batch" + ((page++)) + done + printf '%s\n' "${repos[@]}" +} + +get_gitea_repos() { + local page=1 repos=() + while true; do + local batch + batch=$(_gitea_api GET "/repos/search?limit=50&page=$page" \ + | python3 -c " +import sys, json +for r in json.load(sys.stdin).get('data', []): + if r.get('private') and not $($PUSH_PRIVATE && echo True || echo False): + continue + print(r['full_name'] + '|' + r['clone_url'] + '|' + str(r.get('private',False)).lower()) +" 2>/dev/null) || break + [[ -z "$batch" ]] && break + while IFS= read -r line; do repos+=("$line"); done <<< "$batch" + ((page++)) + done + printf '%s\n' "${repos[@]}" +} + +is_excluded() { + local repo="$1" + for pat in "${EXCLUDE_REPOS[@]}"; do + [[ "$repo" == $pat ]] && return 0 + done + return 1 +} + +# ── sync: GitHub → Gitea (pull) ─────────────────────────────────────────── +sync_github_to_gitea() { + local full_name="$1" clone_url="$2" is_private="$3" + local repo_name="${full_name#*/}" + local local_path="$WORK_DIR/$full_name" + + # Clone or fetch from GitHub + if [[ -d "$local_path" ]]; then + info "Fetching $full_name from GitHub..." + git -C "$local_path" fetch --all --prune --quiet 2>/dev/null || { + err "Failed to fetch $full_name"; return 1; } + else + info "Cloning $full_name from GitHub..." + mkdir -p "$(dirname "$local_path")" + local auth_url="${clone_url/https:\/\//https:\/\/$GITHUB_TOKEN@}" + git clone --bare --quiet "$auth_url" "$local_path" 2>/dev/null || { + err "Failed to clone $full_name"; return 1; } + fi + + # Ensure repo exists on Gitea + local gitea_check + gitea_check=$(_gitea_api GET "/repos/$GITEA_USER/$repo_name" 2>/dev/null) || true + if ! echo "$gitea_check" | python3 -c "import sys,json; json.load(sys.stdin)['id']" &>/dev/null; then + info "Creating $repo_name on Gitea..." + _gitea_api POST "/user/repos" \ + -d "{\"name\":\"$repo_name\",\"private\":$is_private,\"description\":\"Mirror of $full_name from GitHub\"}" \ + >/dev/null || { err "Failed to create $repo_name on Gitea"; return 1; } + fi + + # Push to Gitea + local gitea_push_url="${GITEA_URL/https:\/\//https:\/\/$GITEA_USER:$GITEA_TOKEN@}" + gitea_push_url="${gitea_push_url/http:\/\//http:\/\/$GITEA_USER:$GITEA_TOKEN@}" + gitea_push_url="$gitea_push_url/$GITEA_USER/$repo_name.git" + + git -C "$local_path" push --mirror "$gitea_push_url" --quiet 2>/dev/null || { + err "Failed to push $full_name to Gitea"; return 1; } + ok "GitHub → Gitea: $full_name" + _log "PULL $full_name OK" +} + +# ── sync: Gitea → GitHub (push) ─────────────────────────────────────────── +sync_gitea_to_github() { + local full_name="$1" clone_url="$2" is_private="$3" + local repo_name="${full_name#*/}" + local local_path="$WORK_DIR/gitea/$full_name" + + # Clone or fetch from Gitea + local gitea_auth_url="${clone_url/https:\/\//https:\/\/$GITEA_USER:$GITEA_TOKEN@}" + gitea_auth_url="${gitea_auth_url/http:\/\//http:\/\/$GITEA_USER:$GITEA_TOKEN@}" + + if [[ -d "$local_path" ]]; then + info "Fetching $full_name from Gitea..." + git -C "$local_path" fetch --all --prune --quiet 2>/dev/null || { + err "Failed to fetch $full_name from Gitea"; return 1; } + else + info "Cloning $full_name from Gitea..." + mkdir -p "$(dirname "$local_path")" + git clone --bare --quiet "$gitea_auth_url" "$local_path" 2>/dev/null || { + err "Failed to clone $full_name from Gitea"; return 1; } + fi + + # Ensure repo exists on GitHub + local gh_check + gh_check=$(_github_api GET "/repos/$GITHUB_USER/$repo_name" 2>/dev/null) || true + if ! echo "$gh_check" | python3 -c "import sys,json; json.load(sys.stdin)['id']" &>/dev/null; then + info "Creating $repo_name on GitHub..." + _github_api POST "/user/repos" \ + -d "{\"name\":\"$repo_name\",\"private\":$is_private,\"description\":\"Mirror from Gitea\"}" \ + >/dev/null || { err "Failed to create $repo_name on GitHub"; return 1; } + fi + + # Push to GitHub + local github_push_url="https://$GITHUB_TOKEN@github.com/$GITHUB_USER/$repo_name.git" + git -C "$local_path" push --mirror "$github_push_url" --quiet 2>/dev/null || { + err "Failed to push $full_name to GitHub"; return 1; } + ok "Gitea → GitHub: $full_name" + _log "PUSH $full_name OK" +} + +# ── list (dry run) ───────────────────────────────────────────────────────── +do_list() { + echo -e "\n${BOLD}Repos that would sync:${NC}\n" + + if [[ ${#SYNC_REPOS[@]} -gt 0 ]]; then + echo -e "${CYAN}Explicit list:${NC}" + printf ' %s\n' "${SYNC_REPOS[@]}" + else + if [[ "$MODE" != "push" ]]; then + echo -e "${CYAN}GitHub → Gitea (pull):${NC}" + get_github_repos | while IFS='|' read -r name url priv; do + is_excluded "$name" && echo " $name (excluded)" && continue + echo " $name $([ "$priv" = "true" ] && echo "[private]")" + done + fi + echo "" + if [[ "$MODE" != "pull" ]]; then + echo -e "${CYAN}Gitea → GitHub (push):${NC}" + get_gitea_repos | while IFS='|' read -r name url priv; do + is_excluded "$name" && echo " $name (excluded)" && continue + echo " $name $([ "$priv" = "true" ] && echo "[private]")" + done + fi + fi + echo "" +} + +# ── main sync ────────────────────────────────────────────────────────────── +do_sync() { + local pull_count=0 push_count=0 fail_count=0 + + _log "=== Sync started (mode=$MODE) ===" + + # GitHub → Gitea + if [[ "$MODE" == "all" || "$MODE" == "pull" ]]; then + info "Discovering GitHub repos..." + while IFS='|' read -r name url priv; do + [[ -z "$name" ]] && continue + [[ -n "$SINGLE_REPO" && "$name" != "$SINGLE_REPO" ]] && continue + is_excluded "$name" && continue + if sync_github_to_gitea "$name" "$url" "$priv"; then + ((pull_count++)) + else + ((fail_count++)) + fi + done < <(get_github_repos) + fi + + # Gitea → GitHub + if [[ "$MODE" == "all" || "$MODE" == "push" ]]; then + info "Discovering Gitea repos..." + while IFS='|' read -r name url priv; do + [[ -z "$name" ]] && continue + [[ -n "$SINGLE_REPO" && "${name#*/}" != "${SINGLE_REPO#*/}" ]] && continue + is_excluded "$name" && continue + # Skip repos that came from GitHub (already mirrored) + local repo_name="${name#*/}" + if [[ -d "$WORK_DIR/$GITHUB_USER/$repo_name" ]]; then + info "Skipping $name (already a GitHub mirror)" + continue + fi + if sync_gitea_to_github "$name" "$url" "$priv"; then + ((push_count++)) + else + ((fail_count++)) + fi + done < <(get_gitea_repos) + fi + + echo "" + ok "Sync complete: ${pull_count} pulled, ${push_count} pushed, ${fail_count} failed" + _log "=== Sync complete: pull=$pull_count push=$push_count fail=$fail_count ===" +} + +# ── systemd timer ────────────────────────────────────────────────────────── +install_timer() { + local service_file="/etc/systemd/system/gitea-github-sync.service" + local timer_file="/etc/systemd/system/gitea-github-sync.timer" + local script_path + script_path="$(readlink -f "$0")" + + info "Installing systemd timer (interval: $TIMER_INTERVAL)..." + + sudo tee "$service_file" > /dev/null << EOF +[Unit] +Description=Gitea-GitHub Mirror Sync +After=network-online.target docker.service +Wants=network-online.target + +[Service] +Type=oneshot +User=$USER +ExecStart=$script_path +Environment=HOME=$HOME +StandardOutput=append:$LOG_FILE +StandardError=append:$LOG_FILE +EOF + + sudo tee "$timer_file" > /dev/null << EOF +[Unit] +Description=Gitea-GitHub Sync Timer + +[Timer] +OnBootSec=5min +OnUnitActiveSec=$TIMER_INTERVAL +Persistent=true + +[Install] +WantedBy=timers.target +EOF + + sudo systemctl daemon-reload + sudo systemctl enable --now gitea-github-sync.timer + ok "Timer installed: every $TIMER_INTERVAL" + ok "Check status: systemctl status gitea-github-sync.timer" + ok "Run now: sudo systemctl start gitea-github-sync.service" + ok "Logs: $LOG_FILE" +} + +remove_timer() { + info "Removing systemd timer..." + sudo systemctl disable --now gitea-github-sync.timer 2>/dev/null || true + sudo rm -f /etc/systemd/system/gitea-github-sync.{service,timer} + sudo systemctl daemon-reload + ok "Timer removed" +} + +# ── preflight checks ────────────────────────────────────────────────────── +preflight() { + local ok=true + if [[ -z "$GITEA_TOKEN" || "$GITEA_TOKEN" == "your-gitea-token-here" ]]; then + err "GITEA_TOKEN not set. Edit $ENV_FILE"; ok=false + fi + if [[ -z "$GITHUB_TOKEN" || "$GITHUB_TOKEN" == "your-github-token-here" ]]; then + err "GITHUB_TOKEN not set. Edit $ENV_FILE"; ok=false + fi + if [[ -z "$GITEA_USER" || -z "$GITHUB_USER" ]]; then + err "Run --init first to configure usernames"; ok=false + fi + $ok || exit 1 +} + +# ── main ─────────────────────────────────────────────────────────────────── +load_config + +case "$MODE" in + init) do_init ;; + install-timer) install_timer ;; + remove-timer) remove_timer ;; + list) preflight; do_list ;; + *) preflight; do_sync ;; +esac diff --git a/laptop_full_setup.sh b/laptop_full_setup.sh index cd37f63..ba9f608 100755 --- a/laptop_full_setup.sh +++ b/laptop_full_setup.sh @@ -39,30 +39,42 @@ LOCAL_IP=$(ip route get 1.1.1.1 2>/dev/null | grep -oP 'src \K\S+' \ # Models (defaults, may be adjusted below based on VRAM) EMBED_MODEL="nomic-embed-text" -CHAT_MODEL="qwen2.5:14b" -CODE_MODEL="qwen2.5-coder:7b" -FAST_MODEL="qwen2.5:7b" +CHAT_MODEL="qwen3.5:9b" +CODE_MODEL="qwen3.5:9b" +FAST_MODEL="qwen3.5:4b" # ── detect GPU ──────────────────────────────────────────────────────────────── VRAM_GB=$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits 2>/dev/null \ | head -1 | awk '{printf "%d", $1/1024}' 2>/dev/null || echo "0") +GPU_COUNT=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | wc -l || echo "0") GPU_NAME=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | head -1 || echo "None") +TOTAL_VRAM=$((VRAM_GB * GPU_COUNT)) -if [[ "$VRAM_GB" -ge 14 ]]; then - CHAT_MODEL="qwen2.5:14b"; CODE_MODEL="qwen2.5-coder:14b" - GPU_TIER="16GB VRAM — 14B models" -elif [[ "$VRAM_GB" -ge 8 ]]; then - CHAT_MODEL="qwen2.5:14b"; CODE_MODEL="qwen2.5-coder:7b" - GPU_TIER="8GB VRAM — 14B chat, 7B code" -elif [[ "$VRAM_GB" -ge 4 ]]; then - CHAT_MODEL="qwen2.5:7b"; CODE_MODEL="qwen2.5-coder:7b" - GPU_TIER="6GB VRAM — 7B models" -elif [[ "$VRAM_GB" -gt 0 ]]; then - CHAT_MODEL="qwen2.5:7b"; CODE_MODEL="qwen2.5-coder:7b" - GPU_TIER="${VRAM_GB}GB VRAM — 7B models" +# Ollama optimization flags (stacked — see docs/gpu-setup-research.md) +OLLAMA_KV_CACHE="q8_0" # halves KV cache VRAM (q4_0 for aggressive) +OLLAMA_FLASH="1" # flash attention: less VRAM, no quality loss + +if [[ "$TOTAL_VRAM" -ge 40 ]]; then + CHAT_MODEL="qwen3.5:27b"; CODE_MODEL="qwen3.5:27b" + CTX=131072; GPU_TIER="${TOTAL_VRAM}GB VRAM — 27B dense, 128K context" +elif [[ "$TOTAL_VRAM" -ge 28 ]]; then + CHAT_MODEL="qwen3.5-35b-a3b"; CODE_MODEL="qwen3.5-35b-a3b" + CTX=131072; GPU_TIER="${TOTAL_VRAM}GB VRAM — 35B MoE, 128K context" +elif [[ "$TOTAL_VRAM" -ge 14 ]]; then + CHAT_MODEL="qwen3.5-35b-a3b"; CODE_MODEL="qwen3.5-35b-a3b" + CTX=65536; GPU_TIER="${TOTAL_VRAM}GB VRAM — 35B MoE + KV quant, 64K context" +elif [[ "$TOTAL_VRAM" -ge 8 ]]; then + CHAT_MODEL="qwen3.5:9b"; CODE_MODEL="qwen3.5:9b" + CTX=32768; GPU_TIER="${TOTAL_VRAM}GB VRAM — 9B dense, 32K context" +elif [[ "$TOTAL_VRAM" -ge 4 ]]; then + CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" + CTX=16384; GPU_TIER="${TOTAL_VRAM}GB VRAM — 4B models, 16K context" +elif [[ "$TOTAL_VRAM" -gt 0 ]]; then + CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" + CTX=8192; GPU_TIER="${TOTAL_VRAM}GB VRAM — 4B models" else - CHAT_MODEL="qwen2.5:7b"; CODE_MODEL="qwen2.5-coder:7b" - GPU_TIER="CPU only — 7B models (slow)" + CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" + CTX=4096; OLLAMA_KV_CACHE="q4_0"; GPU_TIER="CPU only — 4B models (slow)" fi # ── new vs update ───────────────────────────────────────────────────────────── @@ -485,19 +497,19 @@ if $INSTALL_AI; then printf " %-4s %-8s %-42s %s\n" "3)" "22B" "phi4:14b + codestral:22b" "$(speed_label 13)" printf " %-4s %-8s %-42s %s\n" "4)" "70B" "llama3.3:70b + codestral:22b" "$(speed_label 41)" ;; - 2) # Performance - _TIER_NAMES=(7B 14B 32B 72B) - printf " %-4s %-8s %-42s %s\n" "1)" "7B" "qwen2.5:7b + qwen2.5-coder:7b" "$(speed_label 4)" - printf " %-4s %-8s %-42s %s\n" "2)" "14B" "qwen2.5:14b + qwen2.5-coder:14b" "$(speed_label 9)" - printf " %-4s %-8s %-42s %s\n" "3)" "32B" "qwen2.5:14b + qwen2.5-coder:32b" "$(speed_label 19)" - printf " %-4s %-8s %-42s %s\n" "4)" "72B" "qwen2.5:72b + qwen2.5-coder:32b" "$(speed_label 41)" + 2) # Performance (Qwen 3.5 — Feb 2026) + _TIER_NAMES=(4B 9B 35B 27B) + printf " %-4s %-8s %-42s %s\n" "1)" "4B" "qwen3.5:4b (chat+code)" "$(speed_label 2)" + printf " %-4s %-8s %-42s %s\n" "2)" "9B" "qwen3.5:9b (chat+code)" "$(speed_label 5)" + printf " %-4s %-8s %-42s %s\n" "3)" "35B" "qwen3.5-35b-a3b (MoE, 3B active)" "$(speed_label 12)" + printf " %-4s %-8s %-42s %s\n" "4)" "27B" "qwen3.5:27b (dense, A- quality)" "$(speed_label 17)" ;; - 3) # Mixed - _TIER_NAMES=(7B 14B 32B 70B) - printf " %-4s %-8s %-42s %s\n" "1)" "7B" "mistral:7b + qwen2.5-coder:7b" "$(speed_label 4)" - printf " %-4s %-8s %-42s %s\n" "2)" "14B" "phi4:14b + qwen2.5-coder:14b" "$(speed_label 9)" - printf " %-4s %-8s %-42s %s\n" "3)" "32B" "phi4:14b + qwen2.5-coder:32b" "$(speed_label 19)" - printf " %-4s %-8s %-42s %s\n" "4)" "70B" "llama3.3:70b + qwen2.5-coder:32b" "$(speed_label 41)" + 3) # Mixed (Western chat + Qwen 3.5 code) + _TIER_NAMES=(7B 14B 35B 70B) + printf " %-4s %-8s %-42s %s\n" "1)" "7B" "mistral:7b + qwen3.5:4b" "$(speed_label 4)" + printf " %-4s %-8s %-42s %s\n" "2)" "14B" "phi4:14b + qwen3.5:9b" "$(speed_label 9)" + printf " %-4s %-8s %-42s %s\n" "3)" "35B" "phi4:14b + qwen3.5-35b-a3b" "$(speed_label 19)" + printf " %-4s %-8s %-42s %s\n" "4)" "70B" "llama3.3:70b + qwen3.5-35b-a3b" "$(speed_label 41)" ;; esac @@ -535,16 +547,16 @@ if $INSTALL_AI; then 1:14B) FAST_MODEL="mistral:7b"; CHAT_MODEL="phi4:14b"; CODE_MODEL="starcoder2:15b"; REASON_MODEL="phi4:14b" ;; 1:22B) FAST_MODEL="mistral:7b"; CHAT_MODEL="phi4:14b"; CODE_MODEL="codestral:22b"; REASON_MODEL="phi4:14b" ;; 1:70B) FAST_MODEL="mistral:7b"; CHAT_MODEL="llama3.3:70b"; CODE_MODEL="codestral:22b"; REASON_MODEL="llama3.3:70b" ;; - # Performance-first - 2:7B) FAST_MODEL="qwen2.5:7b"; CHAT_MODEL="qwen2.5:7b"; CODE_MODEL="qwen2.5-coder:7b"; REASON_MODEL="" ;; - 2:14B) FAST_MODEL="qwen2.5:7b"; CHAT_MODEL="qwen2.5:14b"; CODE_MODEL="qwen2.5-coder:14b"; REASON_MODEL="deepseek-r1:14b" ;; - 2:32B) FAST_MODEL="qwen2.5:7b"; CHAT_MODEL="qwen2.5:14b"; CODE_MODEL="qwen2.5-coder:32b"; REASON_MODEL="deepseek-r1:14b" ;; - 2:72B) FAST_MODEL="qwen2.5:7b"; CHAT_MODEL="qwen2.5:72b"; CODE_MODEL="qwen2.5-coder:32b"; REASON_MODEL="deepseek-r1:14b" ;; - # Mixed - 3:7B) FAST_MODEL="mistral:7b"; CHAT_MODEL="mistral:7b"; CODE_MODEL="qwen2.5-coder:7b"; REASON_MODEL="" ;; - 3:14B) FAST_MODEL="mistral:7b"; CHAT_MODEL="phi4:14b"; CODE_MODEL="qwen2.5-coder:14b"; REASON_MODEL="phi4:14b" ;; - 3:32B) FAST_MODEL="mistral:7b"; CHAT_MODEL="phi4:14b"; CODE_MODEL="qwen2.5-coder:32b"; REASON_MODEL="phi4:14b" ;; - 3:70B) FAST_MODEL="mistral:7b"; CHAT_MODEL="llama3.3:70b"; CODE_MODEL="qwen2.5-coder:32b"; REASON_MODEL="llama3.3:70b" ;; + # Performance-first (Qwen 3.5) + 2:4B) FAST_MODEL="qwen3.5:4b"; CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b"; REASON_MODEL="" ;; + 2:9B) FAST_MODEL="qwen3.5:4b"; CHAT_MODEL="qwen3.5:9b"; CODE_MODEL="qwen3.5:9b"; REASON_MODEL="" ;; + 2:35B) FAST_MODEL="qwen3.5:4b"; CHAT_MODEL="qwen3.5-35b-a3b"; CODE_MODEL="qwen3.5-35b-a3b"; REASON_MODEL="" ;; + 2:27B) FAST_MODEL="qwen3.5:9b"; CHAT_MODEL="qwen3.5:27b"; CODE_MODEL="qwen3.5:27b"; REASON_MODEL="" ;; + # Mixed (Western chat + Qwen 3.5 code) + 3:7B) FAST_MODEL="mistral:7b"; CHAT_MODEL="mistral:7b"; CODE_MODEL="qwen3.5:4b"; REASON_MODEL="" ;; + 3:14B) FAST_MODEL="mistral:7b"; CHAT_MODEL="phi4:14b"; CODE_MODEL="qwen3.5:9b"; REASON_MODEL="phi4:14b" ;; + 3:35B) FAST_MODEL="mistral:7b"; CHAT_MODEL="phi4:14b"; CODE_MODEL="qwen3.5-35b-a3b"; REASON_MODEL="phi4:14b" ;; + 3:70B) FAST_MODEL="mistral:7b"; CHAT_MODEL="llama3.3:70b"; CODE_MODEL="qwen3.5-35b-a3b"; REASON_MODEL="llama3.3:70b" ;; *) warn "Unrecognised tier '$TIER_PICK' — keeping detected defaults" ;; @@ -763,8 +775,11 @@ WEBUI_URL= # Gitea — generate at http://$LOCAL_IP:3001/user/settings/applications GITEA_TOKEN=your-gitea-token-here -# GitHub — optional, for GitHub API access via MCP +# GitHub — optional, for GitHub API access via MCP and Gitea↔GitHub sync GITHUB_TOKEN=your-github-token-here + +# Gitea URL — used by sync script (default: http://localhost:3001) +GITEA_URL=http://$LOCAL_IP:3001 ENV ok "Created .env — add your tokens before using MCP Gitea/GitHub tools" else @@ -816,10 +831,12 @@ services: volumes: ${OLLAMA_VOLUME_LINE} environment: - - OLLAMA_NUM_GPU=999 # use all available VRAM (auto-detects GPU size) - - OLLAMA_NUM_CTX=8192 # lower to 4096 if you hit OOM + - OLLAMA_NUM_GPU=999 # use all available VRAM (auto-detects GPU size) + - OLLAMA_NUM_CTX=$CTX # auto-set by detected VRAM - OLLAMA_KEEP_ALIVE=24h - OLLAMA_MAX_LOADED_MODELS=1 + - OLLAMA_KV_CACHE_TYPE=$OLLAMA_KV_CACHE # q8_0 halves KV cache; q4_0 = 1/3 size + - OLLAMA_FLASH_ATTENTION=$OLLAMA_FLASH # tiled attention: less VRAM, no quality loss deploy: resources: reservations: @@ -885,7 +902,7 @@ ${OLLAMA_VOLUME_LINE} - OLLAMA_URL=http://ollama:11434 - CHROMA_URL=http://chromadb:8000 - EMBED_MODEL=nomic-embed-text - - CHAT_MODEL=qwen2.5:14b + - CHAT_MODEL=$CHAT_MODEL - PAPERS_DIR=/papers - REPOS_DIR=/repos command: > @@ -913,6 +930,7 @@ ${OLLAMA_VOLUME_LINE} - $BASE/repos:/repos - $BASE/mcp_server.py:/app/mcp_server.py - $BASE/mcp_requirements.txt:/app/mcp_requirements.txt + - $SCRIPT_DIR/gitea-github-sync.sh:/app/gitea-github-sync.sh:ro working_dir: /app env_file: $BASE/.env environment: @@ -922,7 +940,7 @@ ${OLLAMA_VOLUME_LINE} - RAG_URL=http://rag-server:8001 command: > bash -c "apt-get update -qq && - apt-get install -y --no-install-recommends git ripgrep && + apt-get install -y --no-install-recommends git ripgrep curl && pip install --no-cache-dir -r mcp_requirements.txt && python mcp_server.py" depends_on: @@ -1360,6 +1378,14 @@ if $INSTALL_AI; then echo -e " ${YELLOW}Add API tokens to:${NC} $BASE/.env" $SVC_RAG && echo -e " ${YELLOW}Drop PDFs into:${NC} $BASE/papers/" echo -e " ${YELLOW}Your workspace:${NC} $BASE/workspace/" + $SVC_GITEA && { + echo "" + echo -e " ${YELLOW}Gitea ↔ GitHub sync:${NC}" + echo " First time: $SCRIPT_DIR/gitea-github-sync.sh --init" + echo " Sync now: $SCRIPT_DIR/gitea-github-sync.sh" + echo " Auto (6h): $SCRIPT_DIR/gitea-github-sync.sh --install-timer" + echo " Via MCP: gitea_github_sync(mode='all')" + } fi if $SVC_KIWIX && [[ "$ZIM_CHOICE" == "3" ]]; then echo -e " ${YELLOW}ZIM downloads:${NC} ./kiwix_download.sh (not started)" diff --git a/local-ai-setup.sh b/local-ai-setup.sh index 2ac56bc..ab175cd 100755 --- a/local-ai-setup.sh +++ b/local-ai-setup.sh @@ -23,19 +23,31 @@ IS_UPDATE=false; [[ -f "$BASE/docker-compose.yml" ]] && IS_UPDATE=true # ── detect VRAM and set models accordingly ──────────────────────────────────── VRAM_GB=$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits 2>/dev/null \ | head -1 | awk '{printf "%d", $1/1024}' 2>/dev/null || echo "0") +GPU_COUNT=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | wc -l || echo "0") +TOTAL_VRAM=$((VRAM_GB * GPU_COUNT)) -if [[ "$VRAM_GB" -ge 14 ]]; then - CHAT_MODEL="qwen2.5:14b"; CODE_MODEL="qwen2.5-coder:14b" - CTX=32768; TIER="16GB — 14B models + 32k context" -elif [[ "$VRAM_GB" -ge 8 ]]; then - CHAT_MODEL="qwen2.5:14b"; CODE_MODEL="qwen2.5-coder:7b" - CTX=16384; TIER="8-16GB — 14B chat, 7B code, 16k context" -elif [[ "$VRAM_GB" -ge 4 ]]; then - CHAT_MODEL="qwen2.5:7b"; CODE_MODEL="qwen2.5-coder:7b" - CTX=8192; TIER="6GB — 7B models, 8k context" +# Ollama optimization flags (stacked — see docs/gpu-setup-research.md) +OLLAMA_KV_CACHE="q8_0" # halves KV cache VRAM (q4_0 for aggressive) +OLLAMA_FLASH="1" # flash attention: less VRAM, no quality loss + +if [[ "$TOTAL_VRAM" -ge 40 ]]; then + CHAT_MODEL="qwen3.5:27b"; CODE_MODEL="qwen3.5:27b" + CTX=131072; TIER="${TOTAL_VRAM}GB — 27B dense, 128K context" +elif [[ "$TOTAL_VRAM" -ge 28 ]]; then + CHAT_MODEL="qwen3.5-35b-a3b"; CODE_MODEL="qwen3.5-35b-a3b" + CTX=131072; TIER="${TOTAL_VRAM}GB — 35B MoE, 128K context" +elif [[ "$TOTAL_VRAM" -ge 14 ]]; then + CHAT_MODEL="qwen3.5-35b-a3b"; CODE_MODEL="qwen3.5-35b-a3b" + CTX=65536; TIER="${TOTAL_VRAM}GB — 35B MoE + KV quant, 64K context" +elif [[ "$TOTAL_VRAM" -ge 8 ]]; then + CHAT_MODEL="qwen3.5:9b"; CODE_MODEL="qwen3.5:9b" + CTX=32768; TIER="${TOTAL_VRAM}GB — 9B dense, 32K context" +elif [[ "$TOTAL_VRAM" -ge 4 ]]; then + CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" + CTX=16384; TIER="${TOTAL_VRAM}GB — 4B models, 16K context" else - CHAT_MODEL="qwen2.5:7b"; CODE_MODEL="qwen2.5-coder:7b" - CTX=4096; TIER="CPU-only — 7B models, 4k context" + CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" + CTX=4096; OLLAMA_KV_CACHE="q4_0"; TIER="CPU-only — 4B models, 4K context" fi EMBED_MODEL="nomic-embed-text" @@ -488,6 +500,7 @@ if [[ ! -f "$BASE/.env" ]]; then # Local AI Stack — edit to add your API tokens GITEA_TOKEN=your-gitea-token-here GITHUB_TOKEN=your-github-token-here +GITEA_URL=http://$LOCAL_IP:3001 ENV ok "Created .env" else @@ -527,9 +540,11 @@ services: volumes: [ollama-models:/root/.ollama] environment: - OLLAMA_NUM_GPU=999 - - OLLAMA_NUM_CTX= + - OLLAMA_NUM_CTX=$CTX - OLLAMA_KEEP_ALIVE=24h - OLLAMA_MAX_LOADED_MODELS=1 + - OLLAMA_KV_CACHE_TYPE=$OLLAMA_KV_CACHE + - OLLAMA_FLASH_ATTENTION=$OLLAMA_FLASH deploy: resources: reservations: @@ -586,7 +601,7 @@ services: - OLLAMA_URL=http://ollama:11434 - CHROMA_URL=http://chromadb:8000 - EMBED_MODEL=nomic-embed-text - - CHAT_MODEL= + - CHAT_MODEL=$CHAT_MODEL command: > bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git && pip install --no-cache-dir -r requirements.txt && @@ -605,6 +620,7 @@ services: - $BASE/repos:/repos - $BASE/mcp_server.py:/app/mcp_server.py - $BASE/mcp_requirements.txt:/app/mcp_requirements.txt + - $SCRIPT_DIR/gitea-github-sync.sh:/app/gitea-github-sync.sh:ro working_dir: /app env_file: $BASE/.env environment: @@ -613,7 +629,7 @@ services: - GITEA_URL=http://gitea:3000 - RAG_URL=http://rag-server:8001 command: > - bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git ripgrep && + bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git ripgrep curl && pip install --no-cache-dir -r mcp_requirements.txt && python mcp_server.py" depends_on: [rag-server] diff --git a/mcp_server.py b/mcp_server.py index 7003ac9..5832ce6 100644 --- a/mcp_server.py +++ b/mcp_server.py @@ -180,6 +180,25 @@ def github_api(method: str, endpoint: str, body: str = "") -> str: except Exception: return r.text +# ── Gitea ↔ GitHub sync ────────────────────────────────────────────────────── +@mcp.tool() +def gitea_github_sync(mode: str = "all", repo: str = "") -> str: + """Run Gitea↔GitHub mirror sync. mode: all|pull|push|list. repo: optional owner/name.""" + cmd = ["/app/gitea-github-sync.sh"] + if mode == "pull": cmd.append("--pull-only") + elif mode == "push": cmd.append("--push-only") + elif mode == "list": cmd.append("--list") + if repo: + cmd.extend(["--repo", repo]) + try: + r = subprocess.run(cmd, capture_output=True, text=True, timeout=600, + env={**os.environ, "SYNC_ENV": "/app/.env"}) + return (r.stdout + r.stderr).strip() or "Sync completed (no output)" + except subprocess.TimeoutExpired: + return "Sync timed out after 10 minutes" + except Exception as e: + return f"Sync failed: {e}" + # ── RAG ingest ──────────────────────────────────────────────────────────────── @mcp.tool() def ingest_repo(url: str, name: str = "", branch: str = "main") -> str: diff --git a/ubuntu-post-install.sh b/ubuntu-post-install.sh index b260202..5d038e5 100644 --- a/ubuntu-post-install.sh +++ b/ubuntu-post-install.sh @@ -1106,19 +1106,31 @@ IS_UPDATE=false; [[ -f "$BASE/docker-compose.yml" ]] && IS_UPDATE=true # ── detect VRAM and set models accordingly ──────────────────────────────────── VRAM_GB=$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits 2>/dev/null \ | head -1 | awk '{printf "%d", $1/1024}' 2>/dev/null || echo "0") +GPU_COUNT=$(nvidia-smi --query-gpu=name --format=csv,noheader 2>/dev/null | wc -l || echo "0") +TOTAL_VRAM=$((VRAM_GB * GPU_COUNT)) -if [[ "$VRAM_GB" -ge 14 ]]; then - CHAT_MODEL="qwen2.5:14b"; CODE_MODEL="qwen2.5-coder:14b" - CTX=32768; TIER="16GB — 14B models + 32k context" -elif [[ "$VRAM_GB" -ge 8 ]]; then - CHAT_MODEL="qwen2.5:14b"; CODE_MODEL="qwen2.5-coder:7b" - CTX=16384; TIER="8-16GB — 14B chat, 7B code, 16k context" -elif [[ "$VRAM_GB" -ge 4 ]]; then - CHAT_MODEL="qwen2.5:7b"; CODE_MODEL="qwen2.5-coder:7b" - CTX=8192; TIER="6GB — 7B models, 8k context" +# Ollama optimization flags (stacked — see docs/gpu-setup-research.md) +OLLAMA_KV_CACHE="q8_0" # halves KV cache VRAM (q4_0 for aggressive) +OLLAMA_FLASH="1" # flash attention: less VRAM, no quality loss + +if [[ "$TOTAL_VRAM" -ge 40 ]]; then + CHAT_MODEL="qwen3.5:27b"; CODE_MODEL="qwen3.5:27b" + CTX=131072; TIER="${TOTAL_VRAM}GB — 27B dense, 128K context" +elif [[ "$TOTAL_VRAM" -ge 28 ]]; then + CHAT_MODEL="qwen3.5-35b-a3b"; CODE_MODEL="qwen3.5-35b-a3b" + CTX=131072; TIER="${TOTAL_VRAM}GB — 35B MoE, 128K context" +elif [[ "$TOTAL_VRAM" -ge 14 ]]; then + CHAT_MODEL="qwen3.5-35b-a3b"; CODE_MODEL="qwen3.5-35b-a3b" + CTX=65536; TIER="${TOTAL_VRAM}GB — 35B MoE + KV quant, 64K context" +elif [[ "$TOTAL_VRAM" -ge 8 ]]; then + CHAT_MODEL="qwen3.5:9b"; CODE_MODEL="qwen3.5:9b" + CTX=32768; TIER="${TOTAL_VRAM}GB — 9B dense, 32K context" +elif [[ "$TOTAL_VRAM" -ge 4 ]]; then + CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" + CTX=16384; TIER="${TOTAL_VRAM}GB — 4B models, 16K context" else - CHAT_MODEL="qwen2.5:7b"; CODE_MODEL="qwen2.5-coder:7b" - CTX=4096; TIER="CPU-only — 7B models, 4k context" + CHAT_MODEL="qwen3.5:4b"; CODE_MODEL="qwen3.5:4b" + CTX=4096; OLLAMA_KV_CACHE="q4_0"; TIER="CPU-only — 4B models, 4K context" fi EMBED_MODEL="nomic-embed-text" @@ -1611,9 +1623,11 @@ services: volumes: [ollama-models:/root/.ollama] environment: - OLLAMA_NUM_GPU=999 - - OLLAMA_NUM_CTX= + - OLLAMA_NUM_CTX=$CTX - OLLAMA_KEEP_ALIVE=24h - OLLAMA_MAX_LOADED_MODELS=1 + - OLLAMA_KV_CACHE_TYPE=$OLLAMA_KV_CACHE + - OLLAMA_FLASH_ATTENTION=$OLLAMA_FLASH deploy: resources: reservations: From ba1c7fd68d49e7297803822ca7a294dc761e7a9e Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 18:13:49 +0000 Subject: [PATCH 14/15] Add search layer: Kiwix offline docs + DuckDuckGo web search MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The model can now search before it generates: MCP tools (available via Open WebUI + Claude Code): - search_docs(query): searches Kiwix ZIM files (Wikipedia, Stack Overflow, DevDocs, Arch Wiki) — instant, offline, no rate limits - read_doc(path): reads full article content from Kiwix results - web_search(query): DuckDuckGo search, no API key needed Open WebUI native search: - ENABLE_RAG_WEB_SEARCH=true + RAG_WEB_SEARCH_ENGINE=duckduckgo - Switched from SearXNG (not in stack) to DDG (zero config) Search priority: Kiwix first (offline, fast) → DDG fallback (live web) All 3 setup scripts updated: - duckduckgo-search added to mcp_requirements.txt - KIWIX_URL=http://kiwix:80 added to MCP container env - curl added to MCP container deps (for sync script) - Open WebUI DDG search enabled by default https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- laptop_full_setup.sh | 4 ++ local-ai-setup.sh | 4 ++ mcp_server.py | 86 +++++++++++++++++++++++++++++++++++++++++- ubuntu-post-install.sh | 6 ++- 4 files changed, 97 insertions(+), 3 deletions(-) diff --git a/laptop_full_setup.sh b/laptop_full_setup.sh index ba9f608..4fa43f9 100755 --- a/laptop_full_setup.sh +++ b/laptop_full_setup.sh @@ -753,6 +753,7 @@ mcp[cli] fastapi uvicorn[standard] httpx +duckduckgo-search REQ ok "mcp_requirements.txt" @@ -867,6 +868,8 @@ ${OLLAMA_VOLUME_LINE} - ENABLE_TOOL_SERVERS=true - WEBUI_AUTH=true - WEBUI_URL=${WEBUI_URL:-} + - ENABLE_RAG_WEB_SEARCH=true + - RAG_WEB_SEARCH_ENGINE=duckduckgo depends_on: ollama: condition: service_healthy @@ -938,6 +941,7 @@ ${OLLAMA_VOLUME_LINE} - REPOS_DIR=/repos - GITEA_URL=http://gitea:3000 - RAG_URL=http://rag-server:8001 + - KIWIX_URL=http://kiwix:80 command: > bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git ripgrep curl && diff --git a/local-ai-setup.sh b/local-ai-setup.sh index ab175cd..03f15b7 100755 --- a/local-ai-setup.sh +++ b/local-ai-setup.sh @@ -489,6 +489,7 @@ mcp[cli] fastapi uvicorn[standard] httpx +duckduckgo-search REQ ok "requirements.txt + mcp_requirements.txt" @@ -569,6 +570,8 @@ services: - ENABLE_OPENAI_API=true - ENABLE_TOOL_SERVERS=true - WEBUI_AUTH=true + - ENABLE_RAG_WEB_SEARCH=true + - RAG_WEB_SEARCH_ENGINE=duckduckgo depends_on: ollama: {condition: service_healthy} @@ -628,6 +631,7 @@ services: - REPOS_DIR=/repos - GITEA_URL=http://gitea:3000 - RAG_URL=http://rag-server:8001 + - KIWIX_URL=http://kiwix:80 command: > bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git ripgrep curl && pip install --no-cache-dir -r mcp_requirements.txt && diff --git a/mcp_server.py b/mcp_server.py index 5832ce6..a35ff2b 100644 --- a/mcp_server.py +++ b/mcp_server.py @@ -1,7 +1,8 @@ #!/usr/bin/env python3 """ MCP Server — Claude Code-equivalent tools for Open WebUI / Claude Code CLI. -Tools: bash, file read/write/list, code search, git ops, Gitea API, repo ingest. +Tools: bash, file read/write/list, code search, git ops, Gitea API, repo ingest, + offline doc search (Kiwix), web search (DuckDuckGo), Gitea↔GitHub sync. Connects via SSE on port 8002 — add to Open WebUI Tools or ~/.claude/mcp.json """ import os, subprocess, textwrap @@ -16,6 +17,7 @@ GITEA_URL = os.getenv("GITEA_URL", "http://gitea:3000") GITEA_TOKEN = os.getenv("GITEA_TOKEN", "") GITHUB_TOKEN= os.getenv("GITHUB_TOKEN","") RAG_URL = os.getenv("RAG_URL", "http://rag-server:8001") +KIWIX_URL = os.getenv("KIWIX_URL", "http://kiwix:80") mcp = FastMCP("local-dev-tools") @@ -127,6 +129,88 @@ def git_checkout(branch: str, repo: str = "", create: bool = False) -> str: args = ["checkout", "-b", branch] if create else ["checkout", branch] return _git(args, repo) +# ── Search: Kiwix (offline docs) + DuckDuckGo (web) ───────────────────────── +@mcp.tool() +def search_docs(query: str, limit: int = 5) -> str: + """Search offline docs (Wikipedia, Stack Overflow, DevDocs, Arch Wiki) via Kiwix. + Returns article titles, snippets, and URLs. Always try this before web_search.""" + import re + try: + # Kiwix full-text search returns HTML — parse the results + r = httpx.get(f"{KIWIX_URL}/search", params={"pattern": query, "pageLength": limit}, + timeout=15, follow_redirects=True) + if r.status_code != 200: + return f"Kiwix returned {r.status_code}. Is kiwix running with ZIM files loaded?" + html = r.text + results = [] + # Parse search result entries from Kiwix HTML + # Kiwix wraps results in
tags or links with snippets + articles = re.findall( + r']+href="(/[^"]+)"[^>]*>\s*]*>([^<]*).*?' + r'(?:]*>([^<]*))?.*?' + r'(?:]*>(.*?)

)?', + html, re.DOTALL + ) + if not articles: + # Fallback: grab any links with text from the results + articles = re.findall(r']+href="(/[^"]+)"[^>]*>([^<]+)
', html) + for path, title in articles[:limit]: + results.append(f"**{title.strip()}**\n URL: {KIWIX_URL}{path}\n") + else: + for path, title, cite, snippet in articles[:limit]: + snippet_clean = re.sub(r'<[^>]+>', '', snippet or '').strip() + entry = f"**{title.strip()}**" + if cite: + entry += f" ({cite.strip()})" + if snippet_clean: + entry += f"\n {snippet_clean[:300]}" + entry += f"\n URL: {KIWIX_URL}{path}" + results.append(entry) + if not results: + return f"No results for '{query}' in offline docs. Try web_search instead." + return "\n\n".join(results) + except httpx.ConnectError: + return "Kiwix not reachable. Is the kiwix container running with ZIM files?" + except Exception as e: + return f"Kiwix search error: {e}" + +@mcp.tool() +def read_doc(path: str) -> str: + """Read a full article from Kiwix by its path (from search_docs results). + Example: read_doc('/wikipedia_en_all/A/Python_(programming_language)')""" + try: + r = httpx.get(f"{KIWIX_URL}{path}", timeout=15, follow_redirects=True) + if r.status_code != 200: + return f"Not found: {path} (HTTP {r.status_code})" + import re + # Strip HTML tags, keep text content + text = re.sub(r']*>.*?', '', r.text, flags=re.DOTALL) + text = re.sub(r']*>.*?', '', text, flags=re.DOTALL) + text = re.sub(r'<[^>]+>', ' ', text) + text = re.sub(r'\s+', ' ', text).strip() + # Truncate to ~8K chars to fit in model context + if len(text) > 8000: + text = text[:8000] + "\n\n[... truncated — article continues ...]" + return text + except Exception as e: + return f"Error reading doc: {e}" + +@mcp.tool() +def web_search(query: str, num_results: int = 5) -> str: + """Search the live web via DuckDuckGo. No API key needed. + Use search_docs first for programming/wiki topics — it's faster and offline.""" + try: + from duckduckgo_search import DDGS + results = [] + with DDGS() as ddgs: + for r in ddgs.text(query, max_results=num_results): + results.append(f"**{r['title']}**\n {r['body']}\n {r['href']}") + return "\n\n".join(results) if results else f"No web results for '{query}'" + except ImportError: + return "duckduckgo-search not installed. Add it to mcp_requirements.txt." + except Exception as e: + return f"Web search error: {e}" + # ── Gitea API ───────────────────────────────────────────────────────────────── def _gitea(method: str, path: str, body: dict = {}) -> dict: if not GITEA_TOKEN: diff --git a/ubuntu-post-install.sh b/ubuntu-post-install.sh index 5d038e5..eeda5ad 100644 --- a/ubuntu-post-install.sh +++ b/ubuntu-post-install.sh @@ -1572,6 +1572,7 @@ mcp[cli] fastapi uvicorn[standard] httpx +duckduckgo-search REQ ok "requirements.txt + mcp_requirements.txt" @@ -1651,7 +1652,7 @@ services: - OPENAI_API_KEY=local-rag - ENABLE_OPENAI_API=true - ENABLE_RAG_WEB_SEARCH=true - - RAG_WEB_SEARCH_ENGINE=searxng + - RAG_WEB_SEARCH_ENGINE=duckduckgo - SEARXNG_QUERY_URL=http://searxng:8080/search?q=&format=json - WEBUI_AUTH=false depends_on: @@ -1712,8 +1713,9 @@ services: - REPOS_DIR=/repos - GITEA_URL=http://gitea:3000 - RAG_URL=http://rag-server:8001 + - KIWIX_URL=http://kiwix:80 command: > - bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git ripgrep && + bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git ripgrep curl && pip install --no-cache-dir -r mcp_requirements.txt && python mcp_server.py" depends_on: [rag-server] From b5a749224de4e4c2137c85a4b7d1dbb3cee7b301 Mon Sep 17 00:00:00 2001 From: Claude Date: Sun, 22 Mar 2026 18:17:51 +0000 Subject: [PATCH 15/15] Unified search: Kiwix offline + DDG live with freshness detection Replaced three separate tools (search_docs, web_search, search) with a smart unified search() that: - Always hits Kiwix first (instant, offline, no rate limit) - Checks query for freshness keywords (latest, 2026, release, CVE, etc) - If time-sensitive: also hits DDG, flags "prefer live results" - If timeless (algorithms, docs, concepts): Kiwix only, skips web hit - If Kiwix returns nothing: falls back to DDG automatically Model sees both result sets with clear guidance on which to trust. Keeps read_doc() for reading full Kiwix articles and web_search() for explicit live-only queries when verifying offline currency. 130GB of ZIMs earn their keep on timeless topics (no rate limit, instant, complete articles). DDG covers everything else. https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu --- mcp_server.py | 168 ++++++++++++++++++++++++++++++++++---------------- 1 file changed, 114 insertions(+), 54 deletions(-) diff --git a/mcp_server.py b/mcp_server.py index a35ff2b..ce5aca0 100644 --- a/mcp_server.py +++ b/mcp_server.py @@ -129,66 +129,132 @@ def git_checkout(branch: str, repo: str = "", create: bool = False) -> str: args = ["checkout", "-b", branch] if create else ["checkout", branch] return _git(args, repo) -# ── Search: Kiwix (offline docs) + DuckDuckGo (web) ───────────────────────── -@mcp.tool() -def search_docs(query: str, limit: int = 5) -> str: - """Search offline docs (Wikipedia, Stack Overflow, DevDocs, Arch Wiki) via Kiwix. - Returns article titles, snippets, and URLs. Always try this before web_search.""" - import re +# ── Search: unified (Kiwix offline + DuckDuckGo live) ─────────────────────── +# Kiwix ZIMs have complete, high-quality articles but may be months old. +# DDG has live results but lower signal-to-noise. The unified search tool +# checks both and lets the model see freshness info to judge which to trust. +# +# Heuristic: topics that change fast (releases, CVEs, "latest X") get flagged +# as potentially stale in offline results. Timeless topics (algorithms, language +# docs, math) are fine from Kiwix and skip the web hit entirely. + +import re as _re +from datetime import datetime as _dt + +# Words that suggest the query needs fresh data +_FRESH_KEYWORDS = _re.compile( + r'\b(latest|newest|recent|2025|2026|update|release|version|changelog|CVE|vulnerability|' + r'breaking change|deprecat|current|today|this year|this month|announce|just released)\b', + _re.IGNORECASE +) + +def _kiwix_search(query: str, limit: int = 5) -> list[dict]: + """Search Kiwix, return list of {title, snippet, path, source}.""" try: - # Kiwix full-text search returns HTML — parse the results - r = httpx.get(f"{KIWIX_URL}/search", params={"pattern": query, "pageLength": limit}, + r = httpx.get(f"{KIWIX_URL}/search", + params={"pattern": query, "pageLength": limit}, timeout=15, follow_redirects=True) if r.status_code != 200: - return f"Kiwix returned {r.status_code}. Is kiwix running with ZIM files loaded?" + return [] html = r.text results = [] - # Parse search result entries from Kiwix HTML - # Kiwix wraps results in
tags or links with snippets - articles = re.findall( + # Try structured parse first + articles = _re.findall( r']+href="(/[^"]+)"[^>]*>\s*]*>([^<]*).*?' r'(?:]*>([^<]*))?.*?' r'(?:]*>(.*?)

)?', - html, re.DOTALL + html, _re.DOTALL ) - if not articles: - # Fallback: grab any links with text from the results - articles = re.findall(r']+href="(/[^"]+)"[^>]*>([^<]+)
', html) - for path, title in articles[:limit]: - results.append(f"**{title.strip()}**\n URL: {KIWIX_URL}{path}\n") - else: + if articles: for path, title, cite, snippet in articles[:limit]: - snippet_clean = re.sub(r'<[^>]+>', '', snippet or '').strip() - entry = f"**{title.strip()}**" - if cite: - entry += f" ({cite.strip()})" - if snippet_clean: - entry += f"\n {snippet_clean[:300]}" - entry += f"\n URL: {KIWIX_URL}{path}" - results.append(entry) - if not results: - return f"No results for '{query}' in offline docs. Try web_search instead." - return "\n\n".join(results) - except httpx.ConnectError: - return "Kiwix not reachable. Is the kiwix container running with ZIM files?" - except Exception as e: - return f"Kiwix search error: {e}" + snippet_clean = _re.sub(r'<[^>]+>', '', snippet or '').strip()[:300] + results.append({"title": title.strip(), "snippet": snippet_clean, + "path": path, "source": cite.strip() if cite else "kiwix"}) + else: + # Fallback: grab any links + for path, title in _re.findall(r']+href="(/[^"]+)"[^>]*>([^<]+)', html)[:limit]: + results.append({"title": title.strip(), "snippet": "", + "path": path, "source": "kiwix"}) + return results + except Exception: + return [] + +def _ddg_search(query: str, limit: int = 5) -> list[dict]: + """Search DuckDuckGo, return list of {title, snippet, url}.""" + try: + from duckduckgo_search import DDGS + results = [] + with DDGS() as ddgs: + for r in ddgs.text(query, max_results=limit): + results.append({"title": r["title"], "snippet": r["body"], "url": r["href"]}) + return results + except Exception: + return [] + +@mcp.tool() +def search(query: str, limit: int = 5) -> str: + """Unified search: checks offline docs (Kiwix) AND live web (DuckDuckGo). + Returns results from both with freshness guidance. + For timeless topics (algorithms, docs): offline results are sufficient. + For time-sensitive topics (releases, CVEs): live results are flagged as preferred.""" + needs_fresh = bool(_FRESH_KEYWORDS.search(query)) + output_parts = [] + + # Always search Kiwix (fast, local) + kiwix_results = _kiwix_search(query, limit) + if kiwix_results: + header = "## Offline Docs (Kiwix)" + if needs_fresh: + header += " ⚠️ POSSIBLY STALE — query looks time-sensitive, prefer live results below" + output_parts.append(header) + for i, r in enumerate(kiwix_results, 1): + entry = f"{i}. **{r['title']}**" + if r["source"] and r["source"] != "kiwix": + entry += f" ({r['source']})" + if r["snippet"]: + entry += f"\n {r['snippet']}" + entry += f"\n → read_doc('{r['path']}')" + output_parts.append(entry) + + # Search DDG if: query needs fresh data, OR Kiwix returned nothing, OR always (to compare) + do_web = needs_fresh or not kiwix_results + ddg_results = [] + if do_web: + ddg_results = _ddg_search(query, limit) + + if ddg_results: + header = "## Live Web (DuckDuckGo)" + if needs_fresh: + header += " ✓ PREFER THESE for this query" + output_parts.append(header) + for i, r in enumerate(ddg_results, 1): + output_parts.append(f"{i}. **{r['title']}**\n {r['snippet']}\n {r['url']}") + elif do_web: + output_parts.append("## Live Web (DuckDuckGo)\n(no results or DDG unreachable)") + + if not kiwix_results and not ddg_results: + return f"No results for '{query}' from either offline docs or web search." + + # Freshness note + if kiwix_results and not needs_fresh and not ddg_results: + output_parts.append("\n_Offline results look sufficient for this topic. " + "Use web_search() if you need to verify currency._") + + return "\n\n".join(output_parts) @mcp.tool() def read_doc(path: str) -> str: - """Read a full article from Kiwix by its path (from search_docs results). + """Read a full article from Kiwix by its path (from search results). Example: read_doc('/wikipedia_en_all/A/Python_(programming_language)')""" try: r = httpx.get(f"{KIWIX_URL}{path}", timeout=15, follow_redirects=True) if r.status_code != 200: return f"Not found: {path} (HTTP {r.status_code})" - import re # Strip HTML tags, keep text content - text = re.sub(r']*>.*?', '', r.text, flags=re.DOTALL) - text = re.sub(r']*>.*?', '', text, flags=re.DOTALL) - text = re.sub(r'<[^>]+>', ' ', text) - text = re.sub(r'\s+', ' ', text).strip() - # Truncate to ~8K chars to fit in model context + text = _re.sub(r']*>.*?', '', r.text, flags=_re.DOTALL) + text = _re.sub(r']*>.*?', '', text, flags=_re.DOTALL) + text = _re.sub(r'<[^>]+>', ' ', text) + text = _re.sub(r'\s+', ' ', text).strip() if len(text) > 8000: text = text[:8000] + "\n\n[... truncated — article continues ...]" return text @@ -197,19 +263,13 @@ def read_doc(path: str) -> str: @mcp.tool() def web_search(query: str, num_results: int = 5) -> str: - """Search the live web via DuckDuckGo. No API key needed. - Use search_docs first for programming/wiki topics — it's faster and offline.""" - try: - from duckduckgo_search import DDGS - results = [] - with DDGS() as ddgs: - for r in ddgs.text(query, max_results=num_results): - results.append(f"**{r['title']}**\n {r['body']}\n {r['href']}") - return "\n\n".join(results) if results else f"No web results for '{query}'" - except ImportError: - return "duckduckgo-search not installed. Add it to mcp_requirements.txt." - except Exception as e: - return f"Web search error: {e}" + """Search ONLY the live web via DuckDuckGo. Use search() instead for most queries — + it checks both offline and live. Use this directly only when you specifically need + live-only results (e.g., verifying if offline info is current).""" + results = _ddg_search(query, num_results) + if not results: + return f"No web results for '{query}'" + return "\n\n".join(f"**{r['title']}**\n {r['snippet']}\n {r['url']}" for r in results) # ── Gitea API ───────────────────────────────────────────────────────────────── def _gitea(method: str, path: str, body: dict = {}) -> dict: