#!/usr/bin/env bash # Local AI Stack — single script, new install or update # Usage: ./local-ai-setup.sh [--force] [--no-pull] set -euo pipefail RED='\033[0;31m'; GREEN='\033[0;32m'; YELLOW='\033[1;33m' CYAN='\033[0;36m'; BOLD='\033[1m'; NC='\033[0m' info() { echo -e "${CYAN}[..]${NC} $*"; } ok() { echo -e "${GREEN}[OK]${NC} $*"; } warn() { echo -e "${YELLOW}[!!]${NC} $*"; } section() { echo -e "\n${BOLD}━━━ $* ━━━${NC}"; } FORCE=false; NO_PULL=false for a in "$@"; do [[ "$a" == "--force" ]] && FORCE=true; [[ "$a" == "--no-pull" ]] && NO_PULL=true; done BASE="$HOME/docker/ai-stack" SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" LOCAL_IP=$(ip route get 1.1.1.1 2>/dev/null | grep -oP 'src \K\S+' || hostname -I | awk '{print $1}') [[ -z "$LOCAL_IP" ]] && read -rp "Enter LAN IP: " LOCAL_IP IS_UPDATE=false; [[ -f "$BASE/docker-compose.yml" ]] && IS_UPDATE=true # ── detect VRAM and set models accordingly ──────────────────────────────────── VRAM_GB=$(nvidia-smi --query-gpu=memory.total --format=csv,noheader,nounits 2>/dev/null \ | head -1 | awk '{printf "%d", $1/1024}' 2>/dev/null || echo "0") if [[ "$VRAM_GB" -ge 14 ]]; then CHAT_MODEL="qwen2.5:14b"; CODE_MODEL="qwen2.5-coder:14b" CTX=32768; TIER="16GB — 14B models + 32k context" elif [[ "$VRAM_GB" -ge 8 ]]; then CHAT_MODEL="qwen2.5:14b"; CODE_MODEL="qwen2.5-coder:7b" CTX=16384; TIER="8-16GB — 14B chat, 7B code, 16k context" elif [[ "$VRAM_GB" -ge 4 ]]; then CHAT_MODEL="qwen2.5:7b"; CODE_MODEL="qwen2.5-coder:7b" CTX=8192; TIER="6GB — 7B models, 8k context" else CHAT_MODEL="qwen2.5:7b"; CODE_MODEL="qwen2.5-coder:7b" CTX=4096; TIER="CPU-only — 7B models, 4k context" fi EMBED_MODEL="nomic-embed-text" section "Local AI Stack — $($IS_UPDATE && echo UPDATE || echo NEW INSTALL)" info "Base : $BASE" info "IP : $LOCAL_IP" info "GPU : ${VRAM_GB}GB VRAM → $TIER" # ── SearXNG safe-search & engine selection ──────────────────────────────────── SEARXNG_DISABLE_ENGINES="" SEARXNG_ENABLE_ENGINES="" echo "" echo " SearXNG safe-search level:" echo " 0) None — all results, no filtering (default)" echo " 1) Moderate — filter explicit content" echo " 2) Strict — block all explicit content" echo "" read -rp " Choice [0]: " _SAFE_PICK case "${_SAFE_PICK:-0}" in 1) SEARXNG_SAFE_LEVEL="moderate"; SEARXNG_SAFE_INT=1 ;; 2) SEARXNG_SAFE_LEVEL="strict"; SEARXNG_SAFE_INT=2 ;; *) SEARXNG_SAFE_LEVEL="none"; SEARXNG_SAFE_INT=0 ;; esac # Default ON/OFF for an engine given the chosen safe-search level. # Engines without safe-search support default OFF for moderate/strict. _sxst() { [[ "$SEARXNG_SAFE_LEVEL" == "none" ]] && echo ON && return case "$1" in mojeek|"mojeek images"|"mojeek news"|\ yandex|baidu|naver|\ invidious|piped|peertube|sepiasearch|\ vimeo|bitchute|rumble|odysee|\ imgur|deviantart|artstation) echo OFF ;; *) echo ON ;; esac } # Whiptail checklist — prints selected engine names (one per line). # ESC returns the engines that were pre-checked ON (keeps defaults unchanged). _sxmenu() { local _t="$1"; shift local -a _n=() _s=() _a=(); local _i=1 while [[ $# -ge 3 ]]; do _n+=("$1"); _s+=("$3") _a+=("$_i" "$(printf '%-26s %s' "$1" "$2")" "$3") _i=$(( _i + 1 )); shift 3 done local _h=$(( ${#_n[@]} + 9 > 24 ? 24 : ${#_n[@]} + 9 )) local _lh=$(( ${#_n[@]} < 14 ? ${#_n[@]} : 14 )) local _o _j if ! _o=$(whiptail --backtitle "SearXNG — $SEARXNG_SAFE_LEVEL" \ --title "$_t" --checklist \ "SPACE = toggle ENTER = confirm ESC = use defaults" \ "$_h" 76 "$_lh" "${_a[@]}" 3>&1 1>&2 2>&3 2>/dev/null); then for _j in "${!_n[@]}"; do [[ "${_s[$_j]}" == "ON" ]] && printf '%s\n' "${_n[$_j]}" done return fi for _id in $_o; do _id="${_id//\"/}"; printf '%s\n' "${_n[$(( _id-1 ))]}"; done } _SX_AUTOFF=(mojeek yandex baidu naver invidious piped peertube sepiasearch) if command -v whiptail &>/dev/null && [[ -t 0 ]]; then # ── Web search ───────────────────────────────────────────────────────────── _SX_W=$(_sxmenu "Web Search Engines" \ "google" "Google" ON \ "bing" "Microsoft Bing" ON \ "duckduckgo" "DuckDuckGo" ON \ "brave" "Brave Search" ON \ "startpage" "Startpage (Google proxy)" ON \ "qwant" "Qwant" ON \ "yahoo" "Yahoo" ON \ "ecosia" "Ecosia" ON \ "presearch" "Presearch" ON \ "yep" "Yep" ON \ "wiby" "Wiby (classic/indie web)" ON \ "mojeek" "Mojeek [no safe-search]" "$(_sxst mojeek)" \ "yandex" "Yandex [unreliable filter]" "$(_sxst yandex)" \ "baidu" "Baidu [no safe-search]" "$(_sxst baidu)" \ "naver" "Naver (Korean) [no safe-search]" "$(_sxst naver)") # ── Images ───────────────────────────────────────────────────────────────── _SX_I=$(_sxmenu "Image Search Engines" \ "google images" "Google Images" ON \ "bing images" "Bing Images" ON \ "duckduckgo images" "DuckDuckGo Images" ON \ "brave images" "Brave Images" ON \ "qwant images" "Qwant Images" ON \ "startpage images" "Startpage Images" ON \ "mojeek images" "Mojeek Images [no safe-search]" "$(_sxst 'mojeek images')" \ "presearch images" "Presearch Images" ON \ "openverse" "Openverse (open license)" ON \ "unsplash" "Unsplash (stock photos)" ON \ "pexels" "Pexels (stock photos)" ON \ "pixabay images" "Pixabay Images" ON \ "pinterest" "Pinterest" ON \ "flickr" "Flickr" ON \ "wikcommons.images" "Wikimedia Commons Images" ON \ "imgur" "Imgur [adult content]" "$(_sxst imgur)" \ "deviantart" "DeviantArt [adult content]" "$(_sxst deviantart)" \ "artstation" "ArtStation [adult content]" "$(_sxst artstation)" \ "adobe stock" "Adobe Stock" ON) # ── Videos ───────────────────────────────────────────────────────────────── _SX_V=$(_sxmenu "Video Search Engines" \ "youtube" "YouTube" ON \ "bing videos" "Bing Videos" ON \ "brave videos" "Brave Videos" ON \ "duckduckgo videos" "DuckDuckGo Videos" ON \ "google videos" "Google Videos" ON \ "qwant videos" "Qwant Videos" ON \ "dailymotion" "Dailymotion" ON \ "media.ccc.de" "media.ccc.de (tech talks)" ON \ "wikcommons.videos" "Wikimedia Commons Videos" ON \ "vimeo" "Vimeo [no safe-search]" "$(_sxst vimeo)" \ "odysee" "Odysee [no safe-search]" "$(_sxst odysee)" \ "rumble" "Rumble [no safe-search]" "$(_sxst rumble)" \ "bitchute" "BitChute [no safe-search]" "$(_sxst bitchute)" \ "invidious" "Invidious (YT) [no safe-search]" "$(_sxst invidious)" \ "piped" "Piped (YT) [no safe-search]" "$(_sxst piped)" \ "peertube" "PeerTube [no safe-search]" "$(_sxst peertube)" \ "sepiasearch" "SepiaSearch [no safe-search]" "$(_sxst sepiasearch)") # ── News ─────────────────────────────────────────────────────────────────── _SX_N=$(_sxmenu "News Search Engines" \ "google news" "Google News" ON \ "bing news" "Bing News" ON \ "duckduckgo news" "DuckDuckGo News" ON \ "brave news" "Brave News" ON \ "qwant news" "Qwant News" ON \ "startpage news" "Startpage News" ON \ "presearch news" "Presearch News" ON \ "mojeek news" "Mojeek News [no safe-search]" "$(_sxst 'mojeek news')" \ "reuters" "Reuters" ON \ "yahoo news" "Yahoo News" ON \ "wikinews" "WikiNews" ON \ "yep news" "Yep News" ON) # ── Reference & Knowledge ────────────────────────────────────────────────── _SX_R=$(_sxmenu "Reference & Knowledge" \ "wikipedia" "Wikipedia" ON \ "wikidata" "Wikidata" ON \ "wolframalpha" "Wolfram|Alpha" ON \ "ask" "Ask.com" ON \ "ddg definitions" "DuckDuckGo Definitions" ON \ "encyclopsearch" "Encyclopsearch" ON \ "wikibooks" "Wikibooks" ON \ "wikiquote" "Wikiquote" ON \ "wikisource" "Wikisource" ON \ "wikispecies" "Wikispecies" ON \ "wikiversity" "Wikiversity" ON \ "wikivoyage" "Wikivoyage" ON) # ── Science ──────────────────────────────────────────────────────────────── _SX_S=$(_sxmenu "Science / Academic" \ "arxiv" "arXiv (preprints)" ON \ "semantic scholar" "Semantic Scholar" ON \ "pubmed" "PubMed (medical)" ON \ "crossref" "Crossref (DOI/papers)" ON \ "base" "BASE (open access)" ON) _SX_ALL=( # Web "google" "bing" "duckduckgo" "brave" "startpage" "qwant" "yahoo" "ecosia" "presearch" "yep" "wiby" "mojeek" "yandex" "baidu" "naver" # Images "google images" "bing images" "duckduckgo images" "brave images" "qwant images" "startpage images" "mojeek images" "presearch images" "openverse" "unsplash" "pexels" "pixabay images" "pinterest" "flickr" "wikcommons.images" "imgur" "deviantart" "artstation" "adobe stock" # Videos "youtube" "bing videos" "brave videos" "duckduckgo videos" "google videos" "qwant videos" "dailymotion" "media.ccc.de" "wikcommons.videos" "vimeo" "odysee" "rumble" "bitchute" "invidious" "piped" "peertube" "sepiasearch" # News "google news" "bing news" "duckduckgo news" "brave news" "qwant news" "startpage news" "presearch news" "mojeek news" "reuters" "yahoo news" "wikinews" "yep news" # Reference "wikipedia" "wikidata" "wolframalpha" "ask" "ddg definitions" "encyclopsearch" "wikibooks" "wikiquote" "wikisource" "wikispecies" "wikiversity" "wikivoyage" # Science "arxiv" "semantic scholar" "pubmed" "crossref" "base" ) _SX_SEL=$(printf '%s\n' "$_SX_W" "$_SX_I" "$_SX_V" "$_SX_N" "$_SX_R" "$_SX_S") for _e in "${_SX_ALL[@]}"; do if printf '%s\n' "$_SX_SEL" | grep -qxF "$_e"; then for _ao in "${_SX_AUTOFF[@]}"; do if [[ "$_e" == "$_ao" ]]; then SEARXNG_ENABLE_ENGINES+="${SEARXNG_ENABLE_ENGINES:+,}${_e}" break fi done else SEARXNG_DISABLE_ENGINES+="${SEARXNG_DISABLE_ENGINES:+,}${_e}" fi done else echo "" echo " (whiptail not available — text entry)" echo " Web: google bing duckduckgo brave startpage qwant yahoo presearch" echo " Imgs: google_images bing_images duckduckgo_images flickr imgur" echo " Vids: youtube dailymotion vimeo rumble odysee bitchute" echo " News: google_news bing_news reuters yahoo_news wikinews" read -rp " Engines to disable (space-separated, Enter for defaults): " _ENGS SEARXNG_DISABLE_ENGINES="${_ENGS// /,}" fi echo "" write_if_new() { local dest="$1"; local body; body=$(cat) if [[ ! -f "$dest" ]] || $FORCE; then printf '%s\n' "$body" > "$dest"; ok "Wrote $(basename "$dest")" else info "Kept $(basename "$dest") (--force to overwrite)" fi } # ── prereqs (new install only) ──────────────────────────────────────────────── if ! $IS_UPDATE; then section "Prerequisites" if ! command -v docker &>/dev/null; then info "Installing Docker..." curl -fsSL https://get.docker.com | sh sudo usermod -aG docker "$USER" warn "Run: newgrp docker (or log out/in)" else ok "Docker: $(docker --version | cut -d' ' -f3)" fi if command -v nvidia-smi &>/dev/null && ! dpkg -l 2>/dev/null | grep -q nvidia-container-toolkit; then info "Installing NVIDIA Container Toolkit..." curl -fsSL https://nvidia.github.io/libnvidia-container/gpgkey \ | sudo gpg --dearmor --yes -o /usr/share/keyrings/nvidia-container-toolkit-keyring.gpg curl -fsSL https://nvidia.github.io/libnvidia-container/stable/deb/nvidia-container-toolkit.list \ | sed 's#deb https://#deb [signed-by=/usr/share/keyrings/nvidia-container-toolkit-keyring.gpg] https://#g' \ | sudo tee /etc/apt/sources.list.d/nvidia-container-toolkit.list sudo apt-get update -qq && sudo apt-get install -y nvidia-container-toolkit sudo nvidia-ctk runtime configure --runtime=docker && sudo systemctl restart docker ok "NVIDIA Container Toolkit installed" fi command -v rg &>/dev/null || sudo apt-get install -y ripgrep fi # ── directories ─────────────────────────────────────────────────────────────── section "Directories" for d in papers repos workspace index searxng invokeai-data invokeai-outputs kiwix gitea portainer-data logs; do mkdir -p "$BASE/$d" done ok "Ready under $BASE" # ============================================================================= section "Writing server.py (RAG)" # ============================================================================= write_if_new "$BASE/server.py" << 'PY' import ast, fnmatch, hashlib, json, logging, os, re, subprocess, threading, time from pathlib import Path from typing import Any import chromadb, httpx from chromadb.utils.embedding_functions import OllamaEmbeddingFunction from fastapi import FastAPI, HTTPException, Request, BackgroundTasks from fastapi.middleware.cors import CORSMiddleware from fastapi.responses import StreamingResponse from pydantic import BaseModel logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s") log = logging.getLogger("rag") OLLAMA_URL = os.getenv("OLLAMA_URL", "http://ollama:11434") CHROMA_URL = os.getenv("CHROMA_URL", "http://chromadb:8000") EMBED_MODEL = os.getenv("EMBED_MODEL", "nomic-embed-text") CHAT_MODEL = os.getenv("CHAT_MODEL", "qwen2.5:14b") PAPERS_DIR = Path(os.getenv("PAPERS_DIR", "/papers")) REPOS_DIR = Path(os.getenv("REPOS_DIR", "/repos")) TOP_K = int(os.getenv("TOP_K", "6")) CODE_EXTS = {".py",".js",".ts",".tsx",".jsx",".go",".rs",".java",".c",".cpp", ".h",".cs",".rb",".sh",".yaml",".yml",".toml",".sql",".md"} SKIP_DIRS = {"node_modules",".git","__pycache__","dist","build",".venv","venv","target"} MAX_BYTES = 400_000 app = FastAPI(title="RAG Server") app.add_middleware(CORSMiddleware, allow_origins=["*"], allow_methods=["*"], allow_headers=["*"]) def _embed_fn(): return OllamaEmbeddingFunction(url=f"{OLLAMA_URL}/api/embeddings", model_name=EMBED_MODEL) def _chroma(): host, port = CHROMA_URL.replace("http://","").split(":") return chromadb.HttpClient(host=host, port=int(port)) def get_col(name): return _chroma().get_or_create_collection(name, embedding_function=_embed_fn()) def _doc_id(text, key): return hashlib.md5(f"{key}|{text[:200]}".encode()).hexdigest() def _sliding(text, size=1000, overlap=150): chunks, i = [], 0 while i < len(text): chunks.append(text[i:i+size]); i += size - overlap return [c for c in chunks if c.strip()] def _chunk_python(src): try: tree = ast.parse(src) except SyntaxError: return [] lines = src.splitlines(); out = [] for node in ast.iter_child_nodes(tree): if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)): out.append((node.name, "\n".join(lines[node.lineno-1:node.end_lineno])[:4000])) return out def _chunk_file(path, src): if path.suffix == ".py": pairs = _chunk_python(src) if pairs: return pairs pat = re.compile(r'(?:^|\n)(?=(?:export\s+)?(?:async\s+)?(?:function|class)|^func |^type |^impl |^pub fn |^fn )', re.M) parts = [p.strip() for p in pat.split(src) if p.strip()] if len(parts) > 1: return [(f"s{i}", p[:4000]) for i,p in enumerate(parts)] return [(f"c{i}", c) for i,c in enumerate(_sliding(src, 1200, 200))] def ingest_file(col, fpath, repo=""): if fpath.stat().st_size > MAX_BYTES: return if any(fnmatch.fnmatch(fpath.name, p) for p in ("*.min.js","*.map","package-lock.json","yarn.lock")): return try: src = fpath.read_text(encoding="utf-8", errors="ignore") except: return if not src.strip(): return rel = str(fpath) pairs = _chunk_file(fpath, src) if fpath.suffix in CODE_EXTS else \ [(f"c{i}",c) for i,c in enumerate(_sliding(src))] ids,docs,metas = [],[],[] for label,chunk in pairs: if not chunk.strip(): continue ids.append(_doc_id(chunk, rel+label)); docs.append(chunk) metas.append({"source":rel,"label":label,"repo":repo,"lang":fpath.suffix.lstrip(".")}) if ids: col.upsert(ids=ids, documents=docs, metadatas=metas) def ingest_dir(col, directory, repo=""): count = 0 for f in directory.rglob("*"): if not f.is_file() or any(p in f.parts for p in SKIP_DIRS): continue ingest_file(col, f, repo); count += 1 log.info("Indexed %d files from %s", count, directory); return count def ingest_pdfs(col): try: import pypdf except ImportError: return 0 n = 0 for pdf in PAPERS_DIR.glob("*.pdf"): try: text = "\n".join(p.extract_text() or "" for p in pypdf.PdfReader(str(pdf)).pages) for i,chunk in enumerate(_sliding(text)): col.upsert(ids=[_doc_id(chunk,str(pdf)+str(i))], documents=[chunk], metadatas=[{"source":str(pdf),"label":f"p{i}","repo":"","lang":"pdf"}]) n += 1 except Exception as e: log.warning("PDF %s: %s", pdf.name, e) return n def _startup(): for _ in range(40): try: r = httpx.get(f"{OLLAMA_URL}/api/tags", timeout=5) if any(EMBED_MODEL in m["name"] for m in r.json().get("models",[])): break except: pass log.info("Waiting for embed model..."); time.sleep(5) code_col = get_col("code"); papers_col = get_col("papers") for d in REPOS_DIR.iterdir(): if d.is_dir(): ingest_dir(code_col, d, d.name) ingest_pdfs(papers_col) for f in PAPERS_DIR.glob("*.txt"): ingest_file(papers_col, f) log.info("Startup index done") @app.on_event("startup") async def on_startup(): threading.Thread(target=_startup, daemon=True).start() @app.get("/health") async def health(): try: cc = _chroma() return {"status":"ok", "code": cc.get_collection("code", embedding_function=_embed_fn()).count(), "papers": cc.get_collection("papers", embedding_function=_embed_fn()).count()} except Exception as e: return {"status":"error","detail":str(e)} class RepoReq(BaseModel): url: str; name: str = ""; branch: str = "main" @app.post("/ingest/repo") async def ingest_repo(req: RepoReq): name = req.name or req.url.rstrip("/").split("/")[-1].removesuffix(".git") dest = REPOS_DIR / name try: if dest.exists(): subprocess.run(["git","pull"], cwd=dest, check=True, timeout=120) else: subprocess.run(["git","clone","--depth=1","-b",req.branch,req.url,str(dest)], check=True, timeout=300) except subprocess.CalledProcessError as e: raise HTTPException(400, str(e)) return {"status":"ok","repo":name,"files":ingest_dir(get_col("code"),dest,name)} @app.post("/ingest/papers") async def trigger_papers(bg: BackgroundTasks): bg.add_task(ingest_pdfs, get_col("papers")); return {"status":"queued"} async def _webhook(payload): repo = payload.get("repository") or {} url = repo.get("clone_url") or repo.get("html_url",""); name = repo.get("name","unknown") if not url: return {"status":"ignored"} dest = REPOS_DIR / name if dest.exists(): subprocess.run(["git","pull"], cwd=dest, timeout=120) else: subprocess.run(["git","clone","--depth=1",url,str(dest)], timeout=300) return {"status":"ok","repo":name,"files":ingest_dir(get_col("code"),dest,name)} @app.post("/webhook/gitea") async def wh_gitea(r: Request): return await _webhook(await r.json()) @app.post("/webhook/github") async def wh_github(r: Request): return await _webhook(await r.json()) def _ctx(query, cols): parts = [] for cn in cols: try: col = get_col(cn) if col.count() == 0: continue res = col.query(query_texts=[query], n_results=min(TOP_K, col.count())) for doc,meta in zip(res["documents"][0], res["metadatas"][0]): parts.append(f"### {meta.get('source','')}:{meta.get('label','')}\n```{meta.get('lang','')}\n{doc}\n```") except Exception as e: log.warning("col %s: %s", cn, e) return "\n\n".join(parts) class ChatReq(BaseModel): model: str = CHAT_MODEL; messages: list[dict[str,Any]] stream: bool = False; collections: list[str] = ["code","papers"] @app.post("/v1/chat/completions") async def chat(req: ChatReq): query = next((m["content"] for m in reversed(req.messages) if m.get("role")=="user"), "") ctx = _ctx(query, req.collections) msgs = [{"role":"system","content":f"You are a coding assistant. Use context below.\n\n## Context\n{ctx}"}] + req.messages payload = {"model":req.model,"messages":msgs,"stream":req.stream} if req.stream: async def gen(): async with httpx.AsyncClient(timeout=300) as c: async with c.stream("POST",f"{OLLAMA_URL}/v1/chat/completions",json=payload) as r: async for chunk in r.aiter_bytes(): yield chunk return StreamingResponse(gen(), media_type="text/event-stream") async with httpx.AsyncClient(timeout=300) as c: r = await c.post(f"{OLLAMA_URL}/v1/chat/completions", json=payload) return r.json() if __name__ == "__main__": import uvicorn; uvicorn.run("server:app", host="0.0.0.0", port=8001, reload=False) PY # ============================================================================= section "Writing mcp_server.py" # ============================================================================= write_if_new "$BASE/mcp_server.py" << 'PY' import json, os, re, subprocess from pathlib import Path import httpx from mcp.server.fastmcp import FastMCP WORKSPACE = Path(os.getenv("WORKSPACE_DIR", "/workspace")) REPOS_DIR = Path(os.getenv("REPOS_DIR", "/repos")) GITEA_URL = os.getenv("GITEA_URL", "http://gitea:3000") GITEA_TOKEN = os.getenv("GITEA_TOKEN", "") GITHUB_TOKEN = os.getenv("GITHUB_TOKEN","") RAG_URL = os.getenv("RAG_URL", "http://rag-server:8001") mcp = FastMCP("local-dev-tools") @mcp.tool() def bash(command: str, cwd: str = "") -> str: """Run a shell command. Default cwd is /workspace.""" work = Path(cwd) if cwd else WORKSPACE work.mkdir(parents=True, exist_ok=True) try: r = subprocess.run(command, shell=True, cwd=work, timeout=120, capture_output=True, text=True) out = r.stdout + (f"\n[stderr]\n{r.stderr}" if r.stderr else "") if r.returncode != 0: out += f"\n[exit {r.returncode}]" return out or "(no output)" except subprocess.TimeoutExpired: return "[timeout]" except Exception as e: return f"[error] {e}" @mcp.tool() def read_file(path: str) -> str: """Read a file. Absolute or relative to /workspace.""" p = Path(path) if Path(path).is_absolute() else WORKSPACE / path if not p.exists(): return f"[not found] {p}" if p.stat().st_size > 500_000: return f"[too large: {p.stat().st_size//1024}KB]" return p.read_text(encoding="utf-8", errors="replace") @mcp.tool() def write_file(path: str, content: str) -> str: """Write content to a file. Relative to /workspace.""" p = Path(path) if Path(path).is_absolute() else WORKSPACE / path p.parent.mkdir(parents=True, exist_ok=True) p.write_text(content, encoding="utf-8") return f"Wrote {len(content)} chars to {p}" @mcp.tool() def list_files(path: str = "", pattern: str = "**/*") -> str: """List files matching a glob pattern.""" base = Path(path) if path else WORKSPACE if not base.exists(): return f"[not found] {base}" files = sorted(str(f.relative_to(base)) for f in base.glob(pattern) if f.is_file()) return "\n".join(files[:500]) or "(empty)" @mcp.tool() def search_code(query: str, path: str = "", glob: str = "", case_sensitive: bool = False) -> str: """Search file contents with ripgrep.""" base = path or str(WORKSPACE) cmd = ["rg", "--line-number", "--no-heading"] if not case_sensitive: cmd.append("-i") if glob: cmd += ["-g", glob] cmd += [query, base] try: r = subprocess.run(cmd, capture_output=True, text=True, timeout=30) lines = r.stdout.strip().splitlines() if len(lines) > 200: lines = lines[:200] + [f"...({len(r.stdout.splitlines())-200} more)"] return "\n".join(lines) or "(no matches)" except FileNotFoundError: r = subprocess.run(["grep","-rn",query,base], capture_output=True, text=True, timeout=30) return r.stdout[:8000] or "(no matches)" @mcp.tool() def fetch_url(url: str, extract_text: bool = True) -> str: """Fetch the content of a URL.""" try: r = httpx.get(url, timeout=30, follow_redirects=True, headers={"User-Agent":"Mozilla/5.0"}) content = r.text if extract_text: content = re.sub(r']*>.*?', '', content, flags=re.DOTALL) content = re.sub(r']*>.*?', '', content, flags=re.DOTALL) content = re.sub(r'<[^>]+>', '', content) content = re.sub(r'\n{3,}', '\n\n', content).strip() return content[:20000] except Exception as e: return f"[error] {e}" def _git(args, repo=""): cwd = Path(repo) if repo else WORKSPACE r = subprocess.run(["git"]+args, cwd=cwd, capture_output=True, text=True, timeout=60) return (r.stdout + r.stderr).strip() or "(no output)" @mcp.tool() def git_status(repo: str = "") -> str: """Show git status.""" return _git(["status","--short"], repo) @mcp.tool() def git_diff(repo: str = "", cached: bool = False) -> str: """Show git diff.""" flag = ["--cached"] if cached else [] return _git(["diff","--stat"]+flag, repo) + "\n\n" + _git(["diff"]+flag, repo) @mcp.tool() def git_log(repo: str = "", n: int = 10) -> str: """Show last n commits.""" return _git(["log",f"-{n}","--oneline","--decorate"], repo) @mcp.tool() def git_commit(message: str, repo: str = "", add_all: bool = True) -> str: """Stage all and commit.""" if add_all: _git(["add","-A"], repo) return _git(["commit","-m",message], repo) @mcp.tool() def git_checkout(branch: str, repo: str = "", create: bool = False) -> str: """Checkout or create a branch.""" return _git(["checkout","-b",branch] if create else ["checkout",branch], repo) def _gitea(method, path, body=None): if not GITEA_TOKEN: return {"error":"GITEA_TOKEN not set in .env"} r = httpx.request(method, f"{GITEA_URL}/api/v1{path}", json=body, headers={"Authorization":f"token {GITEA_TOKEN}"}, timeout=30) try: return r.json() except: return {"status":r.status_code,"text":r.text} @mcp.tool() def gitea_list_repos() -> str: """List your Gitea repos.""" d = _gitea("GET", "/repos/search?limit=50") if "error" in d: return d["error"] return "\n".join(f"{r['full_name']} — {r.get('description','')}" for r in d.get("data",[])) @mcp.tool() def gitea_create_repo(name: str, private: bool = True, description: str = "") -> str: """Create a Gitea repo.""" r = _gitea("POST","/user/repos",{"name":name,"private":private,"description":description,"auto_init":True,"default_branch":"main"}) return r.get("html_url") or str(r) @mcp.tool() def gitea_create_issue(repo: str, title: str, body: str = "") -> str: """Create a Gitea issue (owner/repo).""" r = _gitea("POST",f"/repos/{repo}/issues",{"title":title,"body":body}) return r.get("html_url") or str(r) @mcp.tool() def github_api(method: str, endpoint: str, body: str = "") -> str: """Call GitHub REST API. endpoint e.g. /repos/owner/repo/issues""" if not GITHUB_TOKEN: return "GITHUB_TOKEN not set in .env" r = httpx.request(method.upper(), f"https://api.github.com{endpoint}", json=json.loads(body) if body else None, headers={"Authorization":f"Bearer {GITHUB_TOKEN}","Accept":"application/vnd.github+json"}, timeout=30) try: return json.dumps(r.json(), indent=2) except: return r.text @mcp.tool() def ingest_repo(url: str, name: str = "", branch: str = "main") -> str: """Clone a repo and index it in RAG.""" r = httpx.post(f"{RAG_URL}/ingest/repo", json={"url":url,"name":name,"branch":branch}, timeout=300) return r.text @mcp.tool() def rag_health() -> str: """Check RAG server status.""" try: return httpx.get(f"{RAG_URL}/health", timeout=10).text except Exception as e: return f"RAG unreachable: {e}" if __name__ == "__main__": import uvicorn uvicorn.run(mcp.get_asgi_app(), host="0.0.0.0", port=8002) PY # ============================================================================= section "Requirements" # ============================================================================= cat > "$BASE/requirements.txt" << 'REQ' fastapi uvicorn[standard] httpx pydantic chromadb pypdf python-multipart REQ cat > "$BASE/mcp_requirements.txt" << 'REQ' mcp[cli] fastapi uvicorn[standard] httpx REQ ok "requirements.txt + mcp_requirements.txt" # ============================================================================= section "SearXNG Config" # ============================================================================= mkdir -p "$BASE/searxng" _SXARGS=("$SEARXNG_SAFE_LEVEL") [[ -n "$SEARXNG_DISABLE_ENGINES" ]] && _SXARGS+=(--disable-engines "$SEARXNG_DISABLE_ENGINES") [[ -n "$SEARXNG_ENABLE_ENGINES" ]] && _SXARGS+=(--enable-engines "$SEARXNG_ENABLE_ENGINES") BASE="$BASE" bash "$SCRIPT_DIR/configure-searxng-safesearch.sh" "${_SXARGS[@]}" # ============================================================================= section ".env (tokens — never overwritten)" # ============================================================================= if [[ ! -f "$BASE/.env" ]]; then cat > "$BASE/.env" << ENV # Local AI Stack — edit to add your API tokens GITEA_TOKEN=your-gitea-token-here GITHUB_TOKEN=your-github-token-here ENV ok "Created .env" else info "Kept .env" fi # ============================================================================= section "Docker Compose" # ============================================================================= cat > "$BASE/docker-compose.yml" << COMPOSE # Local AI Stack — 2026-03-20 # GPU: OLLAMA_NUM_GPU=999 uses all available VRAM automatically (V100/RTX/any) # Context: OLLAMA_NUM_CTX= set by detected VRAM (GB) services: ollama: image: ollama/ollama:latest container_name: ollama restart: unless-stopped ports: ["0.0.0.0:11434:11434"] volumes: [ollama-models:/root/.ollama] environment: - OLLAMA_NUM_GPU=999 - OLLAMA_NUM_CTX= - OLLAMA_KEEP_ALIVE=24h - OLLAMA_MAX_LOADED_MODELS=1 deploy: resources: reservations: devices: - driver: nvidia count: all capabilities: [gpu] healthcheck: test: ["CMD","ollama","list"] interval: 30s; timeout: 10s; retries: 5 open-webui: image: ghcr.io/open-webui/open-webui:main container_name: open-webui restart: unless-stopped ports: ["0.0.0.0:3000:8080"] volumes: [open-webui-data:/app/backend/data] environment: - OLLAMA_BASE_URL=http://ollama:11434 - OPENAI_API_BASE_URL=http://rag-server:8001/v1 - OPENAI_API_KEY=local-rag - ENABLE_OPENAI_API=true - ENABLE_RAG_WEB_SEARCH=true - RAG_WEB_SEARCH_ENGINE=searxng - SEARXNG_QUERY_URL=http://searxng:8080/search?q=&format=json&safesearch=${SEARXNG_SAFE_INT} - RAG_WEB_SEARCH_RESULT_COUNT=5 - RAG_WEB_SEARCH_CONCURRENT_REQUESTS=10 - ENABLE_TOOL_SERVERS=true - WEBUI_AUTH=true depends_on: ollama: {condition: service_healthy} chromadb: image: chromadb/chroma:latest container_name: chromadb restart: unless-stopped ports: ["0.0.0.0:8000:8000"] volumes: [$BASE/index:/chroma/chroma] environment: - IS_PERSISTENT=TRUE - ANONYMIZED_TELEMETRY=FALSE healthcheck: test: ["CMD-SHELL","wget -qO- http://localhost:8000/api/v2/heartbeat || exit 1"] interval: 15s; timeout: 5s; retries: 5 rag-server: image: python:3.11-slim container_name: rag-server restart: unless-stopped ports: ["0.0.0.0:8001:8001"] volumes: - $BASE/papers:/papers - $BASE/repos:/repos - $BASE/index:/index - $BASE/server.py:/app/server.py - $BASE/requirements.txt:/app/requirements.txt working_dir: /app environment: - OLLAMA_URL=http://ollama:11434 - CHROMA_URL=http://chromadb:8000 - EMBED_MODEL=nomic-embed-text - CHAT_MODEL= command: > bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git && pip install --no-cache-dir -r requirements.txt && uvicorn server:app --host 0.0.0.0 --port 8001" depends_on: chromadb: {condition: service_healthy} ollama: {condition: service_healthy} mcp-server: image: python:3.11-slim container_name: mcp-server restart: unless-stopped ports: ["0.0.0.0:8002:8002"] volumes: - $BASE/workspace:/workspace - $BASE/repos:/repos - $BASE/mcp_server.py:/app/mcp_server.py - $BASE/mcp_requirements.txt:/app/mcp_requirements.txt working_dir: /app env_file: $BASE/.env environment: - WORKSPACE_DIR=/workspace - REPOS_DIR=/repos - GITEA_URL=http://gitea:3000 - RAG_URL=http://rag-server:8001 command: > bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git ripgrep && pip install --no-cache-dir -r mcp_requirements.txt && python mcp_server.py" depends_on: [rag-server] searxng: image: searxng/searxng:latest container_name: searxng restart: unless-stopped ports: ["0.0.0.0:8888:8080"] volumes: [$BASE/searxng:/etc/searxng] cap_drop: [ALL] cap_add: [CHOWN, SETGID, SETUID] kiwix: image: ghcr.io/kiwix/kiwix-serve:latest container_name: kiwix restart: unless-stopped ports: ["0.0.0.0:8181:8080"] volumes: [$BASE/kiwix:/data] command: "*.zim" gitea: image: gitea/gitea:latest container_name: gitea restart: unless-stopped ports: ["0.0.0.0:3001:3000","0.0.0.0:2222:22"] volumes: - $BASE/gitea:/data - /etc/timezone:/etc/timezone:ro - /etc/localtime:/etc/localtime:ro environment: - USER_UID=1000 - USER_GID=1000 - GITEA__database__DB_TYPE=sqlite3 - GITEA__database__PATH=/data/gitea/gitea.db - GITEA__webhook__ALLOWED_HOST_LIST=rag-server,mcp-server invokeai: image: ghcr.io/invoke-ai/invokeai:latest container_name: invokeai restart: unless-stopped ports: ["0.0.0.0:9090:9090"] volumes: - invokeai-models:/invokeai/models - $BASE/invokeai-outputs:/invokeai/outputs - $BASE/invokeai-data:/invokeai/databases environment: - INVOKEAI_HOST=0.0.0.0 - INVOKEAI_PORT=9090 - INVOKEAI_PRECISION=float16 deploy: resources: reservations: devices: - driver: nvidia count: all capabilities: [gpu] portainer: image: portainer/portainer-ce:latest container_name: portainer restart: unless-stopped ports: ["0.0.0.0:9000:9000","0.0.0.0:9443:9443"] volumes: - /var/run/docker.sock:/var/run/docker.sock - $BASE/portainer-data:/data volumes: ollama-models: open-webui-data: invokeai-models: COMPOSE ok "docker-compose.yml" # ============================================================================= section "Firewall" # ============================================================================= if command -v ufw &>/dev/null && [[ ! -f "$BASE/.ufw-done" ]] || $FORCE; then read -rp " LAN subnet [192.168.1.0/24]: " LAN; LAN="${LAN:-192.168.1.0/24}" [[ "$LAN" =~ /[0-9]+$ ]] || LAN="${LAN}/24" for pc in "3000:Open WebUI" "11434:Ollama" "8001:RAG" "8002:MCP" \ "8000:ChromaDB" "8888:SearXNG" "8181:Kiwix" \ "3001:Gitea" "2222:Gitea SSH" "9090:InvokeAI" \ "9000:Portainer" "9443:Portainer S"; do sudo ufw allow from "$LAN" to any port "${pc%%:*}" proto tcp comment "${pc##*:}" >/dev/null done sudo ufw reload >/dev/null; ok "UFW rules set for $LAN"; touch "$BASE/.ufw-done" fi # ============================================================================= section "Helper Scripts" # ============================================================================= cat > "$BASE/start.sh" << STARTSH #!/bin/bash cd "$BASE" docker compose pull --quiet 2>/dev/null docker compose up -d echo "" echo " Open WebUI → http://$LOCAL_IP:3000" echo " InvokeAI → http://$LOCAL_IP:9090" echo " SearXNG → http://$LOCAL_IP:8888" echo " Kiwix → http://$LOCAL_IP:8181 (run kiwix_download.sh first)" echo " Gitea → http://$LOCAL_IP:3001" echo " RAG → http://$LOCAL_IP:8001/health" echo " MCP SSE → http://$LOCAL_IP:8002/sse" echo " Portainer → https://$LOCAL_IP:9443" echo "" echo " claude mcp add local http://$LOCAL_IP:8002/sse" STARTSH chmod +x "$BASE/start.sh" cat > "$BASE/stop.sh" << STOPSH #!/bin/bash cd "$BASE" && docker compose down STOPSH chmod +x "$BASE/stop.sh" cat > "$BASE/status.sh" << 'STATUSSH' #!/bin/bash echo "=== GPU ===" && nvidia-smi --query-gpu=name,memory.used,memory.total \ --format=csv,noheader 2>/dev/null || echo "(no GPU)" echo "" && echo "=== Containers ===" && docker ps --format "table {{.Names}}\t{{.Status}}\t{{.Ports}}" echo "" && echo "=== Ollama ===" && docker exec ollama ollama ps 2>/dev/null || echo "(not running)" echo "" && echo "=== RAG ===" && curl -s http://localhost:8001/health | python3 -m json.tool 2>/dev/null STATUSSH chmod +x "$BASE/status.sh" cat > "$BASE/pull-models.sh" << PULLSH #!/bin/bash echo "Waiting for Ollama..." until docker exec ollama ollama list &>/dev/null; do sleep 3; done echo "Embed model (RAG — required)..." docker exec ollama ollama pull $EMBED_MODEL echo "Fast chat model..." docker exec ollama ollama pull qwen2.5:7b echo "Smart chat model..." docker exec ollama ollama pull $CHAT_MODEL echo "Code model..." docker exec ollama ollama pull $CODE_MODEL echo "Reasoning model (DeepSeek-R1 14B — optional)..." read -rp "Pull DeepSeek-R1:14b for planning/reasoning? [y/N]: " DR [[ "\${DR,,}" == "y" ]] && docker exec ollama ollama pull deepseek-r1:14b echo "" && docker exec ollama ollama list PULLSH chmod +x "$BASE/pull-models.sh" ok "start.sh stop.sh status.sh pull-models.sh" # ============================================================================= section "Systemd" # ============================================================================= sudo tee /etc/systemd/system/local-ai.service >/dev/null << SYSD [Unit] Description=Local AI Stack After=docker.service network-online.target Requires=docker.service [Service] Type=oneshot RemainAfterExit=yes User=$USER WorkingDirectory=$BASE ExecStart=/bin/bash $BASE/start.sh ExecStop=/bin/bash $BASE/stop.sh TimeoutStartSec=300 [Install] WantedBy=multi-user.target SYSD sudo systemctl daemon-reload && sudo systemctl enable local-ai.service ok "Systemd: local-ai.service enabled" # ============================================================================= section "Starting Stack" # ============================================================================= cd "$BASE" info "Pulling images..." docker compose pull --quiet docker compose up -d ok "Stack running" if ! $IS_UPDATE && ! $NO_PULL; then echo "" read -rp "Pull Ollama models now? (~15-30 min) [Y/n]: " DO_PULL [[ "${DO_PULL,,}" != "n" ]] && bash "$BASE/pull-models.sh" fi # ============================================================================= echo "" echo -e "${GREEN}${BOLD}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}" echo -e "${GREEN}${BOLD} Done! GPU: ${VRAM_GB}GB → $TIER${NC}" echo -e "${GREEN}${BOLD}━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━━${NC}" echo "" echo -e " ${CYAN}Open WebUI${NC} → http://$LOCAL_IP:3000" echo -e " ${CYAN}InvokeAI${NC} → http://$LOCAL_IP:9090" echo -e " ${CYAN}SearXNG${NC} → http://$LOCAL_IP:8888" echo -e " ${CYAN}Kiwix${NC} → http://$LOCAL_IP:8181" echo -e " ${CYAN}Gitea${NC} → http://$LOCAL_IP:3001" echo -e " ${CYAN}RAG${NC} → http://$LOCAL_IP:8001/health" echo -e " ${CYAN}MCP SSE${NC} → http://$LOCAL_IP:8002/sse" echo -e " ${CYAN}Portainer${NC} → https://$LOCAL_IP:9443" echo "" echo -e " ${YELLOW}Add MCP to Claude Code:${NC}" echo " claude mcp add local http://$LOCAL_IP:8002/sse" echo "" echo -e " ${YELLOW}Tokens:${NC} $BASE/.env" echo -e " ${YELLOW}PDFs:${NC} $BASE/papers/" echo -e " ${YELLOW}Workspace:${NC} $BASE/workspace/" echo -e " ${YELLOW}ZIMs:${NC} ./kiwix_download.sh" echo "" echo -e " ${YELLOW}Save Claude usage:${NC} use local 14B for boilerplate, docs," echo " simple fixes. Use Claude Sonnet 4.6 for hard bugs," echo " multi-file refactoring, architecture decisions." echo ""