Files
local-ai/mcp_server.py
T
Claude ba1c7fd68d Add search layer: Kiwix offline docs + DuckDuckGo web search
The model can now search before it generates:

MCP tools (available via Open WebUI + Claude Code):
- search_docs(query): searches Kiwix ZIM files (Wikipedia, Stack Overflow,
  DevDocs, Arch Wiki) — instant, offline, no rate limits
- read_doc(path): reads full article content from Kiwix results
- web_search(query): DuckDuckGo search, no API key needed

Open WebUI native search:
- ENABLE_RAG_WEB_SEARCH=true + RAG_WEB_SEARCH_ENGINE=duckduckgo
- Switched from SearXNG (not in stack) to DDG (zero config)

Search priority: Kiwix first (offline, fast) → DDG fallback (live web)

All 3 setup scripts updated:
- duckduckgo-search added to mcp_requirements.txt
- KIWIX_URL=http://kiwix:80 added to MCP container env
- curl added to MCP container deps (for sync script)
- Open WebUI DDG search enabled by default

https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu
2026-03-22 18:13:49 +00:00

307 lines
14 KiB
Python

#!/usr/bin/env python3
"""
MCP Server — Claude Code-equivalent tools for Open WebUI / Claude Code CLI.
Tools: bash, file read/write/list, code search, git ops, Gitea API, repo ingest,
offline doc search (Kiwix), web search (DuckDuckGo), Gitea↔GitHub sync.
Connects via SSE on port 8002 — add to Open WebUI Tools or ~/.claude/mcp.json
"""
import os, subprocess, textwrap
from pathlib import Path
import httpx
from mcp.server.fastmcp import FastMCP
WORKSPACE = Path(os.getenv("WORKSPACE_DIR", "/workspace"))
REPOS_DIR = Path(os.getenv("REPOS_DIR", "/repos"))
GITEA_URL = os.getenv("GITEA_URL", "http://gitea:3000")
GITEA_TOKEN = os.getenv("GITEA_TOKEN", "")
GITHUB_TOKEN= os.getenv("GITHUB_TOKEN","")
RAG_URL = os.getenv("RAG_URL", "http://rag-server:8001")
KIWIX_URL = os.getenv("KIWIX_URL", "http://kiwix:80")
mcp = FastMCP("local-dev-tools")
# ── bash ──────────────────────────────────────────────────────────────────────
@mcp.tool()
def bash(command: str, cwd: str = "") -> str:
"""Run a shell command. Default cwd is /workspace."""
work = Path(cwd) if cwd else WORKSPACE
work.mkdir(parents=True, exist_ok=True)
try:
r = subprocess.run(command, shell=True, cwd=work, timeout=120,
capture_output=True, text=True)
out = r.stdout + (f"\n[stderr]\n{r.stderr}" if r.stderr else "")
if r.returncode != 0:
out += f"\n[exit {r.returncode}]"
return out or "(no output)"
except subprocess.TimeoutExpired:
return "[timeout after 120s]"
except Exception as e:
return f"[error] {e}"
# ── file ops ──────────────────────────────────────────────────────────────────
@mcp.tool()
def read_file(path: str) -> str:
"""Read a file. Use absolute path or relative to /workspace."""
p = Path(path) if Path(path).is_absolute() else WORKSPACE / path
if not p.exists():
return f"[not found] {p}"
if p.stat().st_size > 500_000:
return f"[too large — {p.stat().st_size//1024}KB]"
return p.read_text(encoding="utf-8", errors="replace")
@mcp.tool()
def write_file(path: str, content: str) -> str:
"""Write content to a file (creates parent dirs). Relative to /workspace."""
p = Path(path) if Path(path).is_absolute() else WORKSPACE / path
p.parent.mkdir(parents=True, exist_ok=True)
p.write_text(content, encoding="utf-8")
return f"Wrote {len(content)} chars to {p}"
@mcp.tool()
def list_files(path: str = "", pattern: str = "**/*") -> str:
"""List files matching a glob pattern."""
base = Path(path) if path else WORKSPACE
if not base.exists():
return f"[not found] {base}"
files = sorted(str(f.relative_to(base)) for f in base.glob(pattern) if f.is_file())
return "\n".join(files[:500]) or "(empty)"
@mcp.tool()
def search_code(query: str, path: str = "", glob: str = "",
case_sensitive: bool = False) -> str:
"""Search file contents with ripgrep. Returns file:line matches."""
base = path or str(WORKSPACE)
cmd = ["rg", "--line-number", "--no-heading"]
if not case_sensitive:
cmd.append("-i")
if glob:
cmd += ["-g", glob]
cmd += [query, base]
try:
r = subprocess.run(cmd, capture_output=True, text=True, timeout=30)
lines = r.stdout.strip().splitlines()
if len(lines) > 200:
lines = lines[:200] + [f"… ({len(r.stdout.splitlines())-200} more)"]
return "\n".join(lines) or "(no matches)"
except FileNotFoundError:
# ripgrep not installed, fall back to grep
r = subprocess.run(["grep", "-rn", query, base],
capture_output=True, text=True, timeout=30)
return r.stdout[:8000] or "(no matches)"
except Exception as e:
return f"[error] {e}"
# ── git ───────────────────────────────────────────────────────────────────────
def _git(args: list[str], repo: str = "") -> str:
cwd = Path(repo) if repo else WORKSPACE
r = subprocess.run(["git"] + args, cwd=cwd,
capture_output=True, text=True, timeout=60)
return (r.stdout + r.stderr).strip() or "(no output)"
@mcp.tool()
def git_status(repo: str = "") -> str:
"""Show git status of a repo (default: /workspace)."""
return _git(["status", "--short"], repo)
@mcp.tool()
def git_diff(repo: str = "", cached: bool = False) -> str:
"""Show git diff (staged if cached=True)."""
args = ["diff", "--stat", "--cached"] if cached else ["diff", "--stat"]
return _git(args, repo) + "\n\n" + _git(
["diff", "--cached"] if cached else ["diff"], repo)
@mcp.tool()
def git_log(repo: str = "", n: int = 10) -> str:
"""Show last n git commits."""
return _git(["log", f"-{n}", "--oneline", "--decorate"], repo)
@mcp.tool()
def git_commit(message: str, repo: str = "", add_all: bool = True) -> str:
"""Stage all changes and create a commit."""
if add_all:
_git(["add", "-A"], repo)
return _git(["commit", "-m", message], repo)
@mcp.tool()
def git_checkout(branch: str, repo: str = "", create: bool = False) -> str:
"""Checkout a branch, optionally creating it."""
args = ["checkout", "-b", branch] if create else ["checkout", branch]
return _git(args, repo)
# ── Search: Kiwix (offline docs) + DuckDuckGo (web) ─────────────────────────
@mcp.tool()
def search_docs(query: str, limit: int = 5) -> str:
"""Search offline docs (Wikipedia, Stack Overflow, DevDocs, Arch Wiki) via Kiwix.
Returns article titles, snippets, and URLs. Always try this before web_search."""
import re
try:
# Kiwix full-text search returns HTML — parse the results
r = httpx.get(f"{KIWIX_URL}/search", params={"pattern": query, "pageLength": limit},
timeout=15, follow_redirects=True)
if r.status_code != 200:
return f"Kiwix returned {r.status_code}. Is kiwix running with ZIM files loaded?"
html = r.text
results = []
# Parse search result entries from Kiwix HTML
# Kiwix wraps results in <article> tags or <a> links with snippets
articles = re.findall(
r'<a[^>]+href="(/[^"]+)"[^>]*>\s*<span[^>]*>([^<]*)</span>.*?'
r'(?:<cite[^>]*>([^<]*)</cite>)?.*?'
r'(?:<p[^>]*>(.*?)</p>)?',
html, re.DOTALL
)
if not articles:
# Fallback: grab any links with text from the results
articles = re.findall(r'<a[^>]+href="(/[^"]+)"[^>]*>([^<]+)</a>', html)
for path, title in articles[:limit]:
results.append(f"**{title.strip()}**\n URL: {KIWIX_URL}{path}\n")
else:
for path, title, cite, snippet in articles[:limit]:
snippet_clean = re.sub(r'<[^>]+>', '', snippet or '').strip()
entry = f"**{title.strip()}**"
if cite:
entry += f" ({cite.strip()})"
if snippet_clean:
entry += f"\n {snippet_clean[:300]}"
entry += f"\n URL: {KIWIX_URL}{path}"
results.append(entry)
if not results:
return f"No results for '{query}' in offline docs. Try web_search instead."
return "\n\n".join(results)
except httpx.ConnectError:
return "Kiwix not reachable. Is the kiwix container running with ZIM files?"
except Exception as e:
return f"Kiwix search error: {e}"
@mcp.tool()
def read_doc(path: str) -> str:
"""Read a full article from Kiwix by its path (from search_docs results).
Example: read_doc('/wikipedia_en_all/A/Python_(programming_language)')"""
try:
r = httpx.get(f"{KIWIX_URL}{path}", timeout=15, follow_redirects=True)
if r.status_code != 200:
return f"Not found: {path} (HTTP {r.status_code})"
import re
# Strip HTML tags, keep text content
text = re.sub(r'<script[^>]*>.*?</script>', '', r.text, flags=re.DOTALL)
text = re.sub(r'<style[^>]*>.*?</style>', '', text, flags=re.DOTALL)
text = re.sub(r'<[^>]+>', ' ', text)
text = re.sub(r'\s+', ' ', text).strip()
# Truncate to ~8K chars to fit in model context
if len(text) > 8000:
text = text[:8000] + "\n\n[... truncated — article continues ...]"
return text
except Exception as e:
return f"Error reading doc: {e}"
@mcp.tool()
def web_search(query: str, num_results: int = 5) -> str:
"""Search the live web via DuckDuckGo. No API key needed.
Use search_docs first for programming/wiki topics — it's faster and offline."""
try:
from duckduckgo_search import DDGS
results = []
with DDGS() as ddgs:
for r in ddgs.text(query, max_results=num_results):
results.append(f"**{r['title']}**\n {r['body']}\n {r['href']}")
return "\n\n".join(results) if results else f"No web results for '{query}'"
except ImportError:
return "duckduckgo-search not installed. Add it to mcp_requirements.txt."
except Exception as e:
return f"Web search error: {e}"
# ── Gitea API ─────────────────────────────────────────────────────────────────
def _gitea(method: str, path: str, body: dict = {}) -> dict:
if not GITEA_TOKEN:
return {"error": "GITEA_TOKEN not set in .env"}
url = f"{GITEA_URL}/api/v1{path}"
headers = {"Authorization": f"token {GITEA_TOKEN}",
"Content-Type": "application/json"}
r = httpx.request(method, url, json=body or None, headers=headers, timeout=30)
try:
return r.json()
except Exception:
return {"status": r.status_code, "text": r.text}
@mcp.tool()
def gitea_list_repos() -> str:
"""List your Gitea repos."""
repos = _gitea("GET", "/repos/search?limit=50")
if "error" in repos:
return repos["error"]
return "\n".join(f"{r['full_name']}{r.get('description','')}"
for r in repos.get("data", []))
@mcp.tool()
def gitea_create_repo(name: str, private: bool = True, description: str = "") -> str:
"""Create a new Gitea repository."""
r = _gitea("POST", "/user/repos",
{"name": name, "private": private, "description": description,
"auto_init": True, "default_branch": "main"})
return r.get("html_url") or str(r)
@mcp.tool()
def gitea_create_issue(repo: str, title: str, body: str = "") -> str:
"""Create an issue on a Gitea repo (format: owner/repo)."""
r = _gitea("POST", f"/repos/{repo}/issues", {"title": title, "body": body})
return r.get("html_url") or str(r)
# ── GitHub API ────────────────────────────────────────────────────────────────
@mcp.tool()
def github_api(method: str, endpoint: str, body: str = "") -> str:
"""Call the GitHub REST API. endpoint e.g. /repos/owner/repo/issues"""
if not GITHUB_TOKEN:
return "GITHUB_TOKEN not set in .env"
import json as _json
headers = {"Authorization": f"Bearer {GITHUB_TOKEN}",
"Accept": "application/vnd.github+json"}
r = httpx.request(method.upper(), f"https://api.github.com{endpoint}",
json=_json.loads(body) if body else None,
headers=headers, timeout=30)
try:
return _json.dumps(r.json(), indent=2)
except Exception:
return r.text
# ── Gitea ↔ GitHub sync ──────────────────────────────────────────────────────
@mcp.tool()
def gitea_github_sync(mode: str = "all", repo: str = "") -> str:
"""Run Gitea↔GitHub mirror sync. mode: all|pull|push|list. repo: optional owner/name."""
cmd = ["/app/gitea-github-sync.sh"]
if mode == "pull": cmd.append("--pull-only")
elif mode == "push": cmd.append("--push-only")
elif mode == "list": cmd.append("--list")
if repo:
cmd.extend(["--repo", repo])
try:
r = subprocess.run(cmd, capture_output=True, text=True, timeout=600,
env={**os.environ, "SYNC_ENV": "/app/.env"})
return (r.stdout + r.stderr).strip() or "Sync completed (no output)"
except subprocess.TimeoutExpired:
return "Sync timed out after 10 minutes"
except Exception as e:
return f"Sync failed: {e}"
# ── RAG ingest ────────────────────────────────────────────────────────────────
@mcp.tool()
def ingest_repo(url: str, name: str = "", branch: str = "main") -> str:
"""Clone a git repo and index it in the RAG code collection."""
r = httpx.post(f"{RAG_URL}/ingest/repo",
json={"url": url, "name": name, "branch": branch}, timeout=300)
return r.text
@mcp.tool()
def rag_health() -> str:
"""Check RAG server status and indexed document counts."""
try:
r = httpx.get(f"{RAG_URL}/health", timeout=10)
return r.text
except Exception as e:
return f"RAG server unreachable: {e}"
if __name__ == "__main__":
import uvicorn
app = mcp.sse_app()
uvicorn.run(app, host="0.0.0.0", port=8002)