Add search layer: Kiwix offline docs + DuckDuckGo web search
The model can now search before it generates: MCP tools (available via Open WebUI + Claude Code): - search_docs(query): searches Kiwix ZIM files (Wikipedia, Stack Overflow, DevDocs, Arch Wiki) — instant, offline, no rate limits - read_doc(path): reads full article content from Kiwix results - web_search(query): DuckDuckGo search, no API key needed Open WebUI native search: - ENABLE_RAG_WEB_SEARCH=true + RAG_WEB_SEARCH_ENGINE=duckduckgo - Switched from SearXNG (not in stack) to DDG (zero config) Search priority: Kiwix first (offline, fast) → DDG fallback (live web) All 3 setup scripts updated: - duckduckgo-search added to mcp_requirements.txt - KIWIX_URL=http://kiwix:80 added to MCP container env - curl added to MCP container deps (for sync script) - Open WebUI DDG search enabled by default https://claude.ai/code/session_01PtYTPherSJaxDEVPgF6Nxu
This commit is contained in:
@@ -753,6 +753,7 @@ mcp[cli]
|
|||||||
fastapi
|
fastapi
|
||||||
uvicorn[standard]
|
uvicorn[standard]
|
||||||
httpx
|
httpx
|
||||||
|
duckduckgo-search
|
||||||
REQ
|
REQ
|
||||||
ok "mcp_requirements.txt"
|
ok "mcp_requirements.txt"
|
||||||
|
|
||||||
@@ -867,6 +868,8 @@ ${OLLAMA_VOLUME_LINE}
|
|||||||
- ENABLE_TOOL_SERVERS=true
|
- ENABLE_TOOL_SERVERS=true
|
||||||
- WEBUI_AUTH=true
|
- WEBUI_AUTH=true
|
||||||
- WEBUI_URL=${WEBUI_URL:-}
|
- WEBUI_URL=${WEBUI_URL:-}
|
||||||
|
- ENABLE_RAG_WEB_SEARCH=true
|
||||||
|
- RAG_WEB_SEARCH_ENGINE=duckduckgo
|
||||||
depends_on:
|
depends_on:
|
||||||
ollama:
|
ollama:
|
||||||
condition: service_healthy
|
condition: service_healthy
|
||||||
@@ -938,6 +941,7 @@ ${OLLAMA_VOLUME_LINE}
|
|||||||
- REPOS_DIR=/repos
|
- REPOS_DIR=/repos
|
||||||
- GITEA_URL=http://gitea:3000
|
- GITEA_URL=http://gitea:3000
|
||||||
- RAG_URL=http://rag-server:8001
|
- RAG_URL=http://rag-server:8001
|
||||||
|
- KIWIX_URL=http://kiwix:80
|
||||||
command: >
|
command: >
|
||||||
bash -c "apt-get update -qq &&
|
bash -c "apt-get update -qq &&
|
||||||
apt-get install -y --no-install-recommends git ripgrep curl &&
|
apt-get install -y --no-install-recommends git ripgrep curl &&
|
||||||
|
|||||||
@@ -489,6 +489,7 @@ mcp[cli]
|
|||||||
fastapi
|
fastapi
|
||||||
uvicorn[standard]
|
uvicorn[standard]
|
||||||
httpx
|
httpx
|
||||||
|
duckduckgo-search
|
||||||
REQ
|
REQ
|
||||||
ok "requirements.txt + mcp_requirements.txt"
|
ok "requirements.txt + mcp_requirements.txt"
|
||||||
|
|
||||||
@@ -569,6 +570,8 @@ services:
|
|||||||
- ENABLE_OPENAI_API=true
|
- ENABLE_OPENAI_API=true
|
||||||
- ENABLE_TOOL_SERVERS=true
|
- ENABLE_TOOL_SERVERS=true
|
||||||
- WEBUI_AUTH=true
|
- WEBUI_AUTH=true
|
||||||
|
- ENABLE_RAG_WEB_SEARCH=true
|
||||||
|
- RAG_WEB_SEARCH_ENGINE=duckduckgo
|
||||||
depends_on:
|
depends_on:
|
||||||
ollama: {condition: service_healthy}
|
ollama: {condition: service_healthy}
|
||||||
|
|
||||||
@@ -628,6 +631,7 @@ services:
|
|||||||
- REPOS_DIR=/repos
|
- REPOS_DIR=/repos
|
||||||
- GITEA_URL=http://gitea:3000
|
- GITEA_URL=http://gitea:3000
|
||||||
- RAG_URL=http://rag-server:8001
|
- RAG_URL=http://rag-server:8001
|
||||||
|
- KIWIX_URL=http://kiwix:80
|
||||||
command: >
|
command: >
|
||||||
bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git ripgrep curl &&
|
bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git ripgrep curl &&
|
||||||
pip install --no-cache-dir -r mcp_requirements.txt &&
|
pip install --no-cache-dir -r mcp_requirements.txt &&
|
||||||
|
|||||||
+85
-1
@@ -1,7 +1,8 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""
|
"""
|
||||||
MCP Server — Claude Code-equivalent tools for Open WebUI / Claude Code CLI.
|
MCP Server — Claude Code-equivalent tools for Open WebUI / Claude Code CLI.
|
||||||
Tools: bash, file read/write/list, code search, git ops, Gitea API, repo ingest.
|
Tools: bash, file read/write/list, code search, git ops, Gitea API, repo ingest,
|
||||||
|
offline doc search (Kiwix), web search (DuckDuckGo), Gitea↔GitHub sync.
|
||||||
Connects via SSE on port 8002 — add to Open WebUI Tools or ~/.claude/mcp.json
|
Connects via SSE on port 8002 — add to Open WebUI Tools or ~/.claude/mcp.json
|
||||||
"""
|
"""
|
||||||
import os, subprocess, textwrap
|
import os, subprocess, textwrap
|
||||||
@@ -16,6 +17,7 @@ GITEA_URL = os.getenv("GITEA_URL", "http://gitea:3000")
|
|||||||
GITEA_TOKEN = os.getenv("GITEA_TOKEN", "")
|
GITEA_TOKEN = os.getenv("GITEA_TOKEN", "")
|
||||||
GITHUB_TOKEN= os.getenv("GITHUB_TOKEN","")
|
GITHUB_TOKEN= os.getenv("GITHUB_TOKEN","")
|
||||||
RAG_URL = os.getenv("RAG_URL", "http://rag-server:8001")
|
RAG_URL = os.getenv("RAG_URL", "http://rag-server:8001")
|
||||||
|
KIWIX_URL = os.getenv("KIWIX_URL", "http://kiwix:80")
|
||||||
|
|
||||||
mcp = FastMCP("local-dev-tools")
|
mcp = FastMCP("local-dev-tools")
|
||||||
|
|
||||||
@@ -127,6 +129,88 @@ def git_checkout(branch: str, repo: str = "", create: bool = False) -> str:
|
|||||||
args = ["checkout", "-b", branch] if create else ["checkout", branch]
|
args = ["checkout", "-b", branch] if create else ["checkout", branch]
|
||||||
return _git(args, repo)
|
return _git(args, repo)
|
||||||
|
|
||||||
|
# ── Search: Kiwix (offline docs) + DuckDuckGo (web) ─────────────────────────
|
||||||
|
@mcp.tool()
|
||||||
|
def search_docs(query: str, limit: int = 5) -> str:
|
||||||
|
"""Search offline docs (Wikipedia, Stack Overflow, DevDocs, Arch Wiki) via Kiwix.
|
||||||
|
Returns article titles, snippets, and URLs. Always try this before web_search."""
|
||||||
|
import re
|
||||||
|
try:
|
||||||
|
# Kiwix full-text search returns HTML — parse the results
|
||||||
|
r = httpx.get(f"{KIWIX_URL}/search", params={"pattern": query, "pageLength": limit},
|
||||||
|
timeout=15, follow_redirects=True)
|
||||||
|
if r.status_code != 200:
|
||||||
|
return f"Kiwix returned {r.status_code}. Is kiwix running with ZIM files loaded?"
|
||||||
|
html = r.text
|
||||||
|
results = []
|
||||||
|
# Parse search result entries from Kiwix HTML
|
||||||
|
# Kiwix wraps results in <article> tags or <a> links with snippets
|
||||||
|
articles = re.findall(
|
||||||
|
r'<a[^>]+href="(/[^"]+)"[^>]*>\s*<span[^>]*>([^<]*)</span>.*?'
|
||||||
|
r'(?:<cite[^>]*>([^<]*)</cite>)?.*?'
|
||||||
|
r'(?:<p[^>]*>(.*?)</p>)?',
|
||||||
|
html, re.DOTALL
|
||||||
|
)
|
||||||
|
if not articles:
|
||||||
|
# Fallback: grab any links with text from the results
|
||||||
|
articles = re.findall(r'<a[^>]+href="(/[^"]+)"[^>]*>([^<]+)</a>', html)
|
||||||
|
for path, title in articles[:limit]:
|
||||||
|
results.append(f"**{title.strip()}**\n URL: {KIWIX_URL}{path}\n")
|
||||||
|
else:
|
||||||
|
for path, title, cite, snippet in articles[:limit]:
|
||||||
|
snippet_clean = re.sub(r'<[^>]+>', '', snippet or '').strip()
|
||||||
|
entry = f"**{title.strip()}**"
|
||||||
|
if cite:
|
||||||
|
entry += f" ({cite.strip()})"
|
||||||
|
if snippet_clean:
|
||||||
|
entry += f"\n {snippet_clean[:300]}"
|
||||||
|
entry += f"\n URL: {KIWIX_URL}{path}"
|
||||||
|
results.append(entry)
|
||||||
|
if not results:
|
||||||
|
return f"No results for '{query}' in offline docs. Try web_search instead."
|
||||||
|
return "\n\n".join(results)
|
||||||
|
except httpx.ConnectError:
|
||||||
|
return "Kiwix not reachable. Is the kiwix container running with ZIM files?"
|
||||||
|
except Exception as e:
|
||||||
|
return f"Kiwix search error: {e}"
|
||||||
|
|
||||||
|
@mcp.tool()
|
||||||
|
def read_doc(path: str) -> str:
|
||||||
|
"""Read a full article from Kiwix by its path (from search_docs results).
|
||||||
|
Example: read_doc('/wikipedia_en_all/A/Python_(programming_language)')"""
|
||||||
|
try:
|
||||||
|
r = httpx.get(f"{KIWIX_URL}{path}", timeout=15, follow_redirects=True)
|
||||||
|
if r.status_code != 200:
|
||||||
|
return f"Not found: {path} (HTTP {r.status_code})"
|
||||||
|
import re
|
||||||
|
# Strip HTML tags, keep text content
|
||||||
|
text = re.sub(r'<script[^>]*>.*?</script>', '', r.text, flags=re.DOTALL)
|
||||||
|
text = re.sub(r'<style[^>]*>.*?</style>', '', text, flags=re.DOTALL)
|
||||||
|
text = re.sub(r'<[^>]+>', ' ', text)
|
||||||
|
text = re.sub(r'\s+', ' ', text).strip()
|
||||||
|
# Truncate to ~8K chars to fit in model context
|
||||||
|
if len(text) > 8000:
|
||||||
|
text = text[:8000] + "\n\n[... truncated — article continues ...]"
|
||||||
|
return text
|
||||||
|
except Exception as e:
|
||||||
|
return f"Error reading doc: {e}"
|
||||||
|
|
||||||
|
@mcp.tool()
|
||||||
|
def web_search(query: str, num_results: int = 5) -> str:
|
||||||
|
"""Search the live web via DuckDuckGo. No API key needed.
|
||||||
|
Use search_docs first for programming/wiki topics — it's faster and offline."""
|
||||||
|
try:
|
||||||
|
from duckduckgo_search import DDGS
|
||||||
|
results = []
|
||||||
|
with DDGS() as ddgs:
|
||||||
|
for r in ddgs.text(query, max_results=num_results):
|
||||||
|
results.append(f"**{r['title']}**\n {r['body']}\n {r['href']}")
|
||||||
|
return "\n\n".join(results) if results else f"No web results for '{query}'"
|
||||||
|
except ImportError:
|
||||||
|
return "duckduckgo-search not installed. Add it to mcp_requirements.txt."
|
||||||
|
except Exception as e:
|
||||||
|
return f"Web search error: {e}"
|
||||||
|
|
||||||
# ── Gitea API ─────────────────────────────────────────────────────────────────
|
# ── Gitea API ─────────────────────────────────────────────────────────────────
|
||||||
def _gitea(method: str, path: str, body: dict = {}) -> dict:
|
def _gitea(method: str, path: str, body: dict = {}) -> dict:
|
||||||
if not GITEA_TOKEN:
|
if not GITEA_TOKEN:
|
||||||
|
|||||||
@@ -1572,6 +1572,7 @@ mcp[cli]
|
|||||||
fastapi
|
fastapi
|
||||||
uvicorn[standard]
|
uvicorn[standard]
|
||||||
httpx
|
httpx
|
||||||
|
duckduckgo-search
|
||||||
REQ
|
REQ
|
||||||
ok "requirements.txt + mcp_requirements.txt"
|
ok "requirements.txt + mcp_requirements.txt"
|
||||||
|
|
||||||
@@ -1651,7 +1652,7 @@ services:
|
|||||||
- OPENAI_API_KEY=local-rag
|
- OPENAI_API_KEY=local-rag
|
||||||
- ENABLE_OPENAI_API=true
|
- ENABLE_OPENAI_API=true
|
||||||
- ENABLE_RAG_WEB_SEARCH=true
|
- ENABLE_RAG_WEB_SEARCH=true
|
||||||
- RAG_WEB_SEARCH_ENGINE=searxng
|
- RAG_WEB_SEARCH_ENGINE=duckduckgo
|
||||||
- SEARXNG_QUERY_URL=http://searxng:8080/search?q=<query>&format=json
|
- SEARXNG_QUERY_URL=http://searxng:8080/search?q=<query>&format=json
|
||||||
- WEBUI_AUTH=false
|
- WEBUI_AUTH=false
|
||||||
depends_on:
|
depends_on:
|
||||||
@@ -1712,8 +1713,9 @@ services:
|
|||||||
- REPOS_DIR=/repos
|
- REPOS_DIR=/repos
|
||||||
- GITEA_URL=http://gitea:3000
|
- GITEA_URL=http://gitea:3000
|
||||||
- RAG_URL=http://rag-server:8001
|
- RAG_URL=http://rag-server:8001
|
||||||
|
- KIWIX_URL=http://kiwix:80
|
||||||
command: >
|
command: >
|
||||||
bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git ripgrep &&
|
bash -c "apt-get update -qq && apt-get install -y --no-install-recommends git ripgrep curl &&
|
||||||
pip install --no-cache-dir -r mcp_requirements.txt &&
|
pip install --no-cache-dir -r mcp_requirements.txt &&
|
||||||
python mcp_server.py"
|
python mcp_server.py"
|
||||||
depends_on: [rag-server]
|
depends_on: [rag-server]
|
||||||
|
|||||||
Reference in New Issue
Block a user