]*>(.*?)
)?', html, _re.DOTALL ) if articles: for path, title, cite, snippet in articles[:limit]: snippet_clean = _re.sub(r'<[^>]+>', '', snippet or '').strip()[:300] results.append({"title": title.strip(), "snippet": snippet_clean, "path": path, "source": cite.strip() if cite else "kiwix"}) else: # Fallback: grab any links for path, title in _re.findall(r']+href="(/[^"]+)"[^>]*>([^<]+)', html)[:limit]: results.append({"title": title.strip(), "snippet": "", "path": path, "source": "kiwix"}) return results except Exception: return [] def _ddg_search(query: str, limit: int = 5) -> list[dict]: """Search DuckDuckGo, return list of {title, snippet, url}.""" try: from duckduckgo_search import DDGS results = [] with DDGS() as ddgs: for r in ddgs.text(query, max_results=limit): results.append({"title": r["title"], "snippet": r["body"], "url": r["href"]}) return results except Exception: return [] @mcp.tool() def search(query: str, limit: int = 5) -> str: """Unified search: checks offline docs (Kiwix) AND live web (DuckDuckGo). Returns results from both with freshness guidance. For timeless topics (algorithms, docs): offline results are sufficient. For time-sensitive topics (releases, CVEs): live results are flagged as preferred.""" needs_fresh = bool(_FRESH_KEYWORDS.search(query)) output_parts = [] # Always search Kiwix (fast, local) kiwix_results = _kiwix_search(query, limit) if kiwix_results: header = "## Offline Docs (Kiwix)" if needs_fresh: header += " ⚠️ POSSIBLY STALE — query looks time-sensitive, prefer live results below" output_parts.append(header) for i, r in enumerate(kiwix_results, 1): entry = f"{i}. **{r['title']}**" if r["source"] and r["source"] != "kiwix": entry += f" ({r['source']})" if r["snippet"]: entry += f"\n {r['snippet']}" entry += f"\n → read_doc('{r['path']}')" output_parts.append(entry) # Search DDG if: query needs fresh data, OR Kiwix returned nothing, OR always (to compare) do_web = needs_fresh or not kiwix_results ddg_results = [] if do_web: ddg_results = _ddg_search(query, limit) if ddg_results: header = "## Live Web (DuckDuckGo)" if needs_fresh: header += " ✓ PREFER THESE for this query" output_parts.append(header) for i, r in enumerate(ddg_results, 1): output_parts.append(f"{i}. **{r['title']}**\n {r['snippet']}\n {r['url']}") elif do_web: output_parts.append("## Live Web (DuckDuckGo)\n(no results or DDG unreachable)") if not kiwix_results and not ddg_results: return f"No results for '{query}' from either offline docs or web search." # Freshness note if kiwix_results and not needs_fresh and not ddg_results: output_parts.append("\n_Offline results look sufficient for this topic. " "Use web_search() if you need to verify currency._") return "\n\n".join(output_parts) @mcp.tool() def read_doc(path: str) -> str: """Read a full article from Kiwix by its path (from search results). Example: read_doc('/wikipedia_en_all/A/Python_(programming_language)')""" try: r = httpx.get(f"{KIWIX_URL}{path}", timeout=15, follow_redirects=True) if r.status_code != 200: return f"Not found: {path} (HTTP {r.status_code})" # Strip HTML tags, keep text content text = _re.sub(r'', '', r.text, flags=_re.DOTALL) text = _re.sub(r'', '', text, flags=_re.DOTALL) text = _re.sub(r'<[^>]+>', ' ', text) text = _re.sub(r'\s+', ' ', text).strip() if len(text) > 8000: text = text[:8000] + "\n\n[... truncated — article continues ...]" return text except Exception as e: return f"Error reading doc: {e}" @mcp.tool() def web_search(query: str, num_results: int = 5) -> str: """Search ONLY the live web via DuckDuckGo. Use search() instead for most queries — it checks both offline and live. Use this directly only when you specifically need live-only results (e.g., verifying if offline info is current).""" results = _ddg_search(query, num_results) if not results: return f"No web results for '{query}'" return "\n\n".join(f"**{r['title']}**\n {r['snippet']}\n {r['url']}" for r in results) # ── Gitea API ───────────────────────────────────────────────────────────────── def _gitea(method: str, path: str, body: dict = {}) -> dict: if not GITEA_TOKEN: return {"error": "GITEA_TOKEN not set in .env"} url = f"{GITEA_URL}/api/v1{path}" headers = {"Authorization": f"token {GITEA_TOKEN}", "Content-Type": "application/json"} r = httpx.request(method, url, json=body or None, headers=headers, timeout=30) try: return r.json() except Exception: return {"status": r.status_code, "text": r.text} @mcp.tool() def gitea_list_repos() -> str: """List your Gitea repos.""" repos = _gitea("GET", "/repos/search?limit=50") if "error" in repos: return repos["error"] return "\n".join(f"{r['full_name']} — {r.get('description','')}" for r in repos.get("data", [])) @mcp.tool() def gitea_create_repo(name: str, private: bool = True, description: str = "") -> str: """Create a new Gitea repository.""" r = _gitea("POST", "/user/repos", {"name": name, "private": private, "description": description, "auto_init": True, "default_branch": "main"}) return r.get("html_url") or str(r) @mcp.tool() def gitea_create_issue(repo: str, title: str, body: str = "") -> str: """Create an issue on a Gitea repo (format: owner/repo).""" r = _gitea("POST", f"/repos/{repo}/issues", {"title": title, "body": body}) return r.get("html_url") or str(r) # ── GitHub API ──────────────────────────────────────────────────────────────── @mcp.tool() def github_api(method: str, endpoint: str, body: str = "") -> str: """Call the GitHub REST API. endpoint e.g. /repos/owner/repo/issues""" if not GITHUB_TOKEN: return "GITHUB_TOKEN not set in .env" import json as _json headers = {"Authorization": f"Bearer {GITHUB_TOKEN}", "Accept": "application/vnd.github+json"} r = httpx.request(method.upper(), f"https://api.github.com{endpoint}", json=_json.loads(body) if body else None, headers=headers, timeout=30) try: return _json.dumps(r.json(), indent=2) except Exception: return r.text # ── Gitea ↔ GitHub sync ────────────────────────────────────────────────────── @mcp.tool() def gitea_github_sync(mode: str = "all", repo: str = "") -> str: """Run Gitea↔GitHub mirror sync. mode: all|pull|push|list. repo: optional owner/name.""" cmd = ["/app/gitea-github-sync.sh"] if mode == "pull": cmd.append("--pull-only") elif mode == "push": cmd.append("--push-only") elif mode == "list": cmd.append("--list") if repo: cmd.extend(["--repo", repo]) try: r = subprocess.run(cmd, capture_output=True, text=True, timeout=600, env={**os.environ, "SYNC_ENV": "/app/.env"}) return (r.stdout + r.stderr).strip() or "Sync completed (no output)" except subprocess.TimeoutExpired: return "Sync timed out after 10 minutes" except Exception as e: return f"Sync failed: {e}" # ── RAG ingest ──────────────────────────────────────────────────────────────── @mcp.tool() def ingest_repo(url: str, name: str = "", branch: str = "main") -> str: """Clone a git repo and index it in the RAG code collection.""" r = httpx.post(f"{RAG_URL}/ingest/repo", json={"url": url, "name": name, "branch": branch}, timeout=300) return r.text @mcp.tool() def rag_health() -> str: """Check RAG server status and indexed document counts.""" try: r = httpx.get(f"{RAG_URL}/health", timeout=10) return r.text except Exception as e: return f"RAG server unreachable: {e}" if __name__ == "__main__": import uvicorn app = mcp.sse_app() uvicorn.run(app, host="0.0.0.0", port=8002)