#!/usr/bin/env python3 """ Sandbox MCP Server with Python Execution, SearXNG Search, Playwright & Local Workspace Tools Provides: 1. Python 3 Code Execution Sandbox 2. SearXNG Metasearch Engine Connection 3. Playwright Browser Automation (Stateful navigation, typing, clicking, reading) 4. Local Workspace Access (File structure, read, write, patch) 5. Git Version Control Integration 6. Smart Document Parsing (markitdown) 7. Semantic Layer Logging Middleware 8. Browser-Compliant CORS Configuration (for llama.cpp Web UI) """ import os import sys import io import re import fnmatch import contextlib import logging import subprocess from dataclasses import dataclass from typing import Optional, List import requests from fastmcp import FastMCP from fastmcp.server.middleware import Middleware, MiddlewareContext from starlette.middleware import Middleware as StarletteMiddleware from starlette.middleware.cors import CORSMiddleware from starlette.responses import JSONResponse # ===================================================================== # 1. Logging & Global Configuration # ===================================================================== logging.basicConfig( level=logging.INFO, format='%(asctime)s [%(levelname)s] %(name)s - %(message)s', handlers=[logging.StreamHandler(sys.stdout)] ) logger = logging.getLogger("Sandbox-Server") SEARXNG_INSTANCE_URL = os.environ.get("SEARXNG_INSTANCE_URL", "http://localhost:8000") SEARCH_TIMEOUT = int(os.environ.get("SEARXNG_TIMEOUT", "15")) MAX_RESULTS = int(os.environ.get("SEARXNG_MAX_RESULTS", "10")) WORKSPACE_ROOT = os.path.abspath(os.environ.get("WORKSPACE_ROOT", os.getcwd())) logger.info(f"SearXNG instance: {SEARXNG_INSTANCE_URL}") logger.info(f"Workspace Root Sandbox: {WORKSPACE_ROOT}") # Initialize FastMCP Server renamed to reflect its multipurpose nature mcp = FastMCP("MCP-Multi-Server") # Global lifecycle state manager for tracking a continuous Playwright browser context PLAYWRIGHT_MANAGER = { "playwright": None, "browser": None, "context": None, "page": None } def _resolve_safe_path(relative_path: str) -> str: """Resolves a relative path against WORKSPACE_ROOT and blocks traversal attacks.""" target_path = os.path.abspath(os.path.join(WORKSPACE_ROOT, relative_path)) if not target_path.startswith(WORKSPACE_ROOT): raise PermissionError("Access denied: Attempted to escape WORKSPACE_ROOT sandbox.") return target_path # Create /health endpoint for docker health checks @mcp.custom_route("/health", methods=["GET"]) async def health_check(request): return JSONResponse({"status": "healthy", "service": "mcp-server"}) # ===================================================================== # 2. MCP Semantic Logging Middleware # ===================================================================== class ToolObserverMiddleware(Middleware): """Logs incoming LLM tool invocations and server responses.""" async def on_call_tool(self, context: MiddlewareContext, call_next): tool_name = getattr(context.message, "name", "unknown") arguments = getattr(context.message, "arguments", {}) logger.info(f"📥 [LLM REQUEST] Invoking tool: '{tool_name}'") logger.info(f" ↳ Arguments: {arguments}") try: result = await call_next(context) content = getattr(result, "content", result) logger.info(f"📤 [SERVER RESPONSE] Tool '{tool_name}' completed successfully.") logger.info(f" ↳ Content: {str(content)[:500]}... (truncated)\n") return result except Exception as e: logger.error(f"❌ [SERVER ERROR] Tool '{tool_name}' crashed: {str(e)}\n") raise e mcp.add_middleware(ToolObserverMiddleware()) # ===================================================================== # 3. Browser-Compliant CORS Configuration # ===================================================================== middleware_config = [ StarletteMiddleware( CORSMiddleware, allow_origins=["*"], allow_methods=["GET", "POST", "DELETE", "OPTIONS"], allow_headers=[ "mcp-protocol-version", "mcp-session-id", "Authorization", "Content-Type", ], expose_headers=["mcp-session-id"], ) ] # ===================================================================== # 4. SearXNG Data Models & Helper Functions # ===================================================================== @dataclass class SearchResult: """Represents a single search result from SearXNG.""" title: str url: str content: str engine: str category: str score: Optional[float] = None def to_markdown(self) -> str: """Format the result as a readable markdown string with clean emojis.""" lines = [ f"**{self.title}**", f"🔗 {self.url}", ] if self.content: lines.append(f"📝 {self.content}") lines.append(f"🔍 Engine: {self.engine} | Category: {self.category}") if self.score: lines.append(f"⭐ Score: {self.score:.2f}") return "\n".join(lines) def _format_search_results(results: list, max_display: int = 10) -> str: """Format search results into a readable string.""" if not results: return "No results found." display_results = results[:max_display] lines = [f"Found {len(results)} result(s):\n"] for i, result in enumerate(display_results, 1): lines.append(f"--- Result {i} ---") lines.append(result.to_markdown()) lines.append("") if len(results) > max_display: lines.append(f"*(Showing {max_display} of {len(results)} results)*)") return "\n".join(lines) def search_searxng( query: str, categories: str = "general", engines: Optional[List[str]] = None, language: str = "all", pageno: int = 1, safesearch: int = 0, time_range: Optional[str] = None, max_results: int = MAX_RESULTS, ) -> list: """Queries the SearXNG API and extracts raw elements into SearchResult objects.""" if not query or not query.strip(): raise ValueError("Query cannot be empty") url = f"{SEARXNG_INSTANCE_URL}/search" params = { "q": query.strip(), "format": "json", "categories": categories, "pageno": pageno, "safesearch": safesearch, "language": language, } if engines: params["engines"] = ",".join(engines) if time_range: params["time_range"] = time_range headers = { "User-Agent": "MCP-SearXNG-Search/1.0", "Accept": "application/json", } try: response = requests.get( url, params=params, headers=headers, timeout=SEARCH_TIMEOUT, ) response.raise_for_status() data = response.json() raw_results = data.get("results", []) results = [] for item in raw_results: result = SearchResult( title=item.get("title", ""), url=item.get("url", ""), content=item.get("content", ""), engine=item.get("engine", ""), category=item.get("category", ""), score=item.get("score"), ) results.append(result) return results[:max_results] except requests.exceptions.Timeout: raise requests.exceptions.Timeout(f"Request to {SEARXNG_INSTANCE_URL} timed out.") except requests.exceptions.ConnectionError: raise requests.exceptions.ConnectionError(f"Cannot connect to {SEARXNG_INSTANCE_URL}.") # ===================================================================== # 5. Playwright Session Controller # ===================================================================== async def get_active_page(): """Lazily spins up and returns a continuous, stateful Chromium page context.""" global PLAYWRIGHT_MANAGER if PLAYWRIGHT_MANAGER["page"] is None: from playwright.async_api import async_playwright logger.info("Starting fresh headless Playwright session...") pw = await async_playwright().start() browser = await pw.chromium.launch(headless=True) context = await browser.new_context( viewport={"width": 1280, "height": 720}, user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36" ) page = await context.new_page() PLAYWRIGHT_MANAGER["playwright"] = pw PLAYWRIGHT_MANAGER["browser"] = browser PLAYWRIGHT_MANAGER["context"] = context PLAYWRIGHT_MANAGER["page"] = page return PLAYWRIGHT_MANAGER["page"] # ===================================================================== # 6. MCP Registered Tools - Core Stack & Playwright # ===================================================================== @mcp.tool() def execute_python_code(code: str) -> str: """Executes arbitrary Python 3 code in this isolated sandbox container.""" stdout_buffer = io.StringIO() stderr_buffer = io.StringIO() with contextlib.redirect_stdout(stdout_buffer), contextlib.redirect_stderr(stderr_buffer): try: exec(code, {}, {}) except Exception as e: sys.stderr.write(f"Runtime Error: {str(e)}") output = stdout_buffer.getvalue() errors = stderr_buffer.getvalue() if errors: return f"Errors caught during execution:\n{errors}\nPartial Output:\n{output}" return output if output else "Code executed successfully with no returned stdout." @mcp.tool() def search_searxng_tool(query: str, categories: str = "general", max_results: int = 10) -> str: """Search the web using a SearXNG metasearch instance.""" try: results = search_searxng(query=query, categories=categories, max_results=max_results) return _format_search_results(results, max_display=max_results) except Exception as e: return f"Error: An unexpected error occurred: {str(e)}" @mcp.tool() async def playwright_navigate(url: str) -> str: """Navigate the persistent browser context to a specified URL.""" try: if not url.startswith(("http://", "https://")): url = "https://" + url page = await get_active_page() response = await page.goto(url, wait_until="networkidle", timeout=30000) return f"Successfully loaded {url}\nPage Title: {await page.title()}" except Exception as e: return f"Navigation Error: {str(e)}" @mcp.tool() async def playwright_get_page_text() -> str: """Retrieve the raw human-readable text block currently visible on the browser page.""" try: page = PLAYWRIGHT_MANAGER["page"] if not page: return "Error: No active browser session. Run playwright_navigate first." text_content = await page.evaluate("() => document.body.innerText") return f"=== CURRENT BROWSER TEXT SCREEN ===\n\n{text_content.strip()}" except Exception as e: return f"Extraction Error: {str(e)}" # ===================================================================== # 7. MCP Registered Tools - Module A: Local Workspace Filesystem # ===================================================================== @mcp.tool() def file_list_directory_tree(relative_path: str = ".", depth: int = 3) -> str: """Provides a structural map of the workspace directory layout.""" try: target_dir = _resolve_safe_path(relative_path) if not os.path.isdir(target_dir): return f"Error: Directory not found at {relative_path}" tree = [] base_depth = target_dir.rstrip(os.sep).count(os.sep) for root, dirs, files in os.walk(target_dir): # Exclude standard hidden/binary directories dirs[:] = [d for d in dirs if d not in {".git", "__pycache__", "node_modules", ".venv"}] current_depth = root.count(os.sep) - base_depth if current_depth >= depth: del dirs[:] continue indent = " " * current_depth folder_name = os.path.basename(root) if current_depth > 0 else relative_path tree.append(f"{indent}📂 {folder_name}/") sub_indent = " " * (current_depth + 1) for f in files: tree.append(f"{sub_indent}📄 {f}") return "\n".join(tree) except Exception as e: return f"Error listing directory: {str(e)}" @mcp.tool() def file_view_lines(relative_path: str, start_line: int = 1, end_line: int = 100) -> str: """Allows precise reading of file blocks to prevent context window explosions.""" try: target_file = _resolve_safe_path(relative_path) if not os.path.isfile(target_file): return f"Error: File not found at {relative_path}" with open(target_file, "r", encoding="utf-8") as f: lines = f.readlines() start_idx = max(0, start_line - 1) end_idx = min(len(lines), end_line) output = [f"File: {relative_path} (Lines {start_idx + 1}-{end_idx} of {len(lines)})"] for i, line in enumerate(lines[start_idx:end_idx], start=start_idx + 1): line_content = line.rstrip('\n') output.append(f"{i:4d} | {line_content}") return "\n".join(output) except Exception as e: return f"Error reading file: {str(e)}" @mcp.tool() def file_write(relative_path: str, content: str) -> str: """Overwrites or creates a file with a clean string payload.""" try: target_file = _resolve_safe_path(relative_path) os.makedirs(os.path.dirname(target_file), exist_ok=True) with open(target_file, "w", encoding="utf-8") as f: f.write(content) return f"Success: File written to {relative_path}" except Exception as e: return f"Error writing file: {str(e)}" @mcp.tool() def file_patch(relative_path: str, search_block: str, replace_block: str) -> str: """Implements atomic modifications using Search & Replace blocks.""" try: target_file = _resolve_safe_path(relative_path) if not os.path.isfile(target_file): return f"Error: File not found at {relative_path}" with open(target_file, "r", encoding="utf-8") as f: content = f.read() if search_block not in content: return "Error: The exact 'search_block' was not found in the target file. Check for whitespace/indentation differences." updated_content = content.replace(search_block, replace_block) with open(target_file, "w", encoding="utf-8") as f: f.write(updated_content) return f"Success: File {relative_path} patched successfully." except Exception as e: return f"Error patching file: {str(e)}" # ===================================================================== # 8. MCP Registered Tools - Module B: Version Control (Git) # ===================================================================== @mcp.tool() def git_status() -> str: """Lists all staged, unstaged, and untracked modifications.""" try: result = subprocess.run( ["git", "status", "--porcelain"], cwd=WORKSPACE_ROOT, capture_output=True, text=True, check=False ) if result.returncode != 0: return f"Git Error:\n{result.stderr}" return result.stdout if result.stdout else "Working directory clean." except Exception as e: return f"Error executing git status: {str(e)}" @mcp.tool() def git_diff(relative_path: Optional[str] = None) -> str: """Exposes structural code updates via unified diffs.""" try: cmd = ["git", "diff"] if relative_path: cmd.append(_resolve_safe_path(relative_path)) result = subprocess.run( cmd, cwd=WORKSPACE_ROOT, capture_output=True, text=True, check=False ) if result.returncode != 0: return f"Git Error:\n{result.stderr}" return result.stdout if result.stdout else "No changes detected." except Exception as e: return f"Error executing git diff: {str(e)}" @mcp.tool() def git_commit(message: str) -> str: """Lets the agent checkpoint its progress once sub-tasks pass verification.""" try: # Note: Depending on your specific flow, you might need `git add .` first, # or require the agent to commit pre-staged files. We run a commit -am for ease. subprocess.run(["git", "add", "-A"], cwd=WORKSPACE_ROOT, capture_output=True, check=False) result = subprocess.run( ["git", "commit", "-m", message], cwd=WORKSPACE_ROOT, capture_output=True, text=True, check=False ) if result.returncode != 0: return f"Git Commit Error:\n{result.stderr}\n{result.stdout}" return f"Success:\n{result.stdout}" except Exception as e: return f"Error executing git commit: {str(e)}" # ===================================================================== # 9. MCP Registered Tools - Module C: Smart Parsing & Extraction # ===================================================================== @mcp.tool() def extract_to_markdown(relative_path: str) -> str: """Automatically cleans layout-heavy formats into standard Markdown.""" target_path = _resolve_safe_path(relative_path) if not os.path.isfile(target_path): return f"Error: File not found at {relative_path}" try: # Imported defensively inside the context block from markitdown import MarkItDown mid = MarkItDown() result = mid.convert(target_path) return result.text_content except ImportError: return "Error: markitdown library is not installed in the container environment." except Exception as e: return f"Extraction Error: {str(e)}" @mcp.tool() def file_regex_search(pattern: str, file_extension: str = "*") -> str: """Fast grep-style file query scanner across workspace files.""" try: regex = re.compile(pattern) matches = [] for root, dirs, files in os.walk(WORKSPACE_ROOT): dirs[:] = [d for d in dirs if d not in {".git", "__pycache__", "node_modules", ".venv"}] for file in files: if fnmatch.fnmatch(file, file_extension): filepath = os.path.join(root, file) rel_path = os.path.relpath(filepath, WORKSPACE_ROOT) try: with open(filepath, "r", encoding="utf-8") as f: for line_num, line in enumerate(f, 1): if regex.search(line): matches.append(f"{rel_path}:{line_num}: {line.strip()}") except UnicodeDecodeError: continue # Skip binary files silently if not matches: return f"No matches found for pattern: {pattern}" return "Found Matches:\n" + "\n".join(matches) except re.error as e: return f"Invalid regex pattern: {str(e)}" except Exception as e: return f"Error during search: {str(e)}" # ===================================================================== # Execution Block # ===================================================================== if __name__ == "__main__": mcp.run(transport="http", host="0.0.0.0", port=8000, middleware=middleware_config)