Files
python-sandbox-mcp/server.py
T
2026-08-20 17:26:11 +01:00

522 lines
20 KiB
Python

#!/usr/bin/env python3
"""
Sandbox MCP Server with Python Execution, SearXNG Search, Playwright & Local Workspace Tools
Provides:
1. Python 3 Code Execution Sandbox
2. SearXNG Metasearch Engine Connection
3. Playwright Browser Automation (Stateful navigation, typing, clicking, reading)
4. Local Workspace Access (File structure, read, write, patch)
5. Git Version Control Integration
6. Smart Document Parsing (markitdown)
7. Semantic Layer Logging Middleware
8. Browser-Compliant CORS Configuration (for llama.cpp Web UI)
"""
import os
import sys
import io
import re
import fnmatch
import contextlib
import logging
import subprocess
from dataclasses import dataclass
from typing import Optional, List
import requests
from fastmcp import FastMCP
from fastmcp.server.middleware import Middleware, MiddlewareContext
from starlette.middleware import Middleware as StarletteMiddleware
from starlette.middleware.cors import CORSMiddleware
from starlette.responses import JSONResponse
# =====================================================================
# 1. Logging & Global Configuration
# =====================================================================
logging.basicConfig(
level=logging.INFO,
format='%(asctime)s [%(levelname)s] %(name)s - %(message)s',
handlers=[logging.StreamHandler(sys.stdout)]
)
logger = logging.getLogger("Sandbox-Server")
SEARXNG_INSTANCE_URL = os.environ.get("SEARXNG_INSTANCE_URL", "http://localhost:8000")
SEARCH_TIMEOUT = int(os.environ.get("SEARXNG_TIMEOUT", "15"))
MAX_RESULTS = int(os.environ.get("SEARXNG_MAX_RESULTS", "10"))
WORKSPACE_ROOT = os.path.abspath(os.environ.get("WORKSPACE_ROOT", os.getcwd()))
logger.info(f"SearXNG instance: {SEARXNG_INSTANCE_URL}")
logger.info(f"Workspace Root Sandbox: {WORKSPACE_ROOT}")
# Initialize FastMCP Server renamed to reflect its multipurpose nature
mcp = FastMCP("MCP-Multi-Server")
# Global lifecycle state manager for tracking a continuous Playwright browser context
PLAYWRIGHT_MANAGER = {
"playwright": None,
"browser": None,
"context": None,
"page": None
}
def _resolve_safe_path(relative_path: str) -> str:
"""Resolves a relative path against WORKSPACE_ROOT and blocks traversal attacks."""
target_path = os.path.abspath(os.path.join(WORKSPACE_ROOT, relative_path))
if not target_path.startswith(WORKSPACE_ROOT):
raise PermissionError("Access denied: Attempted to escape WORKSPACE_ROOT sandbox.")
return target_path
# Create /health endpoint for docker health checks
@mcp.custom_route("/health", methods=["GET"])
async def health_check(request):
return JSONResponse({"status": "healthy", "service": "mcp-server"})
# =====================================================================
# 2. MCP Semantic Logging Middleware
# =====================================================================
class ToolObserverMiddleware(Middleware):
"""Logs incoming LLM tool invocations and server responses."""
async def on_call_tool(self, context: MiddlewareContext, call_next):
tool_name = getattr(context.message, "name", "unknown")
arguments = getattr(context.message, "arguments", {})
logger.info(f"📥 [LLM REQUEST] Invoking tool: '{tool_name}'")
logger.info(f" ↳ Arguments: {arguments}")
try:
result = await call_next(context)
content = getattr(result, "content", result)
logger.info(f"📤 [SERVER RESPONSE] Tool '{tool_name}' completed successfully.")
logger.info(f" ↳ Content: {str(content)[:500]}... (truncated)\n")
return result
except Exception as e:
logger.error(f"❌ [SERVER ERROR] Tool '{tool_name}' crashed: {str(e)}\n")
raise e
mcp.add_middleware(ToolObserverMiddleware())
# =====================================================================
# 3. Browser-Compliant CORS Configuration
# =====================================================================
middleware_config = [
StarletteMiddleware(
CORSMiddleware,
allow_origins=["*"],
allow_methods=["GET", "POST", "DELETE", "OPTIONS"],
allow_headers=[
"mcp-protocol-version",
"mcp-session-id",
"Authorization",
"Content-Type",
],
expose_headers=["mcp-session-id"],
)
]
# =====================================================================
# 4. SearXNG Data Models & Helper Functions
# =====================================================================
@dataclass
class SearchResult:
"""Represents a single search result from SearXNG."""
title: str
url: str
content: str
engine: str
category: str
score: Optional[float] = None
def to_markdown(self) -> str:
"""Format the result as a readable markdown string with clean emojis."""
lines = [
f"**{self.title}**",
f"🔗 {self.url}",
]
if self.content:
lines.append(f"📝 {self.content}")
lines.append(f"🔍 Engine: {self.engine} | Category: {self.category}")
if self.score:
lines.append(f"⭐ Score: {self.score:.2f}")
return "\n".join(lines)
def _format_search_results(results: list, max_display: int = 10) -> str:
"""Format search results into a readable string."""
if not results:
return "No results found."
display_results = results[:max_display]
lines = [f"Found {len(results)} result(s):\n"]
for i, result in enumerate(display_results, 1):
lines.append(f"--- Result {i} ---")
lines.append(result.to_markdown())
lines.append("")
if len(results) > max_display:
lines.append(f"*(Showing {max_display} of {len(results)} results)*)")
return "\n".join(lines)
def search_searxng(
query: str,
categories: str = "general",
engines: Optional[List[str]] = None,
language: str = "all",
pageno: int = 1,
safesearch: int = 0,
time_range: Optional[str] = None,
max_results: int = MAX_RESULTS,
) -> list:
"""Queries the SearXNG API and extracts raw elements into SearchResult objects."""
if not query or not query.strip():
raise ValueError("Query cannot be empty")
url = f"{SEARXNG_INSTANCE_URL}/search"
params = {
"q": query.strip(),
"format": "json",
"categories": categories,
"pageno": pageno,
"safesearch": safesearch,
"language": language,
}
if engines:
params["engines"] = ",".join(engines)
if time_range:
params["time_range"] = time_range
headers = {
"User-Agent": "MCP-SearXNG-Search/1.0",
"Accept": "application/json",
}
try:
response = requests.get(
url,
params=params,
headers=headers,
timeout=SEARCH_TIMEOUT,
)
response.raise_for_status()
data = response.json()
raw_results = data.get("results", [])
results = []
for item in raw_results:
result = SearchResult(
title=item.get("title", ""),
url=item.get("url", ""),
content=item.get("content", ""),
engine=item.get("engine", ""),
category=item.get("category", ""),
score=item.get("score"),
)
results.append(result)
return results[:max_results]
except requests.exceptions.Timeout:
raise requests.exceptions.Timeout(f"Request to {SEARXNG_INSTANCE_URL} timed out.")
except requests.exceptions.ConnectionError:
raise requests.exceptions.ConnectionError(f"Cannot connect to {SEARXNG_INSTANCE_URL}.")
# =====================================================================
# 5. Playwright Session Controller
# =====================================================================
async def get_active_page():
"""Lazily spins up and returns a continuous, stateful Chromium page context."""
global PLAYWRIGHT_MANAGER
if PLAYWRIGHT_MANAGER["page"] is None:
from playwright.async_api import async_playwright
logger.info("Starting fresh headless Playwright session...")
pw = await async_playwright().start()
browser = await pw.chromium.launch(headless=True)
context = await browser.new_context(
viewport={"width": 1280, "height": 720},
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
)
page = await context.new_page()
PLAYWRIGHT_MANAGER["playwright"] = pw
PLAYWRIGHT_MANAGER["browser"] = browser
PLAYWRIGHT_MANAGER["context"] = context
PLAYWRIGHT_MANAGER["page"] = page
return PLAYWRIGHT_MANAGER["page"]
# =====================================================================
# 6. MCP Registered Tools - Core Stack & Playwright
# =====================================================================
@mcp.tool()
def execute_python_code(code: str) -> str:
"""Executes arbitrary Python 3 code in this isolated sandbox container."""
stdout_buffer = io.StringIO()
stderr_buffer = io.StringIO()
with contextlib.redirect_stdout(stdout_buffer), contextlib.redirect_stderr(stderr_buffer):
try:
exec(code, {}, {})
except Exception as e:
sys.stderr.write(f"Runtime Error: {str(e)}")
output = stdout_buffer.getvalue()
errors = stderr_buffer.getvalue()
if errors:
return f"Errors caught during execution:\n{errors}\nPartial Output:\n{output}"
return output if output else "Code executed successfully with no returned stdout."
@mcp.tool()
def search_searxng_tool(query: str, categories: str = "general", max_results: int = 10) -> str:
"""Search the web using a SearXNG metasearch instance."""
try:
results = search_searxng(query=query, categories=categories, max_results=max_results)
return _format_search_results(results, max_display=max_results)
except Exception as e:
return f"Error: An unexpected error occurred: {str(e)}"
@mcp.tool()
async def playwright_navigate(url: str) -> str:
"""Navigate the persistent browser context to a specified URL."""
try:
if not url.startswith(("http://", "https://")):
url = "https://" + url
page = await get_active_page()
response = await page.goto(url, wait_until="networkidle", timeout=30000)
return f"Successfully loaded {url}\nPage Title: {await page.title()}"
except Exception as e:
return f"Navigation Error: {str(e)}"
@mcp.tool()
async def playwright_get_page_text() -> str:
"""Retrieve the raw human-readable text block currently visible on the browser page."""
try:
page = PLAYWRIGHT_MANAGER["page"]
if not page: return "Error: No active browser session. Run playwright_navigate first."
text_content = await page.evaluate("() => document.body.innerText")
return f"=== CURRENT BROWSER TEXT SCREEN ===\n\n{text_content.strip()}"
except Exception as e:
return f"Extraction Error: {str(e)}"
# =====================================================================
# 7. MCP Registered Tools - Module A: Local Workspace Filesystem
# =====================================================================
@mcp.tool()
def file_list_directory_tree(relative_path: str = ".", depth: int = 3) -> str:
"""Provides a structural map of the workspace directory layout."""
try:
target_dir = _resolve_safe_path(relative_path)
if not os.path.isdir(target_dir):
return f"Error: Directory not found at {relative_path}"
tree = []
base_depth = target_dir.rstrip(os.sep).count(os.sep)
for root, dirs, files in os.walk(target_dir):
# Exclude standard hidden/binary directories
dirs[:] = [d for d in dirs if d not in {".git", "__pycache__", "node_modules", ".venv"}]
current_depth = root.count(os.sep) - base_depth
if current_depth >= depth:
del dirs[:]
continue
indent = " " * current_depth
folder_name = os.path.basename(root) if current_depth > 0 else relative_path
tree.append(f"{indent}📂 {folder_name}/")
sub_indent = " " * (current_depth + 1)
for f in files:
tree.append(f"{sub_indent}📄 {f}")
return "\n".join(tree)
except Exception as e:
return f"Error listing directory: {str(e)}"
@mcp.tool()
def file_view_lines(relative_path: str, start_line: int = 1, end_line: int = 100) -> str:
"""Allows precise reading of file blocks to prevent context window explosions."""
try:
target_file = _resolve_safe_path(relative_path)
if not os.path.isfile(target_file):
return f"Error: File not found at {relative_path}"
with open(target_file, "r", encoding="utf-8") as f:
lines = f.readlines()
start_idx = max(0, start_line - 1)
end_idx = min(len(lines), end_line)
output = [f"File: {relative_path} (Lines {start_idx + 1}-{end_idx} of {len(lines)})"]
for i, line in enumerate(lines[start_idx:end_idx], start=start_idx + 1):
line_content = line.rstrip('\n')
output.append(f"{i:4d} | {line_content}")
return "\n".join(output)
except Exception as e:
return f"Error reading file: {str(e)}"
@mcp.tool()
def file_write(relative_path: str, content: str) -> str:
"""Overwrites or creates a file with a clean string payload."""
try:
target_file = _resolve_safe_path(relative_path)
os.makedirs(os.path.dirname(target_file), exist_ok=True)
with open(target_file, "w", encoding="utf-8") as f:
f.write(content)
return f"Success: File written to {relative_path}"
except Exception as e:
return f"Error writing file: {str(e)}"
@mcp.tool()
def file_patch(relative_path: str, search_block: str, replace_block: str) -> str:
"""Implements atomic modifications using Search & Replace blocks."""
try:
target_file = _resolve_safe_path(relative_path)
if not os.path.isfile(target_file):
return f"Error: File not found at {relative_path}"
with open(target_file, "r", encoding="utf-8") as f:
content = f.read()
if search_block not in content:
return "Error: The exact 'search_block' was not found in the target file. Check for whitespace/indentation differences."
updated_content = content.replace(search_block, replace_block)
with open(target_file, "w", encoding="utf-8") as f:
f.write(updated_content)
return f"Success: File {relative_path} patched successfully."
except Exception as e:
return f"Error patching file: {str(e)}"
# =====================================================================
# 8. MCP Registered Tools - Module B: Version Control (Git)
# =====================================================================
@mcp.tool()
def git_status() -> str:
"""Lists all staged, unstaged, and untracked modifications."""
try:
result = subprocess.run(
["git", "status", "--porcelain"],
cwd=WORKSPACE_ROOT, capture_output=True, text=True, check=False
)
if result.returncode != 0:
return f"Git Error:\n{result.stderr}"
return result.stdout if result.stdout else "Working directory clean."
except Exception as e:
return f"Error executing git status: {str(e)}"
@mcp.tool()
def git_diff(relative_path: Optional[str] = None) -> str:
"""Exposes structural code updates via unified diffs."""
try:
cmd = ["git", "diff"]
if relative_path:
cmd.append(_resolve_safe_path(relative_path))
result = subprocess.run(
cmd, cwd=WORKSPACE_ROOT, capture_output=True, text=True, check=False
)
if result.returncode != 0:
return f"Git Error:\n{result.stderr}"
return result.stdout if result.stdout else "No changes detected."
except Exception as e:
return f"Error executing git diff: {str(e)}"
@mcp.tool()
def git_commit(message: str) -> str:
"""Lets the agent checkpoint its progress once sub-tasks pass verification."""
try:
# Note: Depending on your specific flow, you might need `git add .` first,
# or require the agent to commit pre-staged files. We run a commit -am for ease.
subprocess.run(["git", "add", "-A"], cwd=WORKSPACE_ROOT, capture_output=True, check=False)
result = subprocess.run(
["git", "commit", "-m", message],
cwd=WORKSPACE_ROOT, capture_output=True, text=True, check=False
)
if result.returncode != 0:
return f"Git Commit Error:\n{result.stderr}\n{result.stdout}"
return f"Success:\n{result.stdout}"
except Exception as e:
return f"Error executing git commit: {str(e)}"
# =====================================================================
# 9. MCP Registered Tools - Module C: Smart Parsing & Extraction
# =====================================================================
@mcp.tool()
def extract_to_markdown(relative_path: str) -> str:
"""Automatically cleans layout-heavy formats into standard Markdown."""
target_path = _resolve_safe_path(relative_path)
if not os.path.isfile(target_path):
return f"Error: File not found at {relative_path}"
try:
# Imported defensively inside the context block
from markitdown import MarkItDown
mid = MarkItDown()
result = mid.convert(target_path)
return result.text_content
except ImportError:
return "Error: markitdown library is not installed in the container environment."
except Exception as e:
return f"Extraction Error: {str(e)}"
@mcp.tool()
def file_regex_search(pattern: str, file_extension: str = "*") -> str:
"""Fast grep-style file query scanner across workspace files."""
try:
regex = re.compile(pattern)
matches = []
for root, dirs, files in os.walk(WORKSPACE_ROOT):
dirs[:] = [d for d in dirs if d not in {".git", "__pycache__", "node_modules", ".venv"}]
for file in files:
if fnmatch.fnmatch(file, file_extension):
filepath = os.path.join(root, file)
rel_path = os.path.relpath(filepath, WORKSPACE_ROOT)
try:
with open(filepath, "r", encoding="utf-8") as f:
for line_num, line in enumerate(f, 1):
if regex.search(line):
matches.append(f"{rel_path}:{line_num}: {line.strip()}")
except UnicodeDecodeError:
continue # Skip binary files silently
if not matches:
return f"No matches found for pattern: {pattern}"
return "Found Matches:\n" + "\n".join(matches)
except re.error as e:
return f"Invalid regex pattern: {str(e)}"
except Exception as e:
return f"Error during search: {str(e)}"
# =====================================================================
# Execution Block
# =====================================================================
if __name__ == "__main__":
mcp.run(transport="http", host="0.0.0.0", port=8000, middleware=middleware_config)