518 lines
20 KiB
Python
518 lines
20 KiB
Python
#!/usr/bin/env python3
|
|
"""
|
|
Sandbox MCP Server with Python Execution, SearXNG Search, Playwright & Local Workspace Tools
|
|
|
|
Provides:
|
|
1. Python 3 Code Execution Sandbox
|
|
2. SearXNG Metasearch Engine Connection
|
|
3. Playwright Browser Automation (Stateful navigation, typing, clicking, reading)
|
|
4. Local Workspace Access (File structure, read, write, patch)
|
|
5. Git Version Control Integration
|
|
6. Smart Document Parsing (markitdown)
|
|
7. Semantic Layer Logging Middleware
|
|
8. Browser-Compliant CORS Configuration (for llama.cpp Web UI)
|
|
"""
|
|
|
|
import os
|
|
import sys
|
|
import io
|
|
import re
|
|
import fnmatch
|
|
import contextlib
|
|
import logging
|
|
import subprocess
|
|
from dataclasses import dataclass
|
|
from typing import Optional, List
|
|
import requests
|
|
|
|
from fastmcp import FastMCP
|
|
from fastmcp.server.middleware import Middleware, MiddlewareContext
|
|
from starlette.middleware import Middleware as StarletteMiddleware
|
|
from starlette.middleware.cors import CORSMiddleware
|
|
from starlette.responses import JSONResponse
|
|
|
|
# =====================================================================
|
|
# 1. Logging & Global Configuration
|
|
# =====================================================================
|
|
logging.basicConfig(
|
|
level=logging.INFO,
|
|
format='%(asctime)s [%(levelname)s] %(name)s - %(message)s',
|
|
handlers=[logging.StreamHandler(sys.stdout)]
|
|
)
|
|
logger = logging.getLogger("Sandbox-Server")
|
|
|
|
SEARXNG_INSTANCE_URL = os.environ.get("SEARXNG_INSTANCE_URL", "http://localhost:8000")
|
|
SEARCH_TIMEOUT = int(os.environ.get("SEARXNG_TIMEOUT", "15"))
|
|
MAX_RESULTS = int(os.environ.get("SEARXNG_MAX_RESULTS", "10"))
|
|
|
|
WORKSPACE_ROOT = os.path.abspath(os.environ.get("WORKSPACE_ROOT", os.getcwd()))
|
|
|
|
logger.info(f"SearXNG instance: {SEARXNG_INSTANCE_URL}")
|
|
logger.info(f"Workspace Root Sandbox: {WORKSPACE_ROOT}")
|
|
|
|
# Initialize FastMCP Server renamed to reflect its multipurpose nature
|
|
mcp = FastMCP("MCP-Multi-Server")
|
|
|
|
# Global lifecycle state manager for tracking a continuous Playwright browser context
|
|
PLAYWRIGHT_MANAGER = {
|
|
"playwright": None,
|
|
"browser": None,
|
|
"context": None,
|
|
"page": None
|
|
}
|
|
|
|
def _resolve_safe_path(relative_path: str) -> str:
|
|
"""Resolves a relative path against WORKSPACE_ROOT and blocks traversal attacks."""
|
|
target_path = os.path.abspath(os.path.join(WORKSPACE_ROOT, relative_path))
|
|
if not target_path.startswith(WORKSPACE_ROOT):
|
|
raise PermissionError("Access denied: Attempted to escape WORKSPACE_ROOT sandbox.")
|
|
return target_path
|
|
|
|
# Create /health endpoint for docker health checks
|
|
@mcp.custom_route("/health", methods=["GET"])
|
|
async def health_check(request):
|
|
return JSONResponse({"status": "healthy", "service": "mcp-server"})
|
|
|
|
# =====================================================================
|
|
# 2. MCP Semantic Logging Middleware
|
|
# =====================================================================
|
|
class ToolObserverMiddleware(Middleware):
|
|
"""Logs incoming LLM tool invocations and server responses."""
|
|
async def on_call_tool(self, context: MiddlewareContext, call_next):
|
|
tool_name = getattr(context.message, "name", "unknown")
|
|
arguments = getattr(context.message, "arguments", {})
|
|
|
|
logger.info(f"📥 [LLM REQUEST] Invoking tool: '{tool_name}'")
|
|
logger.info(f" ↳ Arguments: {arguments}")
|
|
|
|
try:
|
|
result = await call_next(context)
|
|
content = getattr(result, "content", result)
|
|
logger.info(f"📤 [SERVER RESPONSE] Tool '{tool_name}' completed successfully.")
|
|
logger.info(f" ↳ Content: {str(content)[:500]}... (truncated)\n")
|
|
return result
|
|
except Exception as e:
|
|
logger.error(f"❌ [SERVER ERROR] Tool '{tool_name}' crashed: {str(e)}\n")
|
|
raise e
|
|
|
|
mcp.add_middleware(ToolObserverMiddleware())
|
|
|
|
# =====================================================================
|
|
# 3. Browser-Compliant CORS Configuration
|
|
# =====================================================================
|
|
middleware_config = [
|
|
StarletteMiddleware(
|
|
CORSMiddleware,
|
|
allow_origins=["*"],
|
|
allow_methods=["GET", "POST", "DELETE", "OPTIONS"],
|
|
allow_headers=[
|
|
"mcp-protocol-version",
|
|
"mcp-session-id",
|
|
"Authorization",
|
|
"Content-Type",
|
|
],
|
|
expose_headers=["mcp-session-id"],
|
|
)
|
|
]
|
|
|
|
# =====================================================================
|
|
# 4. SearXNG Data Models & Helper Functions
|
|
# =====================================================================
|
|
@dataclass
|
|
class SearchResult:
|
|
"""Represents a single search result from SearXNG."""
|
|
title: str
|
|
url: str
|
|
content: str
|
|
engine: str
|
|
category: str
|
|
score: Optional[float] = None
|
|
|
|
def to_markdown(self) -> str:
|
|
"""Format the result as a readable markdown string with clean emojis."""
|
|
lines = [
|
|
f"**{self.title}**",
|
|
f"🔗 {self.url}",
|
|
]
|
|
if self.content:
|
|
lines.append(f"📝 {self.content}")
|
|
lines.append(f"🔍 Engine: {self.engine} | Category: {self.category}")
|
|
if self.score:
|
|
lines.append(f"⭐ Score: {self.score:.2f}")
|
|
return "\n".join(lines)
|
|
|
|
|
|
def _format_search_results(results: list, max_display: int = 10) -> str:
|
|
"""Format search results into a readable string."""
|
|
if not results:
|
|
return "No results found."
|
|
|
|
display_results = results[:max_display]
|
|
lines = [f"Found {len(results)} result(s):\n"]
|
|
|
|
for i, result in enumerate(display_results, 1):
|
|
lines.append(f"--- Result {i} ---")
|
|
lines.append(result.to_markdown())
|
|
lines.append("")
|
|
|
|
if len(results) > max_display:
|
|
lines.append(f"*(Showing {max_display} of {len(results)} results)*)")
|
|
|
|
return "\n".join(lines)
|
|
|
|
|
|
def search_searxng(
|
|
query: str,
|
|
categories: str = "general",
|
|
engines: Optional[List[str]] = None,
|
|
language: str = "all",
|
|
pageno: int = 1,
|
|
safesearch: int = 0,
|
|
time_range: Optional[str] = None,
|
|
max_results: int = MAX_RESULTS,
|
|
) -> list:
|
|
"""Queries the SearXNG API and extracts raw elements into SearchResult objects."""
|
|
if not query or not query.strip():
|
|
raise ValueError("Query cannot be empty")
|
|
|
|
url = f"{SEARXNG_INSTANCE_URL}/search"
|
|
|
|
params = {
|
|
"q": query.strip(),
|
|
"format": "json",
|
|
"categories": categories,
|
|
"pageno": pageno,
|
|
"safesearch": safesearch,
|
|
"language": language,
|
|
}
|
|
|
|
if engines:
|
|
params["engines"] = ",".join(engines)
|
|
if time_range:
|
|
params["time_range"] = time_range
|
|
|
|
headers = {
|
|
"User-Agent": "MCP-SearXNG-Search/1.0",
|
|
"Accept": "application/json",
|
|
}
|
|
|
|
try:
|
|
response = requests.get(
|
|
url,
|
|
params=params,
|
|
headers=headers,
|
|
timeout=SEARCH_TIMEOUT,
|
|
)
|
|
response.raise_for_status()
|
|
data = response.json()
|
|
|
|
raw_results = data.get("results", [])
|
|
results = []
|
|
|
|
for item in raw_results:
|
|
result = SearchResult(
|
|
title=item.get("title", ""),
|
|
url=item.get("url", ""),
|
|
content=item.get("content", ""),
|
|
engine=item.get("engine", ""),
|
|
category=item.get("category", ""),
|
|
score=item.get("score"),
|
|
)
|
|
results.append(result)
|
|
|
|
return results[:max_results]
|
|
|
|
except requests.exceptions.Timeout:
|
|
raise requests.exceptions.Timeout(f"Request to {SEARXNG_INSTANCE_URL} timed out.")
|
|
except requests.exceptions.ConnectionError:
|
|
raise requests.exceptions.ConnectionError(f"Cannot connect to {SEARXNG_INSTANCE_URL}.")
|
|
|
|
|
|
# =====================================================================
|
|
# 5. Playwright Session Controller
|
|
# =====================================================================
|
|
async def get_active_page():
|
|
"""Lazily spins up and returns a continuous, stateful Chromium page context."""
|
|
global PLAYWRIGHT_MANAGER
|
|
if PLAYWRIGHT_MANAGER["page"] is None:
|
|
from playwright.async_api import async_playwright
|
|
|
|
logger.info("Starting fresh headless Playwright session...")
|
|
pw = await async_playwright().start()
|
|
browser = await pw.chromium.launch(headless=True)
|
|
context = await browser.new_context(
|
|
viewport={"width": 1280, "height": 720},
|
|
user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
|
|
)
|
|
page = await context.new_page()
|
|
|
|
PLAYWRIGHT_MANAGER["playwright"] = pw
|
|
PLAYWRIGHT_MANAGER["browser"] = browser
|
|
PLAYWRIGHT_MANAGER["context"] = context
|
|
PLAYWRIGHT_MANAGER["page"] = page
|
|
|
|
return PLAYWRIGHT_MANAGER["page"]
|
|
|
|
# =====================================================================
|
|
# 6. MCP Registered Tools - Core Stack & Playwright
|
|
# =====================================================================
|
|
|
|
@mcp.tool()
|
|
def execute_python_code(code: str) -> str:
|
|
stdout_buffer = io.StringIO()
|
|
stderr_buffer = io.StringIO()
|
|
namespace = {} # shared globals + locals
|
|
with contextlib.redirect_stdout(stdout_buffer), contextlib.redirect_stderr(stderr_buffer):
|
|
try:
|
|
exec(code, namespace) # <-- single arg = globals==locals
|
|
except Exception as e:
|
|
sys.stderr.write(f"Runtime Error: {str(e)}")
|
|
output = stdout_buffer.getvalue()
|
|
if stderr_buffer.getvalue():
|
|
return f"Errors caught during execution:\n{stderr_buffer.getvalue()}\nPartial Output:\n{output}"
|
|
return output if output else "Code executed successfully with no returned stdout."
|
|
|
|
@mcp.tool()
|
|
def search_searxng_tool(query: str, categories: str = "general", max_results: int = 10) -> str:
|
|
"""Search the web using a SearXNG metasearch instance."""
|
|
try:
|
|
results = search_searxng(query=query, categories=categories, max_results=max_results)
|
|
return _format_search_results(results, max_display=max_results)
|
|
except Exception as e:
|
|
return f"Error: An unexpected error occurred: {str(e)}"
|
|
|
|
@mcp.tool()
|
|
async def playwright_navigate(url: str) -> str:
|
|
"""Navigate the persistent browser context to a specified URL."""
|
|
try:
|
|
if not url.startswith(("http://", "https://")):
|
|
url = "https://" + url
|
|
|
|
page = await get_active_page()
|
|
response = await page.goto(url, wait_until="networkidle", timeout=30000)
|
|
return f"Successfully loaded {url}\nPage Title: {await page.title()}"
|
|
except Exception as e:
|
|
return f"Navigation Error: {str(e)}"
|
|
|
|
@mcp.tool()
|
|
async def playwright_get_page_text() -> str:
|
|
"""Retrieve the raw human-readable text block currently visible on the browser page."""
|
|
try:
|
|
page = PLAYWRIGHT_MANAGER["page"]
|
|
if not page: return "Error: No active browser session. Run playwright_navigate first."
|
|
text_content = await page.evaluate("() => document.body.innerText")
|
|
return f"=== CURRENT BROWSER TEXT SCREEN ===\n\n{text_content.strip()}"
|
|
except Exception as e:
|
|
return f"Extraction Error: {str(e)}"
|
|
|
|
# =====================================================================
|
|
# 7. MCP Registered Tools - Module A: Local Workspace Filesystem
|
|
# =====================================================================
|
|
|
|
@mcp.tool()
|
|
def file_list_directory_tree(relative_path: str = ".", depth: int = 3) -> str:
|
|
"""Provides a structural map of the workspace directory layout."""
|
|
try:
|
|
target_dir = _resolve_safe_path(relative_path)
|
|
if not os.path.isdir(target_dir):
|
|
return f"Error: Directory not found at {relative_path}"
|
|
|
|
tree = []
|
|
base_depth = target_dir.rstrip(os.sep).count(os.sep)
|
|
|
|
for root, dirs, files in os.walk(target_dir):
|
|
# Exclude standard hidden/binary directories
|
|
dirs[:] = [d for d in dirs if d not in {".git", "__pycache__", "node_modules", ".venv"}]
|
|
|
|
current_depth = root.count(os.sep) - base_depth
|
|
if current_depth >= depth:
|
|
del dirs[:]
|
|
continue
|
|
|
|
indent = " " * current_depth
|
|
folder_name = os.path.basename(root) if current_depth > 0 else relative_path
|
|
tree.append(f"{indent}📂 {folder_name}/")
|
|
|
|
sub_indent = " " * (current_depth + 1)
|
|
for f in files:
|
|
tree.append(f"{sub_indent}📄 {f}")
|
|
|
|
return "\n".join(tree)
|
|
except Exception as e:
|
|
return f"Error listing directory: {str(e)}"
|
|
|
|
@mcp.tool()
|
|
def file_view_lines(relative_path: str, start_line: int = 1, end_line: int = 100) -> str:
|
|
"""Allows precise reading of file blocks to prevent context window explosions."""
|
|
try:
|
|
target_file = _resolve_safe_path(relative_path)
|
|
if not os.path.isfile(target_file):
|
|
return f"Error: File not found at {relative_path}"
|
|
|
|
with open(target_file, "r", encoding="utf-8") as f:
|
|
lines = f.readlines()
|
|
|
|
start_idx = max(0, start_line - 1)
|
|
end_idx = min(len(lines), end_line)
|
|
|
|
output = [f"File: {relative_path} (Lines {start_idx + 1}-{end_idx} of {len(lines)})"]
|
|
for i, line in enumerate(lines[start_idx:end_idx], start=start_idx + 1):
|
|
line_content = line.rstrip('\n')
|
|
output.append(f"{i:4d} | {line_content}")
|
|
|
|
return "\n".join(output)
|
|
except Exception as e:
|
|
return f"Error reading file: {str(e)}"
|
|
|
|
@mcp.tool()
|
|
def file_write(relative_path: str, content: str) -> str:
|
|
"""Overwrites or creates a file with a clean string payload."""
|
|
try:
|
|
target_file = _resolve_safe_path(relative_path)
|
|
os.makedirs(os.path.dirname(target_file), exist_ok=True)
|
|
|
|
with open(target_file, "w", encoding="utf-8") as f:
|
|
f.write(content)
|
|
return f"Success: File written to {relative_path}"
|
|
except Exception as e:
|
|
return f"Error writing file: {str(e)}"
|
|
|
|
@mcp.tool()
|
|
def file_patch(relative_path: str, search_block: str, replace_block: str) -> str:
|
|
"""Implements atomic modifications using Search & Replace blocks."""
|
|
try:
|
|
target_file = _resolve_safe_path(relative_path)
|
|
if not os.path.isfile(target_file):
|
|
return f"Error: File not found at {relative_path}"
|
|
|
|
with open(target_file, "r", encoding="utf-8") as f:
|
|
content = f.read()
|
|
|
|
if search_block not in content:
|
|
return "Error: The exact 'search_block' was not found in the target file. Check for whitespace/indentation differences."
|
|
|
|
updated_content = content.replace(search_block, replace_block)
|
|
|
|
with open(target_file, "w", encoding="utf-8") as f:
|
|
f.write(updated_content)
|
|
|
|
return f"Success: File {relative_path} patched successfully."
|
|
except Exception as e:
|
|
return f"Error patching file: {str(e)}"
|
|
|
|
|
|
# =====================================================================
|
|
# 8. MCP Registered Tools - Module B: Version Control (Git)
|
|
# =====================================================================
|
|
|
|
@mcp.tool()
|
|
def git_status() -> str:
|
|
"""Lists all staged, unstaged, and untracked modifications."""
|
|
try:
|
|
result = subprocess.run(
|
|
["git", "status", "--porcelain"],
|
|
cwd=WORKSPACE_ROOT, capture_output=True, text=True, check=False
|
|
)
|
|
if result.returncode != 0:
|
|
return f"Git Error:\n{result.stderr}"
|
|
return result.stdout if result.stdout else "Working directory clean."
|
|
except Exception as e:
|
|
return f"Error executing git status: {str(e)}"
|
|
|
|
@mcp.tool()
|
|
def git_diff(relative_path: Optional[str] = None) -> str:
|
|
"""Exposes structural code updates via unified diffs."""
|
|
try:
|
|
cmd = ["git", "diff"]
|
|
if relative_path:
|
|
cmd.append(_resolve_safe_path(relative_path))
|
|
|
|
result = subprocess.run(
|
|
cmd, cwd=WORKSPACE_ROOT, capture_output=True, text=True, check=False
|
|
)
|
|
if result.returncode != 0:
|
|
return f"Git Error:\n{result.stderr}"
|
|
return result.stdout if result.stdout else "No changes detected."
|
|
except Exception as e:
|
|
return f"Error executing git diff: {str(e)}"
|
|
|
|
@mcp.tool()
|
|
def git_commit(message: str) -> str:
|
|
"""Lets the agent checkpoint its progress once sub-tasks pass verification."""
|
|
try:
|
|
# Note: Depending on your specific flow, you might need `git add .` first,
|
|
# or require the agent to commit pre-staged files. We run a commit -am for ease.
|
|
subprocess.run(["git", "add", "-A"], cwd=WORKSPACE_ROOT, capture_output=True, check=False)
|
|
|
|
result = subprocess.run(
|
|
["git", "commit", "-m", message],
|
|
cwd=WORKSPACE_ROOT, capture_output=True, text=True, check=False
|
|
)
|
|
if result.returncode != 0:
|
|
return f"Git Commit Error:\n{result.stderr}\n{result.stdout}"
|
|
return f"Success:\n{result.stdout}"
|
|
except Exception as e:
|
|
return f"Error executing git commit: {str(e)}"
|
|
|
|
|
|
# =====================================================================
|
|
# 9. MCP Registered Tools - Module C: Smart Parsing & Extraction
|
|
# =====================================================================
|
|
|
|
@mcp.tool()
|
|
def extract_to_markdown(relative_path: str) -> str:
|
|
"""Automatically cleans layout-heavy formats into standard Markdown."""
|
|
target_path = _resolve_safe_path(relative_path)
|
|
if not os.path.isfile(target_path):
|
|
return f"Error: File not found at {relative_path}"
|
|
|
|
try:
|
|
# Imported defensively inside the context block
|
|
from markitdown import MarkItDown
|
|
mid = MarkItDown()
|
|
result = mid.convert(target_path)
|
|
return result.text_content
|
|
except ImportError:
|
|
return "Error: markitdown library is not installed in the container environment."
|
|
except Exception as e:
|
|
return f"Extraction Error: {str(e)}"
|
|
|
|
@mcp.tool()
|
|
def file_regex_search(pattern: str, file_extension: str = "*") -> str:
|
|
"""Fast grep-style file query scanner across workspace files."""
|
|
try:
|
|
regex = re.compile(pattern)
|
|
matches = []
|
|
|
|
for root, dirs, files in os.walk(WORKSPACE_ROOT):
|
|
dirs[:] = [d for d in dirs if d not in {".git", "__pycache__", "node_modules", ".venv"}]
|
|
|
|
for file in files:
|
|
if fnmatch.fnmatch(file, file_extension):
|
|
filepath = os.path.join(root, file)
|
|
rel_path = os.path.relpath(filepath, WORKSPACE_ROOT)
|
|
|
|
try:
|
|
with open(filepath, "r", encoding="utf-8") as f:
|
|
for line_num, line in enumerate(f, 1):
|
|
if regex.search(line):
|
|
matches.append(f"{rel_path}:{line_num}: {line.strip()}")
|
|
except UnicodeDecodeError:
|
|
continue # Skip binary files silently
|
|
|
|
if not matches:
|
|
return f"No matches found for pattern: {pattern}"
|
|
|
|
return "Found Matches:\n" + "\n".join(matches)
|
|
except re.error as e:
|
|
return f"Invalid regex pattern: {str(e)}"
|
|
except Exception as e:
|
|
return f"Error during search: {str(e)}"
|
|
|
|
|
|
# =====================================================================
|
|
# Execution Block
|
|
# =====================================================================
|
|
if __name__ == "__main__":
|
|
mcp.run(transport="http", host="0.0.0.0", port=8000, middleware=middleware_config)
|