diff --git a/.env b/.env new file mode 100644 index 0000000..61e1fe3 --- /dev/null +++ b/.env @@ -0,0 +1,12 @@ +# Workspace environment config +WORKSPACE=/usr/workspace + +# SearXNG Configuration +# Change this to your SearXNG instance URL if different from the default +SEARXNG_INSTANCE_URL=http://soda.lab.obelisk.cc:8070 + +# Request timeout in seconds +SEARXNG_TIMEOUT=15 + +# Maximum number of results to return +SEARXNG_MAX_RESULTS=10 diff --git a/.env.example b/.env.example new file mode 100644 index 0000000..257bd45 --- /dev/null +++ b/.env.example @@ -0,0 +1,12 @@ +# Workspace environment config +WORKSPACE=/usr/workspace + +# SearXNG Configuration +# Change this to your SearXNG instance URL if different from the default +SEARXNG_INSTANCE_URL=http://localhost:8000 + +# Request timeout in seconds +SEARXNG_TIMEOUT=15 + +# Maximum number of results to return +SEARXNG_MAX_RESULTS=10 diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..4d627f9 --- /dev/null +++ b/.gitignore @@ -0,0 +1,78 @@ +# Byte-compiled / optimized / DLL files +__pycache__/ +*.py[cod] +*$py.class + +# C extensions +*.so +*.pyd +*.dll + +# Distribution / packaging +build/ +dist/ +*.egg-info/ +pip-wheel-metadata/ +*.egg +wheels/ + +# Virtual environments +venv/ +.env/ +.venv/ +env/ +ENV/ +local/ + +# PyInstaller +# Usually these files are written by a python script from a template +*.manifest +*.spec + +# Installer logs +pip-log.txt +pip-delete-this-directory.txt + +# Unit test / coverage +.coverage +.coverage.* +.cache +nosetests.xml +coverage.xml +*.cover +htmlcov/ +.pytest_cache/ + +# MyPy / pylance / pyright +.mypy_cache/ +.pyre/ +.pytype/ + +# IDEs and editors +.vscode/ +.idea/ +*.sublime-project +*.sublime-workspace + +# Jupyter +.ipynb_checkpoints +.ipynb_meta + +# Logs and databases +*.log +*.sqlite3 +*.db + +# OS files +.DS_Store +Thumbs.db +desktop.ini + +# Documentation +docs/_build/ + +# Misc +*.bak +*.swp + +workspace/ diff --git a/Dockerfile b/Dockerfile index dfd644d..89fc4e0 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,8 +1,19 @@ -# Dockerfile FROM python:3.11-slim -COPY --from=ghcr.io/astral-sh/uv:latest /uv /uvx /bin/ -WORKDIR /app -RUN uv pip install --system fastmcp mcp -COPY server.py . + +# Install system dependencies including git for version control operations +RUN apt-get update && apt-get install -y git && rm -rf /var/lib/apt/lists/* && pip install fastmcp mcp requests playwright markitdown && playwright install chromium --with-deps + +RUN groupadd -g 1000 user && \ + useradd -m -u 1000 -g 1000 -s /bin/bash user + +WORKDIR /workspace + +COPY server.py /app/server.py + +USER 1000:1000 + +RUN playwright install + EXPOSE 8000 -CMD ["python", "server.py"] + +CMD ["python", "/app/server.py"] diff --git a/docker-compose.yml b/docker-compose.yml index 6593d6e..89deb24 100644 --- a/docker-compose.yml +++ b/docker-compose.yml @@ -1,8 +1,27 @@ -# docker-compose.yml +--- + +volumes: + workspace: + external: true + name: turnstone_workspace + + services: - python-mcp: - build: . - container_name: python-sandbox-mcp + mcp-server: + build: + context: . + dockerfile: Dockerfile + user: "1000:1000" ports: - - "8000:8000" + - "8010:8000" + volumes: + - ${WORKSPACE_MOUNT:-/workspace}:/workspace + env_file: + - .env restart: unless-stopped + healthcheck: + test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://localhost:8000/health')"] + interval: 30s + timeout: 10s + retries: 3 + start_period: 10s diff --git a/server.py b/server.py index 6994e6d..015d4ef 100644 --- a/server.py +++ b/server.py @@ -1,23 +1,270 @@ -# server.py +#!/usr/bin/env python3 +""" +Sandbox MCP Server with Python Execution, SearXNG Search, Playwright & Local Workspace Tools + +Provides: +1. Python 3 Code Execution Sandbox +2. SearXNG Metasearch Engine Connection +3. Playwright Browser Automation (Stateful navigation, typing, clicking, reading) +4. Local Workspace Access (File structure, read, write, patch) +5. Git Version Control Integration +6. Smart Document Parsing (markitdown) +7. Semantic Layer Logging Middleware +8. Browser-Compliant CORS Configuration (for llama.cpp Web UI) +""" + +import os import sys import io +import re +import fnmatch import contextlib -from mcp.server.fastmcp import FastMCP +import logging +import subprocess +from dataclasses import dataclass +from typing import Optional, List +import requests -mcp = FastMCP("Sandbox-Python") +from fastmcp import FastMCP +from fastmcp.server.middleware import Middleware, MiddlewareContext +from starlette.middleware import Middleware as StarletteMiddleware +from starlette.middleware.cors import CORSMiddleware +from starlette.responses import JSONResponse + +# ===================================================================== +# 1. Logging & Global Configuration +# ===================================================================== +logging.basicConfig( + level=logging.INFO, + format='%(asctime)s [%(levelname)s] %(name)s - %(message)s', + handlers=[logging.StreamHandler(sys.stdout)] +) +logger = logging.getLogger("Sandbox-Server") + +SEARXNG_INSTANCE_URL = os.environ.get("SEARXNG_INSTANCE_URL", "http://localhost:8000") +SEARCH_TIMEOUT = int(os.environ.get("SEARXNG_TIMEOUT", "15")) +MAX_RESULTS = int(os.environ.get("SEARXNG_MAX_RESULTS", "10")) + +WORKSPACE_ROOT = os.path.abspath(os.environ.get("WORKSPACE_ROOT", os.getcwd())) + +logger.info(f"SearXNG instance: {SEARXNG_INSTANCE_URL}") +logger.info(f"Workspace Root Sandbox: {WORKSPACE_ROOT}") + +# Initialize FastMCP Server renamed to reflect its multipurpose nature +mcp = FastMCP("MCP-Multi-Server") + +# Global lifecycle state manager for tracking a continuous Playwright browser context +PLAYWRIGHT_MANAGER = { + "playwright": None, + "browser": None, + "context": None, + "page": None +} + +def _resolve_safe_path(relative_path: str) -> str: + """Resolves a relative path against WORKSPACE_ROOT and blocks traversal attacks.""" + target_path = os.path.abspath(os.path.join(WORKSPACE_ROOT, relative_path)) + if not target_path.startswith(WORKSPACE_ROOT): + raise PermissionError("Access denied: Attempted to escape WORKSPACE_ROOT sandbox.") + return target_path + +# Create /health endpoint for docker health checks +@mcp.custom_route("/health", methods=["GET"]) +async def health_check(request): + return JSONResponse({"status": "healthy", "service": "mcp-server"}) + +# ===================================================================== +# 2. MCP Semantic Logging Middleware +# ===================================================================== +class ToolObserverMiddleware(Middleware): + """Logs incoming LLM tool invocations and server responses.""" + async def on_call_tool(self, context: MiddlewareContext, call_next): + tool_name = getattr(context.message, "name", "unknown") + arguments = getattr(context.message, "arguments", {}) + + logger.info(f"📥 [LLM REQUEST] Invoking tool: '{tool_name}'") + logger.info(f" ↳ Arguments: {arguments}") + + try: + result = await call_next(context) + content = getattr(result, "content", result) + logger.info(f"📤 [SERVER RESPONSE] Tool '{tool_name}' completed successfully.") + logger.info(f" ↳ Content: {str(content)[:500]}... (truncated)\n") + return result + except Exception as e: + logger.error(f"❌ [SERVER ERROR] Tool '{tool_name}' crashed: {str(e)}\n") + raise e + +mcp.add_middleware(ToolObserverMiddleware()) + +# ===================================================================== +# 3. Browser-Compliant CORS Configuration +# ===================================================================== +middleware_config = [ + StarletteMiddleware( + CORSMiddleware, + allow_origins=["*"], + allow_methods=["GET", "POST", "DELETE", "OPTIONS"], + allow_headers=[ + "mcp-protocol-version", + "mcp-session-id", + "Authorization", + "Content-Type", + ], + expose_headers=["mcp-session-id"], + ) +] + +# ===================================================================== +# 4. SearXNG Data Models & Helper Functions +# ===================================================================== +@dataclass +class SearchResult: + """Represents a single search result from SearXNG.""" + title: str + url: str + content: str + engine: str + category: str + score: Optional[float] = None + + def to_markdown(self) -> str: + """Format the result as a readable markdown string with clean emojis.""" + lines = [ + f"**{self.title}**", + f"🔗 {self.url}", + ] + if self.content: + lines.append(f"📝 {self.content}") + lines.append(f"🔍 Engine: {self.engine} | Category: {self.category}") + if self.score: + lines.append(f"⭐ Score: {self.score:.2f}") + return "\n".join(lines) + + +def _format_search_results(results: list, max_display: int = 10) -> str: + """Format search results into a readable string.""" + if not results: + return "No results found." + + display_results = results[:max_display] + lines = [f"Found {len(results)} result(s):\n"] + + for i, result in enumerate(display_results, 1): + lines.append(f"--- Result {i} ---") + lines.append(result.to_markdown()) + lines.append("") + + if len(results) > max_display: + lines.append(f"*(Showing {max_display} of {len(results)} results)*)") + + return "\n".join(lines) + + +def search_searxng( + query: str, + categories: str = "general", + engines: Optional[List[str]] = None, + language: str = "all", + pageno: int = 1, + safesearch: int = 0, + time_range: Optional[str] = None, + max_results: int = MAX_RESULTS, +) -> list: + """Queries the SearXNG API and extracts raw elements into SearchResult objects.""" + if not query or not query.strip(): + raise ValueError("Query cannot be empty") + + url = f"{SEARXNG_INSTANCE_URL}/search" + + params = { + "q": query.strip(), + "format": "json", + "categories": categories, + "pageno": pageno, + "safesearch": safesearch, + "language": language, + } + + if engines: + params["engines"] = ",".join(engines) + if time_range: + params["time_range"] = time_range + + headers = { + "User-Agent": "MCP-SearXNG-Search/1.0", + "Accept": "application/json", + } + + try: + response = requests.get( + url, + params=params, + headers=headers, + timeout=SEARCH_TIMEOUT, + ) + response.raise_for_status() + data = response.json() + + raw_results = data.get("results", []) + results = [] + + for item in raw_results: + result = SearchResult( + title=item.get("title", ""), + url=item.get("url", ""), + content=item.get("content", ""), + engine=item.get("engine", ""), + category=item.get("category", ""), + score=item.get("score"), + ) + results.append(result) + + return results[:max_results] + + except requests.exceptions.Timeout: + raise requests.exceptions.Timeout(f"Request to {SEARXNG_INSTANCE_URL} timed out.") + except requests.exceptions.ConnectionError: + raise requests.exceptions.ConnectionError(f"Cannot connect to {SEARXNG_INSTANCE_URL}.") + + +# ===================================================================== +# 5. Playwright Session Controller +# ===================================================================== +async def get_active_page(): + """Lazily spins up and returns a continuous, stateful Chromium page context.""" + global PLAYWRIGHT_MANAGER + if PLAYWRIGHT_MANAGER["page"] is None: + from playwright.async_api import async_playwright + + logger.info("Starting fresh headless Playwright session...") + pw = await async_playwright().start() + browser = await pw.chromium.launch(headless=True) + context = await browser.new_context( + viewport={"width": 1280, "height": 720}, + user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36" + ) + page = await context.new_page() + + PLAYWRIGHT_MANAGER["playwright"] = pw + PLAYWRIGHT_MANAGER["browser"] = browser + PLAYWRIGHT_MANAGER["context"] = context + PLAYWRIGHT_MANAGER["page"] = page + + return PLAYWRIGHT_MANAGER["page"] + +# ===================================================================== +# 6. MCP Registered Tools - Core Stack & Playwright +# ===================================================================== @mcp.tool() def execute_python_code(code: str) -> str: - """ - Executes arbitrary Python 3 code in this isolated sandbox container. - Returns the standard output (stdout) and error streams (stderr). - """ + """Executes arbitrary Python 3 code in this isolated sandbox container.""" stdout_buffer = io.StringIO() stderr_buffer = io.StringIO() with contextlib.redirect_stdout(stdout_buffer), contextlib.redirect_stderr(stderr_buffer): try: - # Execute within isolated global/local frames exec(code, {}, {}) except Exception as e: sys.stderr.write(f"Runtime Error: {str(e)}") @@ -29,6 +276,246 @@ def execute_python_code(code: str) -> str: return f"Errors caught during execution:\n{errors}\nPartial Output:\n{output}" return output if output else "Code executed successfully with no returned stdout." +@mcp.tool() +def search_searxng_tool(query: str, categories: str = "general", max_results: int = 10) -> str: + """Search the web using a SearXNG metasearch instance.""" + try: + results = search_searxng(query=query, categories=categories, max_results=max_results) + return _format_search_results(results, max_display=max_results) + except Exception as e: + return f"Error: An unexpected error occurred: {str(e)}" + +@mcp.tool() +async def playwright_navigate(url: str) -> str: + """Navigate the persistent browser context to a specified URL.""" + try: + if not url.startswith(("http://", "https://")): + url = "https://" + url + + page = await get_active_page() + response = await page.goto(url, wait_until="networkidle", timeout=30000) + return f"Successfully loaded {url}\nPage Title: {await page.title()}" + except Exception as e: + return f"Navigation Error: {str(e)}" + +@mcp.tool() +async def playwright_get_page_text() -> str: + """Retrieve the raw human-readable text block currently visible on the browser page.""" + try: + page = PLAYWRIGHT_MANAGER["page"] + if not page: return "Error: No active browser session. Run playwright_navigate first." + text_content = await page.evaluate("() => document.body.innerText") + return f"=== CURRENT BROWSER TEXT SCREEN ===\n\n{text_content.strip()}" + except Exception as e: + return f"Extraction Error: {str(e)}" + +# ===================================================================== +# 7. MCP Registered Tools - Module A: Local Workspace Filesystem +# ===================================================================== + +@mcp.tool() +def file_list_directory_tree(relative_path: str = ".", depth: int = 3) -> str: + """Provides a structural map of the workspace directory layout.""" + try: + target_dir = _resolve_safe_path(relative_path) + if not os.path.isdir(target_dir): + return f"Error: Directory not found at {relative_path}" + + tree = [] + base_depth = target_dir.rstrip(os.sep).count(os.sep) + + for root, dirs, files in os.walk(target_dir): + # Exclude standard hidden/binary directories + dirs[:] = [d for d in dirs if d not in {".git", "__pycache__", "node_modules", ".venv"}] + + current_depth = root.count(os.sep) - base_depth + if current_depth >= depth: + del dirs[:] + continue + + indent = " " * current_depth + folder_name = os.path.basename(root) if current_depth > 0 else relative_path + tree.append(f"{indent}📂 {folder_name}/") + + sub_indent = " " * (current_depth + 1) + for f in files: + tree.append(f"{sub_indent}📄 {f}") + + return "\n".join(tree) + except Exception as e: + return f"Error listing directory: {str(e)}" + +@mcp.tool() +def file_view_lines(relative_path: str, start_line: int = 1, end_line: int = 100) -> str: + """Allows precise reading of file blocks to prevent context window explosions.""" + try: + target_file = _resolve_safe_path(relative_path) + if not os.path.isfile(target_file): + return f"Error: File not found at {relative_path}" + + with open(target_file, "r", encoding="utf-8") as f: + lines = f.readlines() + + start_idx = max(0, start_line - 1) + end_idx = min(len(lines), end_line) + + output = [f"File: {relative_path} (Lines {start_idx + 1}-{end_idx} of {len(lines)})"] + for i, line in enumerate(lines[start_idx:end_idx], start=start_idx + 1): + line_content = line.rstrip('\n') + output.append(f"{i:4d} | {line_content}") + + return "\n".join(output) + except Exception as e: + return f"Error reading file: {str(e)}" + +@mcp.tool() +def file_write(relative_path: str, content: str) -> str: + """Overwrites or creates a file with a clean string payload.""" + try: + target_file = _resolve_safe_path(relative_path) + os.makedirs(os.path.dirname(target_file), exist_ok=True) + + with open(target_file, "w", encoding="utf-8") as f: + f.write(content) + return f"Success: File written to {relative_path}" + except Exception as e: + return f"Error writing file: {str(e)}" + +@mcp.tool() +def file_patch(relative_path: str, search_block: str, replace_block: str) -> str: + """Implements atomic modifications using Search & Replace blocks.""" + try: + target_file = _resolve_safe_path(relative_path) + if not os.path.isfile(target_file): + return f"Error: File not found at {relative_path}" + + with open(target_file, "r", encoding="utf-8") as f: + content = f.read() + + if search_block not in content: + return "Error: The exact 'search_block' was not found in the target file. Check for whitespace/indentation differences." + + updated_content = content.replace(search_block, replace_block) + + with open(target_file, "w", encoding="utf-8") as f: + f.write(updated_content) + + return f"Success: File {relative_path} patched successfully." + except Exception as e: + return f"Error patching file: {str(e)}" + + +# ===================================================================== +# 8. MCP Registered Tools - Module B: Version Control (Git) +# ===================================================================== + +@mcp.tool() +def git_status() -> str: + """Lists all staged, unstaged, and untracked modifications.""" + try: + result = subprocess.run( + ["git", "status", "--porcelain"], + cwd=WORKSPACE_ROOT, capture_output=True, text=True, check=False + ) + if result.returncode != 0: + return f"Git Error:\n{result.stderr}" + return result.stdout if result.stdout else "Working directory clean." + except Exception as e: + return f"Error executing git status: {str(e)}" + +@mcp.tool() +def git_diff(relative_path: Optional[str] = None) -> str: + """Exposes structural code updates via unified diffs.""" + try: + cmd = ["git", "diff"] + if relative_path: + cmd.append(_resolve_safe_path(relative_path)) + + result = subprocess.run( + cmd, cwd=WORKSPACE_ROOT, capture_output=True, text=True, check=False + ) + if result.returncode != 0: + return f"Git Error:\n{result.stderr}" + return result.stdout if result.stdout else "No changes detected." + except Exception as e: + return f"Error executing git diff: {str(e)}" + +@mcp.tool() +def git_commit(message: str) -> str: + """Lets the agent checkpoint its progress once sub-tasks pass verification.""" + try: + # Note: Depending on your specific flow, you might need `git add .` first, + # or require the agent to commit pre-staged files. We run a commit -am for ease. + subprocess.run(["git", "add", "-A"], cwd=WORKSPACE_ROOT, capture_output=True, check=False) + + result = subprocess.run( + ["git", "commit", "-m", message], + cwd=WORKSPACE_ROOT, capture_output=True, text=True, check=False + ) + if result.returncode != 0: + return f"Git Commit Error:\n{result.stderr}\n{result.stdout}" + return f"Success:\n{result.stdout}" + except Exception as e: + return f"Error executing git commit: {str(e)}" + + +# ===================================================================== +# 9. MCP Registered Tools - Module C: Smart Parsing & Extraction +# ===================================================================== + +@mcp.tool() +def extract_to_markdown(relative_path: str) -> str: + """Automatically cleans layout-heavy formats into standard Markdown.""" + target_path = _resolve_safe_path(relative_path) + if not os.path.isfile(target_path): + return f"Error: File not found at {relative_path}" + + try: + # Imported defensively inside the context block + from markitdown import MarkItDown + mid = MarkItDown() + result = mid.convert(target_path) + return result.text_content + except ImportError: + return "Error: markitdown library is not installed in the container environment." + except Exception as e: + return f"Extraction Error: {str(e)}" + +@mcp.tool() +def file_regex_search(pattern: str, file_extension: str = "*") -> str: + """Fast grep-style file query scanner across workspace files.""" + try: + regex = re.compile(pattern) + matches = [] + + for root, dirs, files in os.walk(WORKSPACE_ROOT): + dirs[:] = [d for d in dirs if d not in {".git", "__pycache__", "node_modules", ".venv"}] + + for file in files: + if fnmatch.fnmatch(file, file_extension): + filepath = os.path.join(root, file) + rel_path = os.path.relpath(filepath, WORKSPACE_ROOT) + + try: + with open(filepath, "r", encoding="utf-8") as f: + for line_num, line in enumerate(f, 1): + if regex.search(line): + matches.append(f"{rel_path}:{line_num}: {line.strip()}") + except UnicodeDecodeError: + continue # Skip binary files silently + + if not matches: + return f"No matches found for pattern: {pattern}" + + return "Found Matches:\n" + "\n".join(matches) + except re.error as e: + return f"Invalid regex pattern: {str(e)}" + except Exception as e: + return f"Error during search: {str(e)}" + + +# ===================================================================== +# Execution Block +# ===================================================================== if __name__ == "__main__": - # Override standard stdio pipes to expose an HTTP/SSE server - mcp.run(transport="sse", host="0.0.0.0", port=8000) + mcp.run(transport="http", host="0.0.0.0", port=8000, middleware=middleware_config)