diff --git a/.mcp.json b/.mcp.json index c0a6f80b..f3f0e3d0 100644 --- a/.mcp.json +++ b/.mcp.json @@ -116,7 +116,7 @@ "gemma4": { "_comment": "Local Gemma4-12B model via llama-server. Free, fast, good at math/code.", "command": "python3", - "args": ["4-Infrastructure/shim/gemma4_mcp.py"], + "args": ["/home/allaun/SilverSight/python/gemma4_mcp.py"], "env": {} } } diff --git a/4-Infrastructure/shim/gemma4_mcp.py b/4-Infrastructure/shim/gemma4_mcp.py deleted file mode 100644 index 510b350f..00000000 --- a/4-Infrastructure/shim/gemma4_mcp.py +++ /dev/null @@ -1,162 +0,0 @@ -#!/usr/bin/env python3 -""" -gemma4_mcp.py — Local Gemma4-12B MCP server. - -Calls the local llama-server endpoint at http://127.0.0.1:8081/v1. -No API key required. Returns reasoning + content. - -Usage: - python3 gemma4_mcp.py # starts MCP server on stdio -""" - -import json -import sys - -try: - import httpx -except ImportError: - httpx = None - -GEMMA_URL = "http://127.0.0.1:8081/v1/chat/completions" -GEMMA_MODEL = "gemma4-12b" - - -def call_gemma(question: str, system: str = "", max_tokens: int = 2000, - temperature: float = 0.3, enable_thinking: bool = True) -> dict: - """Call local Gemma4-12B model.""" - if httpx is None: - return {"error": "httpx not installed"} - - messages = [] - if system: - messages.append({"role": "system", "content": system}) - messages.append({"role": "user", "content": question}) - - payload = { - "model": GEMMA_MODEL, - "messages": messages, - "max_tokens": max_tokens, - "temperature": temperature, - } - if not enable_thinking: - payload["chat_template_kwargs"] = {"enable_thinking": False} - - try: - with httpx.Client(timeout=120) as client: - resp = client.post( - GEMMA_URL, - json=payload, - headers={ - "Content-Type": "application/json", - "Authorization": "Bearer none", - }, - ) - resp.raise_for_status() - data = resp.json() - - msg = data["choices"][0]["message"] - content = msg.get("content", "") - reasoning = msg.get("reasoning_content", "") - usage = data.get("usage", {}) - timings = data.get("timings", {}) - - return { - "content": content, - "reasoning": reasoning, - "tokens_in": usage.get("prompt_tokens", 0), - "tokens_out": usage.get("completion_tokens", 0), - "prompt_speed": round(timings.get("prompt_per_second", 0), 1), - "generation_speed": round(timings.get("predicted_per_second", 0), 1), - } - except Exception as e: - return {"error": str(e)} - - -def mcp_server(): - for line in sys.stdin: - try: - req = json.loads(line.strip()) - except json.JSONDecodeError: - continue - - method = req.get("method", "") - req_id = req.get("id") - params = req.get("params", {}) - - if method == "initialize": - resp = { - "jsonrpc": "2.0", "id": req_id, - "result": { - "protocolVersion": "2024-11-05", - "capabilities": {"tools": {}}, - "serverInfo": {"name": "gemma4", "version": "0.1.0"}, - }, - } - elif method == "tools/list": - resp = { - "jsonrpc": "2.0", "id": req_id, - "result": { - "tools": [{ - "name": "gemma4", - "description": ( - "Ask a question to the local Gemma4-12B model. " - "Free, fast (~40 tok/s), good at math and code. " - "Use this for quick questions, sanity checks, " - "and mathematical reasoning." - ), - "inputSchema": { - "type": "object", - "properties": { - "question": { - "type": "string", - "description": "The question to ask", - }, - "system": { - "type": "string", - "description": "System prompt (optional)", - }, - "max_tokens": { - "type": "number", - "description": "Max response tokens (default: 2000)", - }, - "enable_thinking": { - "type": "boolean", - "description": "Enable reasoning mode (default: true)", - }, - }, - "required": ["question"], - }, - }], - }, - } - elif method == "tools/call": - tool_name = params.get("name", "") - arguments = params.get("arguments", {}) - - if tool_name == "gemma4": - result = call_gemma( - question=arguments.get("question", ""), - system=arguments.get("system", ""), - max_tokens=arguments.get("max_tokens", 2000), - enable_thinking=arguments.get("enable_thinking", True), - ) - else: - result = {"error": f"Unknown tool: {tool_name}"} - - resp = { - "jsonrpc": "2.0", "id": req_id, - "result": { - "content": [{"type": "text", "text": json.dumps(result, indent=2)}], - }, - } - else: - resp = { - "jsonrpc": "2.0", "id": req_id, - "error": {"code": -32601, "message": f"Method not found: {method}"}, - } - - print(json.dumps(resp), flush=True) - - -if __name__ == "__main__": - mcp_server()