#!/usr/bin/env python3 """MCP stdio server: OCR images posted into opencode via a local Unlimited-OCR backend. Transport: newline-delimited JSON-RPC over stdin/stdout (MCP stdio). Stdlib only. Env overrides: OPENCODE_DB, OCR_URL. """ import base64 import json import os import sqlite3 import sys import urllib.request DB = os.environ.get( 'OPENCODE_DB', os.path.expanduser('~/.local/share/opencode/opencode.db')) OCR_URL = os.environ.get('OCR_URL', 'http://127.0.0.1:11436/v1/chat/completions') SERVER_INFO = {'name': 'ocr-posted', 'version': '1.0.0'} TOOLS = [ { 'name': 'ocr_image', 'description': ( 'OCR an image the user posted into opencode (or any local image file) using the ' 'local Unlimited-OCR VLM server. Use this when the user attaches or pastes an image ' 'that you cannot see directly, e.g. when the message shows ' "'Cannot read (this model does not support image input)'. " 'Returns transcribed text (layout-prefixed lines with prompts that ask for grounding).' ), 'inputSchema': { 'type': 'object', 'properties': { 'target': { 'type': 'string', 'description': ( "'latest' (default): most recent image part in opencode.db. " "Or a part id (prt_...), 'session:', or a local image file path." ), }, 'prompt': { 'type': 'string', 'description': ( "Instruction sent to the OCR model. Default 'Free OCR.' (plain transcription). " "Use '<|grounding|>Convert the document to markdown.' for structured layout, " "'Describe this image in detail.' for visual QA." ), }, 'save': { 'type': 'string', 'description': 'Optional local path to save the decoded image to.', }, }, 'required': [], }, } ] def resolve_data_url(target): """Return (data_url, meta) for the requested target.""" if target and target != 'latest' and not target.startswith('session:'): if os.path.isfile(target): ext = os.path.splitext(target)[1].lower() mime = {'.png': 'image/png', '.jpg': 'image/jpeg', '.jpeg': 'image/jpeg', '.gif': 'image/gif', '.webp': 'image/webp'}.get(ext, 'image/png') b64 = base64.b64encode(open(target, 'rb').read()).decode() return 'data:%s;base64,%s' % (mime, b64), {'source': 'file', 'filename': target} db = sqlite3.connect(DB, timeout=10) row = db.execute('SELECT id, data FROM part WHERE id=?', (target,)).fetchone() if not row: raise ValueError('part id not found: %s' % target) pid, data = row else: db = sqlite3.connect(DB, timeout=10) if target and target.startswith('session:'): row = db.execute( "SELECT id, data FROM part WHERE session_id=? " "AND data LIKE '%\"type\":\"file\"%' AND data LIKE '%data:image%' " "ORDER BY time_created DESC LIMIT 1", (target[8:],)).fetchone() else: row = db.execute( "SELECT id, data FROM part WHERE data LIKE '%\"type\":\"file\"%' " "AND data LIKE '%data:image%' ORDER BY time_created DESC LIMIT 1").fetchone() if not row: scope = ' in session %s' % target[8:] if target and target.startswith('session:') else '' raise ValueError('no image part found%s' % scope) pid, data = row d = json.loads(data) url = d.get('url', '') if not url.startswith('data:image'): raise ValueError('part %s has no inline image data' % pid) return url, {'source': 'opencode.db', 'part': pid, 'filename': d.get('filename')} def run_ocr(data_url, prompt): body = json.dumps({ 'model': 'Unlimited-OCR', 'temperature': 0, 'repeat_penalty': 1.2, 'max_tokens': 4096, 'messages': [{'role': 'user', 'content': [ {'type': 'image_url', 'image_url': {'url': data_url}}, {'type': 'text', 'text': prompt}, ]}], }).encode() req = urllib.request.Request(OCR_URL, data=body, headers={'Content-Type': 'application/json'}) with urllib.request.urlopen(req, timeout=1800) as r: out = json.load(r) return out['choices'][0]['message']['content'] def tool_ocr_image(args): target = args.get('target', 'latest') prompt = args.get('prompt', 'Free OCR.') save = args.get('save') data_url, meta = resolve_data_url(target) if save: open(save, 'wb').write(base64.b64decode(data_url.split(',', 1)[1])) meta['saved_to'] = save text = run_ocr(data_url, prompt) header = 'source=%s part=%s file=%s' % (meta.get('source'), meta.get('part', '-'), meta.get('filename', '-')) if meta.get('saved_to'): header += ' saved_to=%s' % meta['saved_to'] return header + '\n' + text def dispatch(msg): method = msg.get('method') mid = msg.get('id') if method is None: return None if method == 'initialize': params = msg.get('params', {}) return { 'protocolVersion': params.get('protocolVersion', '2025-06-18'), 'capabilities': {'tools': {}}, 'serverInfo': SERVER_INFO, } if method == 'notifications/initialized': return None if method == 'ping': return {} if method == 'tools/list': return {'tools': TOOLS} if method == 'tools/call': params = msg.get('params', {}) name = params.get('name') args = params.get('arguments', {}) or {} if name != 'ocr_image': return None, {'code': -32602, 'message': 'unknown tool: %s' % name} try: text = tool_ocr_image(args) return {'content': [{'type': 'text', 'text': text}], 'isError': False}, None except Exception as e: return {'content': [{'type': 'text', 'text': 'ocr_image failed: %s: %s' % (type(e).__name__, e)}], 'isError': True}, None if method in ('resources/list', 'prompts/list'): return {'resources': []} if method == 'resources/list' else {'prompts': []} if mid is None: return None return None, {'code': -32601, 'message': 'method not supported: %s' % method} def main(): for line in sys.stdin: line = line.strip() if not line: continue try: msg = json.loads(line) except json.JSONDecodeError: continue result = dispatch(msg) if result is None: continue if isinstance(result, tuple): payload, error = result if error is not None: out = {'jsonrpc': '2.0', 'id': msg.get('id'), 'error': error} else: out = {'jsonrpc': '2.0', 'id': msg.get('id'), 'result': payload} else: out = {'jsonrpc': '2.0', 'id': msg.get('id'), 'result': result} sys.stdout.write(json.dumps(out) + '\n') sys.stdout.flush() if __name__ == '__main__': main()