Jarian Cottingham c0fa9148b7 Initial commit: Unlimited-OCR MCP for opencode
- mcp/server.py: stdlib-only MCP stdio server exposing ocr_image tool
- ocr-posted.py: CLI equivalent (DB lookup + OCR via llama-server)
- run-ocr-server.sh: llama-server launcher with tuned sampling flags
- skill/ocr-image: opencode skill teaching agents when/how to call it
- README with full install instructions
2026-08-22 06:24:57 +00:00

191 lines
7.3 KiB
Python
Executable File

#!/usr/bin/env python3
"""MCP stdio server: OCR images posted into opencode via a local Unlimited-OCR backend.
Transport: newline-delimited JSON-RPC over stdin/stdout (MCP stdio).
Stdlib only. Env overrides: OPENCODE_DB, OCR_URL.
"""
import base64
import json
import os
import sqlite3
import sys
import urllib.request
DB = os.environ.get(
'OPENCODE_DB',
os.path.expanduser('~/.local/share/opencode/opencode.db'))
OCR_URL = os.environ.get('OCR_URL', 'http://127.0.0.1:11436/v1/chat/completions')
SERVER_INFO = {'name': 'ocr-posted', 'version': '1.0.0'}
TOOLS = [
{
'name': 'ocr_image',
'description': (
'OCR an image the user posted into opencode (or any local image file) using the '
'local Unlimited-OCR VLM server. Use this when the user attaches or pastes an image '
'that you cannot see directly, e.g. when the message shows '
"'Cannot read <file> (this model does not support image input)'. "
'Returns transcribed text (layout-prefixed lines with prompts that ask for grounding).'
),
'inputSchema': {
'type': 'object',
'properties': {
'target': {
'type': 'string',
'description': (
"'latest' (default): most recent image part in opencode.db. "
"Or a part id (prt_...), 'session:<session_id>', or a local image file path."
),
},
'prompt': {
'type': 'string',
'description': (
"Instruction sent to the OCR model. Default 'Free OCR.' (plain transcription). "
"Use '<|grounding|>Convert the document to markdown.' for structured layout, "
"'Describe this image in detail.' for visual QA."
),
},
'save': {
'type': 'string',
'description': 'Optional local path to save the decoded image to.',
},
},
'required': [],
},
}
]
def resolve_data_url(target):
"""Return (data_url, meta) for the requested target."""
if target and target != 'latest' and not target.startswith('session:'):
if os.path.isfile(target):
ext = os.path.splitext(target)[1].lower()
mime = {'.png': 'image/png', '.jpg': 'image/jpeg', '.jpeg': 'image/jpeg',
'.gif': 'image/gif', '.webp': 'image/webp'}.get(ext, 'image/png')
b64 = base64.b64encode(open(target, 'rb').read()).decode()
return 'data:%s;base64,%s' % (mime, b64), {'source': 'file', 'filename': target}
db = sqlite3.connect(DB, timeout=10)
row = db.execute('SELECT id, data FROM part WHERE id=?', (target,)).fetchone()
if not row:
raise ValueError('part id not found: %s' % target)
pid, data = row
else:
db = sqlite3.connect(DB, timeout=10)
if target and target.startswith('session:'):
row = db.execute(
"SELECT id, data FROM part WHERE session_id=? "
"AND data LIKE '%\"type\":\"file\"%' AND data LIKE '%data:image%' "
"ORDER BY time_created DESC LIMIT 1", (target[8:],)).fetchone()
else:
row = db.execute(
"SELECT id, data FROM part WHERE data LIKE '%\"type\":\"file\"%' "
"AND data LIKE '%data:image%' ORDER BY time_created DESC LIMIT 1").fetchone()
if not row:
scope = ' in session %s' % target[8:] if target and target.startswith('session:') else ''
raise ValueError('no image part found%s' % scope)
pid, data = row
d = json.loads(data)
url = d.get('url', '')
if not url.startswith('data:image'):
raise ValueError('part %s has no inline image data' % pid)
return url, {'source': 'opencode.db', 'part': pid, 'filename': d.get('filename')}
def run_ocr(data_url, prompt):
body = json.dumps({
'model': 'Unlimited-OCR',
'temperature': 0,
'repeat_penalty': 1.2,
'max_tokens': 4096,
'messages': [{'role': 'user', 'content': [
{'type': 'image_url', 'image_url': {'url': data_url}},
{'type': 'text', 'text': prompt},
]}],
}).encode()
req = urllib.request.Request(OCR_URL, data=body,
headers={'Content-Type': 'application/json'})
with urllib.request.urlopen(req, timeout=1800) as r:
out = json.load(r)
return out['choices'][0]['message']['content']
def tool_ocr_image(args):
target = args.get('target', 'latest')
prompt = args.get('prompt', 'Free OCR.')
save = args.get('save')
data_url, meta = resolve_data_url(target)
if save:
open(save, 'wb').write(base64.b64decode(data_url.split(',', 1)[1]))
meta['saved_to'] = save
text = run_ocr(data_url, prompt)
header = 'source=%s part=%s file=%s' % (meta.get('source'), meta.get('part', '-'), meta.get('filename', '-'))
if meta.get('saved_to'):
header += ' saved_to=%s' % meta['saved_to']
return header + '\n' + text
def dispatch(msg):
method = msg.get('method')
mid = msg.get('id')
if method is None:
return None
if method == 'initialize':
params = msg.get('params', {})
return {
'protocolVersion': params.get('protocolVersion', '2025-06-18'),
'capabilities': {'tools': {}},
'serverInfo': SERVER_INFO,
}
if method == 'notifications/initialized':
return None
if method == 'ping':
return {}
if method == 'tools/list':
return {'tools': TOOLS}
if method == 'tools/call':
params = msg.get('params', {})
name = params.get('name')
args = params.get('arguments', {}) or {}
if name != 'ocr_image':
return None, {'code': -32602, 'message': 'unknown tool: %s' % name}
try:
text = tool_ocr_image(args)
return {'content': [{'type': 'text', 'text': text}], 'isError': False}, None
except Exception as e:
return {'content': [{'type': 'text', 'text': 'ocr_image failed: %s: %s' % (type(e).__name__, e)}],
'isError': True}, None
if method in ('resources/list', 'prompts/list'):
return {'resources': []} if method == 'resources/list' else {'prompts': []}
if mid is None:
return None
return None, {'code': -32601, 'message': 'method not supported: %s' % method}
def main():
for line in sys.stdin:
line = line.strip()
if not line:
continue
try:
msg = json.loads(line)
except json.JSONDecodeError:
continue
result = dispatch(msg)
if result is None:
continue
if isinstance(result, tuple):
payload, error = result
if error is not None:
out = {'jsonrpc': '2.0', 'id': msg.get('id'), 'error': error}
else:
out = {'jsonrpc': '2.0', 'id': msg.get('id'), 'result': payload}
else:
out = {'jsonrpc': '2.0', 'id': msg.get('id'), 'result': result}
sys.stdout.write(json.dumps(out) + '\n')
sys.stdout.flush()
if __name__ == '__main__':
main()