diff --git a/scripts/ollama-proxy.py b/scripts/ollama-proxy.py index bbca64b..7b587c5 100644 --- a/scripts/ollama-proxy.py +++ b/scripts/ollama-proxy.py @@ -312,12 +312,12 @@ def _model_entry(tag): "details": { "format": "gguf", "family": "qwen35", - "families": ["qwen35", "qwen2"], + "families": ["qwen35", "qwen2", "clip"], "parameter_size": "27.0B", "quantization_level": "Q4_K_P", "context_length": CTX_SIZE, }, - "capabilities": ["completion", "chat", "tools", "thinking"], + "capabilities": ["completion", "chat", "tools", "thinking", "vision", "multimodal"], } def ollama_tags(): @@ -334,7 +334,7 @@ def ollama_ps(): "details": { "format": "gguf", "family": "qwen35", - "families": ["qwen35", "qwen2"], + "families": ["qwen35", "qwen2", "clip"], "parameter_size": "27.0B", "quantization_level": "Q4_K_P", "context_length": CTX_SIZE, @@ -358,7 +358,7 @@ def ollama_show(payload=None): "details": { "format": "gguf", "family": "qwen35", - "families": ["qwen35", "qwen2"], + "families": ["qwen35", "qwen2", "clip"], "parameter_size": "27.0B", "quantization_level": "Q4_K_P", "context_length": CTX_SIZE, @@ -372,8 +372,14 @@ def ollama_show(payload=None): "qwen35.attention.head_count": 32, "qwen35.attention.head_count_kv": 4, "qwen35.max_output_tokens": MAX_OUTPUT, + "clip.has_vision_encoder": True, + "clip.has_text_encoder": True, }, - "capabilities": ["completion", "chat", "tools", "thinking"], + "projector_info": { + "general.architecture": "clip", + "clip.projector_type": "mlp", + }, + "capabilities": ["completion", "chat", "tools", "thinking", "vision", "multimodal"], } def ollama_status(): @@ -869,26 +875,56 @@ def ollama_pull(payload): # Anthropic Messages API # ---------------------------------------------------------------------------- def _anthropic_content_to_openai(content, is_last=False): - """Convert Anthropic message content (string or block array) to OpenAI string.""" + """Convert Anthropic message content (string or block array) to OpenAI content (string or list).""" if isinstance(content, str): return content - parts = [] + if not isinstance(content, list): + return str(content) + + has_images = any(isinstance(b, dict) and b.get("type") == "image" for b in content) + + if not has_images: + parts = [] + for block in content: + if not isinstance(block, dict): + parts.append(str(block)) + continue + btype = block.get("type") + if btype == "text": + parts.append(block.get("text", "")) + elif btype == "thinking": + if EMIT_THINKING: + parts.append(block.get("thinking", "")) + elif btype == "tool_result": + tc = block.get("content", "") + if isinstance(tc, list): + tc = "".join(b.get("text", "") for b in tc if isinstance(b, dict) and b.get("type") == "text") + parts.append(f"[tool_result: {tc}]") + return "\n".join(parts) + + # Multi-part content with vision images + out_blocks = [] for block in content: + if not isinstance(block, dict): + continue btype = block.get("type") if btype == "text": - parts.append(block.get("text", "")) + out_blocks.append({"type": "text", "text": block.get("text", "")}) elif btype == "thinking": - # keep reasoning in context if EMIT_THINKING: - parts.append(block.get("thinking", "")) + out_blocks.append({"type": "text", "text": block.get("thinking", "")}) elif btype == "tool_result": tc = block.get("content", "") if isinstance(tc, list): - tc = "".join(b.get("text", "") for b in tc if b.get("type") == "text") - parts.append(f"[tool_result: {tc}]") + tc = "".join(b.get("text", "") for b in tc if isinstance(b, dict) and b.get("type") == "text") + out_blocks.append({"type": "text", "text": f"[tool_result: {tc}]"}) elif btype == "image": - parts.append("[image]") - return "\n".join(parts) + source = block.get("source", {}) + if isinstance(source, dict): + data = source.get("data", "") + mtype = source.get("media_type", "image/jpeg") + out_blocks.append({"type": "image_url", "image_url": {"url": f"data:{mtype};base64,{data}"}}) + return out_blocks def _anthropic_tools_to_openai(tools): """Anthropic tools ({name,input_schema}) -> OpenAI tools."""