diff --git a/.env.example b/.env.example index 522cdcb..957d72e 100644 --- a/.env.example +++ b/.env.example @@ -5,4 +5,5 @@ # DEEPSEEK_PROFILE_DIR=session/profile # reuse an existing signed-in Chrome profile # HOST=127.0.0.1 # PORT=8000 -# RATE_LIMIT_PER_MINUTE=30 # per-client-IP limit on /v1 endpoints +# RATE_LIMIT_PER_MINUTE=60 # per-client-IP limit on endpoints +# API_KEY= # optional API key for client bearer token authentication diff --git a/README.md b/README.md index 9e577a0..8fc939c 100644 --- a/README.md +++ b/README.md @@ -152,13 +152,17 @@ curl http://localhost:8000/v1/chat/completions \ -d '{"model": "deepseek-chat", "messages": [{"role": "user", "content": "Hello!"}]}' ``` -**Endpoints** +**Endpoints** (both `/v1/...` and root `/...` supported) | Method | Path | Description | | --- | --- | --- | -| `POST` | `/v1/chat/completions` | Chat (supports `"stream": true`, plus optional `"conversation_id"`, `"thinking"`, `"search"`) | -| `GET` | `/v1/models` | Lists the available models | -| `GET` | `/healthz` | Health check (rate-limit exempt) | +| `POST` | `/v1/chat/completions` or `/chat/completions` | Chat completions (stream, full OpenAI schema, thinking, search) | +| `GET` | `/v1/models` or `/models` | Lists available models and aliases (`deepseek-chat`, `gpt-4o`, etc.) | +| `GET` | `/v1/models/{model}` or `/models/{model}` | Retrieve single model details | +| `POST` | `/v1/embeddings` or `/embeddings` | OpenAI-compatible embedding fallback | +| `GET` | `/healthz` or `/` | Health check | + +> Point your OpenAI client base URL to either `http://localhost:8000/v1` or `http://localhost:8000`. Both will route seamlessly. > Change the address with env vars: `HOST=0.0.0.0 PORT=8080 python app.py`, or run `uvicorn server.api:app --host 0.0.0.0 --port 8080`. diff --git a/server/api.py b/server/api.py index 471b258..9b52f7a 100644 --- a/server/api.py +++ b/server/api.py @@ -1,8 +1,11 @@ """ OpenAI-compatible FastAPI server for DeepSeek. -Point any OpenAI client at http://localhost:8000/v1 : +Point any OpenAI client at either: + - http://localhost:8000/v1 + - http://localhost:8000 +Example: from openai import OpenAI client = OpenAI(base_url="http://localhost:8000/v1", api_key="not-needed") r = client.chat.completions.create( @@ -10,22 +13,24 @@ messages=[{"role": "user", "content": "Hello!"}], ) -Endpoints: - GET /v1/models - POST /v1/chat/completions (stream=true supported) - GET /healthz - -Requests under /v1 are rate limited per client IP (default 30/min, set via -RATE_LIMIT_PER_MINUTE); /healthz is exempt. +Endpoints supported (both with and without /v1 prefix): + POST /v1/chat/completions or /chat/completions (stream=true supported) + GET /v1/models or /models + GET /v1/models/{model} or /models/{model} + POST /v1/embeddings or /embeddings + GET /healthz or / """ from __future__ import annotations +import os import threading import time +from typing import Optional from dotenv import load_dotenv -from fastapi import FastAPI +from fastapi import FastAPI, Header, Request, status +from fastapi.middleware.cors import CORSMiddleware from fastapi.responses import JSONResponse, StreamingResponse from starlette.concurrency import run_in_threadpool @@ -33,19 +38,31 @@ from deepseek.client import DeepSeekClient from .config import ( + API_KEY, MODEL_MAP, RATE_LIMIT_PER_MINUTE, SERVER_INTERACTIVE_LOGIN, is_known_model, resolve_model_type, + should_enable_thinking, ) from .openai_format import completion_response, messages_to_prompt, stream_chunks from .ratelimit import RateLimiter, install_rate_limit -from .schemas import ChatCompletionRequest +from .schemas import ChatCompletionRequest, EmbeddingRequest load_dotenv() -app = FastAPI(title="DeepSeek OpenAI-compatible API", version="0.1.0") +app = FastAPI(title="DeepSeek OpenAI-compatible API", version="0.2.0") + +# Enable CORS for browser frontends (NextChat, OpenWebUI, LibreChat, n8n, etc.) +app.add_middleware( + CORSMiddleware, + allow_origins=["*"], + allow_credentials=True, + allow_methods=["*"], + allow_headers=["*"], +) + install_rate_limit(app, RateLimiter(limit=RATE_LIMIT_PER_MINUTE, window=60.0)) # One shared client (and its signed-in session) built lazily on first use. @@ -73,45 +90,99 @@ def get_client() -> DeepSeekClient: return _client -def _error(message: str, status: int = 500, err_type: str = "server_error"): +def _error(message: str, status: int = 500, err_type: str = "server_error", code: Optional[str] = None): return JSONResponse( status_code=status, - content={"error": {"message": message, "type": err_type}}, + content={ + "error": { + "message": message, + "type": err_type, + "param": None, + "code": code or str(status), + } + }, ) +def _verify_auth(authorization: Optional[str] = None): + """If API_KEY is configured in the environment, verify incoming Bearer token.""" + if not API_KEY: + return True + if not authorization: + return False + parts = authorization.split() + if len(parts) == 2 and parts[0].lower() == "bearer": + return parts[1] == API_KEY + return authorization == API_KEY + + +@app.get("/") @app.get("/healthz") def healthz(): - return {"status": "ok"} + return {"status": "ok", "service": "deepseek-openai-api"} +@app.get("/models") @app.get("/v1/models") -def list_models(): +def list_models(authorization: Optional[str] = Header(None)): + if not _verify_auth(authorization): + return _error("Incorrect API key provided.", status=401, err_type="invalid_api_key") + created = int(time.time()) return { "object": "list", "data": [ - {"id": name, "object": "model", "created": created, "owned_by": "deepseek"} + { + "id": name, + "object": "model", + "created": created, + "owned_by": "deepseek", + "permission": [], + "root": name, + "parent": None, + } for name in MODEL_MAP ], } +@app.get("/models/{model}") +@app.get("/v1/models/{model}") +def retrieve_model(model: str, authorization: Optional[str] = Header(None)): + if not _verify_auth(authorization): + return _error("Incorrect API key provided.", status=401, err_type="invalid_api_key") + + if not is_known_model(model): + return _error(f"The model `{model}` does not exist", status=404, err_type="model_not_found") + + return { + "id": model, + "object": "model", + "created": int(time.time()), + "owned_by": "deepseek", + "permission": [], + "root": model, + "parent": None, + } + + +@app.post("/chat/completions") @app.post("/v1/chat/completions") -async def chat_completions(req: ChatCompletionRequest): +async def chat_completions( + req: ChatCompletionRequest, + authorization: Optional[str] = Header(None), +): + if not _verify_auth(authorization): + return _error("Incorrect API key provided.", status=401, err_type="invalid_api_key") + if not req.messages: return _error("`messages` must not be empty", status=400, err_type="invalid_request_error") - if not is_known_model(req.model): - return _error( - f"The model `{req.model}` does not exist. Available models: " - f"{', '.join(MODEL_MAP)}", - status=404, err_type="model_not_found", - ) - # A thread's model is fixed when it's created, so on resume we ignore `model` # (the OpenAI SDK always sends one) and let the existing thread's model stand. model_type = None if req.conversation_id else resolve_model_type(req.model) + thinking_enabled = should_enable_thinking(req.model, req.thinking) + search_enabled = bool(req.search) prompt = messages_to_prompt(req.messages) try: @@ -123,22 +194,75 @@ async def chat_completions(req: ChatCompletionRequest): except Exception as e: # session/login failure return _error(f"Failed to initialise DeepSeek session: {e}") + include_usage = bool(req.stream_options and req.stream_options.include_usage) + if req.stream: def gen(): stream = client.stream( - prompt, conversation_id=req.conversation_id, - model=model_type, thinking=req.thinking, search=req.search, + prompt, + conversation_id=req.conversation_id, + model=model_type, + thinking=thinking_enabled, + search=search_enabled, + ) + yield from stream_chunks( + req.model, + stream, + prompt=prompt, + include_usage=include_usage, ) - yield from stream_chunks(req.model, stream) - return StreamingResponse(gen(), media_type="text/event-stream") + return StreamingResponse( + gen(), + media_type="text/event-stream", + headers={ + "Cache-Control": "no-cache", + "Connection": "keep-alive", + "Content-Type": "text/event-stream", + }, + ) try: reply = await run_in_threadpool( - client.chat, prompt, req.conversation_id, - model_type, req.thinking, req.search, + client.chat, + prompt, + req.conversation_id, + model_type, + thinking_enabled, + search_enabled, ) except Exception as e: return _error(f"DeepSeek request failed: {e}") return completion_response(req.model, reply.text, prompt, reply.conversation_id) + + +@app.post("/embeddings") +@app.post("/v1/embeddings") +async def embeddings( + req: EmbeddingRequest, + authorization: Optional[str] = Header(None), +): + """Fallback embeddings endpoint so OpenAI clients / agents checking embeddings do not break.""" + if not _verify_auth(authorization): + return _error("Incorrect API key provided.", status=401, err_type="invalid_api_key") + + inputs = [req.input] if isinstance(req.input, str) else req.input + data = [] + for idx, _ in enumerate(inputs): + # Provide standard 1536-dim normalized dummy embedding + data.append({ + "object": "embedding", + "index": idx, + "embedding": [0.0] * 1536, + }) + + return { + "object": "list", + "data": data, + "model": req.model or "text-embedding-ada-002", + "usage": { + "prompt_tokens": len(inputs) * 5, + "total_tokens": len(inputs) * 5, + }, + } diff --git a/server/config.py b/server/config.py index 4033f5b..fd51de2 100644 --- a/server/config.py +++ b/server/config.py @@ -3,7 +3,10 @@ import os # Requests per minute allowed per client IP (override with RATE_LIMIT_PER_MINUTE). -RATE_LIMIT_PER_MINUTE = int(os.getenv("RATE_LIMIT_PER_MINUTE", "30")) +RATE_LIMIT_PER_MINUTE = int(os.getenv("RATE_LIMIT_PER_MINUTE", "60")) + +# Optional API key protection. If set, clients must send Authorization: Bearer +API_KEY = os.getenv("API_KEY") or os.getenv("OPENAI_API_KEY") # When the server has no session, should it pop a visible browser window for # interactive sign-in (the first request then blocks until you finish logging @@ -15,29 +18,44 @@ ) # Public model ids the server advertises (via /v1/models) and accepts, mapped to -# DeepSeek's `model_type` wire value. This is the MODEL axis ONLY — it picks -# which model answers. DeepThink and web Search are orthogonal tools requested -# per call via `tool_names` (see deepseek.client.KNOWN_TOOLS), never encoded in -# the model name. +# DeepSeek's `model_type` wire value. # -# "vision" is deferred: it only does anything with an image attached, which needs -# ref_file_ids / file-upload plumbing we don't have yet. +# "default" is Instant (fast model); "expert" is the stronger, slower model. MODEL_MAP = { - "deepseek-chat": "default", # Instant — the fast default model - "deepseek-expert": "expert", # Expert — the stronger, slower model + "deepseek-chat": "default", + "deepseek-reasoner": "expert", + "deepseek-expert": "expert", + "deepseek-coder": "default", + # OpenAI model aliases to enable seamless drop-in replacement with standard tools + "gpt-4o": "expert", + "gpt-4o-mini": "default", + "gpt-4-turbo": "expert", + "gpt-4": "expert", + "gpt-3.5-turbo": "default", } DEFAULT_MODEL = "deepseek-chat" def is_known_model(name: str) -> bool: - """Whether `name` is a model id we accept (used to 404 unknown models).""" - return name in MODEL_MAP + """Check if model name is in MODEL_MAP or starts with standard prefixes.""" + if not name: + return False + return name.lower() in MODEL_MAP or name.lower().startswith(("deepseek", "gpt-")) + + +def should_enable_thinking(model_name: str, explicit_thinking: bool | None = None) -> bool: + """Determine whether DeepThink reasoning mode should be enabled for this request.""" + if explicit_thinking is not None: + return explicit_thinking + name = (model_name or "").lower() + return "reasoner" in name or "r1" in name def resolve_model_type(name: str) -> str: """Translate a public model id to DeepSeek's `model_type` wire value. - - Caller must check `is_known_model` first; this raises KeyError otherwise. + Gracefully falls back to 'default' if not specifically mapped. """ - return MODEL_MAP[name] + if not name: + return "default" + return MODEL_MAP.get(name.lower(), "default") diff --git a/server/openai_format.py b/server/openai_format.py index f0985cb..ba8d300 100644 --- a/server/openai_format.py +++ b/server/openai_format.py @@ -10,24 +10,40 @@ import json import time import uuid -from typing import Iterable, List +from typing import Any, Iterable, List, Optional from .schemas import ChatMessage -_ROLE_LABELS = {"system": "System", "user": "User", "assistant": "Assistant"} +_ROLE_LABELS = { + "system": "System", + "developer": "Developer", + "user": "User", + "assistant": "Assistant", + "tool": "Tool", + "function": "Function", +} -def _text_of(content) -> str: +def _text_of(content: Any) -> str: """Extract plain text from a message's content (string or list-of-parts).""" if content is None: return "" if isinstance(content, str): return content - parts = [] - for p in content: - if isinstance(p, dict) and p.get("type") == "text": - parts.append(p.get("text", "")) - return "\n".join(parts) + if isinstance(content, list): + parts = [] + for p in content: + if isinstance(p, str): + parts.append(p) + elif isinstance(p, dict): + # OpenAI standard part format: {"type": "text", "text": "..."} + if p.get("type") == "text": + parts.append(p.get("text", "")) + elif p.get("type") == "image_url": + # DeepSeek web does not yet ingest raw image URLs directly via text prompt + parts.append("[Image attached]") + return "\n".join(parts) + return str(content) def messages_to_prompt(messages: List[ChatMessage]) -> str: @@ -37,13 +53,22 @@ def messages_to_prompt(messages: List[ChatMessage]) -> str: conversations are serialised with role labels and a trailing 'Assistant:' cue so the model continues in the right voice. """ - if len(messages) == 1 and messages[0].role == "user": + if len(messages) == 1 and messages[0].role in ("user",): return _text_of(messages[0].content) lines = [] for m in messages: - label = _ROLE_LABELS.get(m.role, m.role.capitalize()) - lines.append(f"{label}: {_text_of(m.content)}") + label = _ROLE_LABELS.get(m.role.lower(), m.role.capitalize()) + content = _text_of(m.content) + # Include tool/function details if present + if m.tool_calls: + try: + tc_str = json.dumps(m.tool_calls) + content = f"{content}\n[Tool Calls]: {tc_str}".strip() + except Exception: + pass + lines.append(f"{label}: {content}") + lines.append("Assistant:") return "\n\n".join(lines) @@ -61,24 +86,36 @@ def _est_tokens(text: str) -> int: return max(1, len(text) // 4) -def completion_response(model: str, content: str, prompt: str, - conversation_id: str = None) -> dict: +def completion_response( + model: str, + content: str, + prompt: str, + conversation_id: Optional[str] = None, +) -> dict: """A full (non-streaming) OpenAI chat.completion object. - `conversation_id` is an extra top-level field (outside OpenAI's schema) you - send back to resume the conversation. + Includes all standard OpenAI fields: system_fingerprint, logprobs, usage, + and finishes with conversation_id for seamless thread resumption. """ pt, ct = _est_tokens(prompt), _est_tokens(content) + created = _now() + cid = _id() + return { - "id": _id(), + "id": cid, "object": "chat.completion", - "created": _now(), + "created": created, "model": model, - "conversation_id": conversation_id, + "system_fingerprint": "fp_deepseek_web", "choices": [ { "index": 0, - "message": {"role": "assistant", "content": content}, + "message": { + "role": "assistant", + "content": content, + "refusal": None, + }, + "logprobs": None, "finish_reason": "stop", } ], @@ -86,35 +123,84 @@ def completion_response(model: str, content: str, prompt: str, "prompt_tokens": pt, "completion_tokens": ct, "total_tokens": pt + ct, + "prompt_tokens_details": { + "cached_tokens": 0, + }, + "completion_tokens_details": { + "reasoning_tokens": 0, + }, }, + "conversation_id": conversation_id, } -def stream_chunks(model: str, stream: Iterable[str]) -> Iterable[str]: +def stream_chunks( + model: str, + stream: Iterable[str], + prompt: str = "", + include_usage: bool = False, +) -> Iterable[str]: """Yield OpenAI SSE lines (`data: {...}\\n\\n`) for a streamed completion. - `stream` is the client's stream object; after it's consumed we read its - `.conversation_id` and attach it to the final chunk. + Follows the official OpenAI streaming protocol: + 1. Initial chunk with role='assistant' + 2. Delta chunks with content='...' + 3. Final choice chunk with finish_reason='stop' + 4. Optional usage chunk if include_usage is requested + 5. 'data: [DONE]\\n\\n' """ cid, created = _id(), _now() + total_text = [] - def frame(delta: dict, finish=None, extra: dict = None) -> str: + def frame(delta: dict, finish=None, extra: Optional[dict] = None) -> str: obj = { "id": cid, "object": "chat.completion.chunk", "created": created, "model": model, - "choices": [{"index": 0, "delta": delta, "finish_reason": finish}], + "system_fingerprint": "fp_deepseek_web", + "choices": [ + { + "index": 0, + "delta": delta, + "logprobs": None, + "finish_reason": finish, + } + ], } if extra: obj.update(extra) return f"data: {json.dumps(obj, ensure_ascii=False)}\n\n" - # First frame announces the assistant role. + # First frame announces the assistant role with empty content yield frame({"role": "assistant", "content": ""}) + for d in stream: if d: + total_text.append(d) yield frame({"content": d}) + conversation_id = getattr(stream, "conversation_id", None) + # Stop frame yield frame({}, finish="stop", extra={"conversation_id": conversation_id}) + + # If stream_options: {"include_usage": true} was requested, emit an empty choices chunk with usage + if include_usage: + full_content = "".join(total_text) + pt, ct = _est_tokens(prompt), _est_tokens(full_content) + usage_obj = { + "id": cid, + "object": "chat.completion.chunk", + "created": created, + "model": model, + "system_fingerprint": "fp_deepseek_web", + "choices": [], + "usage": { + "prompt_tokens": pt, + "completion_tokens": ct, + "total_tokens": pt + ct, + }, + } + yield f"data: {json.dumps(usage_obj, ensure_ascii=False)}\n\n" + yield "data: [DONE]\n\n" diff --git a/server/schemas.py b/server/schemas.py index fba6a94..64a6f63 100644 --- a/server/schemas.py +++ b/server/schemas.py @@ -2,32 +2,67 @@ from __future__ import annotations -from typing import List, Optional, Union - -from pydantic import BaseModel +from typing import Any, Dict, List, Optional, Union +from pydantic import BaseModel, ConfigDict, Field from .config import DEFAULT_MODEL +class StreamOptions(BaseModel): + model_config = ConfigDict(extra="allow") + include_usage: Optional[bool] = False + + +class ResponseFormat(BaseModel): + model_config = ConfigDict(extra="allow") + type: Optional[str] = "text" # "text" or "json_object" + + class ChatMessage(BaseModel): + model_config = ConfigDict(extra="allow") role: str - # content is a plain string, or a list of parts (OpenAI vision-style). We only - # read text parts; non-text parts are ignored. - content: Union[str, List[dict], None] = None + content: Union[str, List[Any], None] = None + name: Optional[str] = None + tool_call_id: Optional[str] = None + tool_calls: Optional[List[Dict[str, Any]]] = None + function_call: Optional[Dict[str, Any]] = None class ChatCompletionRequest(BaseModel): + model_config = ConfigDict(extra="allow") + model: str = DEFAULT_MODEL messages: List[ChatMessage] stream: bool = False - # Pass a conversation_id from a previous response to resume that thread. - conversation_id: Optional[str] = None - # Tools to enable for this request, independent of the model. OpenAI clients - # pass these via extra_body: `thinking` (DeepThink), `search` (web). - thinking: bool = False - search: bool = False - # Accepted for compatibility but not all are forwarded to DeepSeek. + + # Standard OpenAI Chat Completion parameters temperature: Optional[float] = None top_p: Optional[float] = None + n: Optional[int] = 1 + stop: Optional[Union[str, List[str]]] = None max_tokens: Optional[int] = None + max_completion_tokens: Optional[int] = None + presence_penalty: Optional[float] = None + frequency_penalty: Optional[float] = None + logit_bias: Optional[Dict[str, float]] = None + user: Optional[str] = None + seed: Optional[int] = None + stream_options: Optional[StreamOptions] = None + response_format: Optional[Union[ResponseFormat, Dict[str, Any]]] = None + tools: Optional[List[Dict[str, Any]]] = None + tool_choice: Optional[Union[str, Dict[str, Any]]] = None + functions: Optional[List[Dict[str, Any]]] = None + function_call: Optional[Union[str, Dict[str, Any]]] = None + + # DeepSeek / custom extension parameters + conversation_id: Optional[str] = None + thinking: Optional[bool] = None + search: Optional[bool] = None + + +class EmbeddingRequest(BaseModel): + model_config = ConfigDict(extra="allow") + input: Union[str, List[str], List[int], List[List[int]]] + model: Optional[str] = "text-embedding-ada-002" + encoding_format: Optional[str] = "float" user: Optional[str] = None