Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
3 changes: 2 additions & 1 deletion .env.example
Original file line number Diff line number Diff line change
Expand Up @@ -5,4 +5,5 @@
# DEEPSEEK_PROFILE_DIR=session/profile # reuse an existing signed-in Chrome profile
# HOST=127.0.0.1
# PORT=8000
# RATE_LIMIT_PER_MINUTE=30 # per-client-IP limit on /v1 endpoints
# RATE_LIMIT_PER_MINUTE=60 # per-client-IP limit on endpoints
# API_KEY= # optional API key for client bearer token authentication
12 changes: 8 additions & 4 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -152,13 +152,17 @@ curl http://localhost:8000/v1/chat/completions \
-d '{"model": "deepseek-chat", "messages": [{"role": "user", "content": "Hello!"}]}'
```

**Endpoints**
**Endpoints** (both `/v1/...` and root `/...` supported)

| Method | Path | Description |
| --- | --- | --- |
| `POST` | `/v1/chat/completions` | Chat (supports `"stream": true`, plus optional `"conversation_id"`, `"thinking"`, `"search"`) |
| `GET` | `/v1/models` | Lists the available models |
| `GET` | `/healthz` | Health check (rate-limit exempt) |
| `POST` | `/v1/chat/completions` or `/chat/completions` | Chat completions (stream, full OpenAI schema, thinking, search) |
| `GET` | `/v1/models` or `/models` | Lists available models and aliases (`deepseek-chat`, `gpt-4o`, etc.) |
| `GET` | `/v1/models/{model}` or `/models/{model}` | Retrieve single model details |
| `POST` | `/v1/embeddings` or `/embeddings` | OpenAI-compatible embedding fallback |
| `GET` | `/healthz` or `/` | Health check |

> Point your OpenAI client base URL to either `http://localhost:8000/v1` or `http://localhost:8000`. Both will route seamlessly.

> Change the address with env vars: `HOST=0.0.0.0 PORT=8080 python app.py`, or run `uvicorn server.api:app --host 0.0.0.0 --port 8080`.

Expand Down
184 changes: 154 additions & 30 deletions server/api.py
Original file line number Diff line number Diff line change
@@ -1,51 +1,68 @@
"""
OpenAI-compatible FastAPI server for DeepSeek.

Point any OpenAI client at http://localhost:8000/v1 :
Point any OpenAI client at either:
- http://localhost:8000/v1
- http://localhost:8000

Example:
from openai import OpenAI
client = OpenAI(base_url="http://localhost:8000/v1", api_key="not-needed")
r = client.chat.completions.create(
model="deepseek-chat",
messages=[{"role": "user", "content": "Hello!"}],
)

Endpoints:
GET /v1/models
POST /v1/chat/completions (stream=true supported)
GET /healthz

Requests under /v1 are rate limited per client IP (default 30/min, set via
RATE_LIMIT_PER_MINUTE); /healthz is exempt.
Endpoints supported (both with and without /v1 prefix):
POST /v1/chat/completions or /chat/completions (stream=true supported)
GET /v1/models or /models
GET /v1/models/{model} or /models/{model}
POST /v1/embeddings or /embeddings
GET /healthz or /
"""

from __future__ import annotations

import os
import threading
import time
from typing import Optional

from dotenv import load_dotenv
from fastapi import FastAPI
from fastapi import FastAPI, Header, Request, status
from fastapi.middleware.cors import CORSMiddleware
from fastapi.responses import JSONResponse, StreamingResponse
from starlette.concurrency import run_in_threadpool

from deepseek.auth import LoginRequired
from deepseek.client import DeepSeekClient

from .config import (
API_KEY,
MODEL_MAP,
RATE_LIMIT_PER_MINUTE,
SERVER_INTERACTIVE_LOGIN,
is_known_model,
resolve_model_type,
should_enable_thinking,
)
from .openai_format import completion_response, messages_to_prompt, stream_chunks
from .ratelimit import RateLimiter, install_rate_limit
from .schemas import ChatCompletionRequest
from .schemas import ChatCompletionRequest, EmbeddingRequest

load_dotenv()

app = FastAPI(title="DeepSeek OpenAI-compatible API", version="0.1.0")
app = FastAPI(title="DeepSeek OpenAI-compatible API", version="0.2.0")

# Enable CORS for browser frontends (NextChat, OpenWebUI, LibreChat, n8n, etc.)
app.add_middleware(
CORSMiddleware,
allow_origins=["*"],
allow_credentials=True,
allow_methods=["*"],
allow_headers=["*"],
)

install_rate_limit(app, RateLimiter(limit=RATE_LIMIT_PER_MINUTE, window=60.0))

# One shared client (and its signed-in session) built lazily on first use.
Expand Down Expand Up @@ -73,45 +90,99 @@ def get_client() -> DeepSeekClient:
return _client


def _error(message: str, status: int = 500, err_type: str = "server_error"):
def _error(message: str, status: int = 500, err_type: str = "server_error", code: Optional[str] = None):
return JSONResponse(
status_code=status,
content={"error": {"message": message, "type": err_type}},
content={
"error": {
"message": message,
"type": err_type,
"param": None,
"code": code or str(status),
}
},
)


def _verify_auth(authorization: Optional[str] = None):
"""If API_KEY is configured in the environment, verify incoming Bearer token."""
if not API_KEY:
return True
if not authorization:
return False
parts = authorization.split()
if len(parts) == 2 and parts[0].lower() == "bearer":
return parts[1] == API_KEY
return authorization == API_KEY


@app.get("/")
@app.get("/healthz")
def healthz():
return {"status": "ok"}
return {"status": "ok", "service": "deepseek-openai-api"}


@app.get("/models")
@app.get("/v1/models")
def list_models():
def list_models(authorization: Optional[str] = Header(None)):
if not _verify_auth(authorization):
return _error("Incorrect API key provided.", status=401, err_type="invalid_api_key")

created = int(time.time())
return {
"object": "list",
"data": [
{"id": name, "object": "model", "created": created, "owned_by": "deepseek"}
{
"id": name,
"object": "model",
"created": created,
"owned_by": "deepseek",
"permission": [],
"root": name,
"parent": None,
}
for name in MODEL_MAP
],
}


@app.get("/models/{model}")
@app.get("/v1/models/{model}")
def retrieve_model(model: str, authorization: Optional[str] = Header(None)):
if not _verify_auth(authorization):
return _error("Incorrect API key provided.", status=401, err_type="invalid_api_key")

if not is_known_model(model):
return _error(f"The model `{model}` does not exist", status=404, err_type="model_not_found")

return {
"id": model,
"object": "model",
"created": int(time.time()),
"owned_by": "deepseek",
"permission": [],
"root": model,
"parent": None,
}


@app.post("/chat/completions")
@app.post("/v1/chat/completions")
async def chat_completions(req: ChatCompletionRequest):
async def chat_completions(
req: ChatCompletionRequest,
authorization: Optional[str] = Header(None),
):
if not _verify_auth(authorization):
return _error("Incorrect API key provided.", status=401, err_type="invalid_api_key")

if not req.messages:
return _error("`messages` must not be empty", status=400, err_type="invalid_request_error")

if not is_known_model(req.model):
return _error(
f"The model `{req.model}` does not exist. Available models: "
f"{', '.join(MODEL_MAP)}",
status=404, err_type="model_not_found",
)

# A thread's model is fixed when it's created, so on resume we ignore `model`
# (the OpenAI SDK always sends one) and let the existing thread's model stand.
model_type = None if req.conversation_id else resolve_model_type(req.model)
thinking_enabled = should_enable_thinking(req.model, req.thinking)
search_enabled = bool(req.search)
prompt = messages_to_prompt(req.messages)

try:
Expand All @@ -123,22 +194,75 @@ async def chat_completions(req: ChatCompletionRequest):
except Exception as e: # session/login failure
return _error(f"Failed to initialise DeepSeek session: {e}")

include_usage = bool(req.stream_options and req.stream_options.include_usage)

if req.stream:
def gen():
stream = client.stream(
prompt, conversation_id=req.conversation_id,
model=model_type, thinking=req.thinking, search=req.search,
prompt,
conversation_id=req.conversation_id,
model=model_type,
thinking=thinking_enabled,
search=search_enabled,
)
yield from stream_chunks(
req.model,
stream,
prompt=prompt,
include_usage=include_usage,
)
yield from stream_chunks(req.model, stream)

return StreamingResponse(gen(), media_type="text/event-stream")
return StreamingResponse(
gen(),
media_type="text/event-stream",
headers={
"Cache-Control": "no-cache",
"Connection": "keep-alive",
"Content-Type": "text/event-stream",
},
)

try:
reply = await run_in_threadpool(
client.chat, prompt, req.conversation_id,
model_type, req.thinking, req.search,
client.chat,
prompt,
req.conversation_id,
model_type,
thinking_enabled,
search_enabled,
)
except Exception as e:
return _error(f"DeepSeek request failed: {e}")

return completion_response(req.model, reply.text, prompt, reply.conversation_id)


@app.post("/embeddings")
@app.post("/v1/embeddings")
async def embeddings(
req: EmbeddingRequest,
authorization: Optional[str] = Header(None),
):
"""Fallback embeddings endpoint so OpenAI clients / agents checking embeddings do not break."""
if not _verify_auth(authorization):
return _error("Incorrect API key provided.", status=401, err_type="invalid_api_key")

inputs = [req.input] if isinstance(req.input, str) else req.input
data = []
for idx, _ in enumerate(inputs):
# Provide standard 1536-dim normalized dummy embedding
data.append({
"object": "embedding",
"index": idx,
"embedding": [0.0] * 1536,
})

return {
"object": "list",
"data": data,
"model": req.model or "text-embedding-ada-002",
"usage": {
"prompt_tokens": len(inputs) * 5,
"total_tokens": len(inputs) * 5,
},
}
46 changes: 32 additions & 14 deletions server/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,10 @@
import os

# Requests per minute allowed per client IP (override with RATE_LIMIT_PER_MINUTE).
RATE_LIMIT_PER_MINUTE = int(os.getenv("RATE_LIMIT_PER_MINUTE", "30"))
RATE_LIMIT_PER_MINUTE = int(os.getenv("RATE_LIMIT_PER_MINUTE", "60"))

# Optional API key protection. If set, clients must send Authorization: Bearer <key>
API_KEY = os.getenv("API_KEY") or os.getenv("OPENAI_API_KEY")

# When the server has no session, should it pop a visible browser window for
# interactive sign-in (the first request then blocks until you finish logging
Expand All @@ -15,29 +18,44 @@
)

# Public model ids the server advertises (via /v1/models) and accepts, mapped to
# DeepSeek's `model_type` wire value. This is the MODEL axis ONLY — it picks
# which model answers. DeepThink and web Search are orthogonal tools requested
# per call via `tool_names` (see deepseek.client.KNOWN_TOOLS), never encoded in
# the model name.
# DeepSeek's `model_type` wire value.
#
# "vision" is deferred: it only does anything with an image attached, which needs
# ref_file_ids / file-upload plumbing we don't have yet.
# "default" is Instant (fast model); "expert" is the stronger, slower model.
MODEL_MAP = {
"deepseek-chat": "default", # Instant — the fast default model
"deepseek-expert": "expert", # Expert — the stronger, slower model
"deepseek-chat": "default",
"deepseek-reasoner": "expert",
"deepseek-expert": "expert",
"deepseek-coder": "default",
# OpenAI model aliases to enable seamless drop-in replacement with standard tools
"gpt-4o": "expert",
"gpt-4o-mini": "default",
"gpt-4-turbo": "expert",
"gpt-4": "expert",
"gpt-3.5-turbo": "default",
}

DEFAULT_MODEL = "deepseek-chat"


def is_known_model(name: str) -> bool:
"""Whether `name` is a model id we accept (used to 404 unknown models)."""
return name in MODEL_MAP
"""Check if model name is in MODEL_MAP or starts with standard prefixes."""
if not name:
return False
return name.lower() in MODEL_MAP or name.lower().startswith(("deepseek", "gpt-"))


def should_enable_thinking(model_name: str, explicit_thinking: bool | None = None) -> bool:
"""Determine whether DeepThink reasoning mode should be enabled for this request."""
if explicit_thinking is not None:
return explicit_thinking
name = (model_name or "").lower()
return "reasoner" in name or "r1" in name


def resolve_model_type(name: str) -> str:
"""Translate a public model id to DeepSeek's `model_type` wire value.

Caller must check `is_known_model` first; this raises KeyError otherwise.
Gracefully falls back to 'default' if not specifically mapped.
"""
return MODEL_MAP[name]
if not name:
return "default"
return MODEL_MAP.get(name.lower(), "default")
Loading