From f790ccbe047152791e1ed1779cd8b76f1b0a4699 Mon Sep 17 00:00:00 2001 From: Sumit Gautam Date: Fri, 19 Jun 2026 18:54:38 +0530 Subject: [PATCH] Stable Client + OpenAI API --- .gitignore | 1 + README.md | 193 ++++++++++++-------- app.py | 13 ++ copilot/__init__.py | 19 +- copilot/auth.py | 35 +++- copilot/browser.py | 11 +- copilot/client.py | 282 +++++++++++++---------------- copilot/driver.py | 191 +++++++++++++++++++ copilot/session.py | 108 ----------- examples/01_direct_chat.py | 27 +++ examples/02_direct_conversation.py | 31 ++++ examples/03_direct_stream.py | 30 +++ examples/04_server_http.py | 38 ++++ examples/05_server_stream.py | 45 +++++ examples/06_server_openai_sdk.py | 41 +++++ examples/README.md | 24 +++ main.py | 13 -- requirements.txt | 2 + server/__init__.py | 52 ++++++ server/api.py | 100 ++++++++++ server/config.py | 4 + server/openai_format.py | 60 ++++++ server/prompt.py | 47 +++++ server/schemas.py | 26 +++ 24 files changed, 1025 insertions(+), 368 deletions(-) create mode 100644 app.py create mode 100644 copilot/driver.py delete mode 100644 copilot/session.py create mode 100644 examples/01_direct_chat.py create mode 100644 examples/02_direct_conversation.py create mode 100644 examples/03_direct_stream.py create mode 100644 examples/04_server_http.py create mode 100644 examples/05_server_stream.py create mode 100644 examples/06_server_openai_sdk.py create mode 100644 examples/README.md delete mode 100644 main.py create mode 100644 server/__init__.py create mode 100644 server/api.py create mode 100644 server/config.py create mode 100644 server/openai_format.py create mode 100644 server/prompt.py create mode 100644 server/schemas.py diff --git a/.gitignore b/.gitignore index c7a0a3b..329f12b 100644 --- a/.gitignore +++ b/.gitignore @@ -11,3 +11,4 @@ docs/ # Local config / secrets config.json +test.py \ No newline at end of file diff --git a/README.md b/README.md index 501532f..757eca1 100644 --- a/README.md +++ b/README.md @@ -1,108 +1,159 @@ -# Copilot API +# Windows Copilot API: a free LLM API powered by Microsoft Copilot -An unofficial Python wrapper for Microsoft Copilot's consumer chat -(`copilot.microsoft.com`). It replays Copilot's own chat protocol directly over -HTTP — **no browser needed at request time** — solving the proof-of-work -challenge in-process and clearing Cloudflare with Chrome TLS impersonation. +**Using your own Microsoft Copilot account.** No API key, no credits, no paid plan: it turns the free chat at [copilot.microsoft.com](https://copilot.microsoft.com) into an API you can call from code. -> **Deep dive / recreation:** see [docs/IMPLEMENTATION_GUIDE.md](docs/IMPLEMENTATION_GUIDE.md) -> for the full protocol, the hashcash reverse-engineering, and the auth flow. +You can use it in two ways: -## Quick start +- 🐍 **As a Python library:** just call `client.chat("Hi")`. Supports streaming and multi-turn conversations. +- 🔌 **As a local OpenAI-compatible API:** runs a server at `http://localhost:8000/v1` that speaks the OpenAI format, so the official `openai` SDK (and any OpenAI-compatible app) works as a drop-in, with `localhost` in place of OpenAI. -### 1. Install +You sign in once with your Microsoft account in a browser; your session is saved and refreshed automatically after that. -```powershell -.\venv\Scripts\python.exe -m pip install -r requirements.txt -.\venv\Scripts\python.exe -m playwright install chromium # one-time, for sign-in only +> **Unofficial project.** Not affiliated with or endorsed by Microsoft. It automates the consumer Copilot web experience for personal use, so use it responsibly and within Microsoft's terms. + +--- + +## Why use this? + +- **Free:** uses your normal signed-in Copilot, no API billing. +- **Drop-in OpenAI replacement:** point any OpenAI client at `localhost` and it just works. +- **Works everywhere you're signed in:** the signed-in path works even in regions where *anonymous* Copilot is blocked (e.g. India). +- **Streaming + conversations:** token-by-token output and multi-turn threads addressed by `conversation_id`. + +--- + +## Requirements + +- **Python 3.9+** +- A **Microsoft account** (the free one you use for Copilot is fine) +- Works on Windows, macOS, and Linux + +--- + +## Setup (2 minutes) + +```bash +# 1. Clone the project +git clone +cd "Windows Copilot API" + +# 2. Install dependencies +pip install -r requirements.txt + +# 3. Install the browser Playwright needs (one-time) +playwright install chromium + +# 4. Sign in once: a browser opens, log into your Microsoft account +python -m copilot login ``` -### 2. Sign in once +That's it. Your session is saved under `session/` (git-ignored, never shared) and reused on every run. -Microsoft geo-restricts the *anonymous* chat experience (e.g. in India the -anonymous socket returns `chat-service-unavailable`). Signing in with a Microsoft -account works in those regions, so authenticate once: +> 💡 You can even skip step 4: the **first** time you call `chat()` or start the server, it opens the sign-in browser for you automatically. -```powershell -.\venv\Scripts\python.exe -m copilot login -``` +--- -A browser window opens — sign in, wait for the chat to load, then press Enter. -The session is saved under `session/` and reused automatically afterwards. You -only need to repeat this if the login is revoked or expires. +## Usage 1: In Python (no server) -### 3. Chat +The simplest way if your code is already Python. ```python -from copilot import CopilotSession +from copilot import CopilotClient -chat = CopilotSession() # loads your signed-in auth once +client = CopilotClient() # loads your signed-in session -# buffered — full reply as a string -print(chat.ask("Hello!")) +# Get a full reply +reply = client.chat("Say hello in one short sentence.") +print(reply.text) -# streamed — text as it arrives -for chunk in chat.stream("Tell me a joke"): +# Continue the SAME conversation — pass the id back +reply2 = client.chat("And now in French?", reply.conversation_id) +print(reply2.text) + +# Stream the answer as it's typed +for chunk in client.stream("Tell me a short joke"): print(chunk, end="", flush=True) - -chat.reset() # drop context and start a fresh chat ``` -One `CopilotSession` keeps a single conversation, so successive `ask`/`stream` -calls share context like a real chat. The short-lived access token is refreshed -transparently from your saved profile — no need to re-run `login`. +`chat()` returns the full text plus a `conversation_id`; pass that id back to keep the thread going, or omit it to start fresh. `stream()` yields the reply piece by piece. -Run the included example directly: +👉 More: [examples/01_direct_chat.py](examples/01_direct_chat.py), [02_direct_conversation.py](examples/02_direct_conversation.py), [03_direct_stream.py](examples/03_direct_stream.py) -```powershell -.\venv\Scripts\python.exe main.py +--- + +## Usage 2: As an OpenAI-compatible server + +Start a local server that speaks the OpenAI API, so existing OpenAI tools and SDKs work unchanged. + +```bash +python app.py +# -> Copilot OpenAI-compatible API on http://127.0.0.1:8000 ``` -## Options - -**Anonymous (no sign-in)** — works only where consumer Copilot is available: +Then point any OpenAI client at it (the API key is required by the SDK but ignored): ```python -chat = CopilotSession(anonymous=True) +from openai import OpenAI + +client = OpenAI(base_url="http://localhost:8000/v1", api_key="unused") + +resp = client.chat.completions.create( + model="copilot", + messages=[{"role": "user", "content": "Hello!"}], +) +print(resp.choices[0].message.content) ``` -**Via a proxy** — route through a supported region if anonymous chat is blocked -where you are: +Or call it with plain HTTP / `curl`: -```python -chat = CopilotSession(proxy="http://user:pass@host:port") # or socks5:// +```bash +curl http://localhost:8000/v1/chat/completions \ + -H "Content-Type: application/json" \ + -d '{"messages": [{"role": "user", "content": "Hello!"}]}' ``` -## CLI +**Endpoints** -```powershell -.\venv\Scripts\python.exe -m copilot login # interactive sign-in -.\venv\Scripts\python.exe -m copilot ask "hi" # one-shot reply (browser driver) +| Method | Path | Description | +| --- | --- | --- | +| `POST` | `/v1/chat/completions` | Chat (supports `"stream": true` and an optional `"conversation_id"`) | +| `GET` | `/v1/models` | Lists the single `copilot` model | + +> Change the address with env vars: `HOST=0.0.0.0 PORT=8080 python app.py`, or run `uvicorn server.api:app --host 0.0.0.0 --port 8080`. + +👉 More: [examples/04_server_http.py](examples/04_server_http.py), [05_server_stream.py](examples/05_server_stream.py), [06_server_openai_sdk.py](examples/06_server_openai_sdk.py) + +--- + +## Command line + +```bash +python -m copilot login # sign in and save the session +python -m copilot ask "Hello!" # quick one-shot question ``` -## How it works - -`copilot/client.py` (`Copilot`) speaks the protocol directly over -[`curl_cffi`](https://github.com/lexiforest/curl_cffi): - -1. **Cloudflare** — Chrome impersonation clears it; no browser needed. -2. **Conversation** — `POST /c/api/conversations` returns a conversation id. -3. **Proof-of-work** — the chat socket sends a `hashcash` (or `copilot`) - challenge before streaming; it's solved in-process and the message is - re-sent, mirroring the official client. See `copilot/challenges.py`. - -`CopilotSession` (`copilot/session.py`) is the recommended entry point: it wraps -the low-level `Copilot` driver with auth handling and conversation state. The -Playwright driver in `copilot/browser.py` (`BrowserCopilot`) is an optional -fallback used only for sign-in and the `ask` CLI command. +--- ## Project layout -| Path | Purpose | +| Path | What it does | | --- | --- | -| `copilot/session.py` | `CopilotSession` — high-level chat (use this) | -| `copilot/client.py` | `Copilot` — pure-HTTP protocol driver | -| `copilot/auth.py` | signed-in token caching / refresh | -| `copilot/challenges.py` | proof-of-work solvers | -| `copilot/browser.py` | Playwright fallback (login + CLI) | -| `main.py` | runnable example | +| [copilot/](copilot/) | The core library: `CopilotClient`, auth, browser sign-in, HTTP driver | +| [server/](server/) | The FastAPI OpenAI-compatible server | +| [examples/](examples/) | Runnable examples for every feature ([examples/README.md](examples/README.md)) | +| [app.py](app.py) | Starts the server | + +--- + +## Notes & limitations + +- **Sign in once, then reuse.** The cached token refreshes automatically; you only re-sign-in if the session fully expires. +- **No daily limit, but be reasonable.** Microsoft doesn't impose a daily chat cap, but please use it in moderation, and don't spam or hammer it with automated bulk requests. +- **One model.** Copilot has no model picker, so the server advertises a single model named `copilot`. +- **Your session is private.** Everything in `session/` (cookies + token) stays on your machine and is git-ignored. + +--- + +## License + +For personal and educational use. You are responsible for complying with Microsoft's terms of service. diff --git a/app.py b/app.py new file mode 100644 index 0000000..75ccef8 --- /dev/null +++ b/app.py @@ -0,0 +1,13 @@ +"""Start the OpenAI-compatible Copilot server: + + python app.py # serves http://127.0.0.1:8000 (set HOST / PORT to override) + +To bind a custom host/port from the command line, point uvicorn at the ASGI app: + + uvicorn server.api:app --host 0.0.0.0 --port 8080 +""" + +from server import app + +if __name__ == "__main__": + app() # blocks while uvicorn runs diff --git a/copilot/__init__.py b/copilot/__init__.py index d0dd466..2028877 100644 --- a/copilot/__init__.py +++ b/copilot/__init__.py @@ -1,12 +1,14 @@ """ Copilot API - An unofficial Python wrapper for Microsoft Copilot consumer chat. -Basic usage — create one session, reuse it for many requests: +Basic usage — one client, conversations addressed by id: ->>> from copilot import CopilotSession ->>> chat = CopilotSession() ->>> chat.ask("Hello!") # buffered: full reply as a string ->>> for chunk in chat.stream("And again?"): # streamed: text as it arrives +>>> from copilot import CopilotClient +>>> client = CopilotClient() +>>> r = client.chat("Hello!") # new conversation +>>> r.text, r.conversation_id +>>> client.chat("And again?", r.conversation_id) # continue it +>>> for chunk in client.stream("Stream this"): # incremental output ... print(chunk, end="") """ @@ -14,11 +16,12 @@ __version__ = '1.0.0' from .auth import load_auth from .browser import BrowserCopilot -from .client import Copilot -from .session import CopilotSession +from .client import ChatReply, CopilotClient +from .driver import Copilot __all__ = [ - 'CopilotSession', + 'CopilotClient', + 'ChatReply', 'Copilot', 'BrowserCopilot', 'load_auth', diff --git a/copilot/auth.py b/copilot/auth.py index b47b48b..c63f7fd 100644 --- a/copilot/auth.py +++ b/copilot/auth.py @@ -23,13 +23,19 @@ def load_auth( profile_dir: str = DEFAULT_PROFILE_DIR, max_age: int = AUTH_MAX_AGE, proxy: Optional[str] = None, + auto_login: bool = True, ) -> dict: """Return ``{cookies, access_token, saved_at}`` for the signed-in user. Uses the cached snapshot at ``path`` while fresh; otherwise spins up a headless browser against the persistent ``profile_dir`` to read a fresh MSAL token (the profile stays signed in via its long-lived refresh token) and - re-snapshots. Raises ``RuntimeError`` if the profile is not signed in. + re-snapshots. + + When the profile is *not* signed in (e.g. first-ever use) and ``auto_login`` + is true, this opens a visible browser for interactive Microsoft sign-in + instead of failing — so the very first call just works. Set + ``auto_login=False`` (or run headless/CI) to get a ``RuntimeError`` instead. Intended for the pure-HTTP :class:`copilot.client.Copilot` path:: @@ -48,15 +54,30 @@ def load_auth( from .browser import BrowserCopilot + # Try a headless read first: a signed-in profile just needs a fresh token. bot = BrowserCopilot(profile_dir=profile_dir, headless=True, proxy=proxy) try: bot.start() token = bot.access_token() - if not token or bot.region_blocked(): - raise RuntimeError( - "Not signed in (no access token in the browser profile). " - "Run `python -m copilot login` and sign in first." - ) - return bot.export_auth(path=path, stamp=time.time()) + if token and not bot.region_blocked(): + return bot.export_auth(path=path, stamp=time.time()) finally: bot.close() + + # No signed-in session in the profile. + if not auto_login: + raise RuntimeError( + "Not signed in (no access token in the browser profile). " + "Run `python -m copilot login` and sign in first." + ) + + # First-time use: create the session interactively, then return its auth. + print("No saved Copilot session found — opening a browser to sign in...") + auth = BrowserCopilot(profile_dir=profile_dir, headless=False, proxy=proxy).login(path=path) + if not auth.get("access_token"): + raise RuntimeError( + "Sign-in did not complete (no access token captured). " + "Re-run and finish the Microsoft sign-in before pressing Enter, " + "or sign in manually with `python -m copilot login`." + ) + return auth diff --git a/copilot/browser.py b/copilot/browser.py index 182ef8a..5fb6019 100644 --- a/copilot/browser.py +++ b/copilot/browser.py @@ -227,11 +227,12 @@ class BrowserCopilot: # -- auth --------------------------------------------------------------- - def login(self) -> None: + def login(self, path: str = DEFAULT_AUTH_FILE) -> dict: """Open a visible window for interactive Microsoft sign-in. Blocks until you press Enter in the console. The session is persisted in - ``profile_dir``, so subsequent headless runs reuse it. + ``profile_dir`` (and snapshotted to ``path``), so subsequent headless + runs reuse it. Returns the captured auth dict. """ self.close() self.start(headless=False) @@ -245,13 +246,15 @@ class BrowserCopilot: except EOFError: pass # Snapshot fresh auth so the headless curl_cffi path works immediately. + auth: dict = {} try: - self.export_auth(stamp=time.time()) - print(f"Auth snapshot saved to {DEFAULT_AUTH_FILE}") + auth = self.export_auth(path=path, stamp=time.time()) + print(f"Auth snapshot saved to {path}") except Exception as exc: print(f"(could not snapshot auth: {exc})") self.close() print(f"Session saved to {self.profile_dir}") + return auth def access_token(self) -> Optional[str]: """Return the page's MSAL access token, or ``None`` if anonymous.""" diff --git a/copilot/client.py b/copilot/client.py index fb2f018..29d56b3 100644 --- a/copilot/client.py +++ b/copilot/client.py @@ -1,180 +1,148 @@ -"""Pure-HTTP Copilot driver. +"""High-level Copilot client — the recommended entry point. -Speaks Microsoft Copilot's consumer chat protocol directly over a -Cloudflare-impersonating ``curl_cffi`` session — no browser required. See -:mod:`copilot.browser` for the Playwright-backed fallback. +One client, many conversations addressed by id. :meth:`CopilotClient.chat` +returns the full reply plus the conversation id; pass that id back to continue +the same conversation, or omit it to start a fresh one. :meth:`CopilotClient.stream` +is the incremental variant. + + from copilot import CopilotClient + + client = CopilotClient() # loads signed-in auth once + r = client.chat("My name is Tomato. Remember it.") + print(r.text, r.conversation_id) + + r2 = client.chat("What's my name?", r.conversation_id) # continue + print(r2.text) + + for chunk in client.stream("Tell me a joke"): # new conversation, streamed + print(chunk, end="", flush=True) + +The signed-in access token is refreshed transparently; sign in once with +``python -m copilot login``. Pass ``anonymous=True`` to skip sign-in (only where +anonymous consumer chat is available), or ``proxy=...`` to route through a +supported region. """ -import json import time -from typing import Dict, Optional -from urllib.parse import quote +from dataclasses import dataclass, field +from typing import Generator, List, Optional, Union -from curl_cffi.requests import Session, CurlWsFlag - -from .challenges import solve_copilot_challenge, solve_hashcash -from .models import AbstractProvider, Conversation, ImageResponse, ImageType -from .utils import drain_json, is_accepted_format, raise_for_status, to_bytes +from .auth import AUTH_MAX_AGE, load_auth +from .driver import Copilot +from .models import Conversation, ImageResponse -class Copilot(AbstractProvider): - label = "Microsoft Copilot" - url = "https://copilot.microsoft.com" - working = True - supports_stream = True - default_model = "Copilot" - needs_auth = False # consumer chat works anonymously (cookies only) - websocket_url = "wss://copilot.microsoft.com/c/api/chat?api-version=2" - conversation_url = f"{url}/c/api/conversations" +@dataclass +class ChatReply: + """The full result of a :meth:`CopilotClient.chat` call.""" - def create_completion( - self, - prompt: str, - stream: bool = False, - proxy: str = None, - timeout: int = 900, - image: ImageType = None, - conversation: Optional[Conversation] = None, - return_conversation: bool = False, - cookies: Dict[str, str] = None, - access_token: str = None, - **kwargs - ): - """Stream a Copilot reply to ``prompt``. + text: str + conversation_id: Optional[str] + images: List[ImageResponse] = field(default_factory=list) - Runs Copilot's own chat protocol over a Cloudflare-impersonating - ``curl_cffi`` session: ``POST /c/api/conversations`` then a chat - WebSocket (``send`` -> proof-of-work ``challenge`` -> ``appendText``* -> - ``done``). The challenge is solved in-process (see - :mod:`copilot.challenges`); no browser is required. - ``prompt`` is the user message sent straight to the chat socket (the - protocol has no separate system/role channel). Anonymous by default; - pass ``cookies`` and/or ``access_token`` (e.g. exported from a signed-in - browser session) to run as a logged-in user — required where anonymous - consumer chat is region-restricted. - """ - # Resolve auth: explicit args win, else fall back to the conversation's. - if cookies is None and conversation is not None: - cookies = conversation.cookies - if access_token is None and conversation is not None: - access_token = conversation.access_token +class ChatStream: + """Iterable stream of reply chunks that also exposes the conversation id. - websocket_url = self.websocket_url - headers = None - if access_token: - websocket_url = f"{websocket_url}&accessToken={quote(access_token)}" - headers = {"authorization": f"Bearer {access_token}"} + Yields ``str`` text chunks (and :class:`~copilot.models.ImageResponse` for + generated images). ``conversation_id`` is known up front when continuing an + existing conversation, and is populated as soon as iteration begins when a + new conversation is created. + """ - with Session( - timeout=timeout, - proxy=proxy, - impersonate="chrome", - cookies=cookies, - headers=headers, - ) as session: - # Establish cookies + Cloudflare clearance (anonymous is fine). - session.get(f"{self.url}/") + def __init__(self, chunks: Generator, conversation_id: Optional[str]): + self._chunks = chunks + self.conversation_id = conversation_id - if conversation is None: - response = session.post(self.conversation_url) - raise_for_status(response) - conversation_id = response.json().get("id") - if return_conversation: - yield Conversation(conversation_id, session.cookies.jar) + def __iter__(self) -> Generator[Union[str, ImageResponse], None, None]: + for item in self._chunks: + if isinstance(item, Conversation): + self.conversation_id = item.conversation_id else: - conversation_id = conversation.conversation_id + yield item - images = [] - if image is not None: - data = to_bytes(image) - response = session.post( - f"{self.url}/c/api/attachments", - headers={"content-type": is_accepted_format(data)}, - data=data, - ) - raise_for_status(response) - images.append({"type": "image", "url": response.json().get("url")}) - send_frame = json.dumps({ - "event": "send", - "conversationId": conversation_id, - "content": [*images, {"type": "text", "text": prompt}], - "mode": "chat", - }).encode() +class CopilotClient: + """A Copilot client: one object, many conversations addressed by id. - wss = session.ws_connect(websocket_url) - wss.send(send_frame, CurlWsFlag.TEXT) - yield from self._read_stream(wss, send_frame, timeout) + Parameters + ---------- + anonymous: + Skip sign-in and talk to Copilot anonymously. Only works where the + anonymous consumer experience is available (it is geo-blocked in some + regions, e.g. India). Default ``False`` uses the signed-in session. + proxy: + Optional ``scheme://user:pass@host:port`` proxy, applied to both the + auth refresh and every request. + max_age: + Seconds a cached access token is trusted before it is refreshed. + """ - def _read_stream(self, wss, send_frame: bytes, timeout: int): - """Consume chat-socket frames, solving challenges, yielding text/images.""" - buffer = b"" - is_started = False - answered = False - image_prompt = None - last_msg = None + def __init__( + self, + anonymous: bool = False, + proxy: Optional[str] = None, + max_age: int = AUTH_MAX_AGE, + ): + self._driver = Copilot() + self._anonymous = anonymous + self._proxy = proxy + self._max_age = max_age + self._auth: Optional[dict] = None - deadline = time.time() + timeout - while True: - try: - chunk = wss.recv()[0] - except Exception: - break - if not chunk: - if time.time() > deadline: - break - continue + def stream( + self, + prompt: str, + conversation_id: Optional[str] = None, + **kwargs, + ) -> ChatStream: + """Stream the reply to ``prompt`` as a :class:`ChatStream`. - buffer += chunk if isinstance(chunk, (bytes, bytearray)) else chunk.encode() - messages, buffer = drain_json(buffer) - for msg in messages: - last_msg = msg - event = msg.get("event") - if event == "challenge" and not answered: - token = self._solve_challenge(msg) - if token is not None: - wss.send(json.dumps({ - "event": "challengeResponse", - "token": token, - "method": msg.get("method"), - }).encode(), CurlWsFlag.TEXT) - answered = True - # The client re-sends the held message after a challenge. - wss.send(send_frame, CurlWsFlag.TEXT) - elif event == "appendText": - is_started = True - yield msg.get("text") - elif event == "generatingImage": - image_prompt = msg.get("prompt") - elif event == "imageGenerated": - yield ImageResponse(msg.get("url"), image_prompt, {"preview": msg.get("thumbnailUrl")}) - elif event == "done": - return - elif event == "error": - code = msg.get("errorCode") or msg - if code == "chat-service-unavailable": - raise RuntimeError( - "Copilot error: chat-service-unavailable. The chat backend is " - "typically geo-restricted; if you are outside a supported region, " - "retry via a proxy in a supported region, e.g. " - "create_completion(..., proxy='http://user:pass@host:port')." - ) - raise RuntimeError(f"Copilot error: {code}") + Starts a new conversation when ``conversation_id`` is ``None``; otherwise + continues that conversation. Read ``.conversation_id`` on the returned + stream (during/after iteration) to continue the chat later. + """ + auth = self._fresh_auth() + kw = dict( + stream=True, + proxy=self._proxy, + cookies=auth["cookies"] if auth else None, + access_token=auth["access_token"] if auth else None, + **kwargs, + ) + if conversation_id is None: + # New conversation: have the driver hand back its id. + kw["return_conversation"] = True + else: + kw["conversation_id"] = conversation_id - if not is_started: - raise RuntimeError(f"Invalid response: {last_msg}") + chunks = self._driver.create_completion(prompt, **kw) + return ChatStream(chunks, conversation_id) - @staticmethod - def _solve_challenge(msg: dict): - """Return the challenge response token, or None if unsupported.""" - method = msg.get("method") - parameter = msg.get("parameter") - if not parameter: + def chat( + self, + prompt: str, + conversation_id: Optional[str] = None, + **kwargs, + ) -> ChatReply: + """Return the full reply to ``prompt`` as a :class:`ChatReply`. + + Buffers the whole response; use :meth:`stream` for incremental output. + """ + s = self.stream(prompt, conversation_id=conversation_id, **kwargs) + text: List[str] = [] + images: List[ImageResponse] = [] + for item in s: + if isinstance(item, str): + text.append(item) + elif isinstance(item, ImageResponse): + images.append(item) + return ChatReply("".join(text), s.conversation_id, images) + + def _fresh_auth(self) -> Optional[dict]: + """Return current signed-in auth, refreshing it when stale (or None).""" + if self._anonymous: return None - if method == "hashcash": - return solve_hashcash(parameter) - if method == "copilot": - return solve_copilot_challenge(parameter) - # 'cloudflare' (Turnstile) needs a browser-solved token; unsupported here. - return None + if self._auth is None or (time.time() - self._auth.get("saved_at", 0)) >= self._max_age: + self._auth = load_auth(max_age=self._max_age, proxy=self._proxy) + return self._auth diff --git a/copilot/driver.py b/copilot/driver.py new file mode 100644 index 0000000..05c7cd1 --- /dev/null +++ b/copilot/driver.py @@ -0,0 +1,191 @@ +"""Pure-HTTP Copilot driver. + +Speaks Microsoft Copilot's consumer chat protocol directly over a +Cloudflare-impersonating ``curl_cffi`` session — no browser required. This is the +low-level engine; most callers should use :class:`copilot.client.CopilotClient`. +See :mod:`copilot.browser` for the Playwright-backed fallback. +""" + +import json +import time +from typing import Dict, Optional +from urllib.parse import quote + +from curl_cffi.requests import Session, CurlWsFlag + +from .challenges import solve_copilot_challenge, solve_hashcash +from .models import AbstractProvider, Conversation, ImageResponse, ImageType +from .utils import drain_json, is_accepted_format, raise_for_status, to_bytes + + +class Copilot(AbstractProvider): + label = "Microsoft Copilot" + url = "https://copilot.microsoft.com" + working = True + supports_stream = True + default_model = "Copilot" + needs_auth = False # consumer chat works anonymously (cookies only) + websocket_url = "wss://copilot.microsoft.com/c/api/chat?api-version=2" + conversation_url = f"{url}/c/api/conversations" + + def create_completion( + self, + prompt: str, + stream: bool = False, + proxy: str = None, + timeout: int = 900, + image: ImageType = None, + conversation: Optional[Conversation] = None, + conversation_id: str = None, + return_conversation: bool = False, + cookies: Dict[str, str] = None, + access_token: str = None, + **kwargs + ): + """Stream a Copilot reply to ``prompt``. + + Runs Copilot's own chat protocol over a Cloudflare-impersonating + ``curl_cffi`` session: ``POST /c/api/conversations`` then a chat + WebSocket (``send`` -> proof-of-work ``challenge`` -> ``appendText``* -> + ``done``). The challenge is solved in-process (see + :mod:`copilot.challenges`); no browser is required. + + ``prompt`` is the user message sent straight to the chat socket (the + protocol has no separate system/role channel). Anonymous by default; + pass ``cookies`` and/or ``access_token`` (e.g. exported from a signed-in + browser session) to run as a logged-in user — required where anonymous + consumer chat is region-restricted. + + Conversation targeting (first match wins): + * ``conversation`` — reuse an existing :class:`Conversation` object; + * ``conversation_id`` — resume a conversation by its id string (no + create call), e.g. one saved from a previous run; + * neither — create a fresh conversation. With ``return_conversation`` + the new :class:`Conversation` is yielded first. + """ + # Resolve auth: explicit args win, else fall back to the conversation's. + if cookies is None and conversation is not None: + cookies = conversation.cookies + if access_token is None and conversation is not None: + access_token = conversation.access_token + + websocket_url = self.websocket_url + headers = None + if access_token: + websocket_url = f"{websocket_url}&accessToken={quote(access_token)}" + headers = {"authorization": f"Bearer {access_token}"} + + with Session( + timeout=timeout, + proxy=proxy, + impersonate="chrome", + cookies=cookies, + headers=headers, + ) as session: + # Establish cookies + Cloudflare clearance (anonymous is fine). + session.get(f"{self.url}/") + + if conversation is not None: + conversation_id = conversation.conversation_id + elif conversation_id is not None: + pass # resume an existing conversation by id; skip create + else: + response = session.post(self.conversation_url) + raise_for_status(response) + conversation_id = response.json().get("id") + if return_conversation: + yield Conversation(conversation_id, session.cookies.jar) + + images = [] + if image is not None: + data = to_bytes(image) + response = session.post( + f"{self.url}/c/api/attachments", + headers={"content-type": is_accepted_format(data)}, + data=data, + ) + raise_for_status(response) + images.append({"type": "image", "url": response.json().get("url")}) + + send_frame = json.dumps({ + "event": "send", + "conversationId": conversation_id, + "content": [*images, {"type": "text", "text": prompt}], + "mode": "chat", + }).encode() + + wss = session.ws_connect(websocket_url) + wss.send(send_frame, CurlWsFlag.TEXT) + yield from self._read_stream(wss, send_frame, timeout) + + def _read_stream(self, wss, send_frame: bytes, timeout: int): + """Consume chat-socket frames, solving challenges, yielding text/images.""" + buffer = b"" + is_started = False + answered = False + image_prompt = None + last_msg = None + + deadline = time.time() + timeout + while True: + try: + chunk = wss.recv()[0] + except Exception: + break + if not chunk: + if time.time() > deadline: + break + continue + + buffer += chunk if isinstance(chunk, (bytes, bytearray)) else chunk.encode() + messages, buffer = drain_json(buffer) + for msg in messages: + last_msg = msg + event = msg.get("event") + if event == "challenge" and not answered: + token = self._solve_challenge(msg) + if token is not None: + wss.send(json.dumps({ + "event": "challengeResponse", + "token": token, + "method": msg.get("method"), + }).encode(), CurlWsFlag.TEXT) + answered = True + # The client re-sends the held message after a challenge. + wss.send(send_frame, CurlWsFlag.TEXT) + elif event == "appendText": + is_started = True + yield msg.get("text") + elif event == "generatingImage": + image_prompt = msg.get("prompt") + elif event == "imageGenerated": + yield ImageResponse(msg.get("url"), image_prompt, {"preview": msg.get("thumbnailUrl")}) + elif event == "done": + return + elif event == "error": + code = msg.get("errorCode") or msg + if code == "chat-service-unavailable": + raise RuntimeError( + "Copilot error: chat-service-unavailable. The chat backend is " + "typically geo-restricted; if you are outside a supported region, " + "retry via a proxy in a supported region, e.g. " + "create_completion(..., proxy='http://user:pass@host:port')." + ) + raise RuntimeError(f"Copilot error: {code}") + + if not is_started: + raise RuntimeError(f"Invalid response: {last_msg}") + + @staticmethod + def _solve_challenge(msg: dict): + """Return the challenge response token, or None if unsupported.""" + method = msg.get("method") + parameter = msg.get("parameter") + if not parameter: + return None + if method == "hashcash": + return solve_hashcash(parameter) + if method == "copilot": + return solve_copilot_challenge(parameter) + # 'cloudflare' (Turnstile) needs a browser-solved token; unsupported here. + return None diff --git a/copilot/session.py b/copilot/session.py deleted file mode 100644 index 76272b4..0000000 --- a/copilot/session.py +++ /dev/null @@ -1,108 +0,0 @@ -"""High-level, reusable Copilot session — the one way to use this package. - -Create a :class:`CopilotSession` once and call :meth:`ask` / :meth:`stream` as -many times as you like. It loads the signed-in auth a single time (refreshing the -short-lived access token transparently when it goes stale) and keeps one -conversation, so successive turns share context like a real chat. - - from copilot import CopilotSession - - chat = CopilotSession() - print(chat.ask("Hello!")) # buffered: returns the full reply - for chunk in chat.stream("And again?"): # streamed: yields text as it lands - print(chunk, end="", flush=True) - chat.reset() # start a fresh, context-free chat -""" - -import time -from typing import Generator, Optional, Union - -from .auth import AUTH_MAX_AGE, load_auth -from .client import Copilot -from .models import Conversation, ImageResponse - - -class CopilotSession: - """A long-lived Copilot chat: one object, many requests. - - Parameters - ---------- - anonymous: - Skip sign-in and talk to Copilot anonymously. Only works where the - anonymous consumer experience is available (it is geo-blocked in some - regions, e.g. India). Default ``False`` uses the signed-in session. - proxy: - Optional ``scheme://user:pass@host:port`` proxy, applied to both the - auth refresh and every request. - max_age: - Seconds a cached access token is trusted before it is refreshed. - """ - - def __init__( - self, - anonymous: bool = False, - proxy: Optional[str] = None, - max_age: int = AUTH_MAX_AGE, - ): - self._copilot = Copilot() - self._anonymous = anonymous - self._proxy = proxy - self._max_age = max_age - self._auth: Optional[dict] = None - self._conversation: Optional[Conversation] = None - - def stream( - self, - prompt: str, - *, - new_conversation: bool = False, - **kwargs, - ) -> Generator[Union[str, ImageResponse], None, None]: - """Stream the reply to ``prompt``, yielding text chunks as they arrive. - - Continues the running conversation by default; pass - ``new_conversation=True`` (or call :meth:`reset`) to start fresh. - Image results are yielded as :class:`~copilot.models.ImageResponse`. - """ - if new_conversation: - self._conversation = None - - auth = self._fresh_auth() - kw = dict( - stream=True, - proxy=self._proxy, - cookies=auth["cookies"] if auth else None, - access_token=auth["access_token"] if auth else None, - **kwargs, - ) - if self._conversation is None: - # First turn: have the driver hand back a Conversation we can reuse. - kw["return_conversation"] = True - else: - kw["conversation"] = self._conversation - - for item in self._copilot.create_completion(prompt, **kw): - if isinstance(item, Conversation): - self._conversation = item - else: - yield item - - def ask(self, prompt: str, *, new_conversation: bool = False, **kwargs) -> str: - """Return the full reply to ``prompt`` as a single string (text only).""" - return "".join( - chunk - for chunk in self.stream(prompt, new_conversation=new_conversation, **kwargs) - if isinstance(chunk, str) - ) - - def reset(self) -> None: - """Forget the current conversation; the next call starts a fresh one.""" - self._conversation = None - - def _fresh_auth(self) -> Optional[dict]: - """Return current signed-in auth, refreshing it when stale (or None).""" - if self._anonymous: - return None - if self._auth is None or (time.time() - self._auth.get("saved_at", 0)) >= self._max_age: - self._auth = load_auth(max_age=self._max_age, proxy=self._proxy) - return self._auth diff --git a/examples/01_direct_chat.py b/examples/01_direct_chat.py new file mode 100644 index 0000000..0ff180f --- /dev/null +++ b/examples/01_direct_chat.py @@ -0,0 +1,27 @@ +"""Example 1 — the simplest chat, in-process (no server needed). + +Use this when your code IS Python and you just want a reply from Copilot. + +Run it from the project root: + + python examples/01_direct_chat.py + +On the very first run a browser opens for sign-in automatically — sign in, +press Enter in the terminal, and it continues. After that the session is reused. +""" + +# Make the project importable when this file is run directly. +import sys +import pathlib +sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent.parent)) + +from copilot import CopilotClient + +# Create the client once and reuse it. anonymous=False uses your signed-in +# Microsoft account (works everywhere, including regions where anonymous +# Copilot is blocked). +client = CopilotClient(anonymous=False) + +# .chat() waits for the FULL reply, then returns it. +reply = client.chat("Say hello in one short sentence.") +print(reply.text) diff --git a/examples/02_direct_conversation.py b/examples/02_direct_conversation.py new file mode 100644 index 0000000..4d6ce74 --- /dev/null +++ b/examples/02_direct_conversation.py @@ -0,0 +1,31 @@ +"""Example 2 — a multi-turn conversation, in-process. + +Every reply comes with a `conversation_id`. Pass that id back on the next call +to continue the SAME conversation, so Copilot remembers earlier turns. + +Run it from the project root: + + python examples/02_direct_conversation.py +""" + +# Make the project importable when this file is run directly. +import sys +import pathlib +sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent.parent)) + +import time + +from copilot import CopilotClient + +client = CopilotClient(anonymous=False) + +# Turn 1 — no id is given, so this starts a NEW conversation and returns its id. +first = client.chat("My name is Ada. Remember it.") +print("Copilot:", first.text) +print("conversation_id:", first.conversation_id) + +time.sleep(3) # be gentle — Copilot serves one conversation at a time + +# Turn 2 — pass the id back to CONTINUE the same conversation. +second = client.chat("What's my name? Reply with just the name.", first.conversation_id) +print("Copilot:", second.text) # -> recalls "Ada" diff --git a/examples/03_direct_stream.py b/examples/03_direct_stream.py new file mode 100644 index 0000000..68925fb --- /dev/null +++ b/examples/03_direct_stream.py @@ -0,0 +1,30 @@ +"""Example 3 — stream the reply as it is generated, in-process. + +client.stream() yields pieces of text as they arrive, instead of waiting for the +whole reply. Good for showing output live (like a chat UI typing). + +Run it from the project root: + + python examples/03_direct_stream.py +""" + +# Make the project importable when this file is run directly. +import sys +import pathlib +sys.path.insert(0, str(pathlib.Path(__file__).resolve().parent.parent)) + +from copilot import CopilotClient + +client = CopilotClient(anonymous=False) + +# Omit the id to start a fresh conversation. Its id is filled in on +# .conversation_id as the stream runs, so you can read it afterwards. +stream = client.stream("Tell me a short, clean joke.") + +for chunk in stream: + # Text arrives as strings; generated images would arrive as objects. + if isinstance(chunk, str): + print(chunk, end="", flush=True) + +print() # newline after the streamed text +print("conversation_id:", stream.conversation_id) diff --git a/examples/04_server_http.py b/examples/04_server_http.py new file mode 100644 index 0000000..a3edefc --- /dev/null +++ b/examples/04_server_http.py @@ -0,0 +1,38 @@ +"""Example 4 — talk to the server over plain HTTP (OpenAI-compatible). + +Use this when the caller is NOT this Python process: another language, a tool, +or a different machine. The server speaks the OpenAI Chat Completions shape. + +First, start the server in another terminal: + + python app.py + +Then run this from the project root: + + python examples/04_server_http.py + +Each response includes Copilot's `conversation_id` as an extra field; send it +back to continue the same thread. +""" + +import requests + +URL = "http://localhost:8000/v1/chat/completions" + +# Turn 1 — new conversation. +first = requests.post(URL, json={ + "model": "copilot", + "messages": [{"role": "user", "content": "My name is Ada. Remember it."}], +}).json() +print("Copilot:", first["choices"][0]["message"]["content"]) + +cid = first["conversation_id"] +print("conversation_id:", cid) + +# Turn 2 — continue by sending the conversation_id back in the body. +second = requests.post(URL, json={ + "model": "copilot", + "conversation_id": cid, + "messages": [{"role": "user", "content": "What's my name? Just the name."}], +}).json() +print("Copilot:", second["choices"][0]["message"]["content"]) # -> recalls "Ada" diff --git a/examples/05_server_stream.py b/examples/05_server_stream.py new file mode 100644 index 0000000..0151815 --- /dev/null +++ b/examples/05_server_stream.py @@ -0,0 +1,45 @@ +"""Example 5 — stream from the server over HTTP (Server-Sent Events). + +Streaming needs TWO things: "stream": true in the JSON body (so the server sends +SSE), and stream=True on the request (so you can read it piece by piece). + +First, start the server in another terminal: + + python app.py + +Then run this from the project root: + + python examples/05_server_stream.py + +The server sends lines like `data: {...}`; the text is in +choices[0].delta.content, the conversation_id arrives on the final chunk, and +the stream ends with `data: [DONE]`. +""" + +import json + +import requests + +URL = "http://localhost:8000/v1/chat/completions" + +with requests.post(URL, json={ + "model": "copilot", + "stream": True, + "messages": [{"role": "user", "content": "Tell me a short, clean joke."}], +}, stream=True) as response: + conversation_id = None + for line in response.iter_lines(): + if not line or not line.startswith(b"data: "): + continue + payload = line[len("data: "):] + if payload == b"[DONE]": # end of the stream + break + chunk = json.loads(payload) + piece = chunk["choices"][0]["delta"].get("content") + if piece: + print(piece, end="", flush=True) + if chunk.get("conversation_id"): # present on the final chunk + conversation_id = chunk["conversation_id"] + +print() +print("conversation_id:", conversation_id) diff --git a/examples/06_server_openai_sdk.py b/examples/06_server_openai_sdk.py new file mode 100644 index 0000000..a3fdcf0 --- /dev/null +++ b/examples/06_server_openai_sdk.py @@ -0,0 +1,41 @@ +"""Example 6 — use the official OpenAI SDK against the server. + +The whole point of the server is OpenAI compatibility: point any OpenAI client at +it and existing code works unchanged. + +Install the SDK first: + + pip install openai + +Start the server in another terminal: + + python app.py + +Then run this from the project root: + + python examples/06_server_openai_sdk.py +""" + +from openai import OpenAI + +# Point base_url at the server. api_key is required by the SDK but ignored here. +client = OpenAI(base_url="http://localhost:8000/v1", api_key="unused") + +completion = client.chat.completions.create( + model="copilot", + messages=[{"role": "user", "content": "Say hello in one short sentence."}], +) +print(completion.choices[0].message.content) + +# conversation_id is outside OpenAI's schema, so the SDK keeps it in model_extra. +extra = getattr(completion, "model_extra", None) or {} +cid = extra.get("conversation_id") +print("conversation_id:", cid) + +# To continue that conversation, send the id back via extra_body: +# +# client.chat.completions.create( +# model="copilot", +# messages=[{"role": "user", "content": "..."}], +# extra_body={"conversation_id": cid}, +# ) diff --git a/examples/README.md b/examples/README.md new file mode 100644 index 0000000..3c2b1ed --- /dev/null +++ b/examples/README.md @@ -0,0 +1,24 @@ +# Examples + +Runnable examples for both ways to use this project. Run each from the **project +root** (e.g. `python examples/01_direct_chat.py`). + +## Directly, in Python (`CopilotClient`) + +No server needed. On the first run, sign-in opens a browser automatically. + +| File | Shows | +| --- | --- | +| [01_direct_chat.py](01_direct_chat.py) | The simplest one-shot chat | +| [02_direct_conversation.py](02_direct_conversation.py) | Multi-turn — continue with `conversation_id` | +| [03_direct_stream.py](03_direct_stream.py) | Stream the reply as it's generated | + +## Over HTTP (the OpenAI-compatible server) + +Start the server first in another terminal: `python app.py` + +| File | Shows | +| --- | --- | +| [04_server_http.py](04_server_http.py) | Plain HTTP with `requests` (chat + continue) | +| [05_server_stream.py](05_server_stream.py) | Streaming over Server-Sent Events | +| [06_server_openai_sdk.py](06_server_openai_sdk.py) | The official `openai` SDK (`pip install openai`) | diff --git a/main.py b/main.py deleted file mode 100644 index 094656d..0000000 --- a/main.py +++ /dev/null @@ -1,13 +0,0 @@ -from copilot import CopilotSession - -chat = CopilotSession() # loads your signed-in auth once - -# buffered — get the whole reply as a string -print(chat.ask("Hello!")) - -# # streamed — get text as it arrives -# for chunk in chat.stream("Tell me a joke"): -# print(chunk, end="", flush=True) - -# chat.reset() # drop context and start a fresh chat - \ No newline at end of file diff --git a/requirements.txt b/requirements.txt index cf4f783..148af59 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,2 +1,4 @@ curl_cffi>=0.15 playwright>=1.60 +fastapi>=0.110 +uvicorn>=0.29 diff --git a/server/__init__.py b/server/__init__.py new file mode 100644 index 0000000..99fe1d7 --- /dev/null +++ b/server/__init__.py @@ -0,0 +1,52 @@ +"""OpenAI-compatible HTTP server for Microsoft Copilot. + +Start it: + + from server import app + app() + +(`python app.py` in the project root does exactly this.) The server runs on +http://127.0.0.1:8000 — set HOST / PORT to override. It bridges the OpenAI Chat +Completions shape onto :class:`copilot.CopilotClient`; sign in once first with +``python -m copilot login``. + +Code is split by concern: + + config.py constants + schemas.py pydantic request models + prompt.py flatten OpenAI messages -> one Copilot prompt + openai_format.py build OpenAI response/chunk shapes + api.py FastAPI app, routes, upstream serialization +""" + +import os + +from .api import app as _api + + +def app(host="127.0.0.1", port=8000) -> None: + """Start the server (blocks while uvicorn runs). + + On first run (no saved session) this opens a browser for interactive sign-in + before serving, so requests don't fail with a "not signed in" error. + """ + import uvicorn + + from copilot.auth import load_auth + + host = host or os.environ.get("HOST", "127.0.0.1") + port = port or int(os.environ.get("PORT", "8000")) + + # Ensure a signed-in Copilot session exists before we start serving. On the + # very first run this triggers the interactive browser sign-in (instead of + # letting the first HTTP request fail), then caches it for reuse. + try: + load_auth() + except Exception as exc: + print(f"Warning: could not establish a Copilot session: {exc}") + + print(f"Copilot OpenAI-compatible API on http://{host}:{port} (POST /v1/chat/completions)") + uvicorn.run(_api, host=host, port=port) + + +__all__ = ["app"] diff --git a/server/api.py b/server/api.py new file mode 100644 index 0000000..5919d07 --- /dev/null +++ b/server/api.py @@ -0,0 +1,100 @@ +"""FastAPI app wiring Copilot onto the OpenAI Chat Completions API.""" + +import threading +import time + +from fastapi import FastAPI +from fastapi.responses import JSONResponse, StreamingResponse + +from copilot import CopilotClient + +from .config import MODEL_NAME +from .openai_format import ( + completion_response, + new_id, + sse_event, + stream_chunk, +) +from .prompt import messages_to_prompt +from .schemas import ChatCompletionRequest + +app = FastAPI(title="Copilot OpenAI-compatible API", version="1.0.0") +client = CopilotClient() + +# Copilot's per-account chat socket doesn't tolerate concurrent conversations +# from one process (parallel requests error out or hang). This server bridges a +# single signed-in account, so we serialize upstream calls: concurrent HTTP +# requests queue here and run one at a time. Predictable, at the cost of +# parallelism — fine for a personal bridge. +_upstream_lock = threading.Lock() + + +def _stream(prompt: str, model: str, conversation_id=None): + """Yield OpenAI ``chat.completion.chunk`` SSE events for ``prompt``. + + ``conversation_id`` continues an existing Copilot thread; ``None`` starts a + fresh one (its id is emitted on the final chunk). + """ + cid = new_id() + created = int(time.time()) + try: + with _upstream_lock: # one upstream chat at a time (released on disconnect) + yield sse_event(stream_chunk(cid, created, model, {"role": "assistant"})) + stream = client.stream(prompt, conversation_id=conversation_id) + for piece in stream: + if isinstance(piece, str) and piece: + yield sse_event(stream_chunk(cid, created, model, {"content": piece})) + # Copilot's conversation id is known once the stream has run; emit it + # on the final chunk so callers can track the upstream thread. + yield sse_event( + stream_chunk( + cid, created, model, {}, finish="stop", + conversation_id=stream.conversation_id, + ) + ) + except Exception as exc: # surface errors to the client instead of hanging + yield sse_event( + stream_chunk(cid, created, model, {"content": f"\n[error: {exc}]"}, finish="error") + ) + yield "data: [DONE]\n\n" + + +@app.get("/v1/models") +def list_models(): + return { + "object": "list", + "data": [ + {"id": MODEL_NAME, "object": "model", "created": 0, "owned_by": "microsoft"} + ], + } + + +@app.post("/v1/chat/completions") +def chat_completions(req: ChatCompletionRequest): + prompt = messages_to_prompt(req.messages) + if not prompt.strip(): + return JSONResponse( + status_code=400, + content={"error": {"message": "no text content in messages", "type": "invalid_request_error"}}, + ) + model = req.model or MODEL_NAME + + if req.stream: + return StreamingResponse( + _stream(prompt, model, req.conversation_id), media_type="text/event-stream" + ) + + try: + with _upstream_lock: # serialize: one upstream chat at a time + reply = client.chat(prompt, conversation_id=req.conversation_id) + except Exception as exc: + return JSONResponse( + status_code=502, + content={"error": {"message": str(exc), "type": "upstream_error"}}, + ) + return completion_response(reply.text, model, reply.conversation_id) + + +@app.get("/") +def root(): + return {"service": "Copilot OpenAI-compatible API", "endpoints": ["/v1/models", "/v1/chat/completions"]} diff --git a/server/config.py b/server/config.py new file mode 100644 index 0000000..ef4ff21 --- /dev/null +++ b/server/config.py @@ -0,0 +1,4 @@ +"""Server configuration — shared constants.""" + +# The single model id this bridge advertises (Copilot has no model selector). +MODEL_NAME = "copilot" diff --git a/server/openai_format.py b/server/openai_format.py new file mode 100644 index 0000000..70275b7 --- /dev/null +++ b/server/openai_format.py @@ -0,0 +1,60 @@ +"""Builders for the OpenAI wire shapes (completions, SSE chunks).""" + +import json +import time +import uuid + + +def new_id() -> str: + """A fresh ``chatcmpl-...`` id, as the OpenAI API returns.""" + return f"chatcmpl-{uuid.uuid4().hex}" + + +def completion_response(text: str, model: str, conversation_id=None) -> dict: + """A non-streaming ``chat.completion`` object. + + ``conversation_id`` is Copilot's own conversation id, surfaced as an extra + top-level field (not part of OpenAI's schema, so standard clients ignore it) + for callers that want to track the upstream thread. + """ + return { + "id": new_id(), + "object": "chat.completion", + "created": int(time.time()), + "model": model, + "conversation_id": conversation_id, + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": text}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0}, + } + + +def sse_event(payload: dict) -> str: + """Serialize a payload as a Server-Sent Events ``data:`` line.""" + return f"data: {json.dumps(payload)}\n\n" + + +def stream_chunk( + cid: str, created: int, model: str, delta: dict, finish=None, conversation_id=None +) -> dict: + """A single ``chat.completion.chunk`` object for streaming responses. + + ``conversation_id`` (Copilot's upstream id) is added as an extra top-level + field when known — typically only on the final chunk, since a new + conversation's id isn't available until the stream has started. + """ + chunk = { + "id": cid, + "object": "chat.completion.chunk", + "created": created, + "model": model, + "choices": [{"index": 0, "delta": delta, "finish_reason": finish}], + } + if conversation_id is not None: + chunk["conversation_id"] = conversation_id + return chunk diff --git a/server/prompt.py b/server/prompt.py new file mode 100644 index 0000000..8ed8b8c --- /dev/null +++ b/server/prompt.py @@ -0,0 +1,47 @@ +"""Flatten an OpenAI ``messages`` array into a single Copilot prompt. + +Copilot's protocol has no role/system channel — it takes one prompt string per +turn — so we collapse the whole conversation into one piece of text. +""" + +from typing import Any, List, Optional, Union + +from .schemas import ChatMessage + + +def content_text(content: Optional[Union[str, List[Any]]]) -> str: + """Extract plain text from a message's content (string or content-parts).""" + if content is None: + return "" + if isinstance(content, str): + return content + parts = [] + for part in content: + if isinstance(part, dict): + if part.get("type") == "text": + parts.append(part.get("text", "")) + else: + parts.append(str(part)) + return "\n".join(p for p in parts if p) + + +def messages_to_prompt(messages: List[ChatMessage]) -> str: + """Flatten an OpenAI ``messages`` array into a single Copilot prompt.""" + system = "\n\n".join( + content_text(m.content) for m in messages if m.role == "system" and m.content + ) + convo = [m for m in messages if m.role != "system"] + + if len(convo) == 1 and convo[0].role == "user": + body = content_text(convo[0].content) # simple single-turn request + else: + lines = [] + for m in convo: + label = "User" if m.role == "user" else "Assistant" + lines.append(f"{label}: {content_text(m.content)}") + lines.append("Assistant:") # cue Copilot to continue + body = "\n".join(lines) + + if system and body: + return f"{system}\n\n{body}" + return system or body diff --git a/server/schemas.py b/server/schemas.py new file mode 100644 index 0000000..1e78e5a --- /dev/null +++ b/server/schemas.py @@ -0,0 +1,26 @@ +"""Pydantic request models for the OpenAI-compatible endpoints.""" + +from typing import Any, List, Optional, Union + +from pydantic import BaseModel + +from .config import MODEL_NAME + + +class ChatMessage(BaseModel): + role: str + # content is a plain string, or OpenAI "content parts" (list of dicts), or + # null for some tool/assistant messages. + content: Optional[Union[str, List[Any]]] = None + + +class ChatCompletionRequest(BaseModel): + messages: List[ChatMessage] + model: Optional[str] = MODEL_NAME + stream: bool = False + # Copilot's own conversation id (returned in earlier responses). Pass it back + # to continue that thread; omit it to start a fresh conversation. Outside + # OpenAI's schema, but standard clients can set it via extra_body. + conversation_id: Optional[str] = None + # Any other OpenAI fields (temperature, max_tokens, ...) are accepted and + # ignored — Copilot's consumer protocol doesn't expose those knobs.