diff --git a/.vscode/launch.json b/.vscode/launch.json index 30700f70..28d4fb57 100644 --- a/.vscode/launch.json +++ b/.vscode/launch.json @@ -2,39 +2,36 @@ "version": "0.2.0", "configurations": [ { - "name": "Frontend Debug (npm)", + "name": "Frontend (npm)", "type": "node-terminal", "request": "launch", "command": "npm run dev", "cwd": "${workspaceFolder}/frontend" }, { - "name": "Flask Debugger", - "type": "debugpy", - "request": "launch", - "module": "flask", - "env": { - "FLASK_APP": "docsgpt/app.py", - "PYTHONPATH": "${workspaceFolder}", - "FLASK_ENV": "development", - "FLASK_DEBUG": "1", - "FLASK_RUN_PORT": "7091", - "FLASK_RUN_HOST": "0.0.0.0" - - }, - "args": [ - "run", - "--no-debugger" - ], - "cwd": "${workspaceFolder}", + "name": "API (uvicorn)", + "type": "debugpy", + "request": "launch", + "module": "uvicorn", + "env": { + "PYTHONPATH": "${workspaceFolder}" + }, + "args": [ + "docsgpt.asgi:asgi_app", + "--host", + "127.0.0.1", + "--port", + "7091" + ], + "cwd": "${workspaceFolder}" }, { - "name": "Celery Debugger", + "name": "Celery worker", "type": "debugpy", "request": "launch", "module": "celery", "env": { - "PYTHONPATH": "${workspaceFolder}", + "PYTHONPATH": "${workspaceFolder}" }, "args": [ "-A", @@ -47,10 +44,10 @@ "cwd": "${workspaceFolder}" }, { - "name": "Dev Containers (Mongo + Redis)", + "name": "Dev services (Postgres + Redis)", "type": "node-terminal", "request": "launch", - "command": "docker compose -f deployment/docker-compose-dev.yaml up --build", + "command": "docker compose -f deployment/docker-compose-dev.yaml up", "cwd": "${workspaceFolder}" } ], @@ -58,9 +55,9 @@ { "name": "DocsGPT: Full Stack", "configurations": [ - "Frontend Debug (npm)", - "Flask Debugger", - "Celery Debugger" + "Frontend (npm)", + "API (uvicorn)", + "Celery worker" ], "presentation": { "group": "DocsGPT", @@ -68,4 +65,4 @@ } } ] -} \ No newline at end of file +} diff --git a/docs/content/Deploying/Development-Environment.mdx b/docs/content/Deploying/Development-Environment.mdx index 6c01879f..133c5531 100644 --- a/docs/content/Deploying/Development-Environment.mdx +++ b/docs/content/Deploying/Development-Environment.mdx @@ -119,7 +119,34 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a 5. **Run the Backend:** - For local development, run the ASGI composition under uvicorn. It serves the **whole** application, hot-reloads on source changes, and matches the production runtime: + One command runs the API and the worker from this checkout, each restarting when you save a file: + + ```bash + docsgpt dev + ``` + + Both run as children of that terminal, with their output interleaved and labelled, and Ctrl-C stops + them together. Useful flags: + + | Flag | What it does | + | --- | --- | + | `--ui` | also start the Vite dev server, so the whole app runs from one command | + | `--mock-llm` | run `scripts/mock_llm.py` and point DocsGPT at it, so no API key is needed | + | `--no-worker` | leave the worker to you, for instance when debugging it in your editor | + | `--no-reload` | do not restart anything on save | + | `--port` | serve the API somewhere other than 7091 | + + `docsgpt dev` is for a checkout. `docsgpt up --native`, by contrast, installs supervised services + that outlive the shell — see [Run it as services](/Deploying/Pip-Install#run-it-as-services-without-docker). + + + `docsgpt doctor` checks the things that usually break a new setup: whether PostgreSQL answers and + its schema matches this version, whether Redis answers, whether a model provider is configured, + and whether the port is free. Run it first when something does not start. + + + To run the two processes yourself instead, start the ASGI composition under uvicorn. It serves the + **whole** application, hot-reloads on source changes, and matches the production runtime: ```bash uvicorn docsgpt.asgi:asgi_app --host 0.0.0.0 --port 7091 --reload @@ -135,7 +162,7 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a But it serves **only** the WSGI Flask app and omits the native-async routes mounted on the ASGI shell in `docsgpt/asgi.py`: the `/mcp` FastMCP endpoint, the chat reconnect reader `GET /api/messages//events`, the notification stream `GET /api/events`, the remote-device command stream `GET /api/devices/sessions//events`, and artifact downloads `GET /api/artifacts//download`. Under `flask run` those paths return 404 — chat still works (`POST /stream` is a Flask route), but live notifications, stream auto-resume, paired devices and artifact downloads don't. Use `flask run` only when you don't need them. -6. **Start the Celery Worker:** +6. **Start the Celery Worker** (not needed if you used `docsgpt dev`)**:** Open a new terminal window (and activate your virtual environment if you used one). Start the Celery worker to handle background tasks: @@ -153,10 +180,14 @@ To run the DocsGPT backend locally, you'll need to set up a Python environment a **Running in Debugger (VSCode):** -For easier debugging, you can launch the Flask app and Celery worker directly from VSCode's debugger. +For easier debugging, you can launch the API and the Celery worker directly from VSCode's debugger. * Press Shift + Cmd + D (macOS) or Shift + Windows + D (Windows) to open the Run and Debug view. -* You should see configurations named "Flask" and "Celery". Select the desired configuration and click the "Start Debugging" button (green play icon). +* You should see configurations named "API (uvicorn)" and "Celery worker", and a compound "DocsGPT: Full Stack" that starts them with the frontend. Select one and click the "Start Debugging" button (green play icon). + +The API configuration runs the same ASGI app as production, so the routes mounted on the ASGI shell +work under the debugger. It deliberately runs without `--reload`: the reloader restarts the server in +a child process, which your breakpoints would not be attached to. ## 3. Start the Frontend @@ -207,3 +238,16 @@ To run the DocsGPT frontend locally, you'll need Node.js and npm (Node Package M This command will start the Vite development server. The frontend application will typically be accessible at [http://localhost:5173/](http://localhost:5173/). The terminal will display the exact URL where the frontend is running. With both the backend and frontend running, you should now have a fully functional DocsGPT development environment. You can access the application in your browser at [http://localhost:5173/](http://localhost:5173/) and start developing! + +## Working on two branches at once + +Each install keeps its own directory and its own services, so a second branch can run beside the +first as long as it gets its own port: + +```bash +docsgpt dev --port 7092 # a second checkout, second terminal +docsgpt up --native --dir ~/.docsgpt/review --port 7092 # or a second installed copy +``` + +A native install in another directory gets its own service names, so the two never write over each +other's units. `docsgpt status --dir ~/.docsgpt/review` reports on that one alone. diff --git a/docs/content/changelog.mdx b/docs/content/changelog.mdx index ab78c4ad..59fa4ae8 100644 --- a/docs/content/changelog.mdx +++ b/docs/content/changelog.mdx @@ -13,6 +13,17 @@ request, and [Upgrading](/upgrading) covers the steps an existing deployment has ## Unreleased +### A development loop in one command + +`docsgpt dev` runs this checkout's API and worker as children of one terminal, both restarting when +you save, with their output interleaved and Ctrl-C stopping them together. `--ui` adds the Vite dev +server and `--mock-llm` runs the bundled mock model, so a working loop needs no API key. +`docsgpt doctor` checks PostgreSQL, its schema version, Redis, the model provider and the port; +`docsgpt restart` bounces the services without touching settings; `docsgpt logs -f` now follows a +native install; and `docsgpt env set` applies itself to a running native install instead of asking +you to run `docsgpt up` again. See +[Setting up a development environment](/Deploying/Development-Environment). + ### Run DocsGPT without Docker `docsgpt up --native` runs the API and the worker as services on the machine itself, launchd on diff --git a/docsgpt/cli.py b/docsgpt/cli.py index 1ca2a187..d837642d 100644 --- a/docsgpt/cli.py +++ b/docsgpt/cli.py @@ -89,7 +89,17 @@ def _api(args: argparse.Namespace) -> int: if args.reload or sys.platform == "win32": import uvicorn - uvicorn.run("docsgpt.asgi:asgi_app", host=args.host, port=args.port, reload=args.reload) + from docsgpt.core.paths import package_dir + + # Watch the package, not the working directory: a checkout also holds .venv, node_modules + # and the data the app writes (indexes/, inputs/), which restarts the server mid-ingest. + uvicorn.run( + "docsgpt.asgi:asgi_app", + host=args.host, + port=args.port, + reload=args.reload, + reload_dirs=[str(package_dir())] if args.reload else None, + ) return 0 _gunicorn_application(_gunicorn_options(args.host, args.port, args.workers)).run() @@ -246,12 +256,33 @@ def _add_deploy_commands(commands) -> None: restore.add_argument("--timeout", type=int, default=300, help="seconds to wait for the API afterwards (default: 300)") + doctor = stack_command("doctor", "doctor", "check what this machine needs to run DocsGPT") + doctor.add_argument("--postgres-uri", help="check this database instead of the one in .env") + doctor.add_argument("--redis-url", help="check this Redis instead of the one in .env") + + restart = stack_command("restart", "restart", "restart the services, changing nothing else") + restart.add_argument("services", nargs="*", help="services to restart, e.g. api worker") + env = stack_command("env", "env", "show, get or set the stack's settings") env_actions = env.add_subparsers(dest="env_action", metavar="") get = env_actions.add_parser("get", help="print one setting") get.add_argument("key") - set_ = env_actions.add_parser("set", help="set settings (KEY=VALUE ...); run `docsgpt up` to apply") + set_ = env_actions.add_parser("set", help="set settings (KEY=VALUE ...)") set_.add_argument("pairs", nargs="+", metavar="KEY=VALUE") + set_.add_argument("--no-restart", dest="restart", action="store_false", + help="do not restart a running native install afterwards") + + dev = commands.add_parser("dev", help="run this checkout's API, worker and UI with reload") + dev.add_argument("--host", default=DEFAULT_HOST, help="interface for the API (default: localhost)") + dev.add_argument("--port", type=int, default=DEFAULT_PORT, help="port for the API (default: 7091)") + dev.add_argument("--ui", action="store_true", help="also run the frontend dev server") + dev.add_argument("--mock-llm", action="store_true", + help="run the mock LLM and point DocsGPT at it, so no API key is needed") + dev.add_argument("--no-worker", dest="worker", action="store_false", help="do not run the Celery worker") + dev.add_argument("--no-reload", dest="reload", action="store_false", + help="do not restart the API and worker when a file changes") + dev.add_argument("-l", "--loglevel", default="INFO", help="worker log level (default: INFO)") + dev.set_defaults(func=_deploy("dev"), deploy=True) def build_parser() -> argparse.ArgumentParser: diff --git a/docsgpt/deploy/commands.py b/docsgpt/deploy/commands.py index 0d95655f..8b1503a6 100644 --- a/docsgpt/deploy/commands.py +++ b/docsgpt/deploy/commands.py @@ -12,6 +12,7 @@ import socket import subprocess import sys import tempfile +import time import webbrowser from collections.abc import Callable, Mapping from dataclasses import dataclass @@ -623,12 +624,36 @@ def logs(args, context: Optional[Context] = None) -> int: else: print("(nothing logged yet)") if args.follow: - print("Following is not supported in native mode; use `tail -f` on the files above.", file=sys.stderr) + return _follow(logs_dir, wanted) return 0 options = (["--follow"] if args.follow else []) + (["--tail", str(args.tail)] if args.tail else []) return context.docker.compose(directory, "logs", *options, *args.services, check=False).returncode +def _follow(logs_dir: Path, services: list) -> int: + """Print new lines from each service's log until the terminal interrupts, prefixed by service.""" + handles: dict = {} + try: + while True: + for service in services: + if service not in handles: + path = logs_dir / f"{service}.log" + if not path.is_file(): + continue + handle = path.open("r", encoding="utf-8", errors="replace") + handle.seek(0, os.SEEK_END) + handles[service] = handle + for line in handles[service].readlines(): + print(f"{service:<6} | {line.rstrip()}") + sys.stdout.flush() + time.sleep(0.3) + except KeyboardInterrupt: + return 0 + finally: + for handle in handles.values(): + handle.close() + + def token(args, context: Optional[Context] = None) -> int: """Print the access token of a ``simple_jwt`` install.""" directory = stack.stack_dir(args.dir) @@ -681,10 +706,215 @@ def env(args, context: Optional[Context] = None) -> int: envfile.update(env_path, updates) except ValueError as exc: raise DeployError(str(exc)) from exc - print(f"Saved to {env_path}. Run `docsgpt up` to apply.") + print(f"Saved to {env_path}.") + directory = stack.stack_dir(args.dir) + if _mode(directory) == "native" and getattr(args, "restart", True): + context = context or Context.default(args) + services = context.service_manager() + names = _service_names(directory) + if any(services.is_running(name) for name in names): + for name in reversed(names): + services.stop(name) + for name in names: + services.start(name) + print("Restarted the services, so the change is live.") + return 0 + print("Run `docsgpt up` to apply.") return 0 +def dev(args, context: Optional[Context] = None) -> int: + """Run this checkout's API, worker and UI as children of this terminal.""" + from docsgpt.core import paths + from docsgpt.deploy import dev as dev_module + + checkout = paths.checkout_root() + if checkout is None: + raise DeployError( + "`docsgpt dev` runs the code in a source checkout, and this is an installed package. " + "Clone the repository and run it from there, or use `docsgpt up --native` to run this copy." + ) + if not _port_is_free(args.port): + raise DeployError( + f"port {args.port} is already in use, so the API cannot bind it. Stop what is on it " + f"(a previous `docsgpt dev`, or `docsgpt down` for an install), or pass --port." + ) + children = dev_module.plan(args, checkout) + print(f"DocsGPT from {checkout}") + for child in children: + print(f" {child.name:<6} {' '.join(child.command)}") + print(f"\nAPI http://{args.host}:{args.port}") + if getattr(args, "ui", False): + print(f"UI http://localhost:{dev_module.UI_PORT}") + print("Ctrl-C stops everything.\n") + return dev_module.run(children) + + +@dataclass +class Check: + """One line of ``docsgpt doctor``: what was looked at and what came back.""" + + name: str + level: str + detail: str + + +MARKS = {"ok": "ok ", "warn": "warn", "fail": "FAIL"} + + +def _migration_head() -> Optional[str]: + """The newest revision shipped with this package, or None when alembic cannot say.""" + try: + from alembic.config import Config + from alembic.script import ScriptDirectory + except ImportError: + return None + ini = Path(__file__).resolve().parents[1] / "alembic.ini" + if not ini.is_file(): + return None + config = Config(str(ini)) + config.set_main_option("script_location", str(ini.parent / "alembic")) + try: + return ScriptDirectory.from_config(config).get_current_head() + except Exception: # noqa: BLE001 - a broken script directory is a doctor finding, not a crash + return None + + +def _check_postgres(uri: Optional[str]) -> Check: + """Connect, and say whether the schema is the one this version expects.""" + if not uri: + return Check("postgres", "fail", "POSTGRES_URI is not set") + try: + import psycopg + except ImportError: + return Check("postgres", "fail", "the psycopg driver is not installed") + try: + with psycopg.connect(uri, connect_timeout=5) as connection, connection.cursor() as cursor: + cursor.execute("select current_setting('server_version')") + version = cursor.fetchone()[0] + cursor.execute("select to_regclass('public.alembic_version')") + applied = cursor.fetchone()[0] is not None + current = None + if applied: + cursor.execute("select version_num from alembic_version") + row = cursor.fetchone() + current = row[0] if row else None + except (psycopg.Error, OSError, ValueError) as exc: + return Check("postgres", "fail", f"cannot connect: {str(exc).strip()}") + head = _migration_head() + if not current: + return Check("postgres", "fail", f"PostgreSQL {version}, no schema yet; run `docsgpt migrate`") + if head and current != head: + return Check("postgres", "fail", f"PostgreSQL {version} at {current}, this version wants {head}; " + "run `docsgpt migrate`") + return Check("postgres", "ok", f"PostgreSQL {version}, schema at {current}") + + +def _check_redis(urls: Mapping[str, str]) -> Check: + """Ping every Redis the settings name; they are usually one server, three databases.""" + if not urls: + return Check("redis", "fail", "no Redis is configured (CELERY_BROKER_URL)") + try: + import redis + except ImportError: + return Check("redis", "fail", "the redis client is not installed") + for label, url in sorted(urls.items()): + try: + redis.Redis.from_url(url, socket_connect_timeout=3).ping() + except Exception as exc: # noqa: BLE001 - every client error here is the same finding + return Check("redis", "fail", f"{label} ({url}) does not answer: {str(exc).strip()}") + return Check("redis", "ok", f"answering on {len(urls)} database(s)") + + +def _check_provider(env: Mapping[str, str]) -> Check: + """Whether a model provider is set up well enough to answer a question.""" + provider = env.get("LLM_PROVIDER") or "docsgpt" + if provider == "docsgpt": + return Check("provider", "ok", "the DocsGPT public API (no key needed)") + if not (env.get("API_KEY") or env.get("OPENAI_API_KEY")): + return Check("provider", "fail", f"{provider} is configured but no API_KEY is set") + return Check("provider", "ok", f"{provider}{' at ' + env['OPENAI_BASE_URL'] if env.get('OPENAI_BASE_URL') else ''}") + + +def doctor(args, context: Optional[Context] = None) -> int: + """Check what DocsGPT needs on this machine, and say what is missing.""" + from docsgpt.core import paths + + if args.dir: + env_path = stack.stack_dir(args.dir) / ".env" + else: + try: + env_path = paths.env_file() + except FileNotFoundError as exc: + raise DeployError(str(exc)) from exc + env = envfile.read(env_path) + checks = [ + Check("settings", "ok" if env_path.is_file() else "warn", + f"{env_path}" if env_path.is_file() else f"{env_path} does not exist yet; defaults are in use"), + _check_postgres(args.postgres_uri or env.get("POSTGRES_URI")), + _check_redis({ + key: value for key, value in ( + ("broker", args.redis_url or env.get("CELERY_BROKER_URL")), + ("results", env.get("CELERY_RESULT_BACKEND")), + ("cache", env.get("CACHE_REDIS_URL")), + ) if value + }), + _check_provider(env), + ] + + port = int(env.get("DOCSGPT_PORT") or stack.DEFAULT_PORT) + if _port_is_free(port): + checks.append(Check("port", "ok", f"{port} is free")) + else: + checks.append(Check("port", "warn", f"{port} is in use, which is expected if DocsGPT is running")) + + directory = stack.stack_dir(args.dir) + if _mode(directory) == "native": + services = context.service_manager() if context else native.services_for_platform() + names = _service_names(directory) + running = [name for name in names if services.is_running(name)] + level = "ok" if len(running) == len(names) else "warn" + checks.append(Check("services", level, f"{len(running)} of {len(names)} running ({', '.join(names)})")) + + for check in checks: + print(f"[{MARKS[check.level]}] {check.name:<9} {check.detail}") + failed = [check for check in checks if check.level == "fail"] + if failed: + print(f"\n{len(failed)} problem(s) to fix before DocsGPT will work.", file=sys.stderr) + return 1 if failed else 0 + + +def _chosen_services(names: tuple[str, str], wanted: list) -> list: + """The services the user asked for, given either short names (api) or full ones.""" + if not wanted: + return list(names) + chosen = [] + for ask in wanted: + match = [name for name in names if name == ask or name.removeprefix("docsgpt-").startswith(ask)] + if not match: + raise DeployError(f"{ask!r} is not a service of this install; it has {', '.join(names)}.") + chosen.extend(match) + return chosen + + +def restart(args, context: Optional[Context] = None) -> int: + """Restart the services, changing nothing else.""" + context = context or Context.default(args) + directory = stack.stack_dir(args.dir) + if _installed(directory) is None: + return 1 + if _mode(directory) == "native": + services = context.service_manager() + chosen = _chosen_services(_service_names(directory), list(args.services)) + for name in reversed(chosen): + services.stop(name) + for name in chosen: + services.start(name) + print(f"Restarted {', '.join(chosen)}.") + return 0 + return context.docker.compose(directory, "restart", *args.services, check=False).returncode + + def _stack_image(env: Mapping[str, str]) -> str: """The image this install runs; the volume tars go through it, so nothing extra is pulled.""" tag = env.get("DOCSGPT_IMAGE_TAG") or "latest" diff --git a/docsgpt/deploy/dev.py b/docsgpt/deploy/dev.py new file mode 100644 index 00000000..a49dcf07 --- /dev/null +++ b/docsgpt/deploy/dev.py @@ -0,0 +1,214 @@ +"""``docsgpt dev``: this checkout's API, worker and UI as children of one terminal. + +``docsgpt up --native`` installs services meant to outlive the shell. Development wants the +opposite: processes rooted in the checkout, restarting when a file is saved, logging into one +terminal, and gone when Ctrl-C lands. This module decides which processes to run and supervises +them; nothing here imports the app itself. +""" + +from __future__ import annotations + +import os +import shlex +import shutil +import signal +import subprocess +import sys +import threading +import time +from dataclasses import dataclass, field +from pathlib import Path +from typing import Callable, Optional, TextIO + +from docsgpt.deploy.docker import DeployError + +MOCK_LLM_PORT = 8090 +UI_PORT = 5173 +STOP_GRACE = 10.0 + +# One colour per child so a glance at the terminal says who is talking. +COLOURS = {"api": "\033[36m", "worker": "\033[35m", "ui": "\033[32m", "llm": "\033[33m"} +RESET = "\033[0m" +WIDTH = 6 + + +@dataclass +class Child: + """One process ``docsgpt dev`` runs.""" + + name: str + command: list[str] + cwd: Path + env: dict[str, str] = field(default_factory=dict) + + +def watchfiles_available() -> bool: + """Whether the worker can be restarted on save; it arrives with uvicorn's standard extras.""" + try: + import watchfiles # noqa: F401 + except ImportError: + return False + return True + + +def _reloading_command(command: list[str], watched: Path) -> list[str]: + """``command`` under watchfiles, restarted when a Python file under ``watched`` changes.""" + return [sys.executable, "-m", "watchfiles", "--filter", "python", shlex.join(command), str(watched)] + + +def plan( + args, + checkout: Path, + *, + watching: Optional[bool] = None, + launcher: Optional[list[str]] = None, +) -> list[Child]: + """The children to run, in the order they should start.""" + from docsgpt.deploy import stack + + launcher = launcher or [sys.executable, "-m", "docsgpt"] + watching = watchfiles_available() if watching is None else watching + package = checkout / "docsgpt" + environment = {"DOCSGPT_HOME": str(checkout)} + children: list[Child] = [] + + if getattr(args, "mock_llm", False): + script = checkout / "scripts" / "mock_llm.py" + if not script.is_file(): + raise DeployError(f"{script} is missing, so there is no mock LLM to run.") + children.append( + Child( + name="llm", + command=[sys.executable, str(script), "--port", str(MOCK_LLM_PORT)], + cwd=checkout, + env=dict(environment), + ) + ) + # The children read these from the environment, so the checkout's .env is left alone. + chosen = stack.provider_settings( + "openai-compatible", model="mock", base_url=f"http://127.0.0.1:{MOCK_LLM_PORT}/v1" + ) + environment.update({key: value for key, value in chosen.items() if value is not None}) + + api = [*launcher, "api", "--host", args.host, "--port", str(args.port)] + if getattr(args, "reload", True): + api.append("--reload") + children.append(Child(name="api", command=api, cwd=checkout, env=dict(environment))) + + if getattr(args, "worker", True): + worker = [*launcher, "worker", "-l", getattr(args, "loglevel", "INFO")] + if getattr(args, "reload", True) and watching: + worker = _reloading_command(worker, package) + children.append(Child(name="worker", command=worker, cwd=checkout, env=dict(environment))) + + if getattr(args, "ui", False): + frontend = checkout / "frontend" + if not (frontend / "node_modules").is_dir(): + raise DeployError( + f"the frontend has no node_modules yet. Run `npm install --include=dev` in {frontend} " + "and try again, or leave --ui off." + ) + if not shutil.which("npm"): + raise DeployError("npm is not on PATH, so the frontend dev server cannot start.") + children.append(Child(name="ui", command=["npm", "run", "dev"], cwd=frontend, env=dict(environment))) + + return children + + +def _line(name: str, text: str, colour: bool) -> str: + """One output line, prefixed with the child that wrote it.""" + label = name.ljust(WIDTH) + if colour: + return f"{COLOURS.get(name, '')}{label}{RESET} | {text}" + return f"{label} | {text}" + + +def _pump(child: Child, process, out: TextIO, lock: threading.Lock, colour: bool) -> None: + """Copy one child's output to ``out``, a line at a time, prefixed.""" + stream = process.stdout + if stream is None: + return + for text in stream: + with lock: + out.write(_line(child.name, text.rstrip("\n"), colour)) + out.write("\n") + out.flush() + + +def _signal(process, number: int) -> None: + """Signal a child and, on POSIX, everything it started.""" + try: + if os.name == "nt": + process.terminate() + return + os.killpg(os.getpgid(process.pid), number) + except (ProcessLookupError, PermissionError, OSError): + pass + + +def _stop(running: list[tuple[Child, object]], grace: float, sleep: Callable[[float], None]) -> None: + """Interrupt the children, then insist if they are still there. + + A second Ctrl-C lands while this is waiting. It means "stop waiting", not "give up": the wait + ends and the children are killed, rather than the interrupt escaping and leaving them running. + """ + for _, process in running: + if process.poll() is None: + _signal(process, signal.SIGINT) + deadline = time.monotonic() + grace + while time.monotonic() < deadline and any(process.poll() is None for _, process in running): + try: + sleep(0.1) + except KeyboardInterrupt: + break + for _, process in running: + if process.poll() is None: + _signal(process, signal.SIGKILL) + + +def run( + children: list[Child], + *, + out: TextIO = sys.stdout, + spawn: Callable[..., object] = subprocess.Popen, + sleep: Callable[[float], None] = time.sleep, + grace: float = STOP_GRACE, + colour: Optional[bool] = None, +) -> int: + """Start the children and keep them running until one exits or the terminal interrupts.""" + colour = out.isatty() if colour is None else colour + lock = threading.Lock() + running: list[tuple[Child, object]] = [] + try: + for child in children: + process = spawn( + child.command, + cwd=str(child.cwd), + env={**os.environ, **child.env}, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + text=True, + bufsize=1, + # Its own session, so Ctrl-C reaches this process and the children are stopped in order. + start_new_session=os.name != "nt", + ) + running.append((child, process)) + threading.Thread(target=_pump, args=(child, process, out, lock, colour), daemon=True).start() + + while True: + for child, process in running: + code = process.poll() + if code is not None: + with lock: + out.write(_line(child.name, f"exited with {code}", colour)) + out.write("\n") + out.flush() + return code or 1 + sleep(0.2) + except KeyboardInterrupt: + with lock: + out.write("\nStopping ...\n") + out.flush() + return 0 + finally: + _stop(running, grace, sleep) diff --git a/tests/deploy/test_dev.py b/tests/deploy/test_dev.py new file mode 100644 index 00000000..a8344954 --- /dev/null +++ b/tests/deploy/test_dev.py @@ -0,0 +1,180 @@ +"""`docsgpt dev`: the checkout's processes as children of one terminal.""" + +import argparse +import io +import signal +import sys +from pathlib import Path + +import pytest + +from docsgpt.deploy import dev +from docsgpt.deploy.docker import DeployError + + +def _args(**overrides): + values = {"host": "127.0.0.1", "port": 7091, "ui": False, "mock_llm": False, + "worker": True, "reload": True, "loglevel": "INFO"} + values.update(overrides) + return argparse.Namespace(**values) + + +def _named(children): + return [child.name for child in children] + + +class FakeProcess: + """A child that produces the given lines and is done.""" + + def __init__(self, lines=(), code=None): + self.stdout = iter(list(lines)) + self.code = code + self.pid = -1 + + def poll(self): + return self.code + + def terminate(self): + self.code = self.code if self.code is not None else -15 + + +class TestPlan: + def test_the_api_and_worker_run_from_the_checkout(self, tmp_path): + children = dev.plan(_args(), tmp_path, watching=False) + assert _named(children) == ["api", "worker"] + api, worker = children + assert api.command[-6:-1] == ["api", "--host", "127.0.0.1", "--port", "7091"] + assert api.command[-1] == "--reload" + assert worker.command[-2:] == ["-l", "INFO"] + for child in children: + assert child.cwd == tmp_path + assert child.env["DOCSGPT_HOME"] == str(tmp_path), "the checkout is the data home, not ~/.docsgpt" + + def test_the_worker_restarts_on_save_when_watchfiles_is_there(self, tmp_path): + """Celery has no reloader of its own, so it is wrapped in one.""" + worker = dev.plan(_args(), tmp_path, watching=True)[1] + assert "watchfiles" in worker.command + assert str(tmp_path / "docsgpt") in worker.command, "it watches the package, not the whole checkout" + assert "docsgpt worker" in " ".join(worker.command) + + def test_without_watchfiles_the_worker_still_runs(self, tmp_path): + worker = dev.plan(_args(), tmp_path, watching=False)[1] + assert "watchfiles" not in " ".join(worker.command) + + def test_no_reload_leaves_both_alone(self, tmp_path): + children = dev.plan(_args(reload=False), tmp_path, watching=True) + assert "--reload" not in children[0].command + assert "watchfiles" not in " ".join(children[1].command) + + def test_no_worker(self, tmp_path): + assert _named(dev.plan(_args(worker=False), tmp_path, watching=False)) == ["api"] + + def test_the_mock_llm_starts_first_and_the_others_are_pointed_at_it(self, tmp_path): + """A dev loop that needs no API key: the mock has to be up before the API asks it anything.""" + (tmp_path / "scripts").mkdir() + (tmp_path / "scripts" / "mock_llm.py").write_text("", encoding="utf-8") + children = dev.plan(_args(mock_llm=True), tmp_path, watching=False) + assert _named(children) == ["llm", "api", "worker"] + api = children[1] + assert api.env["LLM_PROVIDER"] == "openai" + assert api.env["OPENAI_BASE_URL"] == f"http://127.0.0.1:{dev.MOCK_LLM_PORT}/v1" + assert api.env["API_KEY"], "the client wants some key, even a placeholder" + assert "OPENAI_BASE_URL" not in children[0].env, "the mock itself does not need pointing at itself" + + def test_a_missing_mock_llm_script_is_reported(self, tmp_path): + with pytest.raises(DeployError, match="mock LLM"): + dev.plan(_args(mock_llm=True), tmp_path, watching=False) + + def test_the_ui_needs_its_dependencies(self, tmp_path): + (tmp_path / "frontend").mkdir() + with pytest.raises(DeployError, match="node_modules"): + dev.plan(_args(ui=True), tmp_path, watching=False) + + def test_the_ui_needs_npm_on_path(self, tmp_path, monkeypatch): + (tmp_path / "frontend" / "node_modules").mkdir(parents=True) + monkeypatch.setattr(dev.shutil, "which", lambda name: None) + with pytest.raises(DeployError, match="npm"): + dev.plan(_args(ui=True), tmp_path, watching=False) + + def test_the_ui_runs_in_the_frontend_directory(self, tmp_path, monkeypatch): + (tmp_path / "frontend" / "node_modules").mkdir(parents=True) + monkeypatch.setattr(dev.shutil, "which", lambda name: "/usr/local/bin/npm") + children = dev.plan(_args(ui=True), tmp_path, watching=False) + ui = children[-1] + assert ui.name == "ui" + assert ui.command == ["npm", "run", "dev"] + assert ui.cwd == tmp_path / "frontend" + + +class TestRun: + def _spawn(self, processes): + made = iter(processes) + + def spawn(command, **kwargs): + return next(made) + + return spawn + + def test_output_is_prefixed_with_the_child_that_wrote_it(self, tmp_path): + out = io.StringIO() + children = [dev.Child(name="api", command=["true"], cwd=tmp_path)] + code = dev.run(children, out=out, spawn=self._spawn([FakeProcess(["hello\n"], code=0)]), + sleep=lambda _: None, colour=False) + assert "api | hello" in out.getvalue() + assert code == 1, "a child that ends by itself ends the session, however it exited" + + def test_a_failing_child_returns_its_code(self, tmp_path): + out = io.StringIO() + children = [dev.Child(name="api", command=["false"], cwd=tmp_path)] + code = dev.run(children, out=out, spawn=self._spawn([FakeProcess(code=2)]), + sleep=lambda _: None, colour=False) + assert code == 2 + assert "exited with 2" in out.getvalue() + + def test_an_interrupt_stops_quietly(self, tmp_path): + out = io.StringIO() + + def sleep(_): + raise KeyboardInterrupt + + children = [dev.Child(name="api", command=["sleep"], cwd=tmp_path)] + code = dev.run(children, out=out, spawn=self._spawn([FakeProcess()]), sleep=sleep, colour=False) + assert code == 0 + assert "Stopping" in out.getvalue() + + def test_a_second_interrupt_during_shutdown_still_kills(self, tmp_path, monkeypatch): + """Ctrl-C twice is what you press when it did not die; it must not leave children behind.""" + sent = [] + monkeypatch.setattr(dev, "_signal", lambda process, number: sent.append(number)) + + def sleep(_): + raise KeyboardInterrupt + + children = [dev.Child(name="api", command=["sleep"], cwd=tmp_path)] + code = dev.run(children, out=io.StringIO(), spawn=self._spawn([FakeProcess()]), + sleep=sleep, colour=False, grace=5) + assert code == 0 + assert signal.SIGINT in sent + assert signal.SIGKILL in sent, "the second interrupt escalates instead of escaping" + + def test_children_get_the_checkout_environment(self, tmp_path, monkeypatch): + seen = {} + + def spawn(command, **kwargs): + seen.update(kwargs) + return FakeProcess(code=0) + + monkeypatch.setenv("SOMETHING_ELSE", "kept") + children = [dev.Child(name="api", command=["x"], cwd=tmp_path, env={"DOCSGPT_HOME": str(tmp_path)})] + dev.run(children, out=io.StringIO(), spawn=spawn, sleep=lambda _: None, colour=False) + assert seen["env"]["DOCSGPT_HOME"] == str(tmp_path) + assert seen["env"]["SOMETHING_ELSE"] == "kept", "the shell's environment is kept, not replaced" + assert seen["cwd"] == str(tmp_path) + assert seen["start_new_session"] is (sys.platform != "win32") + + +class TestReloadingCommand: + def test_an_interpreter_path_with_spaces_survives(self): + command = dev._reloading_command(["/opt/my venv/bin/python", "-m", "docsgpt", "worker"], Path("/srv/pkg")) + assert "'/opt/my venv/bin/python' -m docsgpt worker" in command + assert command[-1] == "/srv/pkg" diff --git a/tests/deploy/test_doctor.py b/tests/deploy/test_doctor.py new file mode 100644 index 00000000..1309d512 --- /dev/null +++ b/tests/deploy/test_doctor.py @@ -0,0 +1,167 @@ +"""`docsgpt doctor`, `restart`, following native logs, and settings that apply themselves.""" + +import pytest + +from docsgpt.deploy import commands, envfile +from docsgpt.deploy.docker import DeployError + +from .test_commands import FakeDocker, _context, _run +from .test_native import FakeServices, _names, _native_context + + +def _installed_native(tmp_path, services): + argv = ["up", "--native", "--dir", str(tmp_path), "--yes", "--postgres-uri", "postgresql://localhost/d"] + assert _run(argv, _native_context(services)) == 0 + + +class TestRestart: + def test_it_stops_and_starts_both_without_touching_settings(self, tmp_path): + services = FakeServices() + _installed_native(tmp_path, services) + before = (tmp_path / ".env").read_text(encoding="utf-8") + services.started.clear() + services.stopped.clear() + + assert _run(["restart", "--dir", str(tmp_path)], _native_context(services)) == 0 + assert services.stopped == list(reversed(_names(tmp_path))), "the worker goes down first" + assert services.started == list(_names(tmp_path)) + assert (tmp_path / ".env").read_text(encoding="utf-8") == before + + def test_one_service_by_its_short_name(self, tmp_path): + services = FakeServices() + _installed_native(tmp_path, services) + services.started.clear() + assert _run(["restart", "api", "--dir", str(tmp_path)], _native_context(services)) == 0 + assert services.started == [_names(tmp_path)[0]] + + def test_a_name_this_install_does_not_have(self, tmp_path): + services = FakeServices() + _installed_native(tmp_path, services) + with pytest.raises(DeployError, match="not a service"): + _run(["restart", "frontend", "--dir", str(tmp_path)], _native_context(services)) + + def test_a_docker_install_restarts_its_containers(self, tmp_path): + assert _run(["up", "--yes", "--dir", str(tmp_path)], _context()) == 0 + docker = FakeDocker() + assert _run(["restart", "--dir", str(tmp_path)], _context(docker)) == 0 + assert ["restart"] in [args for _, args in docker.calls] + + +class TestEnvApplies: + def test_a_running_native_install_restarts_itself(self, tmp_path, capsys): + """The old advice was to run `docsgpt up` again, which reruns migrations to change one value.""" + services = FakeServices() + _installed_native(tmp_path, services) + services.started.clear() + + argv = ["env", "--dir", str(tmp_path), "set", "LLM_NAME=gpt-4o"] + assert _run(argv, _native_context(services)) == 0 + assert envfile.read(tmp_path / ".env")["LLM_NAME"] == "gpt-4o" + assert services.started == list(_names(tmp_path)) + assert "Restarted" in capsys.readouterr().out + + def test_no_restart_leaves_the_services_alone(self, tmp_path, capsys): + services = FakeServices() + _installed_native(tmp_path, services) + services.started.clear() + + argv = ["env", "--dir", str(tmp_path), "set", "LLM_NAME=gpt-4o", "--no-restart"] + assert _run(argv, _native_context(services)) == 0 + assert services.started == [] + assert "docsgpt up" in capsys.readouterr().out + + def test_a_stopped_install_is_not_started_by_a_settings_change(self, tmp_path): + services = FakeServices() + _installed_native(tmp_path, services) + for name in _names(tmp_path): + services.stop(name) + services.started.clear() + + argv = ["env", "--dir", str(tmp_path), "set", "LLM_NAME=gpt-4o"] + assert _run(argv, _native_context(services)) == 0 + assert services.started == [], "changing a setting does not start a stopped install" + + +class TestFollowLogs: + def test_it_prints_lines_written_after_it_started(self, tmp_path, capsys, monkeypatch): + logs = tmp_path / "logs" + logs.mkdir() + (logs / "api.log").write_text("old line\n", encoding="utf-8") + + rounds = {"n": 0} + + def sleep(_): + rounds["n"] += 1 + if rounds["n"] == 1: + (logs / "api.log").open("a", encoding="utf-8").write("new line\n") + return + raise KeyboardInterrupt + + monkeypatch.setattr(commands.time, "sleep", sleep) + assert commands._follow(logs, ["api"]) == 0 + printed = capsys.readouterr().out + assert "api | new line" in printed + assert "old line" not in printed, "it starts at the end, like tail -f" + + +class TestChecks: + def test_the_public_api_needs_no_key(self): + check = commands._check_provider({"LLM_PROVIDER": "docsgpt"}) + assert check.level == "ok" + + def test_a_provider_without_a_key_is_a_problem(self): + check = commands._check_provider({"LLM_PROVIDER": "openai"}) + assert check.level == "fail" + assert "API_KEY" in check.detail + + def test_a_provider_with_a_key_and_a_base_url(self): + check = commands._check_provider( + {"LLM_PROVIDER": "openai", "API_KEY": "x", "OPENAI_BASE_URL": "http://localhost:8090/v1"} + ) + assert check.level == "ok" + assert "8090" in check.detail + + def test_services_are_named_for_the_install(self, tmp_path): + names = _names(tmp_path) + assert commands._chosen_services(names, []) == list(names) + assert commands._chosen_services(names, ["worker"]) == [names[1]] + assert commands._chosen_services(names, [names[0]]) == [names[0]] + + +class TestDoctor: + def _only(self, monkeypatch, postgres, redis): + monkeypatch.setattr(commands, "_check_postgres", lambda uri: postgres) + monkeypatch.setattr(commands, "_check_redis", lambda urls: redis) + + def test_it_reports_every_check_and_succeeds_when_they_pass(self, tmp_path, capsys, monkeypatch): + self._only( + monkeypatch, + commands.Check("postgres", "ok", "PostgreSQL 16.2, schema at 0031"), + commands.Check("redis", "ok", "answering on 3 database(s)"), + ) + (tmp_path / ".env").write_text("LLM_PROVIDER=docsgpt\n", encoding="utf-8") + assert _run(["doctor", "--dir", str(tmp_path)], _context()) == 0 + out = capsys.readouterr().out + assert "postgres" in out and "redis" in out and "provider" in out + + def test_a_failing_check_makes_it_exit_one(self, tmp_path, capsys, monkeypatch): + self._only( + monkeypatch, + commands.Check("postgres", "fail", "cannot connect: refused"), + commands.Check("redis", "ok", "answering"), + ) + (tmp_path / ".env").write_text("LLM_PROVIDER=docsgpt\n", encoding="utf-8") + assert _run(["doctor", "--dir", str(tmp_path)], _context()) == 1 + captured = capsys.readouterr() + assert "FAIL" in captured.out + assert "1 problem" in captured.err + + def test_it_says_which_settings_file_it_read(self, tmp_path, capsys, monkeypatch): + self._only( + monkeypatch, + commands.Check("postgres", "ok", "fine"), + commands.Check("redis", "ok", "fine"), + ) + (tmp_path / ".env").write_text("LLM_PROVIDER=docsgpt\n", encoding="utf-8") + _run(["doctor", "--dir", str(tmp_path)], _context()) + assert str(tmp_path / ".env") in capsys.readouterr().out diff --git a/tests/test_cli.py b/tests/test_cli.py index dcad465e..bdb2977d 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -10,6 +10,7 @@ import click import pytest from docsgpt import cli +from docsgpt.core.paths import package_dir from docsgpt.version import __version__ @@ -126,7 +127,28 @@ class TestApi: uvicorn = types.SimpleNamespace(run=MagicMock()) monkeypatch.setitem(sys.modules, "uvicorn", uvicorn) assert cli.main(["api", "--reload", "--host", "127.0.0.1"]) == 0 - uvicorn.run.assert_called_once_with("docsgpt.asgi:asgi_app", host="127.0.0.1", port=7091, reload=True) + uvicorn.run.assert_called_once_with( + "docsgpt.asgi:asgi_app", host="127.0.0.1", port=7091, reload=True, + reload_dirs=[str(package_dir())], + ) + + def test_reload_watches_the_package_not_the_working_directory(self, monkeypatch, tmp_path): + """A checkout also holds .venv, node_modules and the indexes and inputs the app writes to, + so watching the working directory restarts the server mid-ingest.""" + uvicorn = types.SimpleNamespace(run=MagicMock()) + monkeypatch.setitem(sys.modules, "uvicorn", uvicorn) + monkeypatch.chdir(tmp_path) + assert cli.main(["api", "--reload"]) == 0 + watched = uvicorn.run.call_args.kwargs["reload_dirs"] + assert watched == [str(package_dir())] + assert str(tmp_path) not in watched + + def test_without_reload_nothing_is_watched(self, monkeypatch): + uvicorn = types.SimpleNamespace(run=MagicMock()) + monkeypatch.setitem(sys.modules, "uvicorn", uvicorn) + monkeypatch.setattr(sys, "platform", "win32") + assert cli.main(["api"]) == 0 + assert uvicorn.run.call_args.kwargs["reload_dirs"] is None class TestWorker: