commit 78a2043968e18b6be4ef17bbdfc67dd4edb8f1f1 Author: Developer Date: Thu Sep 10 16:15:17 2026 +0000 Add .gitignore and initial project files diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..232d19a --- /dev/null +++ b/.gitignore @@ -0,0 +1,127 @@ +# ============================================================ +# .gitignore for tmux-shot (Python project via uv/hatchling) +# Generated: 2026-09-10 +# ============================================================ + +# ---- Python / Build Artifacts ---- +__pycache__/ +*.py[cod] +*$py.class +*.so +*.egg-info/ +*.egg +dist/ +build/ +_build/ +*.whl + +# Hatch / setuptools build artifacts +.hatch/ +.hatch-temp/ + +# ---- uv (package manager) ---- +uv.lock +.venv/ +.venv.* +!.venv/.gitkeep + +# ---- IDEs ---- +# VS Code +.vscode/ +.vscode-* +*.code-workspace + +# PyCharm / JetBrains +.idea/ +*.iml +*.iws +*.ipr +.idea_modules/ + +# Eclipse +.project +.classpath +.settings/ + +# Sublime Text +*.sublime-project +*.sublime-workspace + +# Vim +*.swp +*.swo +*~ +.vimrc.local + +# ---- OS-generated files ---- +# macOS +.DS_Store +.AppleDouble +.LSOverride +Icon? +._* +Thumbs.db +Desktop.ini + +# Windows +Thumbs.db +ehthumbs.db +Desktop.ini +$RECYCLE.BIN/ +*.lnk +*.dll +*.exe +*.sys +*.bak +*.tmp +*.temp + +# Linux / GNOME +.Trash-* +.gvfs + +# ---- Environment & Secrets ---- +.env +.env.* +!.env.example +secrets/ +credentials/ +*.key +*.pem +*.p12 +*.jks + +# ---- Logs & Debug ---- +*.log +logs/ +dpc.log +*.out + +# ---- Test Artifacts ---- +.pytest_cache/ +.coverage +htmlcov/ +coverage.json +*.cover +.tox/ +noxfile.py/.tox/ + +# ---- Jupyter Notebooks ---- +.ipynb_checkpoints/ +*.ipynb + +# ---- Data / Output directories (project-specific) ---- +out/ +data/ +outputs/ + +# ---- Temporary files ---- +tmp/ +temp/ +*.tmp +*.temp + +# ---- Docker (if added later) ---- +.dockerenv +docker-compose.override.yml +*.pid diff --git a/README.md b/README.md new file mode 100644 index 0000000..c3e325a --- /dev/null +++ b/README.md @@ -0,0 +1,193 @@ +# tmux-shot + +Drive a TUI application inside tmux from Python, and capture pixel-accurate +PNG screenshots of it — no X server, no headless browser, no external +binaries. Pure `tmux` + `pyte` (terminal emulator) + `Pillow`. + +Built for AI-driven workflows: an agent can create a session, type into it, +capture a PNG, look at it, decide what to do next, and repeat. + +## How it works + +1. `tmux_shot.TmuxApp` creates/attaches a tmux session running your TUI and + sends keystrokes to it (`libtmux` under the hood). +2. `tmux capture-pane -e -p` dumps the pane's exact character grid, + including ANSI color/style escape codes. +3. `pyte.Screen` replays that byte stream into an in-memory terminal buffer + (one cell per character, with fg/bg/bold/etc.). +4. `render.py` draws that buffer cell-by-cell with Pillow into a PNG using a + monospace font. + +If the `freeze` CLI (https://github.com/charmbracelet/freeze) is installed, +you can use `render_with_freeze()` instead for nicer-looking output (window +chrome, better font shaping) — it's optional, not a hard dependency. + +## Install + +```bash +uv sync # or: pip install -e . +``` + +## Quick start + +```python +from tmux_shot import TmuxApp + +app = TmuxApp("demo", command="python3 examples/demo_tui.py", width=100, height=30) +app.screenshot("out/1.png") +app.send_keys("j") # drive the app +app.screenshot("out/2.png") +app.send_keys("q") +app.kill() +``` + +Or from the CLI: + +```bash +tmux-shot new demo --command "top" --width 120 --height 40 +tmux-shot shot demo out.png +tmux-shot send demo "q" +tmux-shot kill demo +``` + +## Demo TUI + +`examples/demo_tui.py` is a self-contained curses app (arrow keys to move a +cursor, `q` to quit) used to exercise the pipeline without depending on any +other installed program. + +```bash +uv run python examples/run_demo.py +``` + +produces `out/demo_*.png` frames showing the cursor moving and colors +rendering correctly. + +## AI-driven "photo session" mode + +`tmux-agent` is its own binary (also reachable as `tmux-shot agent`): give it +a plan in plain English and it drives the app itself (arrow keys, waits, +screenshots) via OpenAI-style function calling, looking at each screenshot as +it goes. + +It talks to [devplace.net](https://devplace.net)'s OpenAI-compatible gateway +(`https://devplace.net/openai/v1`, model `molodetz`) by default. Set your key +first: + +```bash +export DEVPLACE_API_KEY= +``` + +The prompt is the only required argument — everything else has a sensible +default: + +```bash +uv run tmux-agent "open the app, find the settings panel, screenshot it" +``` + +Omit the prompt and it asks interactively instead. Every knob from +`tmux-shot`'s other commands is still available when you need it: + +```bash +uv run tmux-agent \ + --session myapp --command "myapp" --width 120 --height 40 \ + --delay 1.5 --max-steps 40 --out-dir out/session1 --gif out/session1.gif \ + "Open the settings menu, navigate to Network, screenshot it as \ + 'network-panel', then go back to the main menu and screenshot that \ + as 'home', then finish." +``` + +Tools exposed to the model: `send_keys` (tmux key names like `Up`/`Enter`/ +`C-c`, or literal text), `wait`, `read_text` (cheap text-only state check), +`screenshot` (saves + shows the model the PNG), and `finish`. `--delay` +controls the automatic pause after every `send_keys` so the TUI has time to +redraw before the next look; the model can also request extra waits itself. +Swap `--model`/`--base-url`/`--app-reference` to point at a different +OpenAI-compatible backend. + +## "N minutes of X" — timed sessions with a guaranteed GIF/video + +Asking for "a gif of 3 minutes of you doing X" doesn't map onto a normal +step-count loop, and a model can misread "video" as "produce a video file" +and just refuse. Two things fix this: + +1. The system prompt is explicit that "video"/"gif"/"movie"/"N minutes of X" + requests are fully in scope — the model's job is to keep taking real + actions (`send_keys`) and periodic `screenshot`s, not to produce a file + itself. +2. `--duration SECONDS` makes the guarantee mechanical instead of relying on + the model cooperating: a background timer (stdlib `threading`, no extra + dependency) takes a screenshot every `--interval` seconds for the entire + requested duration, regardless of what the model does. If the model + stalls or refuses partway through, the run keeps waiting out the clock — + letting the ticker keep capturing — instead of ending early, so a 3-minute + request always produces roughly 3 minutes of real footage. + +```bash +uv run tmux-agent \ + --duration 180 --interval 5 --max-steps 80 \ + --command "grok --cwd /tmp/scratch --always-approve" \ + --gif out/vibecode.gif \ + "actively vibecode something small and fun with grok the whole session" +``` + +`--always-approve` (a `grok` flag, not ours) is worth knowing about here: if +the target app itself asks for permission before editing files or running +commands, an unattended agent has no one to click "yes" for it and will +stall. Only use it against a disposable scratch directory (`--cwd +/tmp/scratch` above) — it removes the safety net that normally stops a +misbehaving app from doing something destructive. + +## Testing + +```bash +uv run pytest +``` + +- `tests/test_render.py`, `tests/test_tmux_control.py`, `tests/test_cli.py` — core + pipeline (colors, reverse video, `send_keys` correctness, screenshots, CLI + round-trips). No network, no API key needed. +- `tests/test_agent.py` — exercises the full LLM tool-calling loop + (`send_keys`/`wait`/`read_text`/`screenshot`/`finish`, image follow-up + messages, the max-steps safety cap) against a scripted fake OpenAI client — + validates the agent harness itself without spending real API calls. +- `tests/test_grok_target.py` — a real-world capability probe against the + `grok` CLI (a third-party TUI), auto-skipped if `grok` isn't installed. + Read-only by design: only ever sends arrow keys, never Enter, so no prompt + is ever submitted to grok's own model. Produces an actual animated GIF of + its welcome screen as a working example of the "movie" pipeline below. + +`TmuxApp.wait_for(text, timeout=10)` polls the pane until a marker string +appears — use it instead of a fixed `sleep()` before screenshotting a TUI +whose startup/redraw time varies (this was discovered as real flakiness +while probing `grok`: its splash screen can take anywhere from ~2s to ~7s +depending on an update check). + +## Making a "movie" (animated GIF) from a sequence of screenshots + +```python +from tmux_shot import frames_to_gif + +frames_to_gif(["01_a.png", "02_b.png", "03_c.png"], "session.gif", duration_ms=500) +``` + +or from the CLI: + +```bash +tmux-shot gif out/01_a.png out/02_b.png out/03_c.png out/session.gif --duration-ms 500 +``` + +`tmux-agent --gif out/session.gif "..."` stitches every screenshot the agent +took during that run into one GIF automatically. + +## Notes for automating this with an AI agent + +- `TmuxApp.screenshot()` returns the path it wrote — feed that straight to a + vision-capable model. +- `TmuxApp.capture_text()` gives you the plain-text grid (no image) when the + agent just needs to read state cheaply instead of "looking" at a picture. +- Sessions are named, so multiple agents/tasks can run against independent + tmux sessions concurrently without colliding. +- Everything is idempotent: `TmuxApp(name, command=...)` attaches to an + existing session with that name instead of erroring if it's already + running. diff --git a/examples/demo_tui.py b/examples/demo_tui.py new file mode 100644 index 0000000..ee7d23e --- /dev/null +++ b/examples/demo_tui.py @@ -0,0 +1,39 @@ +"""Self-contained curses TUI used to exercise the tmux-shot pipeline. + +Arrow keys move the '@' cursor, 'q' quits. No external dependencies beyond +the standard library, so the demo works without any other TUI installed. +""" +import curses + + +def main(stdscr: "curses._CursesWindow") -> None: + curses.curs_set(0) + curses.start_color() + curses.use_default_colors() + for i in range(1, 8): + curses.init_pair(i, i, -1) + + y, x = 5, 10 + while True: + stdscr.erase() + stdscr.addstr(0, 0, "tmux-shot demo TUI -- arrows to move, q to quit", curses.A_BOLD) + for i in range(1, 8): + stdscr.addstr(2, i * 4, f" {i} ", curses.color_pair(i) | curses.A_REVERSE) + stdscr.addstr(y, x, "@", curses.color_pair(2) | curses.A_BOLD) + stdscr.refresh() + + ch = stdscr.getch() + if ch in (ord("q"), ord("Q")): + break + elif ch == curses.KEY_UP: + y = max(3, y - 1) + elif ch == curses.KEY_DOWN: + y += 1 + elif ch == curses.KEY_LEFT: + x = max(0, x - 1) + elif ch == curses.KEY_RIGHT: + x += 1 + + +if __name__ == "__main__": + curses.wrapper(main) diff --git a/examples/run_demo.py b/examples/run_demo.py new file mode 100644 index 0000000..1b4fde2 --- /dev/null +++ b/examples/run_demo.py @@ -0,0 +1,28 @@ +"""Drives examples/demo_tui.py inside tmux and captures a couple of PNG frames. + +Run with: uv run python examples/run_demo.py +""" +import time +from pathlib import Path + +from tmux_shot import TmuxApp + +ROOT = Path(__file__).resolve().parent.parent +OUT = ROOT / "out" +OUT.mkdir(exist_ok=True) +DEMO_SCRIPT = Path(__file__).resolve().parent / "demo_tui.py" + +app = TmuxApp("tmux_shot_demo", command=f"python3 {DEMO_SCRIPT}", width=80, height=24) +time.sleep(0.3) # let curses draw the first frame +app.screenshot(OUT / "demo_1.png") + +for key in ("Down", "Down", "Right", "Right", "Right"): + app.send_keys(key, enter=False) +time.sleep(0.2) +app.screenshot(OUT / "demo_2.png") + +app.send_keys("q") +app.kill() + +print(f"wrote {OUT / 'demo_1.png'}") +print(f"wrote {OUT / 'demo_2.png'}") diff --git a/pyproject.toml b/pyproject.toml new file mode 100644 index 0000000..ad197db --- /dev/null +++ b/pyproject.toml @@ -0,0 +1,35 @@ +[project] +name = "tmux-shot" +version = "0.1.0" +description = "Drive a TUI inside tmux from Python and capture pixel-accurate PNG screenshots of it." +readme = "README.md" +requires-python = ">=3.9" +dependencies = [ + "libtmux>=0.37", + "pyte>=0.8.2", + "pillow>=10.0", + "click>=8.1", + "openai>=2.48.0", +] + +[project.scripts] +tmux-shot = "tmux_shot.cli:main" +tmux-agent = "tmux_shot.agent_cli:main" + +[build-system] +requires = ["hatchling"] +build-backend = "hatchling.build" + +[tool.hatch.build.targets.wheel] +packages = ["src/tmux_shot"] + +[tool.uv] +package = true + +[dependency-groups] +dev = [ + "pytest>=8.4.2", +] + +[tool.pytest.ini_options] +testpaths = ["tests"] diff --git a/src/tmux_shot/__init__.py b/src/tmux_shot/__init__.py new file mode 100644 index 0000000..982c12e --- /dev/null +++ b/src/tmux_shot/__init__.py @@ -0,0 +1,4 @@ +from .render import frames_to_gif, render_ansi_to_png +from .tmux_control import TmuxApp + +__all__ = ["TmuxApp", "render_ansi_to_png", "frames_to_gif"] diff --git a/src/tmux_shot/agent.py b/src/tmux_shot/agent.py new file mode 100644 index 0000000..88420bc --- /dev/null +++ b/src/tmux_shot/agent.py @@ -0,0 +1,623 @@ +from __future__ import annotations + +import base64 +import json +import os +import re +import threading +import time +from pathlib import Path +from typing import Any, Optional + +import click +from openai import APIConnectionError, APITimeoutError, InternalServerError, OpenAI, RateLimitError + +from .tmux_control import TmuxApp + +_TRANSIENT_API_ERRORS = (APIConnectionError, APITimeoutError, InternalServerError, RateLimitError) + +DEFAULT_BASE_URL = "https://devplace.net/openai/v1" +DEFAULT_MODEL = "molodetz" +DEFAULT_APP_REFERENCE = "tmux-shot" + +TOOLS: list[dict] = [ + { + "type": "function", + "function": { + "name": "send_keys", + "description": ( + "Send keystrokes to the TUI running in the tmux pane. Use tmux key " + "names for special keys (Up, Down, Left, Right, Enter, Tab, Escape, " + "C-c, etc.) or literal=true to type plain text verbatim." + ), + "parameters": { + "type": "object", + "properties": { + "keys": {"type": "string", "description": "Keys or text to send."}, + "literal": { + "type": "boolean", + "description": "Send exactly as typed text instead of interpreting tmux key names.", + }, + "enter": { + "type": "boolean", + "description": "Send Enter afterwards.", + }, + }, + "required": ["keys"], + }, + }, + }, + { + "type": "function", + "function": { + "name": "wait", + "description": "Pause for the TUI to redraw or animate before doing anything else.", + "parameters": { + "type": "object", + "properties": { + "seconds": {"type": "number", "description": "How long to wait, in seconds."}, + }, + "required": ["seconds"], + }, + }, + }, + { + "type": "function", + "function": { + "name": "read_text", + "description": "Read the pane's current plain-text contents (cheap, no image) to check exact state.", + "parameters": {"type": "object", "properties": {}}, + }, + }, + { + "type": "function", + "function": { + "name": "screenshot", + "description": ( + "Capture a PNG screenshot of the pane's current state and view it. " + "Use this to look at the TUI before deciding the next action, and to " + "save the specific shots the user asked for." + ), + "parameters": { + "type": "object", + "properties": { + "label": { + "type": "string", + "description": "Short filename-safe label for this screenshot, e.g. 'settings-panel'.", + }, + }, + "required": ["label"], + }, + }, + }, + { + "type": "function", + "function": { + "name": "finish", + "description": "Call this once every step of the requested photo session has been completed.", + "parameters": { + "type": "object", + "properties": { + "summary": { + "type": "string", + "description": "What was done and which screenshots were taken.", + }, + }, + "required": ["summary"], + }, + }, + }, +] + +PLAN_TOOL: dict = { + "type": "function", + "function": { + "name": "propose_plan", + "description": ( + "Break a long task into an ordered list of short, self-contained chunks. Each " + "chunk is later executed by a separate fresh turn that only sees that chunk's " + "description plus a one-line note about what happened before it -- not the full " + "history -- so make every chunk understandable on its own." + ), + "parameters": { + "type": "object", + "properties": { + "chunks": { + "type": "array", + "items": {"type": "string"}, + "description": "Ordered, concrete, independently-actionable chunk descriptions.", + } + }, + "required": ["chunks"], + }, + }, +} + +_REQUIRED_ARGS: dict[str, tuple[str, ...]] = { + "send_keys": ("keys",), + "wait": ("seconds",), + "screenshot": ("label",), + "read_text": (), + "finish": (), +} + +SYSTEM_PROMPT = """You are driving a text-based terminal application (a TUI) running \ +inside a tmux pane, to carry out a "photo session" the user describes: a sequence of \ +navigation/typing actions, optionally ending in one or more screenshots of specific \ +screens/panels. + +You can only interact with the app through the provided tools: +- send_keys: send keystrokes (tmux key names like Up/Down/Left/Right/Enter/Tab/Escape/C-c, \ +or literal text when literal=true) -- this is how you type, navigate menus, run commands, \ +or write code inside the app. This is your only way to actually DO anything. +- wait: pause for redraws/animations +- read_text: cheaply read the pane's plain text to check exact state without spending on vision +- screenshot: capture and view a PNG of the current pane, saved under the given label +- finish: call once you've completed everything the user asked for + +IMPORTANT -- if the user asks for a "video", "gif", "movie", "recording", or "N minutes of you +doing X": you cannot produce a video file yourself, and you must NEVER refuse or say you can't +help with it. That request is fully achievable and is exactly what you're for -- it just doesn't +map onto a single tool call. It means: actually perform the requested activity for the requested +duration using send_keys/wait (type real commands, write real code, navigate the real app -- do +not just sit idle or take one screenshot and quit), while calling screenshot periodically along +the way to capture frames. The calling program automatically stitches every screenshot you take +into the final GIF/video -- your only job is to keep taking real actions and periodic screenshots +until the time is up, then call finish. Treat a duration request as a minimum amount of real, +varied activity to produce, not something to complete in one or two tool calls. + +Always take at least one screenshot before finishing. Keep replies brief, then call a tool. \ +Follow the user's instructions literally and in order — do not skip or shortcut later steps of \ +a multi-step plan just because an earlier one produced a screenshot. If the app doesn't respond \ +as expected, use read_text or screenshot to check the actual state before trying again. You MUST \ +call finish as your last tool call once (and only once) every step the user asked for is done; \ +never stop by simply returning a message with no tool call, and never stop just because you \ +personally can't produce a literal video/audio/binary file -- the tools already handle that.""" + + +def _b64_png(path: Path) -> str: + return base64.b64encode(path.read_bytes()).decode("ascii") + + +def _safe_label(label: str) -> str: + cleaned = "".join(c if c.isalnum() or c in "-_" else "-" for c in label).strip("-") + return cleaned or "shot" + + +_NUMBERED_ITEM_RE = re.compile(r"(?:^|\s)(\d+)[.)]\s+") + + +def _split_run_on_chunk(text: str) -> list[str]: + """Defensive fallback for when the model bundles several numbered steps into one plan + array element ("1. do X. 2. do Y. 3. do Z.") instead of separate elements as instructed -- + a known, observed failure mode of structured output compliance. Splits on the model's own + numbering; returns [text] unchanged if it doesn't look like a bundled list.""" + matches = list(_NUMBERED_ITEM_RE.finditer(text)) + if len(matches) < 2: + return [text] + parts = [] + for idx, m in enumerate(matches): + start = m.end() + end = matches[idx + 1].start() if idx + 1 < len(matches) else len(text) + part = text[start:end].strip() + if part: + parts.append(part) + return parts if len(parts) >= 2 else [text] + + +class AgentRunner: + def __init__( + self, + app: TmuxApp, + client: OpenAI, + model: str, + out_dir: Path, + delay: float, + max_steps: int, + duration: Optional[float] = None, + screenshot_interval: float = 5.0, + chunk_seconds: float = 45.0, + ) -> None: + self.app = app + self.client = client + self.model = model + self.out_dir = out_dir + self.delay = delay + self.max_steps = max_steps + self.duration = duration + self.screenshot_interval = screenshot_interval + self.chunk_seconds = chunk_seconds + self.shots: list[str] = [] + self._step_no = 0 + self._lock = threading.Lock() + self._deadline: Optional[float] = None + self._ticker_stop: Optional[threading.Event] = None + self._ticker_thread: Optional[threading.Thread] = None + + def _next_shot_path(self, label: str) -> Path: + with self._lock: + self._step_no += 1 + n = self._step_no + return self.out_dir / f"{n:03d}_{_safe_label(label)}.png" + + def _take_auto_shot(self) -> None: + path = self._next_shot_path("auto") + try: + self.app.screenshot(path) + except Exception as exc: # pane may be transiently busy/gone; a missed tick isn't fatal + click.echo(f"[agent] auto-screenshot skipped: {exc}") + return + with self._lock: + self.shots.append(str(path)) + + def _ticker_loop(self, stop: threading.Event) -> None: + while not stop.wait(self.screenshot_interval): + self._take_auto_shot() + + def _start_ticker(self) -> None: + if not self.duration: + return + self._ticker_stop = threading.Event() + self._ticker_thread = threading.Thread( + target=self._ticker_loop, args=(self._ticker_stop,), daemon=True + ) + self._ticker_thread.start() + + def _stop_ticker(self) -> None: + if self._ticker_stop: + self._ticker_stop.set() + if self._ticker_thread: + self._ticker_thread.join(timeout=self.screenshot_interval + 1) + + def _time_remaining(self) -> Optional[float]: + if self._deadline is None: + return None + return self._deadline - time.monotonic() + + def _dispatch(self, name: str, args: dict[str, Any]) -> tuple[str, Optional[dict]]: + """Runs one tool call. Returns (tool-message text, optional follow-up image message).""" + if name == "send_keys": + self.app.send_keys( + args["keys"], enter=bool(args.get("enter", False)), literal=bool(args.get("literal", False)) + ) + time.sleep(self.delay) + return f"sent keys: {args['keys']!r}", None + + if name == "wait": + time.sleep(float(args["seconds"])) + return f"waited {args['seconds']}s", None + + if name == "read_text": + return self.app.capture_text(), None + + if name == "screenshot": + label = str(args.get("label", "shot")) + path = self._next_shot_path(label) + self.app.screenshot(path) + with self._lock: + self.shots.append(str(path)) + image_message = { + "role": "user", + "content": [ + {"type": "text", "text": f"Screenshot '{label}' (saved to {path}):"}, + {"type": "image_url", "image_url": {"url": f"data:image/png;base64,{_b64_png(path)}"}}, + ], + } + return f"screenshot saved to {path}", image_message + + if name == "finish": + return str(args.get("summary", "done")), None + + return f"unknown tool: {name}", None + + def _safe_dispatch(self, name: str, args: dict[str, Any]) -> tuple[str, Optional[dict]]: + """Validate-then-dispatch: a malformed or incomplete tool call (missing required + argument, wrong type, transient failure) must never crash the whole run -- it should + come back as an error the model can read and self-correct from, same as any other + tool result. This is the standard fix for the #1 agent production failure mode + (tool misuse: wrong/missing arguments, malformed JSON).""" + missing = [a for a in _REQUIRED_ARGS.get(name, ()) if a not in args] + if missing: + return ( + f"error: {name} call is missing required argument(s) {missing}; " + "retry the call with all required arguments included.", + None, + ) + try: + return self._dispatch(name, args) + except Exception as exc: + return f"error: {name} failed ({type(exc).__name__}: {exc}); adjust arguments and retry.", None + + def _call_model(self, messages: list[dict], tools: list[dict], tool_choice: Any, max_retries: int = 4): + """chat.completions.create with retry-and-backoff on transient failures (connection + drops, timeouts, rate limits, 5xx) -- a long multi-minute session makes many dozens of + API round-trips, so hitting at least one transient hiccup is the expected case, not the + exception. Non-transient errors (bad request, auth, content filter, ...) raise + immediately since retrying them would never succeed.""" + delay = 1.0 + last_exc: Exception | None = None + for attempt in range(1, max_retries + 1): + try: + return self.client.chat.completions.create( + model=self.model, + messages=messages, + tools=tools, + tool_choice=tool_choice, + ) + except _TRANSIENT_API_ERRORS as exc: + last_exc = exc + if attempt < max_retries: + click.echo( + f"[agent] API call failed transiently ({exc}); " + f"retrying in {delay:.0f}s ({attempt}/{max_retries})..." + ) + time.sleep(delay) + delay = min(delay * 2, 30.0) + assert last_exc is not None + raise last_exc + + def _step(self, messages: list[dict], step_label: str) -> tuple[bool, bool, str]: + """One model turn: call the API, dispatch any tool calls, mutate `messages` in place. + + Returns (had_tool_calls, finished, text) -- `text` is this turn's own content, or the + finish summary if it called finish, used as a short carry-forward note between chunks + rather than replaying the full conversation. + """ + response = self._call_model(messages, TOOLS, "auto") + message = response.choices[0].message + messages.append(message.model_dump(exclude_none=True)) + + if message.content: + click.echo(f"[agent] {message.content}") + + if not message.tool_calls: + return False, False, message.content or "" + + followups: list[dict] = [] + finished = False + last_text = message.content or "" + for tool_call in message.tool_calls: + # Some backends leak raw chat-template control tokens (e.g. a "harmony"-style + # model's "<|channel|>commentary") straight into the function name -- observed live + # as "screenshot<|channel|>commentary" instead of "screenshot". Strip anything from + # the first such token onward before matching, so it dispatches on the first try + # instead of falling through to "unknown tool" and wasting a full retry round-trip. + name = tool_call.function.name.split("<|")[0].strip() + raw_args = tool_call.function.arguments or "{}" + try: + args = json.loads(raw_args) + if not isinstance(args, dict): + raise ValueError(f"arguments must be a JSON object, got {type(args).__name__}") + except (json.JSONDecodeError, ValueError) as exc: + click.echo(f"[{step_label}] {name}({raw_args!r}) -- malformed arguments") + text, followup = f"error: could not parse arguments for {name}: {exc}", None + else: + click.echo(f"[{step_label}] {name}({args})") + text, followup = self._safe_dispatch(name, args) + messages.append({"role": "tool", "tool_call_id": tool_call.id, "content": text}) + if followup: + followups.append(followup) + if name == "finish": + finished = True + last_text = text + messages.extend(followups) + return True, finished, last_text + + def _plan_chunks(self, instruction: str, target_count: int) -> list[str]: + """One lightweight planning call: ask for an ordered list of short chunks, the same + way a book gets outlined into chapters before each chapter is drafted separately. + Falls back to no plan (caller then runs unchunked) on any failure -- planning is a + nice-to-have, never a hard requirement.""" + plan_messages = [ + { + "role": "system", + "content": ( + "You break a long TUI automation task into an ordered list of short, " + "concrete, self-contained chunks. Each chunk will later be executed by a " + "separate fresh turn that only sees that chunk's own description plus a " + "one-line note about the previous chunk -- not the full history -- so make " + "every chunk understandable and actionable entirely on its own.\n\n" + "Every chunk MUST be a concrete, hands-on-keyboard action performed live in " + "the actual terminal via keystrokes (e.g. 'create a macro with qa, use it to " + "duplicate a line 5 times with 5@a', 'open a vertical split with :vsp and " + "navigate between panes with Ctrl-w'). Never propose meta/production-style " + "chunks like writing a script/outline, recording, editing footage, adding " + "captions, or exporting a file -- there is no such tool, and no video/script " + "is ever produced separately from the real actions taken. The screenshot " + "capture and final GIF assembly are handled automatically outside these " + "chunks; every chunk here is purely about doing something real on screen.\n\n" + "Each array element must be exactly ONE short chunk (one short sentence, a " + "single focused action or tightly-related handful of keystrokes) -- never " + "bundle several distinct steps or numbered sub-items into one array element. " + "If you find yourself writing '1. ... 2. ... 3. ...' inside a single string, " + "that means it should have been 3 separate array elements instead." + ), + }, + { + "role": "user", + "content": ( + f"Task: {instruction}\n\nBreak this into about {target_count} ordered chunks, " + "each a distinct concrete on-screen action or short sequence of keystrokes." + ), + }, + ] + try: + response = self._call_model( + plan_messages, [PLAN_TOOL], {"type": "function", "function": {"name": "propose_plan"}} + ) + call = response.choices[0].message.tool_calls[0] + args = json.loads(call.function.arguments) + chunks = [str(c).strip() for c in args.get("chunks", []) if str(c).strip()] + if len(chunks) == 1: + split = _split_run_on_chunk(chunks[0]) + if len(split) > 1: + click.echo(f"[agent] plan came back as one bundled item; split into {len(split)} chunks.") + chunks = split + return chunks + except Exception as exc: + click.echo(f"[agent] planning failed ({exc}); running as a single unchunked session.") + return [] + + def _run_flat(self, instruction: str) -> None: + content = instruction + if self.duration: + content += ( + f"\n\n(This is a {self.duration:.0f}-second timed session. Keep taking real " + "actions the whole time -- don't stop early.)" + ) + messages: list[dict] = [ + {"role": "system", "content": SYSTEM_PROMPT}, + {"role": "user", "content": content}, + ] + + stall_nudges = 0 + max_stall_nudges = 5 + step = 0 + while step < self.max_steps: + step += 1 + remaining = self._time_remaining() + if remaining is not None and remaining <= 0: + click.echo("[agent] duration elapsed; stopping.") + return + + had_tools, finished, _ = self._step(messages, f"step {step}") + + if finished: + click.echo("[agent] finished.") + return + + if not had_tools: + remaining = self._time_remaining() + if remaining and remaining > 0 and stall_nudges < max_stall_nudges: + stall_nudges += 1 + click.echo(f"[agent] stopped early with {remaining:.0f}s left; nudging to continue.") + messages.append( + { + "role": "user", + "content": ( + f"You still have about {remaining:.0f} seconds left. Keep taking real " + "actions (send_keys/wait) and periodic screenshots -- do not stop until " + "the time is used or call finish only when truly done." + ), + } + ) + step -= 1 # this turn produced no progress; don't count it against max_steps + continue + if remaining and remaining > 0: + # The model won't cooperate further, but a --duration request is a promise + # about wall-clock footage, not about the model's willingness. The + # background ticker is still running: wait out the clock so the + # requested-length GIF/video gets produced regardless. + click.echo( + f"[agent] model stopped cooperating with {remaining:.0f}s left; " + "letting the automatic screenshot timer fill out the rest." + ) + while True: + remaining = self._time_remaining() + if remaining is None or remaining <= 0: + break + time.sleep(min(remaining, 1.0)) + return + click.echo("[agent] no further tool calls; stopping.") + return + click.echo(f"[agent] hit max-steps ({self.max_steps}) without finishing.") + + def _run_chunked(self, overall_instruction: str, chunks: list[str]) -> None: + """Execute the plan one chunk at a time, each with its OWN fresh, small context + (system prompt + this chunk's description + a one-line note from the previous chunk) + instead of one ever-growing conversation. This is the same fix long-form LLM writing + uses for book-length output: outline once, then generate each part with bounded, + mostly-independent context, rather than one giant continuously-growing prompt where + tool-call reliability degrades turn after turn.""" + per_chunk_steps = max(4, self.max_steps // len(chunks)) + carry = "" + for i, chunk_instruction in enumerate(chunks, start=1): + remaining = self._time_remaining() + if remaining is not None and remaining <= 0: + click.echo(f"[agent] duration elapsed before chunk {i}/{len(chunks)}; stopping.") + break + + click.echo(f"[agent] --- chunk {i}/{len(chunks)}: {chunk_instruction} ---") + seed = f"Overall task: {overall_instruction}\n\nYou are now on step {i} of {len(chunks)}: {chunk_instruction}\n" + if carry: + seed += f"\nWhat just happened (for continuity -- don't repeat it): {carry}\n" + seed += ( + "\nFocus only on this step. When it's done, call finish with a one-sentence " + "summary of what you did -- the next step picks up from there." + ) + messages: list[dict] = [ + {"role": "system", "content": SYSTEM_PROMPT}, + {"role": "user", "content": seed}, + ] + + summary = chunk_instruction + for step in range(1, per_chunk_steps + 1): + remaining = self._time_remaining() + if remaining is not None and remaining <= 0: + summary = "cut short: overall duration elapsed" + break + had_tools, finished, text = self._step(messages, f"chunk {i}/{len(chunks)} step {step}") + if text: + summary = text + if finished or not had_tools: + break # a plain non-tool reply also means this chunk considers itself done + carry = summary + click.echo(f"[agent] chunk {i}/{len(chunks)} done: {summary}") + + def run(self, instruction: str) -> None: + # Plan (if applicable) BEFORE starting the clock/ticker: a --duration request is a + # promise about actual execution time, not wall-clock-including-setup, and planning is + # itself a network round-trip that can take several seconds on its own. + chunks: list[str] = [] + if self.duration and self.duration > self.chunk_seconds * 1.5: + target = max(2, min(8, round(self.duration / self.chunk_seconds))) + click.echo(f"[agent] planning ~{target} chunks for this {self.duration:.0f}s session...") + chunks = self._plan_chunks(instruction, target) + + if self.duration: + self._deadline = time.monotonic() + self.duration + self._start_ticker() + + try: + if chunks: + click.echo(f"[agent] plan ({len(chunks)} chunks):") + for i, c in enumerate(chunks, start=1): + click.echo(f" {i}. {c}") + self._run_chunked(instruction, chunks) + else: + self._run_flat(instruction) + except _TRANSIENT_API_ERRORS as exc: + # Retries inside _call_model already absorbed every transient blip; reaching here + # means the backend is down for longer than the retry budget covers -- a real + # outage, not something more retrying fixes. Degrade gracefully: keep whatever the + # background ticker already captured and stop cleanly instead of a raw traceback + # that would also discard those screenshots. + click.echo( + f"[agent] stopping: the AI backend is still unreachable after retries ({exc}). " + f"Session '{self.app.session_name}' is untouched -- rerun the same command " + "later to pick up where this left off." + ) + finally: + self._stop_ticker() + + self.shots.sort() + if self.shots: + click.echo(f"Screenshots taken ({len(self.shots)}):") + for shot in self.shots: + click.echo(f" - {shot}") + + +def build_client(base_url: str, app_reference: str, timeout: float = 45.0) -> OpenAI: + api_key = os.environ.get("DEVPLACE_API_KEY") + if not api_key: + raise click.ClickException( + "DEVPLACE_API_KEY is not set. Export it first: " + "export DEVPLACE_API_KEY=" + ) + # Without a client-side timeout, one slow/hung call (a complex prompt on a "high reasoning + # effort" model, or a stalled connection) blocks indefinitely -- observed live: a single + # bundled multi-part chunk took well over 150s with no way to interrupt it. A bounded + # timeout turns that into an APITimeoutError, which _call_model already retries/backs off. + return OpenAI( + base_url=base_url, api_key=api_key, default_headers={"X-App-Reference": app_reference}, timeout=timeout + ) diff --git a/src/tmux_shot/agent_cli.py b/src/tmux_shot/agent_cli.py new file mode 100644 index 0000000..f78cff0 --- /dev/null +++ b/src/tmux_shot/agent_cli.py @@ -0,0 +1,150 @@ +from __future__ import annotations + +from pathlib import Path + +import click + +from .render import frames_to_gif +from .tmux_control import TmuxApp + + +@click.command() +@click.argument("prompt", required=False) +@click.option("--session", default="tmux_shot_agent", show_default=True, help="tmux session to create/attach.") +@click.option("--command", default=None, help="Command to launch if the session doesn't exist yet.") +@click.option("--width", default=120, show_default=True) +@click.option("--height", default=40, show_default=True) +@click.option( + "--delay", + default=1.0, + show_default=True, + type=float, + help="Seconds to pause after each send_keys before the model looks again.", +) +@click.option( + "--max-steps", + default=60, + show_default=True, + help="Safety cap on agent tool-call round-trips. Raise this for long --duration runs.", +) +@click.option( + "--duration", + default=None, + type=float, + help=( + "Keep the agent actively working for this many seconds (e.g. 180 for 'three minutes " + "of X') instead of stopping as soon as it thinks it's done. A screenshot is captured " + "automatically every --interval seconds throughout, regardless of what the model does, " + "so a long GIF/video request always produces real footage." + ), +) +@click.option( + "--interval", + default=5.0, + show_default=True, + type=float, + help="Seconds between automatic screenshots when --duration is set.", +) +@click.option( + "--chunk-seconds", + default=45.0, + show_default=True, + type=float, + help=( + "For --duration sessions much longer than this, the task is first split into an " + "ordered plan of roughly duration/chunk-seconds chunks, each run with its own fresh, " + "small context instead of one giant growing conversation -- same fix long-form LLM " + "writing uses (outline once, draft each part separately). Set very high (or use a " + "short --duration) to disable chunking." + ), +) +@click.option("--out-dir", default="out/agent", show_default=True, type=click.Path()) +@click.option("--model", default="molodetz", show_default=True) +@click.option("--base-url", default="https://devplace.net/openai/v1", show_default=True) +@click.option( + "--app-reference", + default="tmux-shot", + show_default=True, + help="Sent as X-App-Reference for cost attribution on the gateway.", +) +@click.option( + "--api-timeout", + default=45.0, + show_default=True, + type=float, + help="Per-request timeout in seconds. A slow/stuck call fails (and retries) instead of blocking forever.", +) +@click.option( + "--gif", + "gif_path", + default=None, + type=click.Path(), + help="Also stitch every screenshot taken during the run into an animated GIF at this path.", +) +@click.option("--gif-duration-ms", default=600, show_default=True, help="Milliseconds per frame in --gif output.") +def agent( + prompt: str | None, + session: str, + command: str | None, + width: int, + height: int, + delay: float, + max_steps: int, + duration: float | None, + interval: float, + chunk_seconds: float, + out_dir: str, + model: str, + base_url: str, + app_reference: str, + api_timeout: float, + gif_path: str | None, + gif_duration_ms: int, +) -> None: + """Give a natural-language PROMPT; an LLM drives a TUI in tmux via tool calls + (send_keys, wait, read_text, screenshot, finish) and takes screenshots. + + Everything is optional except the prompt itself: + + tmux-agent "open the app, find the settings panel, screenshot it" + + For a timed "N minutes of X" style session, add --duration and --gif: + + tmux-agent --duration 180 --interval 5 --gif out/session.gif \\ + --command "grok --cwd /tmp/scratch --always-approve" \\ + "actively vibecode something small and fun with grok the whole time" + + Every knob (--session, --command, --width/--height, --delay, --model, ...) + is still here when you need finer control over a specific run. + """ + if not prompt: + prompt = click.prompt("What should the agent do? (describe the whole photo session)") + + from .agent import AgentRunner, build_client + + tui = TmuxApp(session, command=command, width=width, height=height) + client = build_client(base_url, app_reference, timeout=api_timeout) + runner = AgentRunner( + tui, + client, + model=model, + out_dir=Path(out_dir), + delay=delay, + max_steps=max_steps, + duration=duration, + screenshot_interval=interval, + chunk_seconds=chunk_seconds, + ) + runner.run(prompt) + + if gif_path and runner.shots: + path = frames_to_gif(runner.shots, gif_path, duration_ms=gif_duration_ms) + click.echo(f"gif: {path}") + elif gif_path: + click.echo("gif: skipped, no screenshots were taken") + + +main = agent + +if __name__ == "__main__": + main() diff --git a/src/tmux_shot/cli.py b/src/tmux_shot/cli.py new file mode 100644 index 0000000..35057ff --- /dev/null +++ b/src/tmux_shot/cli.py @@ -0,0 +1,81 @@ +from __future__ import annotations + +import click + +from .agent_cli import agent +from .render import frames_to_gif +from .tmux_control import TmuxApp + + +@click.group() +def main() -> None: + """tmux-shot: drive a TUI inside tmux from the shell and screenshot it.""" + + +@main.command() +@click.argument("name") +@click.option("--command", default=None, help="Command to launch in the new session.") +@click.option("--width", default=120, show_default=True) +@click.option("--height", default=40, show_default=True) +def new(name: str, command: str | None, width: int, height: int) -> None: + """Create (or attach to) a named tmux session.""" + TmuxApp(name, command=command, width=width, height=height) + click.echo(f"session '{name}' ready") + + +@main.command() +@click.argument("name") +@click.argument("output", type=click.Path()) +@click.option("--font-size", default=14, show_default=True) +@click.option("--freeze", "use_freeze", is_flag=True, help="Render via the external `freeze` CLI instead.") +def shot(name: str, output: str, font_size: int, use_freeze: bool) -> None: + """Screenshot a session's current pane state to a PNG.""" + app = TmuxApp(name) + path = app.screenshot_with_freeze(output) if use_freeze else app.screenshot(output, font_size=font_size) + click.echo(path) + + +@main.command() +@click.argument("name") +def text(name: str) -> None: + """Print a session's current pane state as plain text.""" + app = TmuxApp(name) + click.echo(app.capture_text()) + + +@main.command() +@click.argument("name") +@click.argument("keys") +@click.option("--literal/--no-literal", default=False, help="Send keys literally instead of as tmux key names.") +@click.option("--enter/--no-enter", default=True, help="Send Enter after the keys.") +def send(name: str, keys: str, literal: bool, enter: bool) -> None: + """Send keystrokes to a session.""" + app = TmuxApp(name) + app.send_keys(keys, enter=enter, literal=literal) + + +@main.command() +@click.argument("name") +def kill(name: str) -> None: + """Kill a named session.""" + app = TmuxApp(name) + app.kill() + click.echo(f"session '{name}' killed") + + +@main.command() +@click.argument("frames", nargs=-1, required=True, type=click.Path(exists=True)) +@click.argument("output", type=click.Path()) +@click.option("--duration-ms", default=400, show_default=True, help="Milliseconds each frame is shown.") +@click.option("--loop/--no-loop", default=True, help="Loop the animation forever vs play once.") +def gif(frames: tuple[str, ...], output: str, duration_ms: int, loop: bool) -> None: + """Stitch PNG screenshots (in the given order) into an animated GIF "movie".""" + path = frames_to_gif(list(frames), output, duration_ms=duration_ms, loop=0 if loop else 1) + click.echo(path) + + +main.add_command(agent, name="agent") + + +if __name__ == "__main__": + main() diff --git a/src/tmux_shot/render.py b/src/tmux_shot/render.py new file mode 100644 index 0000000..67cb9c1 --- /dev/null +++ b/src/tmux_shot/render.py @@ -0,0 +1,152 @@ +from __future__ import annotations + +from pathlib import Path +from typing import Optional + +import pyte +from PIL import Image, ImageDraw, ImageFont + +_FONT_CANDIDATES = [ + "/usr/share/fonts/truetype/dejavu/DejaVuSansMono.ttf", + "/usr/share/fonts/truetype/liberation/LiberationMono-Regular.ttf", + "/usr/share/fonts/truetype/freefont/FreeMono.ttf", + "/System/Library/Fonts/Menlo.ttc", + "/Library/Fonts/Menlo.ttc", +] + +_BOLD_FONT_CANDIDATES = [ + "/usr/share/fonts/truetype/dejavu/DejaVuSansMono-Bold.ttf", + "/usr/share/fonts/truetype/liberation/LiberationMono-Bold.ttf", + "/usr/share/fonts/truetype/freefont/FreeMonoBold.ttf", +] + +# xterm's default 16-color palette, keyed by the names pyte uses. +_ANSI_COLORS = { + "black": (0, 0, 0), + "red": (205, 0, 0), + "green": (0, 205, 0), + "brown": (205, 205, 0), + "yellow": (205, 205, 0), + "blue": (0, 0, 238), + "magenta": (205, 0, 205), + "cyan": (0, 205, 205), + "white": (229, 229, 229), + "brightblack": (127, 127, 127), + "brightred": (255, 0, 0), + "brightgreen": (0, 255, 0), + "brightbrown": (255, 255, 0), + "brightyellow": (255, 255, 0), + "brightblue": (92, 92, 255), + "brightmagenta": (255, 0, 255), + "brightcyan": (0, 255, 255), + "brightwhite": (255, 255, 255), +} + +DEFAULT_FG = (229, 229, 229) +DEFAULT_BG = (0, 0, 0) + + +def _find_font(candidates: list[str], size: int) -> ImageFont.FreeTypeFont: + for path in candidates: + if Path(path).exists(): + return ImageFont.truetype(path, size) + return ImageFont.load_default() + + +def _resolve_color(value: Optional[str], default: tuple[int, int, int]) -> tuple[int, int, int]: + if not value or value == "default": + return default + if len(value) == 6 and all(c in "0123456789abcdefABCDEF" for c in value): + return (int(value[0:2], 16), int(value[2:4], 16), int(value[4:6], 16)) + return _ANSI_COLORS.get(value, default) + + +def render_ansi_to_png( + ansi_text: str, + output_path: str | Path, + cols: int, + rows: int, + font_path: Optional[str] = None, + font_size: int = 14, +) -> str: + """Replay an ANSI byte stream (e.g. from `tmux capture-pane -e`) into a PNG. + + Uses pyte as a headless VT100 terminal emulator to resolve the escape + codes into a (char, fg, bg, bold, ...) grid, then draws that grid with + Pillow. No X server, browser, or external binary required. + """ + screen = pyte.Screen(cols, rows) + stream = pyte.Stream(screen) + # pyte's Stream treats "\n" as line-feed only (real VT100 semantics: no + # implicit carriage return), so a bare "\n" between tmux's captured rows + # leaves the cursor wherever the previous row's content ended instead of + # returning it to column 0 -- corrupting every row after the first one + # that doesn't happen to end exactly at the pane's edge. + stream.feed(ansi_text.replace("\n", "\r\n")) + + if font_path: + font = ImageFont.truetype(font_path, font_size) + bold_font = font + else: + font = _find_font(_FONT_CANDIDATES, font_size) + bold_font = _find_font(_BOLD_FONT_CANDIDATES, font_size) + + char_w = max(1, round(font.getlength("M"))) + char_h = max(1, round(font_size * 1.4)) + + img = Image.new("RGB", (cols * char_w, rows * char_h), DEFAULT_BG) + draw = ImageDraw.Draw(img) + + for y in range(rows): + row = screen.buffer[y] + for x in range(cols): + char = row[x] + fg = _resolve_color(char.fg, DEFAULT_FG) + bg = _resolve_color(char.bg, DEFAULT_BG) + if char.reverse: + fg, bg = bg, fg + px, py = x * char_w, y * char_h + if bg != DEFAULT_BG: + draw.rectangle([px, py, px + char_w, py + char_h], fill=bg) + if char.data and char.data != " ": + draw.text( + (px, py), + char.data, + font=bold_font if char.bold else font, + fill=fg, + ) + + output_path = Path(output_path) + output_path.parent.mkdir(parents=True, exist_ok=True) + img.save(output_path) + return str(output_path) + + +def frames_to_gif( + frame_paths: list[str | Path], + output_path: str | Path, + duration_ms: int = 400, + loop: int = 0, +) -> str: + """Stitch a sequence of PNG screenshots into one animated GIF ("movie" mode). + + Frames are resized to the first frame's canvas so a run against a + resizable/varying-content pane still produces a valid animation. + """ + if not frame_paths: + raise ValueError("frames_to_gif needs at least one frame") + + frames = [Image.open(p).convert("RGB") for p in frame_paths] + size = frames[0].size + frames = [f if f.size == size else f.resize(size) for f in frames] + + output_path = Path(output_path) + output_path.parent.mkdir(parents=True, exist_ok=True) + frames[0].save( + output_path, + save_all=True, + append_images=frames[1:], + duration=duration_ms, + loop=loop, + ) + return str(output_path) diff --git a/src/tmux_shot/tmux_control.py b/src/tmux_shot/tmux_control.py new file mode 100644 index 0000000..fecc6fd --- /dev/null +++ b/src/tmux_shot/tmux_control.py @@ -0,0 +1,173 @@ +from __future__ import annotations + +import subprocess +import time +from pathlib import Path +from typing import Optional + +import libtmux +from libtmux.pane import Pane +from libtmux.session import Session + +from .render import render_ansi_to_png + + +class TmuxApp: + """Drive a TUI running inside a named tmux session and screenshot it. + + Attaching by name is idempotent: if a session with `session_name` is + already running, it's reused as-is (its own size, command, and state) + instead of being recreated. + """ + + def __init__( + self, + session_name: str, + command: Optional[str] = None, + width: int = 120, + height: int = 40, + server: Optional[libtmux.Server] = None, + ) -> None: + self.session_name = session_name + self.width = width + self.height = height + self.server = server or libtmux.Server() + self.session: Session = self._get_or_create_session(command) + + def _get_or_create_session(self, command: Optional[str]) -> Session: + existing = self.server.sessions.get(session_name=self.session_name, default=None) + if existing is not None: + return existing + + # Known tmux bug (tmux/tmux#4268): a detached `new-session -x -y` is + # silently resized to match whichever client was last active on the + # server, ignoring the requested size, whenever another client is + # attached elsewhere -- and this can happen before the target app has + # even started, so the app itself starts up with (and may cache) the + # wrong terminal size, corrupting anything it later draws with + # absolute cursor positioning (cut-off text, missing borders) even + # after we forcibly resize the pane back afterwards. + # + # Workaround: always start a plain shell first, pin the size (and + # wait for it to actually stick), and only *then* type the real + # command in -- so the target app never observes the wrong size. + session = self.server.new_session( + session_name=self.session_name, + attach=False, + x=self.width, + y=self.height, + ) + session.cmd("set-option", "window-size", "manual") + session.cmd("resize-window", "-x", str(self.width), "-y", str(self.height)) + self._wait_for_pinned_size(session.active_pane) + + if command: + session.active_pane.send_keys(command, literal=True, enter=True) + return session + + def _wait_for_pinned_size(self, pane: Pane, timeout: float = 2.0, interval: float = 0.05) -> None: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + pane.refresh() + if int(pane.width) == self.width and int(pane.height) == self.height: + return + time.sleep(interval) + + @property + def pane(self) -> Pane: + return self.session.active_pane + + def send_keys(self, keys: str, enter: bool = True, literal: bool = False) -> None: + """Send one or more keys to the pane. + + Non-literal mode splits on whitespace and sends each token as its own + tmux key-name argument (matching `tmux send-keys Down Down Right` + CLI semantics) — a single call can drive a multi-step move like + "Down Down Right Right Right". Use literal=True to type text verbatim, + spaces included. + """ + if literal: + self.pane.send_keys(keys, enter=enter, literal=True) + return + tokens = keys.split() + if not tokens: + return + self.pane.cmd("send-keys", *tokens) + if enter: + self.pane.enter() + + def capture_ansi(self) -> str: + """Pane content including ANSI color/style escape codes.""" + lines = self.pane.capture_pane(escape_sequences=True) or [] + return "\n".join(lines) + + def capture_text(self) -> str: + """Plain-text pane content, no styling — cheap to feed to a text-only agent.""" + lines = self.pane.capture_pane() or [] + return "\n".join(lines) + + def wait_for(self, text: str, timeout: float = 10.0, interval: float = 0.2) -> str: + """Poll the pane until `text` shows up in its plain-text content. + + TUI startup/redraw latency varies (network calls, cold caches, etc.), so a + fixed sleep before a screenshot is inherently flaky. Returns the matching + capture; raises TimeoutError if `text` never appears within `timeout`. + """ + deadline = time.monotonic() + timeout + last = "" + while time.monotonic() < deadline: + last = self.capture_text() + if text in last: + return last + time.sleep(interval) + raise TimeoutError(f"timed out after {timeout}s waiting for {text!r} in pane output:\n{last}") + + def live_size(self) -> tuple[int, int]: + """The pane's actual current (cols, rows), refreshed from tmux. + + Can diverge from the width/height passed to the constructor if the pane + was resized after creation (a client attaching, the app itself resizing + it, tmux's smallest-attached-client behavior, etc). Rendering with a + stale size mismatches the cursor-position escape codes the app actually + emitted, producing corrupted-looking screenshots — text cut off + mid-word, missing borders — so screenshot() always re-queries this. + """ + self.pane.refresh() + return int(self.pane.width), int(self.pane.height) + + def screenshot( + self, + path: str | Path, + font_path: Optional[str] = None, + font_size: int = 14, + ) -> str: + """Render the current pane state to a PNG using the built-in pyte+Pillow renderer.""" + ansi = self.capture_ansi() + cols, rows = self.live_size() + return render_ansi_to_png( + ansi, + path, + cols=cols, + rows=rows, + font_path=font_path, + font_size=font_size, + ) + + def screenshot_with_freeze(self, path: str | Path) -> str: + """Render via the external `freeze` CLI (charmbracelet/freeze) if installed. + + Optional alternative renderer with nicer typography/window chrome. + Falls back is not automatic — call screenshot() if freeze isn't installed. + """ + ansi = self.capture_ansi() + path = Path(path) + path.parent.mkdir(parents=True, exist_ok=True) + subprocess.run( + ["freeze", "-o", str(path)], + input=ansi.encode(), + check=True, + ) + return str(path) + + def kill(self) -> None: + self.session.kill() diff --git a/tests/conftest.py b/tests/conftest.py new file mode 100644 index 0000000..6eef00c --- /dev/null +++ b/tests/conftest.py @@ -0,0 +1,35 @@ +from __future__ import annotations + +import uuid +from pathlib import Path + +import pytest + +from tmux_shot import TmuxApp + +REPO_ROOT = Path(__file__).resolve().parent.parent +DEMO_TUI = REPO_ROOT / "examples" / "demo_tui.py" + + +@pytest.fixture +def session_name() -> str: + return f"tmux_shot_test_{uuid.uuid4().hex[:8]}" + + +@pytest.fixture +def demo_app(session_name: str, tmp_path: Path): + app = TmuxApp(session_name, command=f"python3 {DEMO_TUI}", width=80, height=24) + try: + yield app + finally: + try: + app.kill() + except Exception: + pass # already dead (e.g. the test quit it itself) + + +@pytest.fixture +def out_dir(tmp_path: Path) -> Path: + d = tmp_path / "out" + d.mkdir() + return d diff --git a/tests/test_agent.py b/tests/test_agent.py new file mode 100644 index 0000000..29f0bcd --- /dev/null +++ b/tests/test_agent.py @@ -0,0 +1,370 @@ +from __future__ import annotations + +import json +import time +from pathlib import Path +from typing import Any + +from tmux_shot import TmuxApp +from tmux_shot.agent import AgentRunner, _split_run_on_chunk + + +class FakeFunction: + def __init__(self, name: str, arguments: str) -> None: + self.name = name + self.arguments = arguments + + +class FakeToolCall: + def __init__(self, call_id: str, name: str, arguments: str) -> None: + self.id = call_id + self.function = FakeFunction(name, arguments) + + +class FakeMessage: + def __init__(self, content: str | None = None, tool_calls: list[FakeToolCall] | None = None) -> None: + self.content = content + self.tool_calls = tool_calls or [] + + def model_dump(self, exclude_none: bool = True) -> dict[str, Any]: + d: dict[str, Any] = {"role": "assistant", "content": self.content} + if self.tool_calls: + d["tool_calls"] = [ + { + "id": tc.id, + "type": "function", + "function": {"name": tc.function.name, "arguments": tc.function.arguments}, + } + for tc in self.tool_calls + ] + if exclude_none: + d = {k: v for k, v in d.items() if v is not None} + return d + + +class FakeCompletions: + def __init__(self, script: list[FakeMessage]) -> None: + self._script = list(script) + self.requests: list[dict[str, Any]] = [] + + def create(self, **kwargs: Any): + # snapshot messages: it's the same mutable list AgentRunner keeps appending + # to for the rest of the run, so store a copy or every recorded request + # would end up reflecting the final state instead of the state at call time. + self.requests.append({**kwargs, "messages": list(kwargs["messages"])}) + item = self._script.pop(0) + if isinstance(item, Exception): + raise item + return type("R", (), {"choices": [type("C", (), {"message": item})()]})() + + +class FakeChat: + def __init__(self, script: list[FakeMessage]) -> None: + self.completions = FakeCompletions(script) + + +class FakeClient: + def __init__(self, script: list[FakeMessage]) -> None: + self.chat = FakeChat(script) + + +def test_agent_runner_drives_app_via_scripted_tool_calls(demo_app: TmuxApp, out_dir: Path) -> None: + time.sleep(0.3) + + script = [ + FakeMessage( + content="Moving down.", + tool_calls=[FakeToolCall("call_1", "send_keys", '{"keys": "Down Down", "enter": false}')], + ), + FakeMessage( + tool_calls=[FakeToolCall("call_2", "screenshot", '{"label": "moved"}')], + ), + FakeMessage( + tool_calls=[FakeToolCall("call_3", "finish", '{"summary": "moved and captured"}')], + ), + ] + client = FakeClient(script) + + runner = AgentRunner(demo_app, client, model="fake-model", out_dir=out_dir, delay=0.05, max_steps=5) + runner.run("move down twice and screenshot it") + + assert len(runner.shots) == 1 + assert Path(runner.shots[0]).exists() + + # 3 request round-trips: send_keys turn, screenshot turn, finish turn + assert len(client.chat.completions.requests) == 3 + + # the request after the screenshot tool call must include the image as a follow-up + # user message, immediately after that turn's tool-response message + third_request_messages = client.chat.completions.requests[2]["messages"] + roles = [m["role"] for m in third_request_messages] + assert roles.count("tool") == 2 # one per completed tool call so far + image_messages = [ + m for m in third_request_messages if m["role"] == "user" and isinstance(m.get("content"), list) + ] + assert len(image_messages) == 1 + content_types = [part["type"] for part in image_messages[0]["content"]] + assert "image_url" in content_types + + +def test_agent_runner_stops_when_model_returns_no_tool_calls(demo_app: TmuxApp, out_dir: Path) -> None: + script = [FakeMessage(content="Nothing to do here.")] + client = FakeClient(script) + + runner = AgentRunner(demo_app, client, model="fake-model", out_dir=out_dir, delay=0.05, max_steps=5) + runner.run("do nothing") + + assert runner.shots == [] + assert len(client.chat.completions.requests) == 1 + + +def test_agent_runner_keeps_capturing_via_ticker_even_if_model_refuses_outright( + demo_app: TmuxApp, out_dir: Path +) -> None: + """Reproduces a real failure: the model flatly refuses a 'gif of N minutes of X' request + ("I'm sorry, but I can't help with that.") instead of calling any tool. With --duration + set, the background screenshot ticker must still cover the full requested wall-clock time + regardless of the model's cooperation, so a timed GIF/video request is a mechanical + guarantee, not something that depends on the model understanding the ask.""" + refusal = FakeMessage(content="I'm sorry, but I can't help with that.") + client = FakeClient([refusal] * 10) # more than enough for every nudge attempt + + runner = AgentRunner( + demo_app, + client, + model="fake-model", + out_dir=out_dir, + delay=0.0, + max_steps=50, + duration=1.2, + screenshot_interval=0.2, + ) + + start = time.monotonic() + runner.run("I want a gif of you actively using this app") + elapsed = time.monotonic() - start + + assert elapsed >= 1.0 # actually waited out roughly the full duration, not just bailed + assert len(runner.shots) >= 3 # the ticker produced real frames despite total refusal + for shot in runner.shots: + assert Path(shot).exists() + + +def test_agent_runner_survives_malformed_tool_call_instead_of_crashing( + demo_app: TmuxApp, out_dir: Path +) -> None: + """Reproduces a real crash: the model called send_keys({}) -- no 'keys' argument at all -- + which used to raise an unhandled KeyError and kill the whole process. It must instead come + back as a tool-result error the model can read and recover from.""" + script = [ + FakeMessage(tool_calls=[FakeToolCall("call_1", "send_keys", "{}")]), # missing "keys" + FakeMessage(tool_calls=[FakeToolCall("call_2", "send_keys", "not json")]), # malformed JSON + FakeMessage(tool_calls=[FakeToolCall("call_3", "finish", '{"summary": "recovered"}')]), + ] + client = FakeClient(script) + + runner = AgentRunner(demo_app, client, model="fake-model", out_dir=out_dir, delay=0.0, max_steps=5) + runner.run("do something") # must not raise + + third_request_messages = client.chat.completions.requests[2]["messages"] + tool_messages = [m for m in third_request_messages if m["role"] == "tool"] + assert len(tool_messages) == 2 + assert "error" in tool_messages[0]["content"] + assert "missing required argument" in tool_messages[0]["content"] + assert "error" in tool_messages[1]["content"] + + +def test_split_run_on_chunk_splits_bundled_numbered_items() -> None: + bundled = ( + "1. Open vim in a new terminal and create a file. " + "2. Use fzf to search project files and jump to a definition. " + "3. Duplicate a heading with a macro and save." + ) + parts = _split_run_on_chunk(bundled) + assert len(parts) == 3 + assert parts[0].startswith("Open vim") + assert parts[1].startswith("Use fzf") + assert parts[2].startswith("Duplicate a heading") + + +def test_split_run_on_chunk_leaves_ordinary_text_alone() -> None: + assert _split_run_on_chunk("create a macro with qa and replay it with 5@a") == [ + "create a macro with qa and replay it with 5@a" + ] + + +def test_agent_runner_normalizes_a_bundled_plan_into_separate_chunks( + demo_app: TmuxApp, out_dir: Path, capsys +) -> None: + """Reproduces a real failure: the model returned a plan with all 3 steps bundled into a + single array element ('1. ... 2. ... 3. ...') instead of 3 separate elements as instructed. + The runner must recover this into real separate chunks rather than executing one giant, + unbounded mega-chunk.""" + bundled_plan_call = FakeToolCall( + "plan_1", + "propose_plan", + json.dumps( + { + "chunks": [ + "1. Open vim and create a file. 2. Record a macro and replay it 5 times. " + "3. Open a vertical split and copy a line across." + ] + } + ), + ) + script = [ + FakeMessage(tool_calls=[bundled_plan_call]), + FakeMessage(tool_calls=[FakeToolCall("c1", "finish", '{"summary": "opened file"}')]), + FakeMessage(tool_calls=[FakeToolCall("c2", "finish", '{"summary": "did macro"}')]), + FakeMessage(tool_calls=[FakeToolCall("c3", "finish", '{"summary": "did split"}')]), + ] + client = FakeClient(script) + + runner = AgentRunner( + demo_app, + client, + model="fake-model", + out_dir=out_dir, + delay=0.0, + max_steps=12, + duration=6.0, + screenshot_interval=100.0, + chunk_seconds=2.0, + ) + runner.run("give a demo of advanced vim features") + + out = capsys.readouterr().out + assert "split into 3 chunks" in out + assert "chunk 3/3 done: did split" in out + + +def test_agent_runner_chunks_a_long_duration_session_with_status_updates( + demo_app: TmuxApp, out_dir: Path, capsys +) -> None: + """A long --duration task should be planned into chunks (like an outline before drafting + chapters) and each chunk run with its own fresh context, printing a status line per chunk + as it completes.""" + plan_call = FakeToolCall( + "plan_1", "propose_plan", '{"chunks": ["show text objects", "show macros", "show marks"]}' + ) + script = [ + FakeMessage(tool_calls=[plan_call]), + FakeMessage(tool_calls=[FakeToolCall("c1", "finish", '{"summary": "did text objects"}')]), + FakeMessage(tool_calls=[FakeToolCall("c2", "finish", '{"summary": "did macros"}')]), + FakeMessage(tool_calls=[FakeToolCall("c3", "finish", '{"summary": "did marks"}')]), + ] + client = FakeClient(script) + + runner = AgentRunner( + demo_app, + client, + model="fake-model", + out_dir=out_dir, + delay=0.0, + max_steps=12, + duration=6.0, + screenshot_interval=100.0, # long enough the ticker won't fire during this fast fake run + chunk_seconds=2.0, # 6s / 2s -> plans for ~3 chunks + ) + runner.run("give a demo of advanced vim features") + + assert len(client.chat.completions.requests) == 4 # 1 planning call + 3 chunk turns + + out = capsys.readouterr().out + assert "plan (3 chunks)" in out + assert "chunk 1/3 done: did text objects" in out + assert "chunk 2/3 done: did macros" in out + assert "chunk 3/3 done: did marks" in out + + +def test_agent_runner_retries_transient_api_errors_instead_of_crashing( + demo_app: TmuxApp, out_dir: Path +) -> None: + """Reproduces a real crash: the upstream gateway returned a transient 502 + (openai.InternalServerError) mid-session, which used to propagate straight up and kill the + whole run. It must be retried with backoff instead.""" + import httpx2 + from openai import InternalServerError + + request = httpx2.Request("POST", "https://example.invalid/v1/chat/completions") + response = httpx2.Response(502, request=request, json={"detail": "Upstream connection failed"}) + transient_error = InternalServerError("Bad Gateway", response=response, body=None) + + script = [ + transient_error, + transient_error, + FakeMessage(tool_calls=[FakeToolCall("call_1", "finish", '{"summary": "done after retry"}')]), + ] + client = FakeClient(script) + + runner = AgentRunner(demo_app, client, model="fake-model", out_dir=out_dir, delay=0.0, max_steps=5) + runner.run("do something") # must not raise + + assert len(client.chat.completions.requests) == 3 + + +def test_agent_runner_degrades_gracefully_on_persistent_backend_outage( + demo_app: TmuxApp, out_dir: Path, monkeypatch +) -> None: + """Reproduces a real outage: the backend stayed down for longer than the retry budget + (every call 502s). run() must not crash with a raw traceback -- it should stop cleanly and + keep reporting whatever screenshots were already captured.""" + import httpx2 + from openai import InternalServerError + + import tmux_shot.agent as agent_module + + monkeypatch.setattr(agent_module.time, "sleep", lambda _seconds: None) # skip real backoff waits + + request = httpx2.Request("POST", "https://example.invalid/v1/chat/completions") + response = httpx2.Response(502, request=request, json={"detail": "Upstream connection failed"}) + always_down = InternalServerError("Bad Gateway", response=response, body=None) + + client = FakeClient( + [ + FakeMessage(tool_calls=[FakeToolCall("call_1", "screenshot", '{"label": "before-outage"}')]), + always_down, + always_down, + always_down, + always_down, + ] + ) + + runner = AgentRunner(demo_app, client, model="fake-model", out_dir=out_dir, delay=0.0, max_steps=5) + runner.run("do something") # must not raise + + assert len(runner.shots) == 1 # the screenshot taken before the outage is not lost + assert Path(runner.shots[0]).exists() + + +def test_agent_step_strips_leaked_chat_template_tokens_from_tool_name( + demo_app: TmuxApp, out_dir: Path +) -> None: + """Reproduces a real artifact: a backend leaked raw harmony-style channel tokens into the + function name itself ("screenshot<|channel|>commentary" instead of "screenshot"). It must + still dispatch as the real tool on the first try, not fall through to 'unknown tool'.""" + script = [ + FakeMessage(tool_calls=[FakeToolCall("call_1", "read_text<|channel|>commentary", "{}")]), + ] + client = FakeClient(script) + + runner = AgentRunner(demo_app, client, model="fake-model", out_dir=out_dir, delay=0.0, max_steps=5) + messages: list[dict] = [] + had_tools, finished, _ = runner._step(messages, "step 1") + + tool_message = next(m for m in messages if m["role"] == "tool") + assert "unknown tool" not in tool_message["content"] + assert "error" not in tool_message["content"] + + +def test_agent_runner_respects_max_steps_safety_cap(demo_app: TmuxApp, out_dir: Path) -> None: + # every turn asks to wait, never calling finish -> must stop at max_steps, not hang forever + script = [ + FakeMessage(tool_calls=[FakeToolCall(f"call_{i}", "wait", '{"seconds": 0.01}')]) for i in range(10) + ] + client = FakeClient(script) + + runner = AgentRunner(demo_app, client, model="fake-model", out_dir=out_dir, delay=0.0, max_steps=3) + runner.run("wait forever") + + assert len(client.chat.completions.requests) == 3 diff --git a/tests/test_cli.py b/tests/test_cli.py new file mode 100644 index 0000000..69406f0 --- /dev/null +++ b/tests/test_cli.py @@ -0,0 +1,55 @@ +from __future__ import annotations + +import subprocess +import sys +import time +from pathlib import Path + +REPO_ROOT = Path(__file__).resolve().parent.parent +DEMO_TUI = REPO_ROOT / "examples" / "demo_tui.py" + + +def run_cli(*args: str) -> subprocess.CompletedProcess: + return subprocess.run( + [sys.executable, "-m", "tmux_shot.cli", *args], + capture_output=True, + text=True, + timeout=15, + ) + + +def test_cli_full_lifecycle(session_name: str, out_dir: Path) -> None: + try: + result = run_cli("new", session_name, "--command", f"python3 {DEMO_TUI}", "--width", "80", "--height", "24") + assert result.returncode == 0, result.stderr + assert "ready" in result.stdout + + time.sleep(0.3) + + text_result = run_cli("text", session_name) + assert text_result.returncode == 0, text_result.stderr + assert "tmux-shot demo TUI" in text_result.stdout + + shot_path = out_dir / "cli_shot.png" + shot_result = run_cli("shot", session_name, str(shot_path)) + assert shot_result.returncode == 0, shot_result.stderr + assert shot_path.exists() + assert shot_path.stat().st_size > 0 + + send_result = run_cli("send", session_name, "Down", "--no-enter") + assert send_result.returncode == 0, send_result.stderr + finally: + run_cli("kill", session_name) + + +def test_cli_gif_command_stitches_frames(out_dir: Path) -> None: + from tmux_shot.render import render_ansi_to_png + + frames = [ + render_ansi_to_png("frame a", out_dir / "a.png", cols=8, rows=1), + render_ansi_to_png("frame b", out_dir / "b.png", cols=8, rows=1), + ] + gif_path = out_dir / "movie.gif" + result = run_cli("gif", *frames, str(gif_path), "--duration-ms", "150") + assert result.returncode == 0, result.stderr + assert gif_path.exists() diff --git a/tests/test_grok_target.py b/tests/test_grok_target.py new file mode 100644 index 0000000..ae9dc0c --- /dev/null +++ b/tests/test_grok_target.py @@ -0,0 +1,73 @@ +"""Capability probe against a real, complex third-party TUI: the `grok` CLI. + +Everything here is strictly read-only from grok's point of view: we only ever +send Down/Up (menu highlight movement) and never press Enter on its prompt, +so no prompt is ever submitted to a model and no tool/agent action can run. +Skipped entirely when `grok` isn't installed, since it's a local dev tool, +not a project dependency. +""" +from __future__ import annotations + +import shutil +import time +import uuid +from pathlib import Path + +import pytest +from PIL import Image + +from tmux_shot import TmuxApp, frames_to_gif + +pytestmark = pytest.mark.skipif(shutil.which("grok") is None, reason="grok CLI not installed on this machine") + + +def test_grok_doctor_screen_renders_correctly(out_dir: Path) -> None: + session = f"tmux_shot_grok_doctor_{uuid.uuid4().hex[:8]}" + app = TmuxApp(session, command="grok doctor", width=100, height=35) + try: + text = app.wait_for("issue", timeout=15) + assert "Checks not completed" in text + + path = app.screenshot(out_dir / "doctor.png") + img = Image.open(path) + assert img.width == 100 * round(img.width / 100) # sane multiple-of-cols size + assert img.height > 0 + finally: + try: + app.kill() + except Exception: + pass + + +def test_grok_welcome_screen_and_menu_navigation_produce_a_movie(out_dir: Path, tmp_path: Path) -> None: + scratch_cwd = tmp_path / "grok_scratch" + scratch_cwd.mkdir() + + session = f"tmux_shot_grok_welcome_{uuid.uuid4().hex[:8]}" + app = TmuxApp(session, command=f"grok --cwd {scratch_cwd}", width=100, height=35) + frames: list[str] = [] + try: + app.wait_for("Grok Build", timeout=15) + time.sleep(0.3) # let the box-drawing/logo finish painting after the text appears + frames.append(app.screenshot(out_dir / "01_welcome.png")) + + # Move the menu highlight down twice, screenshotting between moves. Arrow + # keys only -- never Enter, so nothing gets submitted to grok's model. + for i, _ in enumerate(range(2), start=2): + app.send_keys("Down", enter=False) + time.sleep(0.3) + frames.append(app.screenshot(out_dir / f"{i:02d}_menu.png")) + + assert len(frames) == 3 + for f in frames: + assert Path(f).stat().st_size > 0 + + movie_path = frames_to_gif(frames, out_dir / "grok_welcome.gif", duration_ms=700) + with Image.open(movie_path) as gif: + assert gif.is_animated + assert gif.n_frames == 3 + finally: + try: + app.kill() + except Exception: + pass diff --git a/tests/test_render.py b/tests/test_render.py new file mode 100644 index 0000000..5ef131b --- /dev/null +++ b/tests/test_render.py @@ -0,0 +1,62 @@ +from __future__ import annotations + +from pathlib import Path + +from PIL import Image + +from tmux_shot.render import frames_to_gif, render_ansi_to_png + + +def test_render_plain_text_produces_correctly_sized_image(out_dir: Path) -> None: + path = render_ansi_to_png("hello world", out_dir / "plain.png", cols=20, rows=3, font_size=14) + img = Image.open(path) + assert img.size[0] > 0 and img.size[1] > 0 + # 20 cols x 3 rows of cells, image should be a whole multiple of cell size in each axis + assert img.width % 20 == 0 + assert img.height % 3 == 0 + + +def test_render_resolves_truecolor_background_and_foreground(out_dir: Path) -> None: + # SGR 38;2 = truecolor fg, 48;2 = truecolor bg. Pure red bg, pure green text. + ansi = "\x1b[48;2;255;0;0m\x1b[38;2;0;255;0mX\x1b[0m" + path = render_ansi_to_png(ansi, out_dir / "color.png", cols=1, rows=1, font_size=20) + img = Image.open(path) + # corner pixel is background-only (no glyph ink there) + corner = img.getpixel((0, 0)) + assert corner == (255, 0, 0) + + +def test_render_reverse_video_swaps_fg_and_bg(out_dir: Path) -> None: + ansi = "\x1b[38;2;255;0;0m\x1b[48;2;0;0;255m\x1b[7mX\x1b[0m" # red fg, blue bg, reversed + path = render_ansi_to_png(ansi, out_dir / "reverse.png", cols=1, rows=1, font_size=20) + img = Image.open(path) + # reversed: background should now be the (originally-foreground) red + assert img.getpixel((0, 0)) == (255, 0, 0) + + +def test_render_handles_ansi_16_color_names(out_dir: Path) -> None: + ansi = "\x1b[42mX\x1b[0m" # classic SGR green background + path = render_ansi_to_png(ansi, out_dir / "named.png", cols=1, rows=1, font_size=20) + img = Image.open(path) + assert img.getpixel((0, 0)) == (0, 205, 0) + + +def test_frames_to_gif_stitches_multiple_pngs_into_one_animation(out_dir: Path) -> None: + frame_paths = [] + for i, label in enumerate(["one", "two", "three"]): + p = render_ansi_to_png(f"frame {label}", out_dir / f"f{i}.png", cols=10, rows=1, font_size=14) + frame_paths.append(p) + + gif_path = frames_to_gif(frame_paths, out_dir / "movie.gif", duration_ms=100) + assert Path(gif_path).exists() + + with Image.open(gif_path) as gif: + assert gif.is_animated + assert gif.n_frames == 3 + + +def test_frames_to_gif_rejects_empty_input(out_dir: Path) -> None: + import pytest + + with pytest.raises(ValueError): + frames_to_gif([], out_dir / "empty.gif") diff --git a/tests/test_tmux_control.py b/tests/test_tmux_control.py new file mode 100644 index 0000000..8b7b03f --- /dev/null +++ b/tests/test_tmux_control.py @@ -0,0 +1,61 @@ +from __future__ import annotations + +import time +from pathlib import Path + +from PIL import Image + +from tmux_shot import TmuxApp + + +def test_session_is_created_and_idempotent_on_reattach(session_name: str, demo_app: TmuxApp) -> None: + again = TmuxApp(session_name) # should attach, not recreate + assert again.session.session_name == demo_app.session.session_name + + +def test_capture_text_shows_initial_frame(demo_app: TmuxApp) -> None: + time.sleep(0.3) + text = demo_app.capture_text() + assert "tmux-shot demo TUI" in text + assert "@" in text + + +def test_send_keys_multi_token_string_moves_cursor(demo_app: TmuxApp) -> None: + """Regression test: a single send_keys("Down Down Right") call must be split into + separate tmux key-name arguments, not sent as one literal string (which tmux would + silently mis-send as garbage keystrokes that don't move anything).""" + time.sleep(0.3) + before = demo_app.capture_text() + + demo_app.send_keys("Down Down Right Right Right", enter=False) + time.sleep(0.3) + after = demo_app.capture_text() + + assert before != after + after_lines = [l for l in after.splitlines() if "@" in l] + assert after_lines, "cursor marker should still be present somewhere" + + +def test_send_keys_literal_types_text_verbatim(demo_app: TmuxApp) -> None: + # 'q' quits the demo curses app back to the underlying shell. + time.sleep(0.3) + assert "tmux-shot demo TUI" in demo_app.capture_text() + demo_app.send_keys("q", literal=True, enter=False) + time.sleep(0.5) + assert "tmux-shot demo TUI" not in demo_app.capture_text() + + +def test_screenshot_produces_a_valid_sized_png(demo_app: TmuxApp, out_dir: Path) -> None: + time.sleep(0.3) + path = demo_app.screenshot(out_dir / "shot.png") + img = Image.open(path) + assert img.width > 0 and img.height > 0 + # 80 cols x 24 rows pane + assert img.width % 80 == 0 + assert img.height % 24 == 0 + + +def test_capture_ansi_includes_escape_codes(demo_app: TmuxApp) -> None: + time.sleep(0.3) + ansi = demo_app.capture_ansi() + assert "\x1b[" in ansi