From ea52dad4097c3792bada38790c29f1cf34fec198 Mon Sep 17 00:00:00 2001 From: Dustin Washington Date: Thu, 8 Oct 2026 08:34:39 -0400 Subject: [PATCH 1/2] #709: Update watcher to catch up on files added during the watcher stop/start cycle --- .gitignore | 25 ++++----- cecli/io.py | 6 ++ cecli/tui/io.py | 5 ++ cecli/watch.py | 92 ++++++++++++++++++++++++++++++- tests/basic/test_watch.py | 113 ++++++++++++++++++++++++++++++++++++++ 5 files changed, 224 insertions(+), 17 deletions(-) diff --git a/.gitignore b/.gitignore index 22b4725523a..d8013b3c2b6 100644 --- a/.gitignore +++ b/.gitignore @@ -1,17 +1,14 @@ -# Ignore everything -* +# Ignore everything at the top level +/* -# But descend into directories -!*/ - -# Recursively allow files under subtree -!/.github/** -!/cecli/** -!/benchmark/** -!/docker/** -!/requirements/** -!/scripts/** -!/tests/** +# Then allow only specific directories; their contents are tracked +!/.github +!/cecli +!/benchmark +!/docker +!/requirements +!/scripts +!/tests # Specific Files !/.dockerignore @@ -47,4 +44,4 @@ __pycache__/ cecli/website/_site/* cecli/website/.sass-cache/* cecli/website/.docmd-*/* -cecli/website/node_modules/* \ No newline at end of file +cecli/website/node_modules/* diff --git a/cecli/io.py b/cecli/io.py index a3bf286315f..c696e4e1c33 100644 --- a/cecli/io.py +++ b/cecli/io.py @@ -988,6 +988,12 @@ def _(event): if not multiline_input: if self.file_watcher: self.file_watcher.start() + + # The catch-up scan may have found files changed while the + # watcher was stopped; handle them before prompting. + if self.file_watcher.changed_files: + return self.file_watcher.process_changes() + if self.clipboard_watcher: self.clipboard_watcher.start() diff --git a/cecli/tui/io.py b/cecli/tui/io.py index 9db636bbb9c..39191c29651 100644 --- a/cecli/tui/io.py +++ b/cecli/tui/io.py @@ -492,6 +492,11 @@ async def get_input( if not self.file_watcher.is_running: self.file_watcher.start() + # Handle files the catch-up scan found before waiting for input + if self.file_watcher.changed_files: + cmd = self.file_watcher.process_changes() + return cmd + # Check if we were interrupted by a file change if self.interrupted: cmd = self.file_watcher.process_changes() diff --git a/cecli/watch.py b/cecli/watch.py index 412e7cffffe..4804e6ac833 100644 --- a/cecli/watch.py +++ b/cecli/watch.py @@ -1,5 +1,7 @@ +import os import re import threading +import time from pathlib import Path from typing import Optional @@ -16,9 +18,10 @@ def load_gitignores(gitignore_paths: list[Path]) -> Optional[PathSpec]: if not gitignore_paths: return None - patterns = [ + always_ignore = [ ".cecli*", ".git", + ".git/", # Git metadata # Common editor backup/temp files "*~", # Emacs/vim backup "*.bak", # Generic backup @@ -45,14 +48,22 @@ def load_gitignores(gitignore_paths: list[Path]) -> Optional[PathSpec]: # Environment files ".env", # Environment variables ".venv/", # Python virtual environments + "venv/", # Python virtual environments + "env/", # Python virtual environments + ".tox/", # Tox environments + "*.egg-info/", # Python package metadata "node_modules/", # Node.js dependencies "vendor/", # Various dependencies # Logs and caches "*.log", # Log files ".cache/", # Cache directories ".pytest_cache/", # Python test cache + ".ruff_cache/", # Ruff cache + ".mypy_cache/", # Mypy cache "coverage/", # Code coverage reports - ] # Always ignore + ] + + patterns = [] for path in gitignore_paths: if path.exists(): try: @@ -61,6 +72,11 @@ def load_gitignores(gitignore_paths: list[Path]) -> Optional[PathSpec]: except Exception: pass # Ignore files that can't be read + # Always-ignore patterns go last: in gitignore semantics the last matching + # pattern wins, so a repo's negations (e.g. "!*/") cannot re-enable heavy + # directories like virtualenvs, caches or .git. + patterns.extend(always_ignore) + return PathSpec.from_lines(GitWildMatchPattern, patterns) if patterns else None @@ -75,13 +91,14 @@ class FileWatcher: def __init__(self, coder, gitignores=None, verbose=False, root=None): self.coder = coder self.io = coder.io - self.root = Path(root) if root else Path(coder.root) + self.root = (Path(root) if root else Path(coder.root)).absolute() self.verbose = verbose self.stop_event = None self.watcher_thread = None self.changed_files = set() self.gitignores = gitignores self.is_running = False + self.last_scan_time = time.time() self.gitignore_spec = load_gitignores( [Path(g) for g in self.gitignores] if self.gitignores else [] @@ -170,6 +187,14 @@ def start(self): self.stop_event = threading.Event() self.changed_files = set() + # Watchfiles only sees changes that happen while it is running, so pick + # up anything created while the watcher was stopped before we begin. + try: + self.catch_up_scan() + except Exception as e: + if self.verbose: + dump(f"File watcher catch-up scan error: {e}") + self.is_running = True self.watcher_thread = threading.Thread(target=self.watch_files, daemon=True) self.watcher_thread.start() @@ -283,6 +308,67 @@ def get_ai_comments(self, filepath): return None, None, None return line_nums, comments, has_action + def catch_up_scan(self): + """Catch AI comments in files changed since the last scan + + watchfiles only reports changes that happen while it is running, so a + file created while the watcher was stopped (or inside a directory created + after watching began) would otherwise be missed. This mtime-gated scan + looks only at files changed since the previous scan and applies the same + gitignore and comment rules as the live watcher. + """ + scan_started = time.time() + threshold = self.last_scan_time + found = set() + + for root in self.get_roots_to_watch(): + root_path = Path(root) + if root_path.is_file(): + try: + candidates = [root_path] if root_path.stat().st_mtime >= threshold else [] + except OSError: + candidates = [] + else: + candidates = self._iter_recent_files(root_path, threshold) + + for path in candidates: + try: + if self.filter_func(None, str(path)): + found.add(str(path.absolute())) + except Exception: + continue + + self.last_scan_time = scan_started + + if found: + self.changed_files.update(found) + + return bool(found) + + def _iter_recent_files(self, directory, threshold): + """Yield files under directory modified at or after threshold""" + for dirpath, dirnames, filenames in os.walk(directory): + rel_dir = os.path.relpath(dirpath, self.root).replace(os.sep, "/") + prefix = "" if rel_dir == "." else rel_dir + "/" + + if self.gitignore_spec: + dirnames[:] = [ + name + for name in dirnames + if not self.gitignore_spec.match_file(prefix + name + "/") + ] + + for name in filenames: + if self.gitignore_spec and self.gitignore_spec.match_file(prefix + name): + continue + + path = Path(dirpath) / name + try: + if path.stat().st_mtime >= threshold: + yield path + except OSError: + continue + def main(): """Example usage of the file watcher""" diff --git a/tests/basic/test_watch.py b/tests/basic/test_watch.py index eebd8b57909..181db3ea91c 100644 --- a/tests/basic/test_watch.py +++ b/tests/basic/test_watch.py @@ -1,3 +1,4 @@ +import time from pathlib import Path from cecli.dump import dump # noqa @@ -164,3 +165,115 @@ def test_ai_comment_pattern(): len(lisp_lines) == lisp_expected ), f"Expected {lisp_expected} AI comments in Lisp fixture, found {len(lisp_lines)}" assert lisp_has_bang == "!", "Expected at least one bang (!) comment in Lisp fixture" + + +def test_catch_up_scan_detects_recent_file(tmp_path): + io = InputOutput(pretty=False, fancy_input=False, yes=False) + coder = MinimalCoder(io) + watcher = FileWatcher(coder, root=tmp_path) + watcher.last_scan_time = 0 + + new_file = tmp_path / "new.py" + new_file.write_text("# ai!\n") + + assert watcher.catch_up_scan() + assert str(new_file.absolute()) in watcher.changed_files + + +def test_catch_up_scan_ignores_old_files(tmp_path): + io = InputOutput(pretty=False, fancy_input=False, yes=False) + coder = MinimalCoder(io) + watcher = FileWatcher(coder, root=tmp_path) + + old_file = tmp_path / "old.py" + old_file.write_text("# ai!\n") + watcher.last_scan_time = time.time() + 5 + + assert not watcher.catch_up_scan() + assert watcher.changed_files == set() + + +def test_catch_up_scan_respects_gitignore(tmp_path): + gitignore = tmp_path / ".gitignore" + gitignore.write_text("ignored/\n") + (tmp_path / "ignored").mkdir() + (tmp_path / "ignored" / "hidden.py").write_text("# ai!\n") + (tmp_path / "kept.py").write_text("# ai!\n") + + io = InputOutput(pretty=False, fancy_input=False, yes=False) + coder = MinimalCoder(io) + watcher = FileWatcher(coder, gitignores=[gitignore], root=tmp_path) + watcher.last_scan_time = 0 + watcher.catch_up_scan() + + assert str((tmp_path / "kept.py").absolute()) in watcher.changed_files + assert str((tmp_path / "ignored" / "hidden.py").absolute()) not in watcher.changed_files + + +def test_catch_up_scan_advances_threshold(tmp_path): + io = InputOutput(pretty=False, fancy_input=False, yes=False) + coder = MinimalCoder(io) + watcher = FileWatcher(coder, root=tmp_path) + watcher.last_scan_time = 0 + (tmp_path / "a.py").write_text("# ai!\n") + + assert watcher.catch_up_scan() + watcher.changed_files = set() + + # Nothing changed since the previous scan, so the same file is not re-found. + assert not watcher.catch_up_scan() + assert watcher.changed_files == set() + + +def test_process_changes_consumes_catch_up(tmp_path): + io = InputOutput(pretty=False, fancy_input=False, yes=False) + coder = MinimalCoder(io) + watcher = FileWatcher(coder, root=tmp_path) + watcher.last_scan_time = 0 + + target = tmp_path / "a.py" + target.write_text("# ai!\n") + + assert watcher.catch_up_scan() + res = watcher.process_changes() + + assert res + assert str(target.absolute()) in coder.abs_fnames + assert not watcher.is_running + + +def test_start_runs_catch_up_scan(tmp_path): + io = InputOutput(pretty=False, fancy_input=False, yes=False) + coder = MinimalCoder(io) + watcher = FileWatcher(coder, root=tmp_path) + watcher.last_scan_time = 0 + + target = tmp_path / "a.py" + target.write_text("# ai!\n") + + watcher.start() + try: + assert str(target.absolute()) in watcher.changed_files + finally: + watcher.stop() + + +def test_catch_up_scan_gates_file_roots(tmp_path): + gitignore = tmp_path / ".gitignore" + gitignore.write_text("ignored/\n") + + top_file = tmp_path / "top.py" + top_file.write_text("# ai!\n") + + io = InputOutput(pretty=False, fancy_input=False, yes=False) + coder = MinimalCoder(io) + watcher = FileWatcher(coder, gitignores=[gitignore], root=tmp_path) + watcher.last_scan_time = 0 + + assert watcher.catch_up_scan() + assert str(top_file.absolute()) in watcher.changed_files + + # top.py is a watched file root, but its mtime predates the new threshold. + watcher.changed_files = set() + assert not watcher.catch_up_scan() + assert watcher.changed_files == set() From 954553a525aef831dc0701bf40ea257b63d7f108 Mon Sep 17 00:00:00 2001 From: Dustin Washington Date: Thu, 8 Oct 2026 08:43:28 -0400 Subject: [PATCH 2/2] Add cecli attribution to openrouter --- cecli/helpers/llms/providers/openrouter.py | 32 ++++++++++++++++++++-- 1 file changed, 29 insertions(+), 3 deletions(-) diff --git a/cecli/helpers/llms/providers/openrouter.py b/cecli/helpers/llms/providers/openrouter.py index 50c8b9a4922..d4bbc6ad17d 100644 --- a/cecli/helpers/llms/providers/openrouter.py +++ b/cecli/helpers/llms/providers/openrouter.py @@ -1,21 +1,47 @@ """OpenRouter provider adapter for the llms package. OpenRouter speaks OpenAI-compatible /v1/chat/completions with Bearer auth, so -the base adapter's defaults apply. Reasoning arrives via ``message.reasoning`` -/ ``message.reasoning_details`` (not ``reasoning_content``); the generic +the base adapter's defaults apply. It additionally sends the app-attribution +headers OpenRouter uses to rank and list cecli (``HTTP-Referer`` / +``X-OpenRouter-Title`` / ``X-OpenRouter-Categories``). Reasoning arrives via +``message.reasoning`` / +``message.reasoning_details`` (not ``reasoning_content``); the generic :func:`cecli.helpers.llms.utils.extract_reasoning` already handles all three shapes, so no normalize override is needed here. """ from __future__ import annotations +from typing import Any, Dict, Optional + from .base import ProviderAdapter +#: App-attribution metadata OpenRouter uses to create/rank the app page. +_APP_URL = "https://cecli.dev" +_APP_TITLE = "cecli" +_APP_CATEGORIES = "cli-agent" + class OpenRouterProvider(ProviderAdapter): - """OpenRouter: Bearer auth + generic reasoning extraction.""" + """OpenRouter: Bearer auth + app attribution + generic reasoning extraction.""" provider: str = "openrouter" + def build_headers( + self, + resolved: Dict[str, Any], + key: Optional[str], + family: str, + headers: Dict[str, str], + ) -> Dict[str, str]: + """Add OpenRouter's app-attribution headers (explicit ones win).""" + merged = super().build_headers(resolved, key, family, headers) + + merged.setdefault("HTTP-Referer", _APP_URL) + merged.setdefault("X-OpenRouter-Title", _APP_TITLE) + merged.setdefault("X-OpenRouter-Categories", _APP_CATEGORIES) + + return merged + __all__ = ["OpenRouterProvider"]