diff --git a/memwalk/__init__.py b/memwalk/__init__.py index d431b28..2e572c9 100644 --- a/memwalk/__init__.py +++ b/memwalk/__init__.py @@ -1,3 +1,3 @@ """memwalk — ask AI about any codebase via cached SSM state.""" -__version__ = "0.3.0" +__version__ = "0.4.0" diff --git a/memwalk/cli.py b/memwalk/cli.py index 0968d77..647fd0f 100644 --- a/memwalk/cli.py +++ b/memwalk/cli.py @@ -77,6 +77,11 @@ def digest( cfg = load_config() source = Path(path).expanduser().resolve() + if n_ctx is None: + from .gpu import auto_n_ctx + chosen = auto_n_ctx() + console.print(f"[dim]Adaptive n_ctx = {chosen:,}[/dim]") + if split: with console.status(f"Discovering subdirectories in {source}…"): results = engine_digest_subdirs(cfg, source, n_ctx=n_ctx, diff --git a/memwalk/engine.py b/memwalk/engine.py index 53a8b01..7e3c513 100644 --- a/memwalk/engine.py +++ b/memwalk/engine.py @@ -12,6 +12,7 @@ from memba import Session from . import cache, corpus from .config import Config +from .gpu import auto_n_ctx # Prompt that frames the ingest call so the assistant turn stored in state # is *substantive* (not "noted") — avoids the contextual inertia bug we @@ -64,7 +65,7 @@ def digest( if not source_path.exists() or not source_path.is_dir(): raise NotADirectoryError(source_path) - n_ctx = n_ctx or cfg.n_ctx + n_ctx = auto_n_ctx(n_ctx if n_ctx else None) files = corpus.collect_files(source_path) if not files: raise RuntimeError(f"No source files found under {source_path}") diff --git a/memwalk/gpu.py b/memwalk/gpu.py new file mode 100644 index 0000000..88ba559 --- /dev/null +++ b/memwalk/gpu.py @@ -0,0 +1,74 @@ +"""GPU VRAM probing and adaptive n_ctx estimation.""" + +from __future__ import annotations + +import shutil +import subprocess + +_NEMOTRON_PROFILE: list[tuple[int, float]] = [ + (8192, 4.5), + (16384, 5.0), + (32768, 5.5), + (65536, 6.5), + (131072, 8.5), + (262144, 12.0), + (524288, 18.0), +] + + +def probe_free_vram_mb() -> int | None: + """Return free VRAM in MiB, or None if no NVIDIA GPU detected.""" + if shutil.which("nvidia-smi"): + try: + out = subprocess.check_output( + [ + "nvidia-smi", + "--query-gpu=memory.free", + "--format=csv,noheader,nounits", + ], + text=True, + timeout=5, + ) + first_line = out.strip().splitlines()[0].strip() + return int(float(first_line)) + except Exception: + pass + + try: + import pynvml # type: ignore[import-untyped] + pynvml.nvmlInit() + handle = pynvml.nvmlDeviceGetHandleByIndex(0) + info = pynvml.nvmlDeviceGetMemoryInfo(handle) + pynvml.nvmlShutdown() + return info.free // (1024 * 1024) + except Exception: + pass + + return None + + +def estimate_max_n_ctx(free_vram_mb: int, *, headroom_mb: int = 1536) -> int: + """Return the largest n_ctx from the profile that fits in free VRAM.""" + usable_mb = free_vram_mb - headroom_mb + if usable_mb <= 0: + return 8192 + + best = 8192 + for n_ctx, needed_gb in _NEMOTRON_PROFILE: + needed_mb = int(needed_gb * 1024) + if needed_mb <= usable_mb: + best = n_ctx + else: + break + return best + + +def auto_n_ctx(preferred: int | None = None) -> int: + """Return n_ctx to use: preferred if given, else GPU-adaptive.""" + if preferred is not None and preferred > 0: + return preferred + + free_mb = probe_free_vram_mb() + if free_mb is None: + return 32768 + return estimate_max_n_ctx(free_mb) diff --git a/pyproject.toml b/pyproject.toml index dfc35e2..35ba8f3 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "memwalk" -version = "0.3.0" +version = "0.4.0" description = "Ask AI about any codebase — local, cached, SSM-state-backed exploration via memba + Nemotron" readme = "README.md" license = { text = "MIT" }