diff --git a/README.md b/README.md index 2293461..49b70d4 100644 --- a/README.md +++ b/README.md @@ -38,7 +38,7 @@ manually from the app (**Start AI** in the sidebar), not at boot. ```powershell # 0. Allow scripts to run (PowerShell blocks unsigned scripts by default, which -# stops Nexus's CLI from working). Process scope covers only one single window; +# stops Nexus's CLI from working). Process scope covers only this window; # LocalMachine makes it permanent so ncp works from every future shell (run # PowerShell as Administrator for the LocalMachine line). Set-ExecutionPolicy -Scope Process -ExecutionPolicy Bypass -Force diff --git a/install-windows.ps1 b/install-windows.ps1 index 18c0e8a..6e17e4f 100644 --- a/install-windows.ps1 +++ b/install-windows.ps1 @@ -178,6 +178,30 @@ try { Write-Warn " $RepoRoot\management" } +# -- Legacy profile shim check -------------------------------------------------- +# An earlier version of this installer registered ncp as a `function ncp` in the +# PowerShell profile instead of management\ncp.cmd (see the "why" above). Profile +# functions resolve before PATH executables, so a leftover one silently shadows +# the real ncp.cmd - `ncp` still runs, just the wrong code, with no error. We do +# not rewrite the user's profile automatically: it is their file, and a blind +# strip risks mangling whatever else lives in it alongside it. Just flag it. +Write-Step "Checking for a stale ncp() in your PowerShell profile" +$DocsRoot = [Environment]::GetFolderPath("MyDocuments") +$ProfileCandidates = @( + (Join-Path $DocsRoot "WindowsPowerShell\Microsoft.PowerShell_profile.ps1"), + (Join-Path $DocsRoot "PowerShell\Microsoft.PowerShell_profile.ps1") +) +$Shadowed = $ProfileCandidates | Where-Object { + (Test-Path $_) -and (Select-String -Path $_ -Pattern '^\s*function\s+ncp\b' -Quiet) +} +if ($Shadowed) { + Write-Warn "Found an old 'function ncp' that will shadow the real ncp.cmd:" + foreach ($p in $Shadowed) { Write-Warn " $p" } + Write-Warn " Remove that function from the file(s) above so ncp resolves to ncp.cmd." +} else { + Write-OK "No stale ncp() in your PowerShell profile" +} + # -- Desktop shortcut ---------------------------------------------------------- Write-Step "Creating desktop shortcut" try { @@ -211,10 +235,11 @@ try { # Read them instead of hardcoding, so the installer can never pull one model # while the backend defaults to another. Push-Location $RepoRoot -$ChatModel = (& $VenvPy -c "from synapse.nexus_config import DEFAULT_CHAT_MODEL as m; print(m)") -$MemModel = (& $VenvPy -c "from synapse.nexus_config import DEFAULT_MEMORY_MODEL as m; print(m)") +$ChatModel = (& $VenvPy -c "from synapse.nexus_config import DEFAULT_CHAT_MODEL as m; print(m)") +$MemModel = (& $VenvPy -c "from synapse.nexus_config import DEFAULT_MEMORY_MODEL as m; print(m)") +$EmbedModel = (& $VenvPy -c "from synapse.nexus_config import DEFAULT_EMBED_MODEL as m; print(m)") Pop-Location -if ($LASTEXITCODE -ne 0 -or -not $ChatModel -or -not $MemModel) { +if ($LASTEXITCODE -ne 0 -or -not $ChatModel -or -not $MemModel -or -not $EmbedModel) { Write-Fail "Could not read the default models from synapse\nexus_config.py - the venv install is broken" } @@ -255,7 +280,7 @@ Pop-Location if ($seedOk) { Write-OK "Default model set to $ChatModel" } else { Write-Warn "Could not persist default model - pick it at the top of the chat instead." } -Write-Step "Pulling models ($ChatModel for chat, $MemModel for memory)" +Write-Step "Pulling models ($ChatModel for chat, $MemModel for memory, $EmbedModel for recall)" # A prompt, not "press Ctrl+C to skip": Ctrl+C in PowerShell 5.1 terminates the # whole script, so the escape hatch the installer advertised was also the one # thing that stopped it finishing - no Ollama cleanup, no summary, no window @@ -277,6 +302,10 @@ if ($pullAnswer -match '^\s*(n|no)\s*$') { ollama pull $MemModel if ($LASTEXITCODE -eq 0) { Write-OK "$MemModel ready (memory curator)" } else { Write-Warn "$MemModel pull skipped/failed - the memory service will fall back to the chat model." } + + ollama pull $EmbedModel + if ($LASTEXITCODE -eq 0) { Write-OK "$EmbedModel ready (conversation recall)" } + else { Write-Warn "$EmbedModel pull skipped/failed - recall will fall back to lexical search." } } # -- Make Ollama manual-start (NexusOS owns the lifecycle) ---------------------- @@ -326,15 +355,32 @@ Write-Host " A process reads PATH once, when it starts. This shell started befo Write-Host " the installer put management\ on PATH, so it will never see ncp no" -ForegroundColor DarkGray Write-Host " matter what you run here. The next terminal you open will." -ForegroundColor DarkGray Write-Host "" -Read-Host " Press Enter to close this window" -# Stop-Process on our own PID, not `exit`: `exit` only ends the window when the -# host was started with -File. Run interactively - `& .\install-windows.ps1` from -# a shell already open, which is the common case - it would just return to the -# prompt in a session whose PATH is permanently stale. Killing the host closes -# the window either way, which is the point: the user cannot accidentally keep -# using a shell where ncp will never resolve. -# Caveat: a Start-Transcript in this window never reaches Stop-Transcript. The -# transcript is flushed as it is written, so the content survives without the -# closing footer. -Stop-Process -Id $PID +# Skip the wait entirely when stdin is redirected: nobody is at the keyboard to +# press anything, and both ReadKey and Read-Host were confirmed (empirically, +# not just in theory) to hang indefinitely in that case instead of failing +# fast - Read-Host only fails fast under an explicit -NonInteractive flag, +# which is not how the self-elevation Start-Process above launches this script. +if (-not [Console]::IsInputRedirected) { + Write-Host " Press any key to close this window..." -ForegroundColor Yellow + try { $null = $Host.UI.RawUI.ReadKey("NoEcho,IncludeKeyDown") } catch { Read-Host | Out-Null } +} + +# $OwnsWindow means this is the elevated instance the self-elevation block above +# launched with `-File`, so a clean `exit` closes it - and does so with exit code +# 0, which is what lets Windows Terminal's closeOnExit actually close the tab +# instead of treating an abrupt kill as a crash and falling back to a fresh +# shell in the pane. Without $OwnsWindow we are in a shell the user already had +# open (they ran an admin PowerShell and invoked the script directly, so it was +# never re-launched with -File) - `exit` there would just return to the prompt +# in a session whose PATH is permanently stale, so kill the host instead; the +# point is the user cannot accidentally keep using a shell where ncp never +# resolves. +# Caveat: a Start-Transcript in this window never reaches Stop-Transcript in the +# Stop-Process path. The transcript is flushed as it is written, so the content +# survives without the closing footer. +if ($OwnsWindow) { + exit 0 +} else { + Stop-Process -Id $PID +} diff --git a/management/ncp.py b/management/ncp.py index 701a273..cad0df5 100644 --- a/management/ncp.py +++ b/management/ncp.py @@ -304,14 +304,19 @@ def stop_service(svc: Service) -> bool: def start_ollama() -> None: """Driven through the backend endpoint (the path the control panel uses) rather than launching the binary, because OllamaManager owns model and GPU - selection. Requires the backend to be up.""" + selection. Requires the backend to be up. + + Long timeout: the endpoint blocks until the model is warmed - weights read + off disk into RAM/VRAM - which the web UI's own "Loading model..." button + state calls out as routinely taking about a minute, not just the Ollama + process launching.""" print("Starting OLLAMA...") req = urllib.request.Request("http://localhost:8000/ollama/start", method="POST") try: - urllib.request.urlopen(req, timeout=30).read(1) + urllib.request.urlopen(req, timeout=180).read(1) print("NEXUS OLLAMA STARTED") - except Exception: - print(" OLLAMA start request failed (backend not reachable on :8000)") + except Exception as e: + print(f" OLLAMA start failed: {e}") def stop_ollama() -> None: @@ -339,6 +344,7 @@ def cmd_start(target) -> None: launch(SERVICES["memory"]); wait_for_port(SERVICES["memory"]) elif target in ("--backend", "-b"): launch(SERVICES["backend"]); wait_for_port(SERVICES["backend"]) + start_ollama() elif target in ("--frontend", "-f"): launch(SERVICES["frontend"]); wait_for_port(SERVICES["frontend"]) elif target in ("--ai", "-a"): @@ -349,6 +355,7 @@ def cmd_start(target) -> None: launch(SERVICES["backend"]) wait_for_port(SERVICES["memory"]) wait_for_port(SERVICES["backend"]) + start_ollama() launch(SERVICES["frontend"]) wait_for_port(SERVICES["frontend"]) else: @@ -656,7 +663,7 @@ Commands: --memory, -m Start only the memory service --frontend,-f Start only the frontend --backend, -b Start only the backend - --ai, -a Start the AI (Ollama) — manual; not started by default + --ai, -a Start only the AI (Ollama); `start`/`start -b` already include it stop Stop ALL Nexus services --memory, -m Stop only the memory service diff --git a/synapse/nexus_config.py b/synapse/nexus_config.py index 9e2bffe..fa32013 100644 --- a/synapse/nexus_config.py +++ b/synapse/nexus_config.py @@ -22,7 +22,7 @@ except Exception: VERSION = "0.0.0" # --- MODEL DEFAULTS --- -# Single source of truth for the two models NexusOS ships with. The installers +# Single source of truth for the three models NexusOS ships with. The installers # pull and pin these, the runtime falls back to them; keeping them in one place # is what stops installer and backend from drifting apart. # @@ -30,8 +30,12 @@ except Exception: # 4 GB GPU into CPU/RAM). Chosen over Qwen3 because Qwen3 is a reasoning model: # smart only with its slow step, weak without it. # Memory: mistral - the curator that extracts facts and titles conversations. +# Embed: nomic-embed-text - powers semantic recall of past conversations +# (OllamaManager.embed / store.semantic_search_conversations). Without it, +# recall silently degrades to lexical substring matching. DEFAULT_CHAT_MODEL = "llama3.1:8b" DEFAULT_MEMORY_MODEL = "mistral:latest" +DEFAULT_EMBED_MODEL = "nomic-embed-text" # --- CORE DIRECTORIES --- DATA_DIR = PROJECT_ROOT / "data" @@ -144,7 +148,7 @@ settings = Settings() # explicit exports for static checkers and IDEs __all__ = ["Settings", "settings", "path", "VERSION", - "DEFAULT_CHAT_MODEL", "DEFAULT_MEMORY_MODEL", + "DEFAULT_CHAT_MODEL", "DEFAULT_MEMORY_MODEL", "DEFAULT_EMBED_MODEL", "PROJECT_ROOT", "DATA_DIR", "MODELS_DIR", "RUNTIME_DIR", "MEMORY_DIR", "LOGS_DIR", "PLAYBOOK_DIR", "UPLOADS_DIR", "EXPORTS_DIR", "MEMORY_DB", diff --git a/synapse/ollama_manager.py b/synapse/ollama_manager.py index f3335e3..27bfdbd 100644 --- a/synapse/ollama_manager.py +++ b/synapse/ollama_manager.py @@ -9,7 +9,7 @@ import signal from pathlib import Path from typing import Any -from .nexus_config import settings, DEFAULT_CHAT_MODEL +from .nexus_config import settings, DEFAULT_CHAT_MODEL, DEFAULT_EMBED_MODEL from .memory.store import store OLLAMA_PORT = 11434 @@ -193,6 +193,7 @@ class OllamaManager: def __init__(self, runtime_dir=None): self.process = None self.running = False + self._available = None # see is_available() self.runtime_dir = Path(runtime_dir) if runtime_dir else Path(__file__).resolve().parent.parent / "runtime" (self.runtime_dir / "logs").mkdir(parents=True, exist_ok=True) @@ -241,17 +242,45 @@ class OllamaManager: except Exception as e: _log.warning("warm: preload failed: %s", e) + def _serve_env(self, gpu_env: dict) -> dict: + """Environment for a `serve` we spawn. One definition - this was + duplicated verbatim in two methods, so the Windows carve-out below had + to be fixed in both places or the two paths would disagree.""" + env = os.environ.copy() + env["OLLAMA_HOST"] = self._api_base + # Every platform uses the project's own model store. Windows used to be + # exempt, because the installer pulled with a bare `ollama pull` into + # %USERPROFILE%\.ollama and a NexusOS-spawned serve pointed elsewhere + # would not have seen those models. The installer sets OLLAMA_MODELS for + # its pulls now (and migrates an existing store), so the exemption just + # meant Windows kept models somewhere `ncp models` could not see. + env["OLLAMA_MODELS"] = str(settings.models_dir) + env.update(gpu_env) + return env + def is_available(self): - bin_path = _ollama_bin() - try: - subprocess.run([bin_path, "--version"], capture_output=True, check=True, timeout=5) - return True - except Exception: - return False + # ponytail: cached for the life of the process. This spawns a subprocess, + # and /status calls it on every poll - the frontend polls continuously, + # so it was a process spawn per tick to answer a question whose answer + # cannot change without someone installing Ollama. Restart to re-detect. + if self._available is None: + bin_path = _ollama_bin() + try: + subprocess.run([bin_path, "--version"], capture_output=True, + check=True, timeout=5) + self._available = True + except Exception: + self._available = False + return self._available def is_running(self): + # 0.5s, not 2s: this is a loopback request to a server that is either + # listening or is not. Ollama ships OFF (the user presses Start AI), so + # the "not running" path is the common one and every /status paid the + # full 2s for it - which pushed /status past the 2s client timeout in + # bin/nexus_window.py and made a healthy backend look dead. try: - r = httpx.get(f"{self._api_base}/api/tags", timeout=2.0) + r = httpx.get(f"{self._api_base}/api/tags", timeout=0.5) return r.status_code == 200 except httpx.RequestError: return False @@ -269,15 +298,7 @@ class OllamaManager: try: backend, gpu_env = _detect_gpu_backend() _log.info("Starting Ollama service via %s (backend: %s)...", _ollama_bin(), backend) - env = os.environ.copy() - env["OLLAMA_HOST"] = self._api_base - # Linux ships a bundled model store under the project. On Windows, - # Ollama is installed system-wide and pulls into its default store, - # so don't override OLLAMA_MODELS or a NexusOS-spawned `serve` won't - # see the models the installer already pulled. - if os.name != "nt": - env["OLLAMA_MODELS"] = str(Path(__file__).resolve().parent.parent / "models") - env.update(gpu_env) + env = self._serve_env(gpu_env) with open(self.log_file, "w") as log: self.process = subprocess.Popen( @@ -318,15 +339,7 @@ class OllamaManager: try: backend, gpu_env = _detect_gpu_backend() _log.info("Starting Ollama service via %s (backend: %s)...", _ollama_bin(), backend) - env = os.environ.copy() - env["OLLAMA_HOST"] = self._api_base - # Linux ships a bundled model store under the project. On Windows, - # Ollama is installed system-wide and pulls into its default store, - # so don't override OLLAMA_MODELS or a NexusOS-spawned `serve` won't - # see the models the installer already pulled. - if os.name != "nt": - env["OLLAMA_MODELS"] = str(Path(__file__).resolve().parent.parent / "models") - env.update(gpu_env) + env = self._serve_env(gpu_env) with open(self.log_file, "w") as log: self.process = subprocess.Popen( @@ -522,7 +535,7 @@ class OllamaManager: _log.exception("chat error after %.3fs: %s", elapsed, e) return None - async def embed(self, text: str, model: str = "nomic-embed-text") -> list[float] | None: + async def embed(self, text: str, model: str = DEFAULT_EMBED_MODEL) -> list[float] | None: """Return an embedding vector for `text` via /api/embeddings. Returns None on any failure so callers can fall back to lexical search —