diff --git a/.github/pr-screenshots/telegram-overflow/topic-final-response-clipped.jpg b/.github/pr-screenshots/telegram-overflow/topic-final-response-clipped.jpg new file mode 100644 index 00000000000..2f3529648e7 Binary files /dev/null and b/.github/pr-screenshots/telegram-overflow/topic-final-response-clipped.jpg differ diff --git a/.github/workflows/deploy-site.yml b/.github/workflows/deploy-site.yml index 9b3e6426652..6e7dc84415d 100644 --- a/.github/workflows/deploy-site.yml +++ b/.github/workflows/deploy-site.yml @@ -11,8 +11,20 @@ on: - 'optional-skills/**' - '.github/workflows/deploy-site.yml' workflow_dispatch: + inputs: + skills_index_run_id: + description: 'Optional Build Skills Index run ID whose skills-index artifact should be deployed' + required: false + type: string + rebuild_skills_index: + description: 'Force a fresh multi-source crawl instead of reusing the latest healthy index' + required: false + default: false + type: boolean permissions: + contents: read + actions: read pages: write id-token: write @@ -44,7 +56,7 @@ jobs: - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 with: - node-version: 20 + node-version: 22 cache: npm cache-dependency-path: website/package-lock.json @@ -55,26 +67,81 @@ jobs: - name: Install PyYAML for skill extraction run: pip install pyyaml==6.0.2 httpx==0.28.1 - - name: Build skills index (unified multi-source catalog) + - name: Prepare skills index (unified multi-source catalog) env: - GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} + GH_TOKEN: ${{ github.token }} + GITHUB_TOKEN: ${{ github.token }} + SKILLS_INDEX_RUN_ID: ${{ github.event.inputs.skills_index_run_id || '' }} + REBUILD_SKILLS_INDEX: ${{ github.event.inputs.rebuild_skills_index || 'false' }} run: | - # Rebuild the unified catalog. The file is gitignored, so a fresh - # checkout starts without it and we want the freshest crawl in - # every deploy. + # The unified external catalog is expensive to crawl and can burn + # through the repository installation's GitHub API quota when several + # docs deploys land close together. Normal docs deploys therefore + # reuse the latest healthy catalog: first the artifact from a + # scheduled skills-index run, then the currently live index. Only a + # manual force rebuild does a fresh crawl here. # - # This MUST be fatal. build_skills_index.py runs a health check and - # exits non-zero WITHOUT writing the output file when a source - # collapses (e.g. a GitHub API rate limit zeroes the github / - # claude-marketplace / well-known taps all at once). Letting the - # deploy continue would either (a) ship a degenerate index missing - # whole hubs — the June 2026 regression where OpenAI/Anthropic/ - # HuggingFace/NVIDIA tabs vanished — or (b) fall through to a - # local-only catalog. Failing here keeps the last good deployment - # live (GitHub Pages serves the previous build) instead of - # publishing a broken catalog. Re-run the workflow once the - # transient rate limit clears. + # If we do crawl, the build remains fatal. build_skills_index.py runs + # the health check BEFORE writing and exits non-zero on source + # collapse, keeping the last good Pages deployment live instead of + # publishing a degenerate catalog. + set -euo pipefail + INDEX_PATH="website/static/api/skills-index.json" + mkdir -p "$(dirname "$INDEX_PATH")" + + validate_index() { + python3 - "$INDEX_PATH" <<'PY' + import json + import sys + from pathlib import Path + + path = Path(sys.argv[1]) + try: + data = json.loads(path.read_text(encoding="utf-8")) + except Exception as exc: + print(f"invalid skills index JSON: {exc}", file=sys.stderr) + sys.exit(1) + skills = data.get("skills") + if not isinstance(skills, list) or len(skills) < 1500: + count = len(skills) if isinstance(skills, list) else "missing" + print(f"skills index too small: {count}", file=sys.stderr) + sys.exit(1) + print(f"skills index ready: {len(skills)} skills") + PY + } + + if [ "$REBUILD_SKILLS_INDEX" = "true" ]; then + python3 scripts/build_skills_index.py + validate_index + exit 0 + fi + + if [ -n "$SKILLS_INDEX_RUN_ID" ]; then + tmpdir="$(mktemp -d)" + echo "Downloading skills-index artifact from run $SKILLS_INDEX_RUN_ID" + if gh run download "$SKILLS_INDEX_RUN_ID" --name skills-index --dir "$tmpdir"; then + candidate="$(find "$tmpdir" -name skills-index.json -type f | head -n 1 || true)" + if [ -n "$candidate" ]; then + cp "$candidate" "$INDEX_PATH" + if validate_index; then + exit 0 + fi + fi + fi + echo "::warning::Could not use skills-index artifact from run $SKILLS_INDEX_RUN_ID; trying live index" + fi + + echo "Downloading currently live skills index" + if curl -fsSL --retry 3 --retry-delay 5 \ + "https://hermes-agent.nousresearch.com/docs/api/skills-index.json" \ + -o "$INDEX_PATH" && validate_index; then + exit 0 + fi + + echo "::warning::Live skills index unavailable or unhealthy; falling back to a fresh crawl" + rm -f "$INDEX_PATH" python3 scripts/build_skills_index.py + validate_index - name: Extract skill metadata for dashboard run: python3 website/scripts/extract-skills.py diff --git a/.github/workflows/docker-publish.yml b/.github/workflows/docker-publish.yml index 2e972cb11c3..c12ad772fa6 100644 --- a/.github/workflows/docker-publish.yml +++ b/.github/workflows/docker-publish.yml @@ -90,7 +90,7 @@ jobs: # (see `_SKIP_PARTS` in scripts/run_tests_parallel.py) because each # shard would otherwise reach the session-scoped ``built_image`` # fixture in ``tests/docker/conftest.py`` and start a 3-7min - # ``docker build`` under a 180s pytest-timeout cap — guaranteed to + # ``docker build`` — guaranteed to # die in fixture setup. # # Piggybacking here avoids a second image build: the smoke test @@ -114,7 +114,7 @@ jobs: run: | uv venv .venv --python 3.11 source .venv/bin/activate - # ``dev`` extra pulls in pytest, pytest-asyncio, pytest-timeout — + # ``dev`` extra pulls in pytest, pytest-asyncio — # everything tests/docker/ needs. We deliberately avoid ``all`` # here because the docker tests only drive the container via # subprocess and don't import hermes_agent's optional deps. diff --git a/.github/workflows/docs-site-checks.yml b/.github/workflows/docs-site-checks.yml index 49111b5ac09..7001c0b7439 100644 --- a/.github/workflows/docs-site-checks.yml +++ b/.github/workflows/docs-site-checks.yml @@ -18,7 +18,7 @@ jobs: - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 with: - node-version: 20 + node-version: 22 cache: npm cache-dependency-path: website/package-lock.json diff --git a/.github/workflows/skills-index.yml b/.github/workflows/skills-index.yml index 72f252b26eb..c6caf098133 100644 --- a/.github/workflows/skills-index.yml +++ b/.github/workflows/skills-index.yml @@ -53,4 +53,4 @@ jobs: - name: Trigger Deploy Site workflow env: GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} - run: gh workflow run deploy-site.yml --repo ${{ github.repository }} + run: gh workflow run deploy-site.yml --repo ${{ github.repository }} -f skills_index_run_id=${{ github.run_id }} diff --git a/.github/workflows/supply-chain-audit.yml b/.github/workflows/supply-chain-audit.yml index 3309de78dae..4bee46a95cd 100644 --- a/.github/workflows/supply-chain-audit.yml +++ b/.github/workflows/supply-chain-audit.yml @@ -29,6 +29,8 @@ jobs: scan: ${{ steps.filter.outputs.scan }} # True when pyproject.toml changed in this PR deps: ${{ steps.filter.outputs.deps }} + # True when the curated MCP catalog / bundled MCP manifests changed. + mcp_catalog: ${{ steps.filter.outputs.mcp_catalog }} steps: - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 with: @@ -54,6 +56,14 @@ jobs: else echo "deps=false" >> "$GITHUB_OUTPUT" fi + MCP_CATALOG_FILES=$(git diff --name-only "$BASE"..."$HEAD" -- \ + 'optional-mcps/**' \ + 'hermes_cli/mcp_catalog.py' || true) + if [ -n "$MCP_CATALOG_FILES" ]; then + echo "mcp_catalog=true" >> "$GITHUB_OUTPUT" + else + echo "mcp_catalog=false" >> "$GITHUB_OUTPUT" + fi scan: name: Scan PR for critical supply chain risks @@ -268,3 +278,50 @@ jobs: runs-on: ubuntu-latest steps: - run: echo "No pyproject.toml changes, skipping dependency bounds check." + + mcp-catalog-review: + name: MCP catalog security review + needs: changes + if: needs.changes.outputs.mcp_catalog == 'true' + runs-on: ubuntu-latest + steps: + - name: Checkout + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + with: + fetch-depth: 0 + + - name: Require explicit MCP catalog review label + env: + GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} + run: | + set -euo pipefail + PR="${{ github.event.pull_request.number }}" + LABELS=$(gh pr view "$PR" --json labels --jq '.labels[].name' || true) + if echo "$LABELS" | grep -Fxq 'mcp-catalog-reviewed'; then + echo "MCP catalog review label present." + exit 0 + fi + + BODY="## ⚠️ MCP catalog security review required + + This PR changes the bundled MCP catalog or MCP catalog installer code. MCP entries can define local commands that users later install into \`mcp_servers\`, so this needs explicit maintainer review before merge. + + A maintainer should verify: + - any new/changed \`optional-mcps/**/manifest.yaml\` command and args are expected, + - stdio transports do not use shell+egress/exfiltration payloads, + - git install refs are pinned and bootstrap commands are minimal, + - requested env vars/secrets match the upstream MCP's documented needs. + + After review, add the \`mcp-catalog-reviewed\` label and re-run this check." + + gh pr comment "$PR" --body "$BODY" || echo "::warning::Could not post PR comment (expected for fork PRs)" + echo "::error::MCP catalog changes require the mcp-catalog-reviewed label." + exit 1 + + mcp-catalog-review-gate: + name: MCP catalog security review + needs: changes + if: always() && needs.changes.outputs.mcp_catalog != 'true' + runs-on: ubuntu-latest + steps: + - run: echo "No MCP catalog changes, skipping MCP catalog security review." diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index cc7d099fd93..a6e7738fa40 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -4,13 +4,13 @@ on: push: branches: [main] paths-ignore: - - '**/*.md' - - 'docs/**' + - "**/*.md" + - "docs/**" pull_request: branches: [main] paths-ignore: - - '**/*.md' - - 'docs/**' + - "**/*.md" + - "docs/**" permissions: contents: read @@ -30,13 +30,17 @@ jobs: slice: [1, 2, 3, 4, 5, 6] steps: - name: Checkout code - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - name: Restore duration cache - uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5 + uses: actions/cache/restore@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5 with: path: test_durations.json - # Single stable key. main always overwrites, PRs always find it. + # main always writes a new suffix, but jobs pick the latest one with the same prefix + # quote from https://docs.github.com/en/actions/reference/workflows-and-actions/dependency-caching#cache-hits-and-misses + # If you provide restore-keys, the cache action sequentially searches for any caches that match the list of restore-keys. + # If there are no exact matches, the action searches for partial matches of the restore keys. + # When the action finds a partial match, the most recent cache is restored to the path directory. key: test-durations - name: Install ripgrep (prebuilt binary) @@ -54,7 +58,7 @@ jobs: rg --version - name: Install uv - uses: astral-sh/setup-uv@d4b2f3b6ecc6e67c4457f6d3e41ec42d3d0fcb86 # v5 + uses: astral-sh/setup-uv@d4b2f3b6ecc6e67c4457f6d3e41ec42d3d0fcb86 # v5 with: # Persist uv's download/wheel cache (~/.cache/uv) across runs. # Keyed on the dependency manifests, so the cache is reused until @@ -115,7 +119,7 @@ jobs: NOUS_API_KEY: "" - name: Upload per-slice durations - uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 with: name: test-durations-slice-${{ matrix.slice }} path: test_durations.json @@ -125,11 +129,11 @@ jobs: # (including PRs) get balanced slicing. save-durations: needs: test - if: always() && github.ref == 'refs/heads/main' + if: needs.test.result == 'success' && github.ref == 'refs/heads/main' runs-on: ubuntu-latest steps: - name: Download all slice durations - uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 + uses: actions/download-artifact@3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c # v8.0.1 with: pattern: test-durations-slice-* path: durations @@ -149,17 +153,17 @@ jobs: " - name: Save merged duration cache - uses: actions/cache/save@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5 + uses: actions/cache/save@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5.0.5 with: path: test_durations.json - key: test-durations + key: test-durations-${{ github.run_id }} e2e: runs-on: ubuntu-latest timeout-minutes: 15 steps: - name: Checkout code - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 - name: Install ripgrep (prebuilt binary) run: | @@ -176,7 +180,7 @@ jobs: rg --version - name: Install uv - uses: astral-sh/setup-uv@d4b2f3b6ecc6e67c4457f6d3e41ec42d3d0fcb86 # v5 + uses: astral-sh/setup-uv@d4b2f3b6ecc6e67c4457f6d3e41ec42d3d0fcb86 # v5 with: # Persist uv's download/wheel cache (~/.cache/uv) across runs. # Keyed on the dependency manifests, so the cache is reused until diff --git a/.github/workflows/typecheck.yml b/.github/workflows/typecheck.yml new file mode 100644 index 00000000000..f3dcc71efdb --- /dev/null +++ b/.github/workflows/typecheck.yml @@ -0,0 +1,25 @@ +# .github/workflows/typecheck.yml +name: Typecheck + +on: + push: + branches: [main] + pull_request: + branches: [main] + +jobs: + typecheck: + runs-on: ubuntu-latest + strategy: + matrix: + package: + [ui-tui, web, apps/bootstrap-installer, apps/desktop, apps/shared] + fail-fast: false # report all failures, not just the first one + steps: + - uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2 + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4 + with: + node-version: 22 + cache: npm + - run: npm ci + - run: npm run --prefix ${{ matrix.package }} typecheck diff --git a/.gitignore b/.gitignore index fa4d64049b7..6d87318e35e 100644 --- a/.gitignore +++ b/.gitignore @@ -89,6 +89,9 @@ website/static/api/skills-index.json # every build). website/static/api/skills.json website/static/api/skills-meta.json +# automation-blueprints-index.json is a build artifact emitted by +# website/scripts/extract-automation-blueprints.py during prebuild. +website/static/api/automation-blueprints-index.json models-dev-upstream/ # Local editor / agent tooling (machine-specific; keep in global config, not the repo) @@ -129,3 +132,7 @@ scripts/out/ # stores the published notes. They are not a build artifact and must never be # committed to the repo root. See the hermes-release skill. RELEASE_v*.md + +# Desktop demo-run scratch output (hermes writes demo/*.txt during recorded +# walkthroughs). Throwaway artifacts, never part of the app. +apps/desktop/demo/ diff --git a/AGENTS.md b/AGENTS.md index 8d9ef1621d8..e032f765447 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -459,7 +459,7 @@ npm install # first time npm run dev # watch mode (rebuilds hermes-ink + tsx --watch) npm start # production npm run build # full build (hermes-ink + tsc) -npm run type-check # typecheck only (tsc --noEmit) +npm run typecheck # typecheck only (tsc --noEmit) npm run lint # eslint npm run fmt # prettier npm test # vitest diff --git a/acp_adapter/server.py b/acp_adapter/server.py index 6901fe28e88..a51db91d4e8 100644 --- a/acp_adapter/server.py +++ b/acp_adapter/server.py @@ -824,6 +824,7 @@ class HermesACPAgent(acp.Agent): try: from model_tools import get_tool_definitions + from agent.memory_manager import inject_memory_provider_tools enabled_toolsets = _expand_acp_enabled_toolsets( getattr(state.agent, "enabled_toolsets", None) or ["hermes-acp"], @@ -839,6 +840,7 @@ class HermesACPAgent(acp.Agent): state.agent.valid_tool_names = { tool["function"]["name"] for tool in state.agent.tools or [] } + inject_memory_provider_tools(state.agent) invalidate = getattr(state.agent, "_invalidate_system_prompt", None) if callable(invalidate): invalidate() @@ -1779,10 +1781,25 @@ class HermesACPAgent(acp.Agent): def _cmd_tools(self, args: str, state: SessionState) -> str: try: from model_tools import get_tool_definitions + from types import SimpleNamespace + from agent.memory_manager import inject_memory_provider_tools + toolsets = _expand_acp_enabled_toolsets( getattr(state.agent, "enabled_toolsets", None) or ["hermes-acp"] ) tools = get_tool_definitions(enabled_toolsets=toolsets, quiet_mode=True) + tool_view = SimpleNamespace( + tools=list(tools or []), + valid_tool_names={ + tool.get("function", {}).get("name") + for tool in tools or [] + if isinstance(tool, dict) + }, + enabled_toolsets=toolsets, + _memory_manager=getattr(state.agent, "_memory_manager", None), + ) + inject_memory_provider_tools(tool_view) + tools = tool_view.tools if not tools: return "No tools available." lines = [f"Available tools ({len(tools)}):"] diff --git a/agent/account_usage.py b/agent/account_usage.py index 2795eb24125..da02af3c478 100644 --- a/agent/account_usage.py +++ b/agent/account_usage.py @@ -145,7 +145,7 @@ def build_nous_credits_snapshot(account_info) -> Optional[AccountUsageSnapshot]: account info to show (fail-open: caller just shows nothing). """ try: - from hermes_cli.nous_account import nous_portal_billing_url + from hermes_cli.nous_account import nous_portal_topup_url if account_info is None or not getattr(account_info, "logged_in", False): return None @@ -213,7 +213,8 @@ def build_nous_credits_snapshot(account_info) -> Optional[AccountUsageSnapshot]: if not windows and not details: return None - details.append(f"Manage / top up: {nous_portal_billing_url(account_info)}") + details.append(f"Top up: {nous_portal_topup_url(account_info)}") + details.append("(or run /credits)") plan = getattr(sub, "plan", None) if sub is not None else None return AccountUsageSnapshot( @@ -337,6 +338,93 @@ def _snapshot_from_credits_state(state) -> Optional[AccountUsageSnapshot]: return None +@dataclass(frozen=True) +class CreditsView: + """Surface-agnostic data for the ``/credits`` command. + + One portal fetch, one parse — consumed identically by the CLI panel, the + gateway button, and any other money surface. Fail-open: when not logged in + or the portal is unreachable, ``logged_in`` is False / ``topup_url`` is None + and callers degrade gracefully. + """ + + logged_in: bool + balance_lines: tuple[str, ...] = () + identity_line: Optional[str] = None + topup_url: Optional[str] = None + depleted: bool = False + + +def build_credits_view(*, markdown: bool = False, timeout: float = 10.0) -> CreditsView: + """Build the /credits view: balance block + identity line + top-up URL. + + Reuses the same account fetch + snapshot + URL builder as the /usage credits + block, so the numbers always match. The balance block is the rendered + snapshot MINUS its trailing top-up/command-hint lines (the /credits surface + supplies its own affordance). Fail-open → ``CreditsView(logged_in=False)``. + """ + not_logged_in = CreditsView(logged_in=False) + try: + from hermes_cli.auth import get_provider_auth_state + + tok = (get_provider_auth_state("nous") or {}).get("access_token") + if not (isinstance(tok, str) and tok.strip()): + return not_logged_in + except Exception: + return not_logged_in + + try: + import concurrent.futures + + from hermes_cli.nous_account import ( + get_nous_portal_account_info, + nous_portal_topup_url, + ) + + with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool: + account = pool.submit(get_nous_portal_account_info, force_fresh=True).result( + timeout=timeout + ) + except Exception: + logger.debug("credits ▸ /credits portal fetch failed (fail-open)", exc_info=True) + return not_logged_in + + if account is None or not getattr(account, "logged_in", False): + return not_logged_in + + snapshot = build_nous_credits_snapshot(account) + # Balance lines = the snapshot block minus the two trailing affordance lines + # ("Top up: " + "(or run /credits)") that build_nous_credits_snapshot + # appends for the /usage surface. /credits renders its own button/panel. + balance_lines: list[str] = [] + if snapshot is not None: + rendered = render_account_usage_lines(snapshot, markdown=markdown) + balance_lines = [ + line + for line in rendered + if not line.lstrip().startswith("Top up:") + and not line.lstrip().startswith("(or run") + ] + + # Identity line — shown before any open (roadmap §4.4). + email = getattr(account, "email", None) + org_name = getattr(account, "org_name", None) + who: list[str] = [] + if email: + who.append(str(email)) + if org_name: + who.append(f"org {org_name}") + identity_line = ("Topping up as " + " / ".join(who)) if who else None + + return CreditsView( + logged_in=True, + balance_lines=tuple(balance_lines), + identity_line=identity_line, + topup_url=nous_portal_topup_url(account), + depleted=getattr(account, "paid_service_access", None) is False, + ) + + def _resolve_codex_usage_url(base_url: str) -> str: normalized = (base_url or "").strip().rstrip("/") if not normalized: diff --git a/agent/agent_init.py b/agent/agent_init.py index 96bfe3d873f..e1594b4585d 100644 --- a/agent/agent_init.py +++ b/agent/agent_init.py @@ -900,6 +900,9 @@ def init_agent( agent.api_key = client_kwargs.get("api_key", "") agent.base_url = client_kwargs.get("base_url", agent.base_url) try: + from agent.ssl_guard import verify_ca_bundle_with_fallback + + verify_ca_bundle_with_fallback() agent.client = agent._create_openai_client(client_kwargs, reason="agent_init", shared=True) if not agent.quiet_mode: print(f"🤖 AI Agent initialized with model: {agent.model}") @@ -1193,38 +1196,8 @@ def init_agent( _ra().logger.warning("Memory provider plugin init failed: %s", _mpe) agent._memory_manager = None - # Inject memory provider tool schemas into the tool surface. - # Skip tools whose names already exist (plugins may register the - # same tools via ctx.register_tool(), which lands in agent.tools - # through _ra().get_tool_definitions()). Duplicate function names cause - # 400 errors on providers that enforce unique names (e.g. Xiaomi - # MiMo via Nous Portal). - # - # Respect the platform's enabled_toolsets configuration (#5544): - # enabled_toolsets is None → no filter, inject (backward compat) - # "memory" in enabled_toolsets → user opted in, inject - # otherwise (incl. []) → user excluded memory, skip injection - # - # Without this gate, `platform_toolsets: telegram: []` still leaks memory - # provider tools (fact_store, etc.) into the tool surface — a 10x latency - # penalty on local models and a frequent trigger of tool-call loops. - if agent._memory_manager and agent.tools is not None and ( - agent.enabled_toolsets is None or "memory" in agent.enabled_toolsets - ): - _existing_tool_names = { - t.get("function", {}).get("name") - for t in agent.tools - if isinstance(t, dict) - } - for _schema in agent._memory_manager.get_all_tool_schemas(): - _tname = _schema.get("name", "") - if _tname and _tname in _existing_tool_names: - continue # already registered via plugin path - _wrapped = {"type": "function", "function": _schema} - agent.tools.append(_wrapped) - if _tname: - agent.valid_tool_names.add(_tname) - _existing_tool_names.add(_tname) + from agent.memory_manager import inject_memory_provider_tools as _inject_memory_provider_tools + _inject_memory_provider_tools(agent) # Skills config: nudge interval for skill creation reminders agent._skill_nudge_interval = 10 diff --git a/agent/agent_runtime_helpers.py b/agent/agent_runtime_helpers.py index daffc025d9b..cae1a685a53 100644 --- a/agent/agent_runtime_helpers.py +++ b/agent/agent_runtime_helpers.py @@ -445,6 +445,45 @@ def repair_message_sequence(agent, messages: List[Dict]) -> int: return repairs +def repair_message_sequence_with_cursor(agent, messages: List[Dict]) -> int: + """Run :func:`repair_message_sequence` and keep the SessionDB flush + cursor consistent with the compacted list (#44837). + + ``repair_message_sequence`` merges/drops messages in place, shrinking + the list. ``_last_flushed_db_idx`` (the DB-write cursor) indexes into + that list, so after compaction it can point past the new end — the + turn-end flush would then skip the assistant/tool chain entirely — or + past unflushed messages shifted to lower indexes. + + Repair preserves object identity for surviving messages, so counting + the survivors from the previously-flushed prefix gives the exact new + cursor even when messages are dropped/merged at indexes *before* the + cursor — a plain ``min()`` clamp would silently skip that many + unflushed rows. Falls back to the clamp when no prefix snapshot is + available. + + Returns the number of repairs made (same as ``repair_message_sequence``). + """ + pre_repair_flushed_ids = None + flush_cursor = getattr(agent, "_last_flushed_db_idx", None) + if isinstance(flush_cursor, int) and flush_cursor > 0: + pre_repair_flushed_ids = {id(m) for m in messages[:flush_cursor]} + + repairs = repair_message_sequence(agent, messages) + + if repairs > 0 and hasattr(agent, "_last_flushed_db_idx"): + if pre_repair_flushed_ids is not None: + agent._last_flushed_db_idx = sum( + 1 for m in messages if id(m) in pre_repair_flushed_ids + ) + else: + agent._last_flushed_db_idx = min( + agent._last_flushed_db_idx, len(messages) + ) + + return repairs + + def strip_think_blocks(agent, content: str) -> str: """Remove reasoning/thinking blocks from content, returning only visible text. @@ -579,12 +618,33 @@ def recover_with_credential_pool( current_provider = (getattr(agent, "provider", "") or "").strip().lower() pool_provider = (getattr(pool, "provider", "") or "").strip().lower() if current_provider and pool_provider and current_provider != pool_provider: - _ra().logger.warning( - "Credential pool provider mismatch: pool=%s, agent=%s — " - "skipping pool mutation to avoid cross-provider contamination", - pool_provider, current_provider, - ) - return False, has_retried_429 + # Custom endpoints use two naming conventions for the SAME provider: + # the agent carries the generic ``custom`` label while the pool is + # keyed ``custom:`` (see CUSTOM_POOL_PREFIX). A literal string + # compare treats them as a mismatch and skips recovery for every + # custom-provider user — 401s/429s then burn the full retry cycle + # with no rotation or refresh. Accept the pair as matching only when + # the agent's CURRENT base_url actually resolves to this pool key, + # so a fallback provider (or a different custom endpoint) still + # triggers the guard. + _custom_match = False + if current_provider == "custom" and pool_provider.startswith("custom:"): + try: + from agent.credential_pool import get_custom_provider_pool_key + _agent_base = (getattr(agent, "base_url", "") or "").strip() + _custom_match = bool(_agent_base) and ( + (get_custom_provider_pool_key(_agent_base) or "").strip().lower() + == pool_provider + ) + except Exception: + _custom_match = False + if not _custom_match: + _ra().logger.warning( + "Credential pool provider mismatch: pool=%s, agent=%s — " + "skipping pool mutation to avoid cross-provider contamination", + pool_provider, current_provider, + ) + return False, has_retried_429 effective_reason = classified_reason if effective_reason is None: @@ -679,15 +739,28 @@ def recover_with_credential_pool( # long-running TUI sessions stuck on stale tokens until the user # exited and reopened. is_entitlement = agent._is_entitlement_failure(error_context, status_code) + _auth_haystack = " ".join( + str(error_context.get(k) or "").lower() + for k in ("message", "reason", "code", "error") + if isinstance(error_context, dict) + ) + if ( + not is_entitlement + and status_code == 403 + and "oauth authentication is currently not allowed for this organization" in _auth_haystack + ): + is_entitlement = True + if ( + not is_entitlement + and status_code == 403 + and (agent.provider or "") == "anthropic" + and getattr(agent, "api_mode", "") == "anthropic_messages" + ): + is_entitlement = True if not is_entitlement and status_code == 403 and (agent.provider or "") == "xai-oauth": - _disambiguator_haystack = " ".join( - str(error_context.get(k) or "").lower() - for k in ("message", "reason", "code", "error") - if isinstance(error_context, dict) - ) _is_xai_auth_failure = ( - "[wke=unauthenticated:" in _disambiguator_haystack - or "oauth2 access token could not be validated" in _disambiguator_haystack + "[wke=unauthenticated:" in _auth_haystack + or "oauth2 access token could not be validated" in _auth_haystack ) if not _is_xai_auth_failure: is_entitlement = True @@ -808,6 +881,8 @@ def try_recover_primary_transport( def drop_thinking_only_and_merge_users( messages: List[Dict[str, Any]], + *, + drop_codex_reasoning_items: bool = True, ) -> List[Dict[str, Any]]: """Drop thinking-only assistant turns; merge any adjacent user messages left behind. @@ -829,7 +904,13 @@ def drop_thinking_only_and_merge_users( return messages # Pass 1: drop thinking-only assistant turns. - kept = [m for m in messages if not _ra().AIAgent._is_thinking_only_assistant(m)] + kept = [ + m for m in messages + if not _ra().AIAgent._is_thinking_only_assistant( + m, + drop_codex_reasoning_items=drop_codex_reasoning_items, + ) + ] dropped = len(messages) - len(kept) if dropped == 0: return messages diff --git a/agent/anthropic_adapter.py b/agent/anthropic_adapter.py index e64bc54bc90..3a2d3f68e17 100644 --- a/agent/anthropic_adapter.py +++ b/agent/anthropic_adapter.py @@ -751,6 +751,9 @@ def build_anthropic_client( from httpx import Timeout normalized_base_url = _normalize_base_url_text(base_url) + if normalized_base_url: + import re as _re + normalized_base_url = _re.sub(r"/v1/?$", "", normalized_base_url.rstrip("/")) _read_timeout = timeout if (isinstance(timeout, (int, float)) and timeout > 0) else 900.0 kwargs = { "timeout": Timeout(timeout=float(_read_timeout), connect=10.0), @@ -1571,6 +1574,15 @@ def _convert_content_part_to_anthropic(part: Any) -> Optional[Dict[str, Any]]: if ptype == "input_text": block: Dict[str, Any] = {"type": "text", "text": part.get("text", "")} + elif ptype == "text": + # A stored Anthropic text block. Rebuild from whitelisted fields only — + # SDK response text blocks carry output-only siblings (parsed_output, + # citations=None) that the Messages INPUT schema rejects with HTTP 400 + # "Extra inputs are not permitted". Do NOT dict(part) it verbatim. + block = {"type": "text", "text": part.get("text", "")} + cits = part.get("citations") + if isinstance(cits, list) and cits: + block["citations"] = cits elif ptype in {"image_url", "input_image"}: image_value = part.get("image_url", {}) url = image_value.get("url", "") if isinstance(image_value, dict) else str(image_value or "") @@ -1685,6 +1697,58 @@ def _content_parts_to_anthropic_blocks(parts: Any) -> List[Dict[str, Any]]: return out +def _sanitize_replay_block(b: Dict[str, Any]) -> Optional[Dict[str, Any]]: + """Strip output-only fields from a stored Anthropic content block so it is + valid as REQUEST input on replay. + + The SDK response objects carry output-only attributes that the Messages + *input* schema forbids ("Extra inputs are not permitted"): text blocks get + ``parsed_output``/``citations`` (when null), tool_use blocks get ``caller``, + etc. ``normalize_response`` captured blocks verbatim via ``_to_plain_data``, + so these leak back as input on the next turn → HTTP 400. + + Whitelist per type (NOT a blacklist) so future SDK output-only fields can't + reintroduce the bug. Returns a clean block, or None to drop it. + """ + if not isinstance(b, dict): + return None + btype = b.get("type") + if btype == "text": + out: Dict[str, Any] = {"type": "text", "text": b.get("text", "")} + # citations is input-valid ONLY when it's a non-empty list; the SDK + # emits citations=None on responses, which the input schema rejects. + cits = b.get("citations") + if isinstance(cits, list) and cits: + out["citations"] = cits + if isinstance(b.get("cache_control"), dict): + out["cache_control"] = b["cache_control"] + return out + if btype == "thinking": + out = {"type": "thinking", "thinking": b.get("thinking", "")} + if b.get("signature"): + out["signature"] = b["signature"] + return out + if btype == "redacted_thinking": + # Only valid with its data payload; drop if missing. + return {"type": "redacted_thinking", "data": b["data"]} if b.get("data") else None + if btype == "tool_use": + out = { + "type": "tool_use", + "id": _sanitize_tool_id(b.get("id", "")), + "name": b.get("name", ""), + "input": b.get("input", {}), + } + if isinstance(b.get("cache_control"), dict): + out["cache_control"] = b["cache_control"] + return out + if btype == "image": + src = b.get("source") + return {"type": "image", "source": src} if isinstance(src, dict) else None + # Unknown/unsupported block type on the input path — drop rather than risk + # another "Extra inputs are not permitted". + return None + + def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]: """Convert an assistant message to Anthropic content blocks. @@ -1692,6 +1756,55 @@ def _convert_assistant_message(m: Dict[str, Any]) -> Dict[str, Any]: reasoning_content injection for Kimi/DeepSeek endpoints. """ content = m.get("content", "") + # Anthropic interleaved-thinking fast path: when this turn carries a + # verbatim, order-preserving block list (set by normalize_response only + # for turns that interleave SIGNED thinking with tool_use), replay it. + # Each block is run through _sanitize_replay_block to strip output-only + # SDK fields (parsed_output, caller, citations=None, …) that the Messages + # INPUT schema forbids — replaying them verbatim caused HTTP 400 "Extra + # inputs are not permitted" (text.parsed_output). Block ORDER is preserved + # (the reason this channel exists); only forbidden sibling fields are + # dropped, leaving thinking signatures and tool_use id/name/input intact. + ordered_blocks = m.get("anthropic_content_blocks") + if isinstance(ordered_blocks, list) and ordered_blocks: + # Re-source each tool_use input from the stored tool_calls map rather + # than the captured block. The ordered-blocks list captures tool_use + # input from the RAW API response (normalize_response), which is NOT + # credential-redacted; tool_calls[].function.arguments IS redacted at + # storage time (build_assistant_message, #19798). Replaying the raw + # block input would resurrect a secret the model inlined into a tool + # call (e.g. terminal(command="curl -H 'Authorization: Bearer sk-...'") + # onto the wire, even though the same value is redacted everywhere else + # in history. Keying by sanitized tool id preserves interleave order + # (the reason this channel exists) while swapping in the redacted + # input. Adapted from #36071 (replay-time tool-input re-sourcing). + redacted_input_by_id: Dict[str, Any] = {} + for tc in m.get("tool_calls", []) or []: + if not isinstance(tc, dict): + continue + fn = tc.get("function", {}) or {} + raw_args = fn.get("arguments", "{}") + try: + parsed_args = json.loads(raw_args) if isinstance(raw_args, str) else raw_args + except (json.JSONDecodeError, ValueError): + parsed_args = {} + redacted_input_by_id[_sanitize_tool_id(tc.get("id", ""))] = parsed_args + replayed: List[Dict[str, Any]] = [] + for b in ordered_blocks: + clean = _sanitize_replay_block(b) + if clean is None: + continue + if clean.get("type") == "tool_use": + # Override raw (un-redacted) input with the redacted copy when + # we have one for this id; fall back to the sanitized block + # input only if the tool_call is missing (shape mismatch). + redacted = redacted_input_by_id.get(clean.get("id", "")) + if redacted is not None: + clean["input"] = redacted + replayed.append(clean) + if replayed: + return {"role": "assistant", "content": replayed} + blocks = _extract_preserved_thinking_blocks(m) if content: if isinstance(content, list): diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index c6e00340e7e..01ea45d7be2 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -1144,7 +1144,8 @@ def _endpoint_speaks_anthropic_messages(base_url: str) -> bool: normalized = (base_url or "").strip().lower().rstrip("/") if not normalized: return False - if normalized.endswith("/anthropic"): + path = urlparse(normalized).path.rstrip("/") + if path.endswith("/anthropic") or path.endswith("/anthropic/v1"): return True hostname = base_url_hostname(normalized) if hostname == "api.anthropic.com": @@ -3190,7 +3191,7 @@ def _resolve_auto(main_runtime: Optional[Dict[str, Any]] = None) -> Tuple[Option if (main_provider and main_model and main_provider not in {"auto", ""}): resolved_provider = main_provider - explicit_base_url = None + explicit_base_url = runtime_base_url or None explicit_api_key = None if runtime_base_url and (main_provider == "custom" or main_provider.startswith("custom:")): resolved_provider = "custom" @@ -5004,7 +5005,7 @@ def _build_call_kwargs( # Provider-specific extra_body merged_extra = dict(extra_body or {}) - if provider == "nous" or auxiliary_is_nous: + if provider == "nous": merged_extra.setdefault("tags", []).extend(_nous_portal_tags()) if merged_extra: kwargs["extra_body"] = merged_extra diff --git a/agent/bedrock_adapter.py b/agent/bedrock_adapter.py index 12c7afb8c18..a09e1bc5d82 100644 --- a/agent/bedrock_adapter.py +++ b/agent/bedrock_adapter.py @@ -208,6 +208,41 @@ def is_stale_connection_error(exc: BaseException) -> bool: return False +def is_streaming_access_denied_error(exc: BaseException) -> bool: + """Return True when AWS denied the ``bedrock:InvokeModelWithResponseStream`` action. + + IAM policies scoped to ``bedrock:InvokeModel`` only (a common least-privilege + setup) reject ``converse_stream()`` with an ``AccessDeniedException`` whose + message names the streaming action, e.g.:: + + User: arn:aws:iam::123456789012:user/x is not authorized to perform: + bedrock:InvokeModelWithResponseStream on resource: ... + + This is permanent for the session — retrying the stream can never succeed — + so callers should flip to the non-streaming ``converse()`` path (which maps + to ``bedrock:InvokeModel``) instead of burning retries. + + Detection is deliberately message-based: boto3 surfaces this as a + ``ClientError`` with ``Error.Code == "AccessDeniedException"``, and the + AnthropicBedrock SDK wraps the same AWS response in its own exception + types, but both preserve the action name in the message. + """ + msg = str(exc).lower() + if "invokemodelwithresponsestream" not in msg: + return False + # ClientError with an explicit access-denied code is the canonical form. + try: + from botocore.exceptions import ClientError + except ImportError: # pragma: no cover — botocore always present with boto3 + ClientError = None # type: ignore[assignment] + if ClientError is not None and isinstance(exc, ClientError): + code = (getattr(exc, "response", None) or {}).get("Error", {}).get("Code", "") + return code in ("AccessDeniedException", "UnauthorizedException") + # Wrapped forms (e.g. AnthropicBedrock SDK PermissionDeniedError) — match + # on the authorization-failure phrasing AWS uses. + return "not authorized" in msg or "accessdenied" in msg + + # --------------------------------------------------------------------------- # AWS credential detection # --------------------------------------------------------------------------- @@ -900,11 +935,14 @@ def build_converse_kwargs( if system_prompt: kwargs["system"] = system_prompt - if temperature is not None: - kwargs["inferenceConfig"]["temperature"] = temperature + from agent.anthropic_adapter import _forbids_sampling_params - if top_p is not None: - kwargs["inferenceConfig"]["topP"] = top_p + if not _forbids_sampling_params(model): + if temperature is not None: + kwargs["inferenceConfig"]["temperature"] = temperature + + if top_p is not None: + kwargs["inferenceConfig"]["topP"] = top_p if stop_sequences: kwargs["inferenceConfig"]["stopSequences"] = stop_sequences @@ -1003,6 +1041,16 @@ def call_converse_stream( try: response = client.converse_stream(**kwargs) except Exception as exc: + if is_streaming_access_denied_error(exc): + # IAM allows bedrock:InvokeModel but not + # InvokeModelWithResponseStream — permanent for this session. + # Fall back to the non-streaming converse() path. + logger.info( + "bedrock: converse_stream denied by IAM on (region=%s, model=%s) — " + "falling back to non-streaming converse().", + region, model, + ) + return normalize_converse_response(client.converse(**kwargs)) if is_stale_connection_error(exc): logger.warning( "bedrock: stale-connection error on converse_stream(region=%s, " diff --git a/agent/chat_completion_helpers.py b/agent/chat_completion_helpers.py index ce066d55640..1ee1702b45e 100644 --- a/agent/chat_completion_helpers.py +++ b/agent/chat_completion_helpers.py @@ -952,6 +952,18 @@ def build_assistant_message(agent, assistant_message, finish_reason: str) -> dic if preserved: msg["reasoning_details"] = preserved + # Anthropic interleaved-thinking replay: when a turn interleaves signed + # thinking blocks with tool_use, the parallel reasoning_details + + # tool_calls fields lose the cross-type ordering, and reconstruction + # front-loads thinking — reordering signed blocks and triggering HTTP 400 + # ("thinking ... blocks in the latest assistant message cannot be + # modified"). Carry the verbatim ordered block list so the adapter can + # replay the latest assistant message unchanged. See + # agent/transports/anthropic.py and agent/anthropic_adapter.py. + ordered_blocks = getattr(assistant_message, "anthropic_content_blocks", None) + if ordered_blocks: + msg["anthropic_content_blocks"] = ordered_blocks + # Codex Responses API: preserve encrypted reasoning items for # multi-turn continuity. These get replayed as input on the next turn. codex_items = getattr(assistant_message, "codex_reasoning_items", None) @@ -1603,6 +1615,8 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta= _get_bedrock_runtime_client, invalidate_runtime_client, is_stale_connection_error, + is_streaming_access_denied_error, + normalize_converse_response, stream_converse_with_callbacks, ) region = api_kwargs.pop("__bedrock_region__", "us-east-1") @@ -1611,6 +1625,29 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta= try: raw_response = client.converse_stream(**api_kwargs) except Exception as _bedrock_exc: + # IAM policies scoped to bedrock:InvokeModel only (no + # InvokeModelWithResponseStream) reject converse_stream() + # with AccessDeniedException. That denial is permanent for + # the session — fall back to the non-streaming converse() + # inline (it maps to bedrock:InvokeModel) and disable + # streaming for subsequent calls so we don't re-fail every + # turn. + if is_streaming_access_denied_error(_bedrock_exc): + agent._disable_streaming = True + agent._safe_print( + "\n⚠ AWS IAM denied bedrock:InvokeModelWithResponseStream — " + "falling back to non-streaming InvokeModel.\n" + " Grant that action to restore streaming output.\n" + ) + logger.info( + "bedrock: converse_stream denied by IAM (%s) — " + "using non-streaming converse() for this session.", + type(_bedrock_exc).__name__, + ) + result["response"] = normalize_converse_response( + client.converse(**api_kwargs) + ) + return # Evict the cached client on stale-connection failures # so the outer retry loop builds a fresh client/pool. if is_stale_connection_error(_bedrock_exc): @@ -1698,6 +1735,14 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta= # poll loop uses this to detect stale connections that keep receiving # SSE keep-alive pings but no actual data. last_chunk_time = {"t": time.time()} + # Stale-stream patience, shared between the httpx socket read timeout + # (built in ``_call_chat_completions`` below) and the stale-stream detector + # (computed further down, before the worker thread starts). Initialized + # here so the read-timeout builder can floor itself at the stale value and + # never fire before the detector. ``None`` until the detector value is + # resolved, so the builder degrades to its plain default if it ever runs + # first. + _stream_stale_timeout = None def _fire_first_delta(): if not first_delta_fired["done"] and on_first_delta: @@ -1734,6 +1779,26 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta= "Local provider detected (%s) — stream read timeout raised to %.0fs", agent.base_url, _stream_read_timeout, ) + elif ( + _stream_read_timeout == 120.0 + and _stream_stale_timeout is not None + and _stream_stale_timeout != float("inf") + and _stream_stale_timeout > _stream_read_timeout + ): + # Cloud reasoning models (e.g. Opus) routinely pause mid-stream + # for minutes during extended thinking. The stale-stream + # detector is deliberately scaled up to tolerate this (180–300s, + # see the stale-timeout block below), but the raw httpx socket + # read timeout defaulted to a flat 120s and fired *first* — + # tearing down a healthy reasoning stream before the stale + # detector (which owns retry + diagnostics) could act. Keep the + # socket read timeout in step with the detector so it no longer + # preempts it. + _stream_read_timeout = _stream_stale_timeout + logger.debug( + "Cloud reasoning stream — read timeout raised to %.0fs to " + "match stale-stream detector", _stream_read_timeout, + ) # Cap connect/pool at 60s even when provider timeout is higher. # connect/pool cover TCP handshake, not model inference. _conn_cap = min(_base_timeout, 60.0) if _provider_timeout_cfg is not None else 30.0 @@ -2384,9 +2449,34 @@ def interruptible_streaming_api_call(agent, api_kwargs: dict, *, on_first_delta= "stream" in _err_lower and "not supported" in _err_lower ) - if _is_stream_unsupported: + # AWS Bedrock (AnthropicBedrock SDK path): IAM policies + # with bedrock:InvokeModel but not + # InvokeModelWithResponseStream reject messages.stream() + # with a permission error naming the streaming action. + # Permanent for the session — flip to non-streaming + # (messages.create() maps to bedrock:InvokeModel). + _is_bedrock_stream_denied = False + if ( + not _is_stream_unsupported + and "invokemodelwithresponsestream" in _err_lower + ): + # Cheap message pre-check before importing the + # adapter — bedrock_adapter triggers a lazy boto3 + # install at import time, which must not run for + # unrelated providers' stream errors. + from agent.bedrock_adapter import ( + is_streaming_access_denied_error, + ) + _is_bedrock_stream_denied = ( + is_streaming_access_denied_error(e) + ) + if _is_stream_unsupported or _is_bedrock_stream_denied: agent._disable_streaming = True agent._safe_print( + "\n⚠ AWS IAM denied bedrock:InvokeModelWithResponseStream. " + "Switching to non-streaming.\n" + " Grant that action to restore streaming output.\n" + if _is_bedrock_stream_denied else "\n⚠ Streaming is not supported for this " "model/provider. Switching to non-streaming.\n" " To avoid this delay, set display.streaming: false " diff --git a/agent/codex_responses_adapter.py b/agent/codex_responses_adapter.py index 943131f5592..b8479141db1 100644 --- a/agent/codex_responses_adapter.py +++ b/agent/codex_responses_adapter.py @@ -127,14 +127,21 @@ def _chat_content_to_responses_parts(content: Any, *, role: str = "user") -> Lis return converted -def _summarize_user_message_for_log(content: Any) -> str: - """Return a short text summary of a user message for logging/trajectory. +def _summarize_user_message_for_log(content: Any, *, sep: str = " ") -> str: + """Flatten message content to a plain-text summary. Multimodal messages arrive as a list of ``{type:"text"|"image_url", ...}`` - parts from the API server. Logging, spinner previews, and trajectory - files all want a plain string — this helper extracts the first chunk of - text and notes any attached images. Returns an empty string for empty - lists and ``str(content)`` for unexpected scalar types. + parts from the API server. Several consumers want a plain string: + + - Logging, spinner previews, and trajectory files (the default ``sep=" "``). + - External memory providers, which feed the text to regexes + (``sanitize_context``) and text APIs — a raw list crashes the sync with + ``expected string or bytes-like object, got 'list'`` (use ``sep="\\n"``). + + Text parts are joined with ``sep``; images become a ``[N image(s)]`` marker + so the turn isn't recorded as if the attachment never existed. Returns an + empty string for empty lists and ``str(content)`` for unexpected scalar + types. """ if content is None: return "" @@ -157,7 +164,7 @@ def _summarize_user_message_for_log(content: Any) -> str: text_bits.append(text) elif ptype in {"image_url", "input_image"}: image_count += 1 - summary = " ".join(text_bits).strip() + summary = sep.join(text_bits).strip() if image_count: note = f"[{image_count} image{'s' if image_count != 1 else ''}]" summary = f"{note} {summary}" if summary else note @@ -1074,6 +1081,7 @@ def _normalize_codex_response( message_items_raw: List[Dict[str, Any]] = [] tool_calls: List[Any] = [] has_incomplete_items = response_status in {"queued", "in_progress", "incomplete"} + saw_streaming_or_item_incomplete = response_status in {"queued", "in_progress"} saw_commentary_phase = False saw_final_answer_phase = False saw_reasoning_item = False @@ -1088,6 +1096,7 @@ def _normalize_codex_response( if item_status in {"queued", "in_progress", "incomplete"}: has_incomplete_items = True + saw_streaming_or_item_incomplete = True if item_type == "message": item_phase = getattr(item, "phase", None) @@ -1245,7 +1254,9 @@ def _normalize_codex_response( finish_reason = "tool_calls" elif leaked_tool_call_text: finish_reason = "incomplete" - elif has_incomplete_items or (saw_commentary_phase and not saw_final_answer_phase): + elif saw_streaming_or_item_incomplete: + finish_reason = "incomplete" + elif (has_incomplete_items or saw_commentary_phase) and not saw_final_answer_phase: finish_reason = "incomplete" elif (reasoning_items_raw or reasoning_parts or saw_reasoning_item) and not final_text: # Response contains only reasoning (encrypted thinking state and/or diff --git a/agent/coding_context.py b/agent/coding_context.py new file mode 100644 index 00000000000..ede0dc1528a --- /dev/null +++ b/agent/coding_context.py @@ -0,0 +1,738 @@ +"""Coding-context awareness — base Hermes, every interactive surface. + +When the user runs Hermes inside a code workspace (CLI, TUI, desktop app, or an +editor over ACP), Hermes shifts into a **coding posture**. This module is the +single place that decides whether we're in that posture and what it implies, +so the rest of the codebase never re-derives "are we coding?" on its own. + +Architecture — one seam, many consumers +---------------------------------------- +The posture is modelled as a frozen :class:`RuntimeMode` selected from a small +:class:`ContextProfile` registry (today: ``coding`` and ``general``). A profile +is *data* — it declares the toolset to collapse to, the operating brief to +inject, and hints for other domains (model routing, memory, subagents). Every +domain reads the same resolved object instead of probing git/config itself: + + * **System prompt** — ``RuntimeMode.system_blocks()`` → the operating brief + + a live git/workspace snapshot (``agent/system_prompt.py``). + * **Toolset** — ``RuntimeMode.toolset_selection()`` → the ``coding`` toolset + plus the user's enabled MCP servers (``cli.py`` / ``tui_gateway``). Only + under the opt-in ``focus`` mode: the default posture is prompt-only and + never touches the user's configured toolsets (toolsets like messaging / + smart-home / music are off-by-default anyway, and someone who explicitly + enabled image-gen or Spotify shouldn't lose it for being in a git repo). + * **Delegation** — subagents inherit the parent's toolset and run through the + same prompt builder, so the coding posture propagates to children for free. + * **Model / memory / compression** — declared on the profile + (``model_hint``, ``memory_policy``) as the extension seam; consumers read + ``mode.profile`` rather than re-deciding. + +Cache safety +------------ +The mode is resolved **once** and is immutable. The workspace snapshot is built +once at prompt-build time and baked into the *stable* system-prompt tier — never +re-probed per turn (that would shatter the prompt cache). Branch and dirty state +drift mid-session, so the brief tells the model to re-check with ``git`` before +acting on the snapshot. A ``/coding`` flip therefore only takes effect next +session (deferred), the same contract as ``/skills install`` vs ``--now``. + +Activation (config ``agent.coding_context``): + + * ``auto`` (default) — posture (brief + snapshot) on an interactive coding + surface sitting in a code workspace (git repo or recognised project root). + Prompt-only; toolsets and the skill index untouched. + * ``focus`` — like ``auto``, but additionally collapses the toolset to the + ``coding`` set + enabled MCP servers and demotes non-coding skill + categories to names-only in the prompt's skill index (no skill is ever + hidden). Explicit opt-in for a lean schema. + * ``on`` — force the posture anywhere (incl. non-workspaces). Prompt-only. + * ``off`` — disable entirely. +""" + +from __future__ import annotations + +import json +import logging +import os +import re +import subprocess +from dataclasses import dataclass +from pathlib import Path +from typing import Any, Optional + +logger = logging.getLogger("hermes.coding_context") + +CODING_TOOLSET = "coding" + +# Surfaces where a coding posture makes sense under ``auto``. Messaging +# platforms (telegram, discord, slack, …) are intentionally absent — a chat bot +# in a group is not pair-programming. +INTERACTIVE_CODING_PLATFORMS = {"cli", "tui", "acp", "desktop", ""} + +# Project-root signals that mark a directory as a code workspace even when it +# isn't (yet) a git repo. Cheap filename checks — no parsing. +_PROJECT_MARKERS = ( + "pyproject.toml", "setup.py", "setup.cfg", "requirements.txt", + "package.json", "tsconfig.json", "deno.json", + "Cargo.toml", "go.mod", "pom.xml", "build.gradle", "build.gradle.kts", + "Gemfile", "composer.json", "mix.exs", "pubspec.yaml", + "CMakeLists.txt", "Makefile", "Dockerfile", + "AGENTS.md", "CLAUDE.md", ".cursorrules", +) + +# Agent-instruction files surfaced separately from manifests in the snapshot. +_CONTEXT_FILES = ("AGENTS.md", "CLAUDE.md", ".cursorrules") + +# Lockfile → package manager, checked in priority order. +_PY_LOCKFILES = (("uv.lock", "uv"), ("poetry.lock", "poetry"), ("Pipfile.lock", "pipenv")) +_JS_LOCKFILES = ( + ("pnpm-lock.yaml", "pnpm"), ("bun.lockb", "bun"), ("bun.lock", "bun"), + ("yarn.lock", "yarn"), ("package-lock.json", "npm"), +) + +# package.json scripts / Makefile targets worth surfacing as verify commands. +_VERIFY_TARGETS = ("test", "tests", "lint", "typecheck", "check", "build", "fmt", "format") +_MAX_VERIFY_COMMANDS = 8 +_MAX_FACT_FILE_BYTES = 256 * 1024 + +_GIT_TIMEOUT = 2.5 + + +# Per-model edit-format steering. Matching the edit tool format to how a model +# was trained reduces mistakes and wasted reasoning (OpenAI/Codex handle +# patch-style diffs best; Anthropic models — and most open-weight coding +# models, whose RL scaffolds use str_replace-style editors — do best with +# string-replacement). Our `patch` tool exposes both: mode="patch" (V4A +# multi-file) and mode="replace" (find-and-swap). We nudge each family toward +# its native format. Unknown families get nothing (the brief's neutral wording +# stands). Substrings match the model id; aligned with TOOL_USE_ENFORCEMENT_MODELS. +# +# GPT/Codex get V4A for ALL edits, single-file included: in codex-rs, +# apply_patch (V4A — apply_patch.lark) is the ONLY file editor, no +# str_replace-style tool exists, and the shipped model prompts say to use +# apply_patch even "for single file edits" — so a replace-mode nudge would +# steer those models toward a format their first-party harness never taught +# them. +_EDIT_FORMAT_GUIDANCE: dict[str, tuple[tuple[str, ...], str]] = { + "patch": ( + ("gpt", "codex"), + "- Edit format: author new files with `write_file`; for edits to " + "existing code use `patch` with `mode='patch'` (V4A diff) — including " + "single-file edits. It's the edit format you handle most reliably.", + ), + "replace": ( + ("claude", "sonnet", "opus", "haiku", + "gemini", "gemma", "deepseek", "qwen", "kimi", "glm", "grok", + "hermes", "llama", "mistral", "devstral", "minimax"), + "- Edit format: author new files with `write_file`; for edits to " + "existing code prefer `patch` in `mode='replace'` — match a unique " + "snippet and swap it. Reach for `mode='patch'` (V4A) only when an edit " + "genuinely spans several files at once.", + ), +} + + +def _model_family(model: Optional[str]) -> Optional[str]: + """Classify a model id into an edit-format family key, or ``None``. + + Used to steer the coding posture toward the edit tool format a model was + trained on. Family-agnostic by design: an unrecognised model gets ``None`` + and the operating brief's neutral edit wording applies. + """ + if not model: + return None + lowered = model.lower() + for family, (needles, _line) in _EDIT_FORMAT_GUIDANCE.items(): + if any(n in lowered for n in needles): + return family + return None + + +def _edit_format_line(model: Optional[str]) -> str: + """The edit-format guidance line for this model's family (``""`` if none).""" + family = _model_family(model) + if family is None: + return "" + return _EDIT_FORMAT_GUIDANCE[family][1] + + +# Operating brief for the coding posture. Tool names referenced here (read_file, +# search_files, patch, write_file, terminal, todo) are in the coding toolset and +# in _HERMES_CORE_TOOLS, so they're present on every surface this fires on. +CODING_AGENT_GUIDANCE = ( + "You are a coding agent pairing with the user inside their codebase. " + "Operate like a careful senior engineer.\n" + "\n" + "Gather context first:\n" + "- Read the relevant files with `read_file` and locate code with " + "`search_files` before changing anything. Trace a symbol to its definition " + "and usages rather than guessing its shape.\n" + "- Batch independent lookups: when several reads/searches don't depend on " + "each other, issue them together in one turn instead of one at a time.\n" + "- Never invent files, symbols, APIs, or imports. If you haven't seen it in " + "the repo, go look. Don't assume a library is available — check the project " + "manifest (pyproject.toml / package.json / Cargo.toml / go.mod) and how " + "neighbouring files import it.\n" + "\n" + "Make changes through the tools, not the chat:\n" + "- Edit with `patch`/`write_file`. Do NOT print code blocks to the user as " + "a substitute for editing — apply the change, then summarise it. Only show " + "code when the user explicitly asks to see it.\n" + "- Match the project's existing style and conventions; AGENTS.md / " + "CLAUDE.md / .cursorrules already in context win over your defaults. Touch " + "only what the task needs — no drive-by refactors, renames, or reformatting " + "— and add any imports/dependencies your code requires.\n" + "- If an edit fails to apply, re-read the file to get the current exact " + "contents before retrying — don't repeat a stale patch. If the same region " + "fails twice, rewrite the enclosing function or file with `write_file` " + "instead of attempting a third patch.\n" + "\n" + "Verify, and know when to stop:\n" + "- Use `terminal` for git, builds, tests, and inspection. Run the relevant " + "tests/linter/build and confirm they pass before claiming the work is done.\n" + "- Terminal state persists across calls: current directory and exported " + "environment variables carry forward. Activate a virtualenv or export setup " + "vars once, then reuse that state instead of re-sourcing it before every " + "test command.\n" + "- Fix root causes, not symptoms: when you find a bug, check sibling call " + "paths for the same flaw and fix the class, not just the reported site.\n" + "- When fixing linter/type errors on a file, stop after about three " + "attempts on the same file and ask the user rather than looping.\n" + "- Track multi-step work with `todo`. Reference code as `path:line` instead " + "of pasting whole files.\n" + "\n" + "Respect the user's repo: don't commit, push, or rewrite history unless " + "asked, and never read, print, or commit secrets — leave `.env` and " + "credential files alone unless the user explicitly asks. The Workspace " + "block below is a snapshot from session start — re-run `git status`/" + "`git branch` before relying on it. Be concise: lead with the change or " + "answer, not a preamble." +) + + +# ── Context profiles (declarative posture definitions) ────────────────────── + + +@dataclass(frozen=True) +class ContextProfile: + """A named operating posture. Pure data — consumers read these fields. + + ``toolset`` — collapse to this toolset (+ enabled MCP) when no explicit + selection is pinned; ``None`` keeps the platform default. + ``guidance`` — operating brief injected into the stable system prompt; + ``""`` injects nothing. + ``model_hint`` — routing preference key for smart model routing + (extension seam; not yet consumed by the router). + ``memory_policy``— memory namespace/weighting hint (extension seam). + ``compact_skill_categories`` — skill categories DEMOTED to names-only in + the system-prompt skill index under the opt-in ``focus`` + mode. Never hidden: every skill name stays visible + (so memory-anchored recall keeps working) — only the + descriptions are dropped to cut index noise. Deny-list + semantics so unknown/custom categories keep full + entries. + """ + + name: str + toolset: Optional[str] = None + guidance: str = "" + model_hint: Optional[str] = None + memory_policy: str = "default" + compact_skill_categories: tuple[str, ...] = () + + +# Skill categories that are clearly not part of a coding workflow. Demoted to +# names-only in the prompt's skill index under the opt-in ``focus`` mode only +# (deny-list — anything not listed here, incl. custom user categories, keeps +# full entries). Coding-adjacent categories (devops, github, mcp, +# data-science, diagramming, research, security, …) are intentionally absent. +_NON_CODING_SKILL_CATEGORIES = ( + "apple", "communication", "cooking", "creative", "email", "finance", + "gaming", "gifs", "health", "media", "music", "note-taking", + "productivity", "shopping", "smart-home", "social-media", "travel", + "yuanbao", +) + + +GENERAL_PROFILE = ContextProfile(name="general") +CODING_PROFILE = ContextProfile( + name="coding", + toolset=CODING_TOOLSET, + guidance=CODING_AGENT_GUIDANCE, + model_hint="coding", + memory_policy="project", + compact_skill_categories=_NON_CODING_SKILL_CATEGORIES, +) + +_PROFILES: dict[str, ContextProfile] = { + GENERAL_PROFILE.name: GENERAL_PROFILE, + CODING_PROFILE.name: CODING_PROFILE, +} + + +def get_profile(name: str) -> ContextProfile: + """Return a registered profile, falling back to ``general``.""" + return _PROFILES.get(name, GENERAL_PROFILE) + + +# ── Helpers ───────────────────────────────────────────────────────────────── + + +def _coding_mode(config: Optional[dict[str, Any]]) -> str: + """Return the normalized ``agent.coding_context`` mode (auto/focus/on/off).""" + if config is None: + try: + from hermes_cli.config import load_config + + config = load_config() + except Exception: + config = {} + raw = ((config or {}).get("agent", {}) or {}).get("coding_context", "auto") + mode = str(raw).strip().lower() + if mode in {"focus", "strict", "lean"}: + return "focus" + if mode in {"on", "true", "yes", "1", "always"}: + return "on" + if mode in {"off", "false", "no", "0", "never"}: + return "off" + return "auto" + + +def _resolve_cwd(cwd: Optional[str | Path]) -> Path: + if cwd: + return Path(cwd).expanduser() + try: + from agent.runtime_cwd import resolve_agent_cwd + + return resolve_agent_cwd() + except Exception: + return Path(os.getcwd()) + + +def _git_root(cwd: Path) -> Optional[Path]: + current = cwd.resolve() + for parent in [current, *current.parents]: + if (parent / ".git").exists(): + return parent + return None + + +def _home() -> Optional[Path]: + try: + return Path.home().resolve() + except (OSError, RuntimeError): + return None + + +def _marker_root(cwd: Path) -> Optional[Path]: + """Nearest ancestor that looks like a project root, or ``None``. + + Walks up at most a few levels so a manifest in the workspace root counts + even when the user is in a subdirectory. ``$HOME`` itself is skipped — a + Makefile or AGENTS.md sitting in the home directory is global user config, + not a project-root signal. + """ + current = cwd.resolve() + home = _home() + for depth, parent in enumerate([current, *current.parents]): + if depth > 6: + break + if parent == home: + continue + for marker in _PROJECT_MARKERS: + if (parent / marker).exists(): + return parent + return None + + +def _detect_profile_name(mode: str, platform: str, cwd_str: str) -> str: + """Resolve which profile applies. + + ``auto``/``focus``: coding when the surface is interactive AND the cwd is a + code workspace (a git repo or a recognised project root). ``on``: always + coding. ``off``: always general. + + A git repo rooted at ``$HOME`` (the dotfiles pattern) is NOT a workspace + signal — without the guard, every session anywhere under a dotfiles-managed + home directory would silently flip to the coding posture. + + Detection is intentionally not memoized: it's a handful of ``stat`` calls, + and callers resolve the mode once per session anyway. Caching here would + risk a stale posture if a long-lived process (gateway/TUI) serves sessions + from different working directories. + """ + if mode == "off": + return GENERAL_PROFILE.name + if mode == "on": + return CODING_PROFILE.name + if platform and platform.strip().lower() not in INTERACTIVE_CODING_PLATFORMS: + return GENERAL_PROFILE.name + cwd = Path(cwd_str) + git_root = _git_root(cwd) + if git_root is not None and git_root == _home(): + git_root = None # dotfiles repo at $HOME — not a code workspace + if git_root is not None or _marker_root(cwd) is not None: + return CODING_PROFILE.name + return GENERAL_PROFILE.name + + +# ── RuntimeMode (the seam) ────────────────────────────────────────────────── + + +@dataclass(frozen=True) +class RuntimeMode: + """The resolved operating posture for a session. Immutable by construction. + + Built once via :func:`resolve_runtime_mode` and consumed by every domain + that cares about the coding/general distinction. Never mutate or re-resolve + mid-session — that would break the prompt cache. + """ + + profile: ContextProfile + surface: str + cwd: Path + # The normalized ``agent.coding_context`` mode this posture was resolved + # under (auto/focus/on/off). Toolset collapse is gated on ``focus``. + config_mode: str = "auto" + # The model id this session runs (e.g. "anthropic/claude-opus-4.8"). Used + # only to steer edit-format guidance toward the model's family — see + # ``_edit_format_line``. Fixed for the session, so cache-safe. + model: Optional[str] = None + + @property + def kind(self) -> str: + return self.profile.name + + @property + def is_coding(self) -> bool: + return self.profile.name == CODING_PROFILE.name + + def toolset_selection(self, config: Optional[dict[str, Any]] = None) -> Optional[list[str]]: + """Toolset list for this posture, or ``None`` to keep the platform default. + + Non-``None`` only under the opt-in ``focus`` mode. The default posture + is prompt-only: most strippable toolsets are off-by-default anyway, and + a user who explicitly enabled one (image-gen for frontend/game assets, + messaging for build notifications, …) keeps it while coding. + + Callers apply this only when the user hasn't pinned an explicit + selection (``--toolsets``, ``HERMES_TUI_TOOLSETS``, …); they never + override a pin. Returns the profile's toolset plus enabled MCP servers. + """ + if self.config_mode != "focus": + return None + if self.profile.toolset is None: + return None + return [self.profile.toolset, *_enabled_mcp_servers(config)] + + def system_blocks(self) -> list[str]: + """Stable system-prompt blocks for this posture (brief + workspace). + + The operating brief carries a model-family edit-format nudge appended + to it (one cached string, not a separate block) so the model is steered + toward the `patch` mode it handles best — see ``_edit_format_line``. + """ + if not self.is_coding: + return [] + blocks: list[str] = [] + if self.profile.guidance: + brief = self.profile.guidance + edit_line = _edit_format_line(self.model) + if edit_line: + brief = f"{brief}\n{edit_line}" + blocks.append(brief) + workspace = build_coding_workspace_block(self.cwd) + if workspace: + blocks.append(workspace) + return blocks + + def compact_skill_categories(self) -> frozenset[str]: + """Skill categories to demote to names-only in the prompt's skill index. + + Gated on the opt-in ``focus`` mode, like the toolset collapse: the + default posture leaves the skill index untouched. Users who didn't ask + for a lean prompt keep full entries for every category — index changes + under ``auto`` proved too surprising in practice, even names-only ones + (a demoted description is information the model no longer weighs when + deciding what to load). + + Demoted — never hidden — even under ``focus``. An earlier revision + fully pruned these categories from the index, which caused silent + capability loss in a real workflow: agent-created skills are the + model's accumulated project memory (server-ops runbooks, learned + pitfalls, …), and models do not reliably reach for ``skills_list`` to + rediscover what the index stopped showing them. Names-only keeps every + skill loadable on recall while still cutting the description noise. + """ + if not self.is_coding or self.config_mode != "focus": + return frozenset() + return frozenset(self.profile.compact_skill_categories) + + +def resolve_runtime_mode( + *, + platform: Optional[str] = None, + cwd: Optional[str | Path] = None, + config: Optional[dict[str, Any]] = None, + model: Optional[str] = None, +) -> RuntimeMode: + """Resolve the operating posture once. Cheap — a handful of ``stat`` calls. + + This is the single entry point every domain should call. The returned + object is immutable and safe to cache for the session. Detection itself is + intentionally *not* memoized (see ``_detect_profile_name``) so a long-lived + process can't pin a stale posture; callers resolve once per session and + hold the result. ``model`` is recorded only to steer edit-format guidance; + it never affects detection. + """ + resolved_cwd = _resolve_cwd(cwd) + mode = _coding_mode(config) + name = _detect_profile_name( + mode, (platform or "").strip().lower(), str(resolved_cwd) + ) + return RuntimeMode( + profile=get_profile(name), + surface=platform or "", + cwd=resolved_cwd, + config_mode=mode, + model=model, + ) + + +# ── Back-compat surface (thin wrappers over RuntimeMode) ──────────────────── + + +def is_coding_context( + *, + platform: Optional[str] = None, + cwd: Optional[str | Path] = None, + config: Optional[dict[str, Any]] = None, +) -> bool: + """Whether Hermes should operate in its coding posture right now.""" + return resolve_runtime_mode(platform=platform, cwd=cwd, config=config).is_coding + + +def coding_selection( + *, + platform: Optional[str] = None, + cwd: Optional[str | Path] = None, + config: Optional[dict[str, Any]] = None, +) -> Optional[list[str]]: + """Toolset selection for the coding posture. + + ``None`` unless the user opted into ``focus`` mode AND the posture is + active — the default coding posture never overrides configured toolsets. + """ + return resolve_runtime_mode( + platform=platform, cwd=cwd, config=config + ).toolset_selection(config) + + +def coding_system_blocks( + *, + platform: Optional[str] = None, + cwd: Optional[str | Path] = None, + config: Optional[dict[str, Any]] = None, + model: Optional[str] = None, +) -> list[str]: + """Stable system-prompt blocks for the current posture (empty when general). + + ``model`` steers the brief's edit-format nudge toward the model's family. + """ + return resolve_runtime_mode( + platform=platform, cwd=cwd, config=config, model=model + ).system_blocks() + + +def coding_compact_skill_categories( + *, + platform: Optional[str] = None, + cwd: Optional[str | Path] = None, + config: Optional[dict[str, Any]] = None, +) -> frozenset[str]: + """Skill categories the active posture demotes to names-only in the index. + + Empty outside the coding posture and outside the opt-in ``focus`` mode — + the default posture never touches the skill index. Under ``focus``, + demoted — never hidden: every skill name stays in the index and remains + loadable via ``skill_view`` / ``skills_list``; only descriptions are + dropped. + """ + return resolve_runtime_mode( + platform=platform, cwd=cwd, config=config + ).compact_skill_categories() + + +def _enabled_mcp_servers(config: Optional[dict[str, Any]]) -> list[str]: + """Names of MCP servers the user has enabled — kept in the coding posture. + + MCP servers (figma, browser, tophat, …) are explicitly configured and part + of the coding workflow, not noise to strip. + """ + try: + from hermes_cli.config import read_raw_config + from hermes_cli.tools_config import _parse_enabled_flag + + servers = read_raw_config().get("mcp_servers") or {} + return [ + str(name) + for name, cfg in servers.items() + if isinstance(cfg, dict) + and _parse_enabled_flag(cfg.get("enabled", True), default=True) + ] + except Exception: + return [] + + +# ── git/workspace probe ───────────────────────────────────────────────────── + + +def _git(cwd: Path, *args: str) -> str: + try: + out = subprocess.run( + ["git", "-C", str(cwd), *args], + capture_output=True, + text=True, + timeout=_GIT_TIMEOUT, + ) + except (OSError, subprocess.SubprocessError): + return "" + return out.stdout.strip() if out.returncode == 0 else "" + + +def _parse_status(porcelain: str) -> tuple[dict[str, str], dict[str, int]]: + """Parse ``git status --porcelain=2 --branch`` into branch + counts.""" + branch: dict[str, str] = {} + counts = {"staged": 0, "modified": 0, "untracked": 0, "conflicts": 0} + for line in porcelain.splitlines(): + if line.startswith("# branch.head"): + branch["head"] = line.split(maxsplit=2)[-1] + elif line.startswith("# branch.upstream"): + branch["upstream"] = line.split(maxsplit=2)[-1] + elif line.startswith("# branch.ab"): + parts = line.split() + branch["ahead"], branch["behind"] = parts[2].lstrip("+"), parts[3].lstrip("-") + elif line.startswith(("1 ", "2 ")): + xy = line.split(maxsplit=2)[1] + if xy[0] != ".": + counts["staged"] += 1 + if xy[1] != ".": + counts["modified"] += 1 + elif line.startswith("u "): + counts["conflicts"] += 1 + elif line.startswith("? "): + counts["untracked"] += 1 + return branch, counts + + +def _read_small(path: Path) -> str: + """Read a small text file, or ``""`` — never raises, never reads huge files.""" + try: + if not path.is_file() or path.stat().st_size > _MAX_FACT_FILE_BYTES: + return "" + return path.read_text(encoding="utf-8", errors="replace") + except OSError: + return "" + + +def _project_facts(root: Path) -> list[str]: + """Detected project facts for the workspace snapshot. + + The point is to hand the model its *verify loop* up front — which manifest, + which package manager, and the exact test/lint/build commands — instead of + making it rediscover them every session. Cheap: stat calls plus reads of a + couple of small files; built once at prompt-build time (cache-safe). + """ + facts: list[str] = [] + + manifests = [m for m in _PROJECT_MARKERS if m not in _CONTEXT_FILES and (root / m).is_file()] + package_managers = [ + pm for lock, pm in (*_PY_LOCKFILES, *_JS_LOCKFILES) if (root / lock).is_file() + ] + if manifests: + line = f"- Project: {', '.join(manifests[:6])}" + if package_managers: + line += f" ({'/'.join(dict.fromkeys(package_managers))})" + facts.append(line) + + verify: list[str] = [] + if (root / "scripts" / "run_tests.sh").is_file(): + verify.append("scripts/run_tests.sh") + if (root / "package.json").is_file(): + try: + scripts = json.loads(_read_small(root / "package.json") or "{}").get("scripts") or {} + except (json.JSONDecodeError, AttributeError): + scripts = {} + js_pm = next((pm for lock, pm in _JS_LOCKFILES if (root / lock).is_file()), "npm") + verify.extend(f"{js_pm} run {name}" for name in _VERIFY_TARGETS if name in scripts) + if (root / "pytest.ini").is_file() or "[tool.pytest" in _read_small(root / "pyproject.toml"): + verify.append("pytest") + makefile = _read_small(root / "Makefile") + if makefile: + verify.extend( + f"make {name}" for name in _VERIFY_TARGETS + if re.search(rf"^{re.escape(name)}\s*:", makefile, re.MULTILINE) + ) + if verify: + deduped = list(dict.fromkeys(verify))[:_MAX_VERIFY_COMMANDS] + facts.append(f"- Verify: {'; '.join(deduped)}") + + context_files = [c for c in _CONTEXT_FILES if (root / c).is_file()] + if context_files: + facts.append(f"- Context files: {', '.join(context_files)}") + + return facts + + +def build_coding_workspace_block(cwd: Optional[str | Path] = None) -> str: + """Workspace snapshot for the system prompt (empty outside a workspace). + + Git state (branch/status/commits) when the cwd is in a repo, plus detected + project facts (manifest, package manager, verify commands, context files) + — so marker-only (non-git) projects still get a snapshot. + """ + resolved = _resolve_cwd(cwd) + git_root = _git_root(resolved) + root = git_root or _marker_root(resolved) + if root is None: + return "" + + lines = ["Workspace (snapshot at session start — re-check with `git` before acting on it):"] + lines.append(f"- Root: {root}") + + if git_root is not None: + branch, counts = _parse_status(_git(root, "status", "--porcelain=2", "--branch")) + head = branch.get("head", "") + if head and head != "(detached)": + line = f"- Branch: {head}" + if branch.get("upstream"): + line += f" \u2192 {branch['upstream']}" + ahead, behind = branch.get("ahead", "0"), branch.get("behind", "0") + if ahead != "0" or behind != "0": + line += f" (ahead {ahead}, behind {behind})" + lines.append(line) + elif head == "(detached)": + lines.append("- Branch: (detached HEAD)") + + # Linked worktree: the per-worktree git dir differs from the shared common dir. + # We surface the fact that it's a worktree (so the model knows branches/stashes + # are shared state) but deliberately do NOT expose the primary tree path — + # giving the model a second absolute path causes it to sometimes run commands + # in the wrong directory. + git_dir, common_dir = _git(root, "rev-parse", "--git-dir"), _git(root, "rev-parse", "--git-common-dir") + if git_dir and common_dir and Path(git_dir).resolve() != Path(common_dir).resolve(): + lines.append("- Worktree: linked (git state shared with primary tree)") + + dirty = [f"{n} {label}" for label, n in ( + ("staged", counts["staged"]), ("modified", counts["modified"]), + ("untracked", counts["untracked"]), ("conflicts", counts["conflicts"]), + ) if n] + lines.append(f"- Status: {', '.join(dirty) if dirty else 'clean'}") + + recent = _git(root, "log", "-3", "--pretty=%h %s") + if recent: + lines.append("- Recent commits:") + lines.extend(f" {c}" for c in recent.splitlines()) + + lines.extend(_project_facts(root)) + return "\n".join(lines) diff --git a/agent/context_compressor.py b/agent/context_compressor.py index 98d226b46af..16db1bedc30 100644 --- a/agent/context_compressor.py +++ b/agent/context_compressor.py @@ -7,7 +7,7 @@ protecting head and tail context. Improvements over v2: - Structured summary template with Resolved/Pending question tracking - Filter-safe summarizer preamble that treats prior turns as source material - - "Remaining Work" replaces "Next Steps" to avoid reading as active instructions + - Historical (reference-only) section headings replace "Next Steps"/"Remaining Work" to avoid reading as active instructions - Clear separator when summary merges into tail message - Iterative summary updates (preserves info across multiple compactions) - Token-budget tail protection instead of fixed message count @@ -34,7 +34,75 @@ from agent.redact import redact_sensitive_text logger = logging.getLogger(__name__) +HISTORICAL_TASK_HEADING = "## Historical Task Snapshot" +HISTORICAL_IN_PROGRESS_HEADING = "## Historical In-Progress State" +HISTORICAL_PENDING_ASKS_HEADING = "## Historical Pending User Asks" +HISTORICAL_REMAINING_WORK_HEADING = "## Historical Remaining Work" + + SUMMARY_PREFIX = ( + "[CONTEXT COMPACTION — REFERENCE ONLY] Earlier turns were compacted " + "into the summary below. This is a handoff from a previous context " + "window — treat it as background reference, NOT as active instructions. " + "Do NOT answer questions or fulfill requests mentioned in this summary; " + "they were already addressed. " + "Respond ONLY to the latest user message that appears AFTER this " + "summary — that message is the single source of truth for what to do " + "right now. " + "Topic overlap with the summary does NOT mean you should resume its " + "task: even on similar topics, the latest user message WINS. Treat ONLY " + "the latest message as the active task and discard stale items from " + f"'{HISTORICAL_TASK_HEADING}' / '{HISTORICAL_IN_PROGRESS_HEADING}' / " + f"'{HISTORICAL_PENDING_ASKS_HEADING}' / " + f"'{HISTORICAL_REMAINING_WORK_HEADING}' entirely — do not 'wrap up' or " + "'finish' work described there unless the latest message explicitly " + "asks for it. " + "Reverse signals in the latest message (e.g. 'stop', 'undo', 'roll " + "back', 'just verify', 'don't do that anymore', 'never mind', a new " + "topic) must immediately end any in-flight work described in the " + "summary; do not re-surface it in later turns. " + "IMPORTANT: Your persistent memory (MEMORY.md, USER.md) in the system " + "prompt is ALWAYS authoritative and active — never ignore or deprioritize " + "memory content due to this compaction note. " + "The current session state (files, config, etc.) may reflect work " + "described here — avoid repeating it:" +) +LEGACY_SUMMARY_PREFIX = "[CONTEXT SUMMARY]:" + +# Metadata key added to context compression summary messages so that frontends +# (CLI, Desktop, gateway, TUI) can distinguish them from real assistant/user +# messages and filter or render them appropriately without content-prefix +# heuristics. See https://github.com/NousResearch/hermes-agent/issues/38389 +# +# Underscore-prefixed ON PURPOSE: the wire sanitizers +# (agent/transports/chat_completions.py convert_messages and the summary-path +# mirror in agent/chat_completion_helpers.py) strip every top-level message +# key starting with "_" before the request leaves the process. Strict +# OpenAI-compatible gateways (Fireworks, Mistral, Moonshot/Kimi, opencode-go) +# reject payloads carrying unknown keys with "Extra inputs are not permitted", +# poisoning every subsequent request in the session — a bare key like +# "is_compressed_summary" would reach the wire and trip exactly that. +COMPRESSED_SUMMARY_METADATA_KEY = "_compressed_summary" + +# Appended to every standalone summary message (and to the merged-into-tail +# prefix) so the model has an unambiguous "summary ends here" boundary. +# Without it, weak models read the verbatim "## Active Task" quote as fresh +# user input (#11475, #14521) or regurgitate an assistant-role summary as +# their own output (#33256). +_SUMMARY_END_MARKER = ( + "--- END OF CONTEXT SUMMARY — " + "respond to the message below, not the summary above ---" +) + +# Handoff prefixes that shipped in earlier releases. A summary persisted under +# one of these can be inherited into a resumed lineage (#35344); when it is +# re-normalized on re-compaction we must strip the OLD prefix too, otherwise the +# stale directive it carried (e.g. "resume exactly from Active Task") survives +# embedded in the body and keeps hijacking replies. Keep newest-first; entries +# are matched literally. Add a frozen copy here whenever SUMMARY_PREFIX changes. +_HISTORICAL_SUMMARY_PREFIXES = ( + # Carveout era (#41607/#38364/#42812): "consistent → use as background" + # licensed stale-task resumption on topic overlap. "[CONTEXT COMPACTION — REFERENCE ONLY] Earlier turns were compacted " "into the summary below. This is a handoff from a previous context " "window — treat it as background reference, NOT as active instructions. " @@ -57,17 +125,7 @@ SUMMARY_PREFIX = ( "prompt is ALWAYS authoritative and active — never ignore or deprioritize " "memory content due to this compaction note. " "The current session state (files, config, etc.) may reflect work " - "described here — avoid repeating it:" -) -LEGACY_SUMMARY_PREFIX = "[CONTEXT SUMMARY]:" - -# Handoff prefixes that shipped in earlier releases. A summary persisted under -# one of these can be inherited into a resumed lineage (#35344); when it is -# re-normalized on re-compaction we must strip the OLD prefix too, otherwise the -# stale directive it carried (e.g. "resume exactly from Active Task") survives -# embedded in the body and keeps hijacking replies. Keep newest-first; entries -# are matched literally. Add a frozen copy here whenever SUMMARY_PREFIX changes. -_HISTORICAL_SUMMARY_PREFIXES = ( + "described here — avoid repeating it:", # Pre-#35344: contained the self-contradicting "resume exactly" directive. "[CONTEXT COMPACTION — REFERENCE ONLY] Earlier turns were compacted " "into the summary below. This is a handoff from a previous context " @@ -110,10 +168,23 @@ _SUMMARY_FAILURE_COOLDOWN_SECONDS = 600 # become another unbounded transcript copy after the LLM summarizer failed. _FALLBACK_SUMMARY_MAX_CHARS = 8_000 _FALLBACK_TURN_MAX_CHARS = 700 +_AUTO_FOCUS_MAX_TURNS = 3 +_AUTO_FOCUS_TURN_MAX_CHARS = 260 +_AUTO_FOCUS_MAX_CHARS = 700 +# Keep a short run of recent messages verbatim even when the token budget is +# already exhausted. The public ``protect_last_n`` default is intentionally +# high for small/light tails, but using all 20 as a hard floor here would bring +# back the old large-tool-output case where nothing can be compacted. +_MAX_TAIL_MESSAGE_FLOOR = 8 _PATH_MENTION_RE = re.compile(r"(?:/|~/?|[A-Za-z]:\\)[^\s`'\")\]}<>]+") +# MEDIA delivery directives must not reach the summarizer — if one leaks into +# the summary, the downstream model may re-emit it as an active directive on +# the next turn, triggering bogus attachment sends (#14665). +_MEDIA_DIRECTIVE_RE = re.compile(r"MEDIA:\S+") + def _dedupe_append(items: list[str], value: str, *, limit: int) -> None: value = value.strip() @@ -974,6 +1045,7 @@ class ContextCompressor(ContextEngine): for msg in turns: role = msg.get("role", "unknown") content = redact_sensitive_text(msg.get("content") or "") + content = _MEDIA_DIRECTIVE_RE.sub("[media attachment]", content) # Tool results: keep enough content for the summarizer if role == "tool": @@ -1155,7 +1227,7 @@ class ContextCompressor(ContextEngine): ) reason_text = f" Summary failure reason: {reason}." if reason else "" - body = f"""## Active Task + body = f"""{HISTORICAL_TASK_HEADING} {active_task} ## Goal @@ -1172,7 +1244,7 @@ Recovered from a deterministic fallback because the LLM context summarizer was u ## Active State Unknown from deterministic fallback. Inspect current repository/session state if needed. -## In Progress +{HISTORICAL_IN_PROGRESS_HEADING} {active_task} ## Blocked @@ -1184,13 +1256,13 @@ None recoverable from deterministic fallback. ## Resolved Questions None recoverable from deterministic fallback. -## Pending User Asks +{HISTORICAL_PENDING_ASKS_HEADING} {active_task} ## Relevant Files {_bullets(relevant_files, limit=12)} -## Remaining Work +{HISTORICAL_REMAINING_WORK_HEADING} Continue from the most recent unfulfilled user ask and protected tail messages. Verify state with tools before making claims. ## Last Dropped Turns @@ -1312,7 +1384,7 @@ Summary generation was unavailable, so this is a best-effort deterministic fallb _temporal_anchoring_rule = "" # Shared structured template (used by both paths). - _template_sections = f"""## Active Task + _template_sections = f"""{HISTORICAL_TASK_HEADING} [THE SINGLE MOST IMPORTANT FIELD. Capture the user's most recent unfulfilled input verbatim — the exact words they used. This includes: - Explicit task assignments ("refactor the auth module") @@ -1359,7 +1431,7 @@ Be specific with file paths, commands, line numbers, and results.] - Any running processes or servers - Environment details that matter] -## In Progress +{HISTORICAL_IN_PROGRESS_HEADING} [Work currently underway — what was being done when compaction fired] ## Blocked @@ -1371,14 +1443,14 @@ Be specific with file paths, commands, line numbers, and results.] ## Resolved Questions [Questions the user asked that were ALREADY answered — include the answer so it is not repeated] -## Pending User Asks -[Questions or requests from the user that have NOT yet been answered or fulfilled. If none, write "None."] +{HISTORICAL_PENDING_ASKS_HEADING} +[Questions or requests from the user that have NOT yet been answered or fulfilled. These are STALE — they were from the compacted turns. Write them here for reference only. The agent must NOT act on them unless the latest user message explicitly requests it. If none, write "None."] ## Relevant Files [Files read, modified, or created — with brief note on each] -## Remaining Work -[What remains to be done — framed as context, not instructions] +{HISTORICAL_REMAINING_WORK_HEADING} +[What remains to be done — framed as STALE context for reference only. The agent must NOT resume this work unless the latest user message explicitly asks for it.] ## Critical Context [Any specific values, error messages, configuration details, or data that would be lost without explicit preservation. NEVER include API keys, tokens, passwords, or credentials — write [REDACTED] instead.] @@ -1421,7 +1493,7 @@ Use this exact structure: prompt += f""" FOCUS TOPIC: "{focus_topic}" -The user has requested that this compaction PRIORITISE preserving all information related to the focus topic above. For content related to "{focus_topic}", include full detail — exact values, file paths, command outputs, error messages, and decisions. For content NOT related to the focus topic, summarise more aggressively (brief one-liners or omit if truly irrelevant). The focus topic sections should receive roughly 60-70% of the summary token budget. Even for the focus topic, NEVER preserve API keys, tokens, passwords, or credentials — use [REDACTED].""" +This compaction should PRIORITISE preserving all information related to the focus topic above. For content related to "{focus_topic}", include full detail — exact values, file paths, command outputs, error messages, and decisions. For content NOT related to the focus topic, summarise more aggressively (brief one-liners or omit if truly irrelevant). The focus topic sections should receive roughly 60-70% of the summary token budget. Even for the focus topic, NEVER preserve API keys, tokens, passwords, or credentials — use [REDACTED].""" try: call_kwargs = { @@ -1574,7 +1646,13 @@ The user has requested that this compaction PRIORITISE preserving all informatio text = (summary or "").strip() for prefix in (SUMMARY_PREFIX, LEGACY_SUMMARY_PREFIX, *_HISTORICAL_SUMMARY_PREFIXES): if text.startswith(prefix): - return text[len(prefix):].lstrip() + text = text[len(prefix):].lstrip() + break + # Strip the trailing end marker too — a rehydrated handoff body that + # keeps it would leak the boundary directive into the iterative-update + # summarizer prompt (and the marker is re-appended on insertion anyway). + if text.endswith(_SUMMARY_END_MARKER): + text = text[: -len(_SUMMARY_END_MARKER)].rstrip() return text @classmethod @@ -1590,6 +1668,52 @@ The user has requested that this compaction PRIORITISE preserving all informatio return True return any(text.startswith(p) for p in _HISTORICAL_SUMMARY_PREFIXES) + @staticmethod + def _has_compressed_summary_metadata(message: Any) -> bool: + """Return True if *message* carries the compressed-summary flag. + + Callers (frontends, CLI, gateway) can use this to distinguish context + compaction summaries from real assistant or user messages without + relying on content-prefix heuristics. The flag is in-process only — + the wire sanitizers strip underscore-prefixed keys before API calls. + """ + if not isinstance(message, dict): + return False + return bool(message.get(COMPRESSED_SUMMARY_METADATA_KEY)) + + @classmethod + def _derive_auto_focus_topic( + cls, + messages: List[Dict[str, Any]], + ) -> Optional[str]: + """Infer a compact focus hint from the most recent real user turns.""" + candidates: list[str] = [] + for idx in range(len(messages) - 1, -1, -1): + msg = messages[idx] + if msg.get("role") != "user": + continue + content = msg.get("content") + if cls._is_context_summary_content(content): + continue + text = redact_sensitive_text(_content_text_for_contains(content).strip()) + if not text: + continue + text = " ".join(text.split()) + if len(text) > _AUTO_FOCUS_TURN_MAX_CHARS: + text = text[: _AUTO_FOCUS_TURN_MAX_CHARS - 1].rstrip() + "…" + candidates.append(text) + if len(candidates) >= _AUTO_FOCUS_MAX_TURNS: + break + + if not candidates: + return None + + candidates.reverse() + focus = "Recent user focus:\n" + "\n".join(f"- {item}" for item in candidates) + if len(focus) > _AUTO_FOCUS_MAX_CHARS: + focus = focus[: _AUTO_FOCUS_MAX_CHARS - 1].rstrip() + "…" + return focus + @classmethod def _find_latest_context_summary( cls, @@ -1742,6 +1866,105 @@ The user has requested that this compaction PRIORITISE preserving all informatio return i return -1 + def _find_last_assistant_message_idx( + self, messages: List[Dict[str, Any]], head_end: int + ) -> int: + """Return the index of the last user-visible assistant reply at or + after *head_end*, or -1. + + A "user-visible reply" is an assistant message with non-empty + textual content — i.e. one that the WebUI / TUI / SessionsPage + rendered as a bubble the operator could read. We deliberately + skip assistant messages that contain only ``tool_calls`` (and + no text), because those render as small "calling tool X" + indicators and aren't what the reporter means by "the output + of the last message you sent" (#29824). + + Falling back to the most recent assistant message of ANY kind + only kicks in when no content-bearing assistant message exists + in the compressible region — typically a fresh session that + just started a multi-step tool sequence with no prior reply + to anchor. In that case the agent fix is a no-op and the + existing user-message anchor carries the load. + """ + last_any = -1 + for i in range(len(messages) - 1, head_end - 1, -1): + msg = messages[i] + if msg.get("role") != "assistant": + continue + if last_any < 0: + last_any = i + content = msg.get("content") + if isinstance(content, str) and content.strip(): + return i + if isinstance(content, list): + # Multimodal / Anthropic-style content: look for any + # text block with non-empty text. + for part in content: + if isinstance(part, dict): + text = part.get("text") or part.get("content") + if isinstance(text, str) and text.strip(): + return i + return last_any + + def _ensure_last_assistant_message_in_tail( + self, + messages: List[Dict[str, Any]], + cut_idx: int, + head_end: int, + ) -> int: + """Guarantee the most recent assistant message is in the protected tail. + + WebUI / TUI / SessionsPage bug (#29824). Without this anchor, + ``_find_tail_cut_by_tokens`` can leave the user's most recent + visible assistant response inside the compressed middle region — + especially when the conversation has a single oversized tool + result or a long stretch of tool-call/result pairs after the + last assistant reply. The summariser then rolls that reply up + into the single ``[CONTEXT COMPACTION — REFERENCE ONLY]`` block + persisted as ``role="user"`` or ``role="assistant"``. From the + operator's perspective the WebUI session viewer + (``web/src/pages/SessionsPage.tsx``) and the TUI chat panel + both suddenly show the opaque "Context compaction" block in the + slot where they were just reading the assistant's actual reply: + + User: "i cant see the output of the last message you + sent, i did see it previously, however now see + 'context compaction'" + + Mirror of ``_ensure_last_user_message_in_tail`` but anchors on + the last assistant-role message. Re-runs the tool-group + alignment so we don't split a ``tool_call`` / ``tool_result`` + group that immediately precedes the anchored message — orphaned + tool messages would otherwise be removed by + ``_sanitize_tool_pairs`` and trigger the same data-loss symptom + we're trying to prevent. + """ + last_asst_idx = self._find_last_assistant_message_idx(messages, head_end) + if last_asst_idx < 0: + # No assistant message in the compressible region — nothing + # to anchor (single-turn pre-reply state, etc.). + return cut_idx + if last_asst_idx >= cut_idx: + # Already in the tail — the token-budget walk did the right + # thing on its own. + return cut_idx + # Pull cut_idx back to the assistant message, then re-align so + # we don't split a tool group that immediately precedes it + # (e.g. an ``assistant(tool_calls)`` → ``tool(result)`` → + # ``assistant(final reply)`` sequence would otherwise leave the + # ``tool`` orphan when cut lands at the final reply). + new_cut = self._align_boundary_backward(messages, last_asst_idx) + if not self.quiet_mode: + logger.debug( + "Anchoring tail cut to last assistant message at index %d " + "(was %d, aligned to %d) to keep the previously-visible " + "reply out of the compaction summary (#29824)", + last_asst_idx, cut_idx, new_cut, + ) + # Safety: never go back into the head region. + return max(new_cut, head_end + 1) + def _ensure_last_user_message_in_tail( self, messages: List[Dict[str, Any]], @@ -1753,7 +1976,7 @@ The user has requested that this compaction PRIORITISE preserving all informatio Context compressor bug (#10896): ``_align_boundary_backward`` can pull ``cut_idx`` past a user message when it tries to keep tool_call/result groups together. If the last user message ends up in the *compressed* - middle region the LLM summariser writes it into "Pending User Asks", + middle region the LLM summariser writes it into "Historical Pending User Asks", but ``SUMMARY_PREFIX`` tells the next model to respond only to user messages *after* the summary — so the task effectively disappears from the active context, causing the agent to stall, repeat completed work, @@ -1800,11 +2023,12 @@ The user has requested that this compaction PRIORITISE preserving all informatio derived from ``summary_target_ratio * context_length``, so it scales automatically with the model's context window. - Token budget is the primary criterion. A hard minimum of 3 messages - is always protected, but the budget is allowed to exceed by up to - 1.5x to avoid cutting inside an oversized message (tool output, file - read, etc.). If even the minimum 3 messages exceed 1.5x the budget - the cut is placed right after the head so compression still runs. + Token budget is the primary criterion. A bounded message-count floor + keeps a short run of recent turns verbatim even when the budget is + exhausted, but the budget is allowed to exceed by up to 1.5x to avoid + cutting inside an oversized message (tool output, file read, etc.). If + even that floor exceeds 1.5x the budget, the cut is placed right after + the head so compression still runs. Never cuts inside a tool_call/result group. Always ensures the most recent user message is in the tail (see ``_ensure_last_user_message_in_tail``). @@ -1812,8 +2036,19 @@ The user has requested that this compaction PRIORITISE preserving all informatio if token_budget is None: token_budget = self.tail_token_budget n = len(messages) - # Hard minimum: always keep at least 3 messages in the tail - min_tail = min(3, n - head_end - 1) if n - head_end > 1 else 0 + # Hard minimum: always keep a bounded recent-message floor in the tail. + # ``protect_last_n`` remains a minimum up to the cap; the cap avoids + # preserving a whole run of bulky tool outputs on every compaction. + available_tail = max(0, n - head_end - 1) + min_tail_floor = max(3, min(self.protect_last_n, _MAX_TAIL_MESSAGE_FLOOR)) + # Leave at least two non-head messages available to summarize on short + # transcripts; otherwise compression can replace a tiny middle with a + # summary and save no messages at all. + compressible_tail_cap = max(3, available_tail - 2) + min_tail = ( + min(min_tail_floor, compressible_tail_cap, available_tail) + if available_tail > 1 else 0 + ) soft_ceiling = int(token_budget * 1.5) accumulated = 0 cut_idx = n # start from beyond the end @@ -1885,6 +2120,13 @@ The user has requested that this compaction PRIORITISE preserving all informatio # active task is never lost to compression (fixes #10896). cut_idx = self._ensure_last_user_message_in_tail(messages, cut_idx, head_end) + # Ensure the most recent assistant message is always in the tail + # so the previously-visible reply isn't silently rolled into the + # ``[CONTEXT COMPACTION — REFERENCE ONLY]`` block (fixes #29824). + # Each anchor only walks ``cut_idx`` backward, so chaining them is + # monotonic — the tail can only grow, never shrink. + cut_idx = self._ensure_last_assistant_message_in_tail(messages, cut_idx, head_end) + return max(cut_idx, head_end + 1) # ------------------------------------------------------------------ @@ -2037,7 +2279,8 @@ The user has requested that this compaction PRIORITISE preserving all informatio ) # Phase 3: Generate structured summary - summary = self._generate_summary(turns_to_summarize, focus_topic=focus_topic) + summary_focus_topic = focus_topic or self._derive_auto_focus_topic(messages) + summary = self._generate_summary(turns_to_summarize, focus_topic=summary_focus_topic) # If summary generation failed, behavior splits on # ``abort_on_summary_failure`` (config: compression.abort_on_summary_failure): @@ -2117,32 +2360,33 @@ The user has requested that this compaction PRIORITISE preserving all informatio # When the summary lands as a standalone role="user" message, # weak models read the verbatim "## Active Task" quote of a past - # user request as fresh input (#11475, #14521). Append the explicit - # end marker — the same one used in the merge-into-tail path — so - # the model has a clear "summary above, not new input" signal. - if not _merge_summary_into_tail and summary_role == "user": - summary = ( - summary - + "\n\n--- END OF CONTEXT SUMMARY — " - "respond to the message below, not the summary above ---" - ) + # user request as fresh input (#11475, #14521). + # When it lands as role="assistant", models may regurgitate the + # summary text as their own output (#33256). In both cases, append + # the explicit end marker so the model has a clear "summary ends + # here, respond to the message below" signal. + if not _merge_summary_into_tail: + summary = summary + "\n\n" + _SUMMARY_END_MARKER if not _merge_summary_into_tail: - compressed.append({"role": summary_role, "content": summary}) + compressed.append({ + "role": summary_role, + "content": summary, + COMPRESSED_SUMMARY_METADATA_KEY: True, + }) for i in range(compress_end, n_messages): msg = messages[i].copy() if _merge_summary_into_tail and i == compress_end: - merged_prefix = ( - summary - + "\n\n--- END OF CONTEXT SUMMARY — " - "respond to the message below, not the summary above ---\n\n" - ) + merged_prefix = summary + "\n\n" + _SUMMARY_END_MARKER + "\n\n" msg["content"] = _append_text_to_content( msg.get("content"), merged_prefix, prepend=True, ) + # Mark the merged message so frontends can identify it as + # containing a compression summary prefix. + msg[COMPRESSED_SUMMARY_METADATA_KEY] = True _merge_summary_into_tail = False compressed.append(msg) diff --git a/agent/conversation_compression.py b/agent/conversation_compression.py index 913c0e25d91..d5469a1b344 100644 --- a/agent/conversation_compression.py +++ b/agent/conversation_compression.py @@ -40,6 +40,16 @@ from agent.model_metadata import estimate_request_tokens_rough logger = logging.getLogger(__name__) +# Stable marker the gateway matches on to re-tag the auto-compaction lifecycle +# status as ``kind="compacting"`` (tui_gateway/server.py::_status_update), so +# drivers like the desktop app can show an explicit "Summarizing…" indicator +# instead of the transcript appearing to silently reset. Keep the marker phrase +# intact if you reword COMPACTION_STATUS. +COMPACTION_STATUS_MARKER = "Compacting context" +COMPACTION_STATUS = ( + f"🗜️ {COMPACTION_STATUS_MARKER} — summarizing earlier conversation so I can continue..." +) + def _compression_lock_holder(agent: Any) -> str: """Build a unique holder id for the lock: pid:tid:agent-instance:uuid. @@ -324,9 +334,7 @@ def compress_context( f"{approx_tokens:,}" if approx_tokens else "unknown", agent.model, focus_topic, ) - agent._emit_status( - "🗜️ Compacting context — summarizing earlier conversation so I can continue..." - ) + agent._emit_status(COMPACTION_STATUS) # ── Compression lock ──────────────────────────────────────────────── # Atomic, state.db-backed lock per session_id. Without this, two @@ -631,7 +639,11 @@ def compress_context( return compressed, new_system_prompt -def try_shrink_image_parts_in_messages(api_messages: list) -> bool: +def try_shrink_image_parts_in_messages( + api_messages: list, + *, + max_dimension: int = 8000, +) -> bool: """Re-encode all native image parts at a smaller size to recover from image-too-large errors (Anthropic 5 MB, unknown other providers). @@ -642,7 +654,8 @@ def try_shrink_image_parts_in_messages(api_messages: list) -> bool: Strategy: look for ``image_url`` / ``input_image`` parts carrying a ``data:image/...;base64,...`` payload. For each one whose encoded size exceeds 4 MB (a safe target that slides under Anthropic's 5 MB - ceiling with header overhead), write the base64 to a tempfile, call + ceiling with header overhead) or whose longest side exceeds + ``max_dimension``, write the base64 to a tempfile, call ``vision_tools._resize_image_for_vision`` to produce a smaller data URL, and substitute it in place. @@ -664,10 +677,9 @@ def try_shrink_image_parts_in_messages(api_messages: list) -> bool: # after a confirmed provider rejection, so the alternative is failure. target_bytes = 4 * 1024 * 1024 # Anthropic enforces an 8000px per-side dimension cap independently of - # the 5 MB byte cap. A tall screenshot can be well under 5 MB yet far - # over 8000px (e.g. 1200×12000 at 0.06 MB). We check pixel dimensions - # even when the byte budget is fine. - max_dimension = 8000 + # the 5 MB byte cap. In many-image requests, the provider can report a + # lower cap (observed: 2000px). The caller passes that parsed ceiling + # when the rejection includes it. changed_count = 0 # Track parts that are over the target but could NOT be shrunk under it. # If any survive, retrying is pointless — the same oversized payload will @@ -684,9 +696,9 @@ def try_shrink_image_parts_in_messages(api_messages: list) -> bool: # Check both byte size AND pixel dimensions. needs_shrink = len(url) > target_bytes # over byte budget if not needs_shrink: - # Even if bytes are fine, check pixel dimensions against - # Anthropic's 8000px cap. A tall image can be tiny in bytes - # yet huge in pixels. + # Even if bytes are fine, check pixel dimensions against the + # provider's reported per-side cap. A screenshot can be tiny in + # bytes yet too large in pixels. try: import base64 as _b64_dim header_d, _, data_d = url.partition(",") @@ -795,6 +807,8 @@ def try_shrink_image_parts_in_messages(api_messages: list) -> bool: __all__ = [ + "COMPACTION_STATUS", + "COMPACTION_STATUS_MARKER", "check_compression_model_feasibility", "replay_compression_warning", "compress_context", diff --git a/agent/conversation_loop.py b/agent/conversation_loop.py index 73bed6b0670..379a038a9e0 100644 --- a/agent/conversation_loop.py +++ b/agent/conversation_loop.py @@ -71,6 +71,35 @@ logger = logging.getLogger(__name__) INTERRUPT_WAITING_FOR_MODEL_PREFIX = "Operation interrupted: waiting for model response (" +def _image_error_max_dimension(error: Exception) -> Optional[int]: + """Extract a provider-reported image dimension ceiling, if present.""" + parts = [] + for value in ( + error, + getattr(error, "message", None), + getattr(error, "body", None), + ): + if value: + try: + parts.append(str(value)) + except Exception: + pass + text = " ".join(parts).lower() + if "image" not in text or "dimension" not in text or "max allowed size" not in text: + return None + + match = re.search(r"max allowed size(?:\s+for [^:]+)?:\s*(\d{3,5})\s*pixels?", text) + if not match: + return None + try: + max_dimension = int(match.group(1)) + except ValueError: + return None + if 512 <= max_dimension <= 8000: + return max_dimension + return None + + def _ollama_context_limit_error(agent: Any, request_tokens: int) -> Optional[str]: """Return a user-facing error when Ollama is loaded with too little context.""" if not getattr(agent, "tools", None): @@ -368,6 +397,42 @@ def _get_continuation_prompt(is_partial_stub: bool, dropped_tools: Optional[List ) +# Shared recovery hint appended to every content-policy refusal message. Both +# the HTTP-200 refusal path (``finish_reason=content_filter``) and the +# exception path (a provider moderation error classified as +# ``content_policy_blocked``) end with the same actionable next steps, so they +# share one trailer to keep the guidance from drifting between the two sites. +_CONTENT_POLICY_RECOVERY_HINT = ( + "Try rephrasing the request, narrowing the context, or " + "adding a fallback provider with `hermes fallback add`." +) + + +def _content_policy_blocked_result( + messages: List[Dict], + api_call_count: int, + *, + final_response: str, + error_detail: str, +) -> Dict[str, Any]: + """Build the terminal turn result for a content-policy block. + + A content-policy refusal is deterministic for the unchanged prompt, so the + turn ends here (no retry). Both the HTTP-200 refusal handler and the + exception-path handler return the identical shape — a failed, non-completed + turn carrying the user-facing message and a ``content_policy_blocked:`` + prefixed error — so they funnel through this one builder. + """ + return { + "final_response": final_response, + "messages": messages, + "api_calls": api_call_count, + "completed": False, + "failed": True, + "error": f"content_policy_blocked: {error_detail}", + } + + def run_conversation( agent, user_message: str, @@ -595,7 +660,11 @@ def run_conversation( # landed after an orphan tool result). Most providers return # empty content on malformed sequences, which would otherwise # retrigger the empty-retry loop indefinitely. - repaired_seq = agent._repair_message_sequence(messages) + # repair_message_sequence_with_cursor also recomputes the SessionDB + # flush cursor (_last_flushed_db_idx) when repair compacts the list, + # so the turn-end flush doesn't skip the assistant/tool chain (#44837). + from agent.agent_runtime_helpers import repair_message_sequence_with_cursor + repaired_seq = repair_message_sequence_with_cursor(agent, messages) if repaired_seq > 0: request_logger.info( "Repaired %s message-alternation violations before request (session=%s)", @@ -703,7 +772,10 @@ def run_conversation( # a thinking-only turn. Runs on the per-call copy only — the # stored conversation history keeps the reasoning block for the # UI transcript and session persistence. - api_messages = agent._drop_thinking_only_and_merge_users(api_messages) + api_messages = agent._drop_thinking_only_and_merge_users( + api_messages, + drop_codex_reasoning_items=agent.api_mode != "codex_responses", + ) # Normalize message whitespace and tool-call JSON for consistent # prefix matching. Ensures bit-perfect prefixes across turns, @@ -1312,6 +1384,106 @@ def run_conversation( ) finish_reason = "length" + # ── Content-policy refusal (HTTP 200) ────────────────── + # The model — or the provider's safety system — returned a + # *successful* response whose stop/finish reason is a refusal: + # Anthropic ``stop_reason="refusal"`` → ``content_filter``; + # OpenAI / portal ``finish_reason="content_filter"`` or a + # populated ``message.refusal`` (mapped in the chat_completions + # transport); Bedrock ``guardrail_intervened``. The content is + # typically empty, so without this branch the response falls + # through to the empty-response / invalid-response retry loops + # and is mis-surfaced as "rate limited" / "no content after + # retries" — burning paid attempts reproducing a deterministic + # refusal. Surface it clearly and stop. Mirrors the + # exception-based ``content_policy_blocked`` recovery: try a + # configured fallback once, otherwise return the refusal. + if finish_reason == "content_filter": + _refusal_transport = agent._get_transport() + if agent.api_mode == "anthropic_messages": + _refusal_result = _refusal_transport.normalize_response( + response, strip_tool_prefix=agent._is_anthropic_oauth + ) + else: + _refusal_result = _refusal_transport.normalize_response(response) + _refusal_text = (getattr(_refusal_result, "content", None) or "").strip() + # Some refusals carry the explanation only in the reasoning + # channel; fall back to it so the user sees *something*. + if not _refusal_text: + _refusal_text = (agent._extract_reasoning(_refusal_result) or "").strip() + + agent._invoke_api_request_error_hook( + task_id=effective_task_id, + turn_id=turn_id, + api_request_id=api_request_id, + api_call_count=api_call_count, + api_start_time=api_start_time, + api_kwargs=api_kwargs, + error_type="ContentPolicyBlocked", + error_message=_refusal_text or "model declined to respond (content_filter)", + status_code=None, + retry_count=retry_count, + max_retries=max_retries, + retryable=False, + reason=FailoverReason.content_policy_blocked.value, + ) + + if thinking_spinner: + thinking_spinner.stop("") + thinking_spinner = None + if agent.thinking_callback: + agent.thinking_callback("") + + # Deterministic for the unchanged prompt — never retry. + # Try a configured fallback once (a different model may not + # refuse); otherwise surface the refusal terminally. + if agent._has_pending_fallback(): + agent._buffer_status( + "⚠️ Model declined to respond (safety refusal) — trying fallback..." + ) + if agent._try_activate_fallback(): + retry_count = 0 + compression_attempts = 0 + _retry.primary_recovery_attempted = False + continue + + agent._flush_status_buffer() + _refusal_log = ( + _refusal_text[:500] + "..." + if len(_refusal_text) > 500 + else _refusal_text + ) + logger.warning( + "%sModel declined to respond (finish_reason=content_filter). " + "model=%s provider=%s refusal=%s", + agent.log_prefix, agent.model, agent.provider, + _refusal_log or "(no text)", + ) + agent._emit_status( + "⚠️ The model declined to respond to this request (safety refusal)." + ) + + _refusal_detail = ( + f"Model's explanation: {_refusal_text}" + if _refusal_text + else "The model returned no explanation." + ) + _refusal_response = ( + "⚠️ The model declined to respond to this request " + "(safety refusal — not a Hermes/gateway failure).\n\n" + f"{_refusal_detail}\n\n" + f"{_CONTENT_POLICY_RECOVERY_HINT}" + ) + + agent._cleanup_task_resources(effective_task_id) + agent._persist_session(messages, conversation_history) + return _content_policy_blocked_result( + messages, + api_call_count, + final_response=_refusal_response, + error_detail=_refusal_text or "model declined (content_filter)", + ) + if finish_reason == "length": if getattr(response, "id", "") == PARTIAL_STREAM_STUB_ID: agent._vprint( @@ -2063,7 +2235,11 @@ def run_conversation( and not _retry.image_shrink_retry_attempted ): _retry.image_shrink_retry_attempted = True - if agent._try_shrink_image_parts_in_messages(api_messages): + image_max_dimension = _image_error_max_dimension(api_error) or 8000 + if agent._try_shrink_image_parts_in_messages( + api_messages, + max_dimension=image_max_dimension, + ): agent._vprint( f"{agent.log_prefix}📐 Image(s) exceeded provider size limit — " f"shrank and retrying...", @@ -2221,30 +2397,54 @@ def run_conversation( print(f"{agent.log_prefix} • Legacy cleanup: hermes config set ANTHROPIC_TOKEN \"\"") print(f"{agent.log_prefix} • Clear stale keys: hermes config set ANTHROPIC_API_KEY \"\"") - # ── Thinking block signature recovery ───────────────── + # Thinking block signature recovery. + # # Anthropic signs thinking blocks against the full turn - # content. Any upstream mutation (context compression, + # content. Any upstream mutation (context compression, # session truncation, message merging) invalidates the - # signature → HTTP 400. Recovery: strip reasoning_details - # from all messages so the next retry sends no thinking - # blocks at all. One-shot — don't retry infinitely. + # signature and the API replies HTTP 400 ("invalid + # signature" or "cannot be modified"). Recovery strips + # ``reasoning_details`` so the retry sends no thinking + # blocks at all. One-shot per outer loop. + # + # The strip targets ``api_messages``, which is the + # API-call-time list that ``_build_api_kwargs`` consumes + # on every retry. ``api_messages`` was populated once at + # the start of the turn from shallow copies of + # ``messages``, so mutating it does not touch the + # canonical store. The previous implementation popped + # ``reasoning_details`` from ``messages`` instead, which + # had two problems: ``api_messages`` carried its own + # reference to the field through the shallow copy, so the + # retry's wire payload still included thinking blocks and + # the recovery never reached the API; and the mutation + # persisted into ``state.db`` through any subsequent + # ``_persist_session`` call, permanently corrupting the + # conversation. Future turns would replay the stripped + # state, hit the same 400, and the agent would terminate + # with ``max_retries_exhausted``, often spawning + # cascading compaction-ended sessions chained off the + # corrupted parent. if ( classified.reason == FailoverReason.thinking_signature and not _retry.thinking_sig_retry_attempted ): _retry.thinking_sig_retry_attempted = True - for _m in messages: - if isinstance(_m, dict): + _api_stripped = 0 + for _m in api_messages: + if isinstance(_m, dict) and "reasoning_details" in _m: _m.pop("reasoning_details", None) + _api_stripped += 1 agent._vprint( - f"{agent.log_prefix}⚠️ Thinking block signature invalid — " - f"stripped all thinking blocks, retrying...", + f"{agent.log_prefix}⚠️ Thinking block signature invalid, " + f"stripped reasoning_details from api_messages for retry...", force=True, ) logger.warning( "%sThinking block signature recovery: stripped " - "reasoning_details from %d messages", - agent.log_prefix, len(messages), + "reasoning_details from %d api_messages " + "(canonical messages unchanged)", + agent.log_prefix, _api_stripped, ) continue @@ -2607,10 +2807,13 @@ def run_conversation( except Exception: pass if _genuine_nous_rate_limit: - # Skip straight to max_retries -- the - # top-of-loop guard will handle fallback or - # bail cleanly. - retry_count = max_retries + # Re-enter the loop exactly once so the + # top-of-loop Nous guard handles fallback or + # bails cleanly. (Setting retry_count to + # max_retries would make the while condition + # false immediately and the guard would never + # run -- no fallback, generic exhaustion error.) + retry_count = max(0, max_retries - 1) continue # Upstream capacity 429: fall through to normal # retry logic. A different model (or the same @@ -3052,20 +3255,17 @@ def run_conversation( if classified.reason == FailoverReason.content_policy_blocked: _summary = agent._summarize_api_error(api_error) _policy_response = ( - f"⚠️ The model provider's safety filter blocked this request " - f"(not a Hermes/gateway failure).\n\n" + "⚠️ The model provider's safety filter blocked this request " + "(not a Hermes/gateway failure).\n\n" f"Provider message: {_summary}\n\n" - f"Try rephrasing the request, narrowing the context, or " - f"adding a fallback provider with `hermes fallback add`." + f"{_CONTENT_POLICY_RECOVERY_HINT}" + ) + return _content_policy_blocked_result( + messages, + api_call_count, + final_response=_policy_response, + error_detail=_summary, ) - return { - "final_response": _policy_response, - "messages": messages, - "api_calls": api_call_count, - "completed": False, - "failed": True, - "error": f"content_policy_blocked: {_summary}", - } return { "final_response": None, "messages": messages, diff --git a/agent/copilot_acp_client.py b/agent/copilot_acp_client.py index b24ddbef5da..e3c03938af4 100644 --- a/agent/copilot_acp_client.py +++ b/agent/copilot_acp_client.py @@ -70,16 +70,6 @@ def _resolve_args() -> list[str]: def _resolve_home_dir() -> str: """Return a stable HOME for child ACP processes.""" - - try: - from hermes_constants import get_subprocess_home - - profile_home = get_subprocess_home() - if profile_home: - return profile_home - except Exception: - pass - home = os.environ.get("HOME", "").strip() if home: return home @@ -105,7 +95,10 @@ def _resolve_home_dir() -> str: def _build_subprocess_env() -> dict[str, str]: env = os.environ.copy() - env["HOME"] = _resolve_home_dir() + home = _resolve_home_dir() + env["HOME"] = home + from hermes_constants import apply_subprocess_home_env + apply_subprocess_home_env(env) return env diff --git a/agent/credits_tracker.py b/agent/credits_tracker.py index 79d05dbb196..19e5b1582da 100644 --- a/agent/credits_tracker.py +++ b/agent/credits_tracker.py @@ -194,17 +194,71 @@ class AgentNotice: id: Optional[str] = None +# ── is_free_tier_model (local-data-only free-model check) ──────────────────── + + +def is_free_tier_model(model: str, base_url: str = "") -> bool: + """Return True when *model* is a Nous free-tier model, using ONLY local data. + + Two signals, both zero-network: + + 1. The ``:free`` suffix — the canonical Nous free SKU marker (e.g. + ``nvidia/nemotron-3-ultra:free``). Free by construction on the API side + (spend is forced to 0 for ``:free`` ids). + 2. A peek into the in-process pricing cache in ``hermes_cli.models`` + (populated when the model picker fetched ``/v1/models`` pricing for + *base_url*). PEEK ONLY — a cache miss never triggers a fetch. This is + CLI/TUI-session best-effort: gateway sessions never run the picker's + pricing fetch, so suppression there rests entirely on the ``:free`` + suffix (which all Nous free SKUs carry). + + Fail-open to False (the depleted notice still shows) on any error: wrongly + showing the warning is recoverable noise; wrongly hiding it on a paid model + would mask a real billing block. + """ + if not model: + return False + if model.endswith(":free"): + return True + if not base_url: + return False + try: + from hermes_cli.models import _is_model_free, _pricing_cache + + # Mirror get_pricing_for_provider's key normalization: the agent's + # Nous base_url is /v1-suffixed (https://inference-api.nousresearch.com/v1) + # but the picker keys _pricing_cache on the pre-/v1 root. + key = base_url.rstrip("/") + if key.endswith("/v1"): + key = key[:-3].rstrip("/") + pricing = _pricing_cache.get(key) + if not pricing: + return False + return _is_model_free(model, pricing) + except Exception: + return False + + # ── evaluate_credits_notices (pure reconciliation function) ────────────────── def evaluate_credits_notices( state: CreditsState, latch: dict, + *, + model_is_free: bool = False, ) -> tuple[list[AgentNotice], list[str]]: """Reconcile credits notices against the latch. Mutates ``latch`` IN PLACE. latch = {"active": set[str], "seen_below_90": bool, "usage_band": Optional[int]}. + ``model_is_free``: True when the session's active model is a Nous free-tier + model (see :func:`is_free_tier_model`). Suppresses the ``credits.depleted`` + notice — a depleted account on a free model can keep inferencing, so the + error banner is noise (and confuses free-tier users who never had credits). + Suppression does NOT emit the "restored" success notice; that fires only on + a genuine ``paid_access`` flip back to True. + Returns ``(to_show: list[AgentNotice], to_clear: list[str])``. Caller emits to_clear FIRST, then to_show. @@ -232,6 +286,16 @@ def evaluate_credits_notices( for band in CREDITS_USAGE_BANDS: # ascending → last match wins = highest if uf >= band[0]: current_band = band + # Top-up suppression: when the account holds purchased (top-up) credits, + # the subscription-cap gauge is the wrong denominator — warning "90% used" + # at a user sitting on $50 of top-up is noise (and it previously stuck + # PERMANENTLY alongside grant_spent at >=100%). Suppress the usage band + # entirely; the cap-reached case is covered by the grant_spent info notice + # below, which already names the remaining top-up balance. A top-up landing + # mid-session flips current_band → None and the clear path below removes + # any showing band line. + if state.purchased_micros > 0: + current_band = None grant_cond = ( state.denominator_kind == "subscription_cap" and uf is not None @@ -284,10 +348,14 @@ def evaluate_credits_notices( active.discard("credits.grant_spent") # ── depleted ───────────────────────────────────────────────────────────── - if depleted_cond and "credits.depleted" not in active: + # Suppressed while the active model is free: inference still works there, + # so the error banner would just alarm users (free-tier users especially, + # who never had paid credits to "lose"). + show_depleted = depleted_cond and not model_is_free + if show_depleted and "credits.depleted" not in active: to_show.append( AgentNotice( - text="✕ Credit access paused · run /usage for balance", + text="✕ Credit access paused · run /credits to top up", level="error", kind=CREDITS_NOTICE_KIND, key="credits.depleted", @@ -295,20 +363,23 @@ def evaluate_credits_notices( ) ) active.add("credits.depleted") - elif "credits.depleted" in active and not depleted_cond: + elif "credits.depleted" in active and not show_depleted: to_clear.append("credits.depleted") active.discard("credits.depleted") - # Recovery: also emit the success notice - to_show.append( - AgentNotice( - text="✓ Credit access restored", - level="success", - kind="ttl", - ttl_ms=CREDITS_RESTORED_TTL_MS, - key="credits.restored", - id="credits.restored", + if not depleted_cond: + # Genuine recovery (paid_access flipped back True): also emit the + # success notice. A clear caused by switching to a free model while + # still depleted must NOT claim access was restored. + to_show.append( + AgentNotice( + text="✓ Credit access restored", + level="success", + kind="ttl", + ttl_ms=CREDITS_RESTORED_TTL_MS, + key="credits.restored", + id="credits.restored", + ) ) - ) return (to_show, to_clear) diff --git a/agent/display.py b/agent/display.py index 8514279888e..84c8509faed 100644 --- a/agent/display.py +++ b/agent/display.py @@ -858,6 +858,20 @@ def _detect_tool_failure(tool_name: str, result: str | None) -> tuple[bool, str] return False, "" +def _used_free_parallel(result: str | None) -> bool: + """True when a web result came from Parallel's free Search MCP. + + Only the keyless Parallel path tags its result with ``provider="parallel"``; + the paid REST path and every other provider omit it. Used to label the tool + line "Parallel search" / "Parallel fetch" exactly when the free MCP served + the call. + """ + if not isinstance(result, str) or '"provider"' not in result: + return False + data = safe_json_loads(result) + return isinstance(data, dict) and str(data.get("provider", "")).lower() == "parallel" + + def get_cute_tool_message( tool_name: str, args: dict, duration: float, result: str | None = None, ) -> str: @@ -895,15 +909,17 @@ def get_cute_tool_message( return f"{line}{failure_suffix}" if tool_name == "web_search": - return _wrap(f"┊ 🔍 search {_trunc(args.get('query', ''), 42)} {dur}") + verb = "Parallel search" if _used_free_parallel(result) else "search" + return _wrap(f"┊ 🔍 {verb:<9} {_trunc(args.get('query', ''), 42)} {dur}") if tool_name == "web_extract": + verb = "Parallel fetch" if _used_free_parallel(result) else "fetch" urls = args.get("urls", []) if urls: url = urls[0] if isinstance(urls, list) else str(urls) domain = url.replace("https://", "").replace("http://", "").split("/")[0] extra = f" +{len(urls)-1}" if len(urls) > 1 else "" - return _wrap(f"┊ 📄 fetch {_trunc(domain, 35)}{extra} {dur}") - return _wrap(f"┊ 📄 fetch pages {dur}") + return _wrap(f"┊ 📄 {verb:<9} {_trunc(domain, 35)}{extra} {dur}") + return _wrap(f"┊ 📄 {verb:<9} pages {dur}") if tool_name == "terminal": return _wrap(f"┊ 💻 $ {_trunc(args.get('command', ''), 42)} {dur}") if tool_name == "process": diff --git a/agent/error_classifier.py b/agent/error_classifier.py index a2045b5f8cd..c39c24a6a5d 100644 --- a/agent/error_classifier.py +++ b/agent/error_classifier.py @@ -549,14 +549,32 @@ def classify_api_error( should_fallback=True, ) - # Anthropic thinking block signature invalid (400). + # Anthropic thinking block recovery (400). Two distinct failure modes, + # same recovery (strip all reasoning_details and retry without thinking + # blocks — see the thinking_signature handler in conversation_loop.py): + # 1. Signature mismatch: a thinking block is signed against the full + # turn content; any upstream mutation (context compression, session + # truncation, message merging) invalidates the signature. + # Pattern: "signature" + "thinking". + # 2. Frozen-block mutation: Anthropic rejects any change to the + # thinking/redacted_thinking blocks in the *latest* assistant + # message — "`thinking` or `redacted_thinking` blocks in the latest + # assistant message cannot be modified. These blocks must remain as + # they were in the original response." This carries no "signature" + # token, so the original pattern missed it and the turn hard-aborted + # as a non-retryable client error instead of self-healing. + # Pattern: "thinking" + ("cannot be modified" | "must remain as they were"). # Don't gate on provider — OpenRouter proxies Anthropic errors, so the # provider may be "openrouter" even though the error is Anthropic-specific. - # The message pattern ("signature" + "thinking") is unique enough. + # The combined patterns are unique enough. if ( status_code == 400 - and "signature" in error_msg and "thinking" in error_msg + and ( + "signature" in error_msg + or "cannot be modified" in error_msg + or "must remain as they were" in error_msg + ) ): return _result( FailoverReason.thinking_signature, diff --git a/agent/errors.py b/agent/errors.py new file mode 100644 index 00000000000..abedd83d29f --- /dev/null +++ b/agent/errors.py @@ -0,0 +1,3 @@ +class SSLConfigurationError(Exception): + """Raised when SSL/TLS certificate bundle configuration fails.""" + pass diff --git a/agent/file_safety.py b/agent/file_safety.py index e9fa487e834..7a70f964125 100644 --- a/agent/file_safety.py +++ b/agent/file_safety.py @@ -46,11 +46,6 @@ def build_write_denied_paths(home: str) -> set[str]: # Top-level Anthropic PKCE credential store remains sensitive even # when a profile is active; default/non-profile sessions still read it. str(hermes_root / ".anthropic_oauth.json"), - os.path.join(home, ".bashrc"), - os.path.join(home, ".zshrc"), - os.path.join(home, ".profile"), - os.path.join(home, ".bash_profile"), - os.path.join(home, ".zprofile"), os.path.join(home, ".netrc"), os.path.join(home, ".pgpass"), os.path.join(home, ".npmrc"), @@ -104,12 +99,6 @@ def is_write_denied(path: str) -> bool: if resolved.startswith(prefix): return True - # Hermes control-plane files: block both the ACTIVE profile's view - # (hermes_home) AND the global root view. Without the root pass, a - # profile-mode session leaves /auth.json + /config.yaml - # writable — letting a prompt-injected write_file overwrite the global - # files that every profile inherits from (same shape as #15981). - control_file_names = ("auth.json", "config.yaml", "webhook_subscriptions.json") mcp_tokens_dir_name = "mcp-tokens" hermes_dirs = [] @@ -122,12 +111,6 @@ def is_write_denied(path: str) -> bool: continue for base_real in hermes_dirs: - for name in control_file_names: - try: - if resolved == os.path.realpath(os.path.join(base_real, name)): - return True - except Exception: - continue try: mcp_real = os.path.realpath(os.path.join(base_real, mcp_tokens_dir_name)) if resolved == mcp_real or resolved.startswith(mcp_real + os.sep): diff --git a/agent/gemini_native_adapter.py b/agent/gemini_native_adapter.py index a0f8e9df548..a79effebba4 100644 --- a/agent/gemini_native_adapter.py +++ b/agent/gemini_native_adapter.py @@ -41,6 +41,16 @@ DEFAULT_GEMINI_BASE_URL = "https://generativelanguage.googleapis.com/v1beta" GEMINI_DEFAULT_MAX_OUTPUT_TOKENS = 65535 +def bare_gemini_model_id(model: str) -> str: + """Strip Gemini's own provider prefix from an aggregator-style model id.""" + name = (model or "").strip() + lowered = name.lower() + for prefix in ("google/", "gemini/"): + if lowered.startswith(prefix): + return name[len(prefix):].strip() or name + return name + + def is_native_gemini_base_url(base_url: str) -> bool: """Return True when the endpoint speaks Gemini's native REST API.""" normalized = str(base_url or "").strip().rstrip("/").lower() @@ -330,7 +340,7 @@ def _build_gemini_contents(messages: List[Dict[str, Any]]) -> tuple[List[Dict[st system_instruction = None joined_system = "\n".join(part for part in system_text_parts if part).strip() if joined_system: - system_instruction = {"parts": [{"text": joined_system}]} + system_instruction = {"role": "system", "parts": [{"text": joined_system}]} return contents, system_instruction @@ -914,6 +924,7 @@ class GeminiNativeClient: thinking_config=thinking_config, ) + model = bare_gemini_model_id(model) if stream: return self._stream_completion(model=model, request=request, timeout=timeout) diff --git a/agent/memory_manager.py b/agent/memory_manager.py index 3cb3a734a8f..240595a4eb3 100644 --- a/agent/memory_manager.py +++ b/agent/memory_manager.py @@ -44,6 +44,66 @@ logger = logging.getLogger(__name__) _SYNC_DRAIN_TIMEOUT_S = 5.0 +def memory_provider_tools_enabled(enabled_toolsets: Optional[List[str]]) -> bool: + """Return whether external memory-provider tools should be exposed.""" + if enabled_toolsets is None: + return True + if not enabled_toolsets: + return False + if "memory" in enabled_toolsets: + return True + + try: + from toolsets import resolve_toolset + + return any("memory" in resolve_toolset(name) for name in enabled_toolsets) + except Exception: + logger.debug("Failed to resolve enabled toolsets for memory-provider tools", exc_info=True) + return False + + +def inject_memory_provider_tools(agent: Any) -> int: + """Append external memory-provider tool schemas to an agent tool surface.""" + memory_manager = getattr(agent, "_memory_manager", None) + tools = getattr(agent, "tools", None) + if not memory_manager or tools is None: + return 0 + + existing_tool_names = { + tool.get("function", {}).get("name") + for tool in tools + if isinstance(tool, dict) + } + if ( + "memory" not in existing_tool_names + and not memory_provider_tools_enabled(getattr(agent, "enabled_toolsets", None)) + ): + return 0 + + get_schemas = getattr(memory_manager, "get_all_tool_schemas", None) + if not callable(get_schemas): + return 0 + + valid_tool_names = getattr(agent, "valid_tool_names", None) + if valid_tool_names is None: + valid_tool_names = set() + agent.valid_tool_names = valid_tool_names + + added = 0 + for schema in get_schemas(): + if not isinstance(schema, dict): + continue + tool_name = schema.get("name", "") + if not tool_name or tool_name in existing_tool_names: + continue + tools.append({"type": "function", "function": schema}) + valid_tool_names.add(tool_name) + existing_tool_names.add(tool_name) + added += 1 + + return added + + # --------------------------------------------------------------------------- # Context fencing helpers # --------------------------------------------------------------------------- diff --git a/agent/model_metadata.py b/agent/model_metadata.py index 3a71e974fdb..8cfec23fe1f 100644 --- a/agent/model_metadata.py +++ b/agent/model_metadata.py @@ -5,6 +5,7 @@ and run_agent.py for pre-flight context checks. """ import ipaddress +import json import logging import os import re @@ -16,7 +17,7 @@ from urllib.parse import urlparse import requests import yaml -from utils import base_url_host_matches, base_url_hostname +from utils import atomic_json_write, base_url_host_matches, base_url_hostname from hermes_constants import OPENROUTER_MODELS_URL @@ -111,6 +112,57 @@ _endpoint_model_metadata_cache: Dict[str, Dict[str, Dict[str, Any]]] = {} _endpoint_model_metadata_cache_time: Dict[str, float] = {} _ENDPOINT_MODEL_CACHE_TTL = 300 + +def _get_model_metadata_cache_path() -> Path: + """Return path to the OpenRouter model metadata disk cache.""" + from hermes_constants import get_hermes_home + return get_hermes_home() / "cache" / "openrouter_model_metadata.json" + + +def _model_metadata_disk_cache_age_seconds() -> Optional[float]: + """Return disk-cache age in seconds, or None if freshness is unknown.""" + try: + cache_path = _get_model_metadata_cache_path() + if not cache_path.exists(): + return None + age = time.time() - cache_path.stat().st_mtime + if age < 0: + return None + return age + except Exception: + return None + + +def _load_model_metadata_disk_cache() -> Dict[str, Dict[str, Any]]: + """Load processed OpenRouter metadata cache from disk.""" + try: + cache_path = _get_model_metadata_cache_path() + with cache_path.open("r", encoding="utf-8") as f: + data = json.load(f) + if not isinstance(data, dict): + return {} + return { + str(key): value + for key, value in data.items() + if isinstance(value, dict) + } + except Exception as e: + logger.debug("Failed to load OpenRouter model metadata disk cache: %s", e) + return {} + + +def _save_model_metadata_disk_cache(data: Dict[str, Dict[str, Any]]) -> None: + """Save processed OpenRouter metadata cache to disk atomically.""" + try: + atomic_json_write( + _get_model_metadata_cache_path(), + data, + indent=0, + separators=(",", ":"), + ) + except Exception as e: + logger.debug("Failed to save OpenRouter model metadata disk cache: %s", e) + # Descending tiers for context length probing when the model is unknown. # We start at 256K (covers GPT-5.x, many current large-context models) and # step down on context-length errors until one works. Tier[0] is also the @@ -627,6 +679,15 @@ def fetch_model_metadata(force_refresh: bool = False) -> Dict[str, Dict[str, Any if not force_refresh and _model_metadata_cache and (time.time() - _model_metadata_cache_time) < _MODEL_CACHE_TTL: return _model_metadata_cache + if not force_refresh: + disk_age = _model_metadata_disk_cache_age_seconds() + if disk_age is not None and disk_age < _MODEL_CACHE_TTL: + disk_cache = _load_model_metadata_disk_cache() + if disk_cache: + _model_metadata_cache = disk_cache + _model_metadata_cache_time = time.time() - disk_age + return _model_metadata_cache + try: response = requests.get(OPENROUTER_MODELS_URL, timeout=10, verify=_resolve_requests_verify()) response.raise_for_status() @@ -648,12 +709,24 @@ def fetch_model_metadata(force_refresh: bool = False) -> Dict[str, Dict[str, Any _model_metadata_cache = cache _model_metadata_cache_time = time.time() + _save_model_metadata_disk_cache(cache) logger.debug("Fetched metadata for %s models from OpenRouter", len(cache)) return cache except Exception as e: logger.warning(f"Failed to fetch model metadata from OpenRouter: {e}") - return _model_metadata_cache or {} + if _model_metadata_cache: + return _model_metadata_cache + disk_cache = _load_model_metadata_disk_cache() + if disk_cache: + _model_metadata_cache = disk_cache + disk_age = _model_metadata_disk_cache_age_seconds() + if disk_age is not None: + _model_metadata_cache_time = time.time() - min(disk_age, _MODEL_CACHE_TTL) + else: + _model_metadata_cache_time = time.time() - _MODEL_CACHE_TTL + 1 + return _model_metadata_cache + return {} def fetch_endpoint_model_metadata( diff --git a/agent/moonshot_schema.py b/agent/moonshot_schema.py index f22176f936e..206ccee1653 100644 --- a/agent/moonshot_schema.py +++ b/agent/moonshot_schema.py @@ -135,7 +135,14 @@ def _repair_schema(node: Any, is_schema: bool = True) -> Any: def _fill_missing_type(node: Dict[str, Any]) -> Dict[str, Any]: """Infer a reasonable ``type`` if this schema node has none.""" - if "type" in node and node["type"] not in {None, ""}: + node_type = node.get("type") + if isinstance(node_type, list): + concrete = next( + (t for t in node_type if isinstance(t, str) and t not in {"", "null"}), + "string", + ) + return {**node, "type": concrete} + if "type" in node and node_type not in {None, ""}: return node # Heuristic: presence of ``properties`` → object, ``items`` → array, ``enum`` diff --git a/agent/prompt_builder.py b/agent/prompt_builder.py index cc62c13f9dd..b11cade39bd 100644 --- a/agent/prompt_builder.py +++ b/agent/prompt_builder.py @@ -489,15 +489,41 @@ PLATFORM_HINTS = { "files arrive as downloadable documents. You can also include image " "URLs in markdown format ![alt](url) and they will be sent as photos." ), + "whatsapp_cloud": ( + "You are on a text messaging communication platform, WhatsApp " + "(via Meta's official Business Cloud API). Standard markdown " + "(**bold**, ~~strike~~, # headers, [links](url)) is auto-converted " + "to WhatsApp's native syntax (*bold*, ~strike~, etc.) — feel free " + "to write in markdown. Tables are NOT supported — prefer bullet " + "lists or labeled key:value pairs. " + "You can send media files natively: include MEDIA:/absolute/path/to/file " + "in your response. Images (.jpg, .png) become photo attachments, " + "videos (.mp4) play inline, audio (.mp3, .ogg) sends as voice/audio " + "messages, other files arrive as documents. Image URLs in markdown " + "format ![alt](url) also work. " + "IMPORTANT: this platform has a 24-hour conversation window — if the " + "user hasn't messaged in 24h, free-form replies are refused by Meta " + "(error 131047). This rarely matters for live chat, but is worth " + "knowing if you're scheduling a delayed message." + ), "telegram": ( "You are on a text messaging communication platform, Telegram. " - "Standard markdown is automatically converted to Telegram format. " + "Standard Markdown is automatically converted to Telegram formatting. " "Supported: **bold**, *italic*, ~~strikethrough~~, ||spoiler||, " "`inline code`, ```code blocks```, [links](url), and ## headers. " - "Telegram has NO table syntax — prefer bullet lists or labeled " - "key: value pairs over pipe tables (any tables you do emit are " - "auto-rewritten into row-group bullets, which you can produce " - "directly for cleaner output). " + "Telegram now supports rich Markdown, so lean into it: whenever it " + "makes the answer clearer or easier to scan, actively reach for real " + "Markdown tables (pipe `| col | col |` syntax), bullet and numbered " + "lists, task lists (`- [ ]` / `- [x]`), headings, nested blockquotes, " + "collapsible details, footnotes/references, math/formulas (`$...$`, " + "`$$...$$`), underline, subscript/superscript, marked (highlighted) " + "text, and anchors. Default to structured formatting over dense " + "paragraphs for any comparison, set of steps, key/value summary, or " + "tabular data. Prefer real Markdown tables and task lists over " + "hand-built bullet substitutes when presenting structured data; these " + "degrade gracefully (tables become readable bullet groups) when rich " + "rendering is unavailable, but advanced constructs like math and " + "collapsible details may render as plain source text in that case. " "You can send media files natively: to deliver a file to the user, " "include MEDIA:/absolute/path/to/file in your response. Images " "(.png, .jpg, .webp) appear as photos, audio (.ogg) sends as voice " @@ -1101,11 +1127,12 @@ def _skill_should_show( def build_skills_system_prompt( available_tools: "set[str] | None" = None, available_toolsets: "set[str] | None" = None, + compact_categories: "frozenset[str] | None" = None, ) -> str: """Build a compact skill index for the system prompt. Two-layer cache: - 1. In-process LRU dict keyed by (skills_dir, tools, toolsets) + 1. In-process LRU dict keyed by (skills_dir, tools, toolsets, hidden) 2. Disk snapshot (``.skills_prompt_snapshot.json``) validated by mtime/size manifest — survives process restarts @@ -1115,6 +1142,12 @@ def build_skills_system_prompt( scanned alongside the local ``~/.hermes/skills/`` directory. External dirs are read-only — they appear in the index but new skills are always created in the local dir. Local skills take precedence when names collide. + + ``compact_categories`` (e.g. from the coding posture — see + agent/coding_context.py) demotes whole categories to a names-only line in + the rendered index. Nothing is ever hidden: every skill name stays + visible and loadable via ``skill_view`` / ``skills_list``; only the + descriptions are dropped, and a footer note explains the demotion. """ skills_dir = get_skills_dir() external_dirs = get_all_skills_dirs()[1:] # skip local (index 0) @@ -1131,7 +1164,7 @@ def build_skills_system_prompt( or get_session_env("HERMES_SESSION_PLATFORM") or "" ) - disabled = get_disabled_skill_names() + disabled = get_disabled_skill_names(_platform_hint or None) cache_key = ( str(skills_dir.resolve()), tuple(str(d) for d in external_dirs), @@ -1139,6 +1172,7 @@ def build_skills_system_prompt( tuple(sorted(str(ts) for ts in (available_toolsets or set()))), _platform_hint, tuple(sorted(disabled)), + tuple(sorted(compact_categories or ())), ) with _SKILLS_PROMPT_CACHE_LOCK: cached = _SKILLS_PROMPT_CACHE.get(cache_key) @@ -1272,18 +1306,44 @@ def build_skills_system_prompt( except Exception as e: logger.debug("Could not read external skill description %s: %s", desc_file, e) + # Posture-driven category demotion (e.g. non-coding skills while pairing + # on code). Demoted categories stay in the index as a single names-only + # line — descriptions are dropped to cut noise, but every skill name + # remains visible so memory-anchored recall ("load ") keeps working. + # NEVER remove entries entirely: agent-created skills are the model's + # project memory, and models don't reach for skills_list to rediscover + # what the index stops showing them. Match on the top-level category + # segment so nested categories ("social-media/twitter") are demoted with + # their parent. + demoted = frozenset( + cat for cat in skills_by_category + if cat.split("/", 1)[0] in (compact_categories or frozenset()) + ) + + hidden_note = "" + if demoted: + hidden_note = ( + "\n(Categories marked [names only] are outside the current coding " + "context, so their descriptions are omitted — the skills work " + "normally and load with skill_view(name) as usual.)" + ) + if not skills_by_category: result = "" else: index_lines = [] for category in sorted(skills_by_category.keys()): + # Deduplicate and sort skills within each category + seen = set() + if category in demoted: + names = sorted({name for name, _ in skills_by_category[category]}) + index_lines.append(f" {category} [names only]: {', '.join(names)}") + continue cat_desc = category_descriptions.get(category, "") if cat_desc: index_lines.append(f" {category}: {cat_desc}") else: index_lines.append(f" {category}:") - # Deduplicate and sort skills within each category - seen = set() for name, desc in sorted(skills_by_category[category], key=lambda x: x[0]): if name in seen: continue @@ -1320,6 +1380,7 @@ def build_skills_system_prompt( "\n" "\n" "Only proceed without loading a skill if genuinely none are relevant to the task." + + hidden_note ) # ── Store in LRU cache ──────────────────────────────────────────── @@ -1383,13 +1444,13 @@ def build_nous_subscription_prompt(valid_tool_names: "set[str] | None" = None) - lines = [ "# Nous Subscription", - "Nous subscription includes managed web tools (Firecrawl), image generation (FAL), OpenAI TTS, and browser automation (Browser Use) by default. Modal execution is optional.", + "Nous subscription includes managed web tools (Firecrawl), image generation (FAL), OpenAI TTS, OpenAI Whisper STT, and browser automation (Browser Use) by default. Modal execution is optional.", "Current capability status:", ] lines.extend(_status_line(feature) for feature in features.items()) lines.extend( [ - "When a Nous-managed feature is active, do not ask the user for Firecrawl, FAL, OpenAI TTS, or Browser-Use API keys.", + "When a Nous-managed feature is active, do not ask the user for Firecrawl, FAL, OpenAI TTS, OpenAI Whisper, or Browser-Use API keys.", "If the user is not subscribed and asks for a capability that Nous subscription would unlock or simplify, suggest Nous subscription as one option alongside direct setup or local alternatives.", "Do not mention subscription unless the user asks about it or it directly solves the current missing capability.", "Useful commands: hermes setup, hermes setup tools, hermes setup terminal, hermes status.", diff --git a/agent/skill_utils.py b/agent/skill_utils.py index 62bcc5a2b4b..6f68d3041b5 100644 --- a/agent/skill_utils.py +++ b/agent/skill_utils.py @@ -272,27 +272,65 @@ def skill_matches_environment(frontmatter: Dict[str, Any]) -> bool: # ── Disabled skills ─────────────────────────────────────────────────────── +_RAW_CONFIG_CACHE: Dict[Tuple[str, int, int], Dict[str, Any]] = {} + + +def _raw_config_cache_clear() -> None: + """Test hook — drop the shared raw config cache.""" + _RAW_CONFIG_CACHE.clear() + + +def _load_raw_config() -> Dict[str, Any]: + """Read config.yaml with a shared mtime+size keyed cache. + + This module intentionally avoids importing ``hermes_cli.config`` on the + skill prompt/build path. A tiny local cache gives the same repeated-read + win without pulling the heavier CLI config stack into startup. + """ + config_path = get_config_path() + if not config_path.exists(): + return {} + try: + stat = config_path.stat() + cache_key = (str(config_path), stat.st_mtime_ns, stat.st_size) + except OSError: + cache_key = None + + if cache_key is not None: + cached = _RAW_CONFIG_CACHE.get(cache_key) + if cached is not None: + return cached + + try: + parsed = yaml_load(config_path.read_text(encoding="utf-8")) + except Exception as e: + logger.debug("Could not read skill config %s: %s", config_path, e) + return {} + if not isinstance(parsed, dict): + return {} + + if cache_key is not None: + _RAW_CONFIG_CACHE.clear() + _RAW_CONFIG_CACHE[cache_key] = parsed + return parsed + + def get_disabled_skill_names(platform: str | None = None) -> Set[str]: """Read disabled skill names from config.yaml. Args: platform: Explicit platform name (e.g. ``"telegram"``). When *None*, resolves from ``HERMES_PLATFORM`` or - ``HERMES_SESSION_PLATFORM`` env vars. Falls back to the - global disabled list when no platform is determined. + ``HERMES_SESSION_PLATFORM`` env vars. Returns the global + disabled list, unioned with the platform-specific list when a + platform is resolved (a globally-disabled skill stays disabled + on every platform). Reads the config file directly (no CLI config imports) to stay lightweight. """ - config_path = get_config_path() - if not config_path.exists(): - return set() - try: - parsed = yaml_load(config_path.read_text(encoding="utf-8")) - except Exception as e: - logger.debug("Could not read skill config %s: %s", config_path, e) - return set() - if not isinstance(parsed, dict): + parsed = _load_raw_config() + if not parsed: return set() skills_cfg = parsed.get("skills") @@ -305,13 +343,14 @@ def get_disabled_skill_names(platform: str | None = None) -> Set[str]: or os.getenv("HERMES_PLATFORM") or get_session_env("HERMES_SESSION_PLATFORM") ) + global_disabled = _normalize_string_set(skills_cfg.get("disabled")) if resolved_platform: platform_disabled = (skills_cfg.get("platform_disabled") or {}).get( resolved_platform ) if platform_disabled is not None: - return _normalize_string_set(platform_disabled) - return _normalize_string_set(skills_cfg.get("disabled")) + return global_disabled | _normalize_string_set(platform_disabled) + return global_disabled def _normalize_string_set(values) -> Set[str]: @@ -336,6 +375,7 @@ _EXTERNAL_DIRS_CACHE: Dict[Tuple[str, int], List[Path]] = {} def _external_dirs_cache_clear() -> None: """Test hook — drop the in-process cache.""" _EXTERNAL_DIRS_CACHE.clear() + _raw_config_cache_clear() def get_external_skills_dirs() -> List[Path]: @@ -368,11 +408,8 @@ def get_external_skills_dirs() -> List[Path]: # Return a copy so callers can't mutate the cached list. return list(cached) - try: - parsed = yaml_load(config_path.read_text(encoding="utf-8")) - except Exception: - return [] - if not isinstance(parsed, dict): + parsed = _load_raw_config() + if not parsed: return [] skills_cfg = parsed.get("skills") @@ -584,15 +621,7 @@ def resolve_skill_config_values( current values (or the declared default if the key isn't set). Path values are expanded via ``os.path.expanduser``. """ - config_path = get_config_path() - config: Dict[str, Any] = {} - if config_path.exists(): - try: - parsed = yaml_load(config_path.read_text(encoding="utf-8")) - if isinstance(parsed, dict): - config = parsed - except Exception: - pass + config = _load_raw_config() resolved: Dict[str, Any] = {} for var in config_vars: diff --git a/agent/ssl_guard.py b/agent/ssl_guard.py new file mode 100644 index 00000000000..557f8566c32 --- /dev/null +++ b/agent/ssl_guard.py @@ -0,0 +1,94 @@ +"""Preventive SSL CA certificate checks for Hermes Agent. + +This module catches broken CA bundle paths before OpenAI/httpx turns them into +opaque ``FileNotFoundError: [Errno 2] No such file or directory`` failures. +""" + +from __future__ import annotations + +import logging +import os +import ssl +from pathlib import Path + +from agent.errors import SSLConfigurationError + +logger = logging.getLogger(__name__) + +_CA_BUNDLE_ENV_VARS = ( + "HERMES_CA_BUNDLE", + "SSL_CERT_FILE", + "REQUESTS_CA_BUNDLE", + "CURL_CA_BUNDLE", +) + +_SKIP_VALUES = {"1", "true", "yes", "on"} + + +def _skip_ssl_guard_enabled() -> bool: + return os.getenv("HERMES_SKIP_SSL_GUARD", "").strip().lower() in _SKIP_VALUES + + +def _repair_hint() -> str: + return ( + "Repair: python -m pip install --force-reinstall certifi openai httpx\n" + "If you configured a custom corporate CA bundle, fix or unset the " + "broken CA bundle environment variable." + ) + + +def _ssl_err(message: str) -> SSLConfigurationError: + """Create a consistent, user-actionable SSL configuration error.""" + return SSLConfigurationError(f"{message}\n{_repair_hint()}") + + +def _validate_bundle_path(label: str, value: str, *, require_substantial: bool = False) -> None: + path = Path(value).expanduser() + if not path.exists(): + raise _ssl_err(f"{label} points to a missing CA bundle: {value}") + if not path.is_file(): + raise _ssl_err(f"{label} does not point to a CA bundle file: {value}") + if require_substantial and path.stat().st_size < 1024: + raise _ssl_err(f"{label} at {value} appears corrupted (too small)") + try: + ctx = ssl.create_default_context(cafile=str(path)) + except Exception as exc: + raise _ssl_err(f"{label} CA bundle at {value} cannot be loaded: {exc}") from exc + if not ctx.get_ca_certs(): + raise _ssl_err(f"{label} CA bundle at {value} did not load any certificates") + + +def verify_ca_bundle() -> None: + """Verify configured and bundled CA certificates are present and loadable. + + Raises: + SSLConfigurationError: If an explicit CA-bundle environment variable + points at a bad path, or if certifi's bundled ``cacert.pem`` is + missing/corrupt. + """ + if _skip_ssl_guard_enabled(): + logger.debug("SSL CA bundle guard skipped via HERMES_SKIP_SSL_GUARD") + return + + for env_var in _CA_BUNDLE_ENV_VARS: + value = os.getenv(env_var) + if value: + _validate_bundle_path(env_var, value) + + try: + import certifi + except Exception as exc: + raise _ssl_err(f"certifi is not importable: {exc}") from exc + + ca_bundle = str(certifi.where()) + _validate_bundle_path("certifi", ca_bundle, require_substantial=True) + + +def verify_ca_bundle_with_fallback() -> None: + """Backward-compatible wrapper for older call sites. + + The old PR name mentioned a platform fallback, but allowing startup with a + broken certifi bundle still leaves httpx/OpenAI and requests call sites + failing later. Keep the wrapper name but enforce the same check. + """ + verify_ca_bundle() diff --git a/agent/system_prompt.py b/agent/system_prompt.py index 4038716df48..76f57dfcdbc 100644 --- a/agent/system_prompt.py +++ b/agent/system_prompt.py @@ -191,9 +191,23 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None) ) if toolset } + # Focus mode (opt-in) demotes non-coding skill categories to + # names-only in the index (never hidden — skill_view/skills_list + # reach everything, and every name stays visible for recall). The + # default coding posture leaves the index untouched. + _compact_cats = frozenset() + try: + from agent.coding_context import coding_compact_skill_categories + + _compact_cats = coding_compact_skill_categories( + platform=agent.platform, cwd=resolve_context_cwd() + ) + except Exception: + _compact_cats = frozenset() skills_prompt = _r.build_skills_system_prompt( available_tools=agent.valid_tool_names, available_toolsets=avail_toolsets, + compact_categories=_compact_cats or None, ) else: skills_prompt = "" @@ -221,6 +235,26 @@ def build_system_prompt_parts(agent: Any, system_message: Optional[str] = None) if _env_hints: stable_parts.append(_env_hints) + # Coding posture (base Hermes, any interactive coding surface in a code + # workspace — see agent/coding_context.py). The operating brief + the live + # git/workspace snapshot are built once here and cached for the session; + # the snapshot is never re-probed per turn (that would break the prompt + # cache), so the brief tells the model to re-check git before relying on it. + if agent.valid_tool_names: + try: + from agent.coding_context import coding_system_blocks + + stable_parts.extend( + coding_system_blocks( + platform=agent.platform, + cwd=resolve_context_cwd(), + model=agent.model, + ) + ) + except Exception: + # Coding-context probing must never block prompt build. + pass + # Local Python toolchain probe — names python/pip/uv/PEP-668 state when # something is non-default so the model can pick the right install # strategy without discovering by failure. Emits a single line; emits diff --git a/agent/tool_executor.py b/agent/tool_executor.py index cd24b63f393..144a2929782 100644 --- a/agent/tool_executor.py +++ b/agent/tool_executor.py @@ -417,7 +417,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe # ── Logging / callbacks ────────────────────────────────────────── tool_names_str = ", ".join(name for _, name, _, _, _, _ in parsed_calls) - if not agent.quiet_mode: + if not agent.quiet_mode and getattr(agent, "tool_progress_mode", "all") != "off": print(f" ⚡ Concurrent: {num_tools} tool calls — {tool_names_str}") for i, (tc, name, args, middleware_trace, block_result, blocked_by_guardrail) in enumerate(parsed_calls, 1): args_str = json.dumps(args, ensure_ascii=False) @@ -702,7 +702,7 @@ def execute_tool_calls_concurrent(agent, assistant_message, messages: list, effe if agent._should_emit_quiet_tool_messages(): cute_msg = _get_cute_tool_message_impl(name, args, tool_duration, result=function_result) agent._safe_print(f" {cute_msg}") - elif getattr(agent, "tool_progress_mode", "all") != "off": + elif not agent.quiet_mode and getattr(agent, "tool_progress_mode", "all") != "off": _preview_str = _multimodal_text_summary(function_result) if agent.verbose_logging: print(f" ✅ Tool {i+1} completed in {tool_duration:.2f}s") @@ -866,7 +866,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe elif function_name == "skill_manage": agent._iters_since_skill = 0 - if not agent.quiet_mode: + if not agent.quiet_mode and getattr(agent, "tool_progress_mode", "all") != "off": args_str = json.dumps(function_args, ensure_ascii=False) if agent.verbose_logging: print(f" 📞 Tool {i}: {function_name}({list(function_args.keys())})") @@ -1384,7 +1384,7 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe # entire batch. The model sees it on the next API iteration. agent._apply_pending_steer_to_tool_results(messages, 1) - if not agent.quiet_mode: + if not agent.quiet_mode and getattr(agent, "tool_progress_mode", "all") != "off": if agent.verbose_logging: print(f" ✅ Tool {i} completed in {tool_duration:.2f}s") print(agent._wrap_verbose("Result: ", function_result)) diff --git a/agent/transports/anthropic.py b/agent/transports/anthropic.py index d77ae63ef32..aad49161385 100644 --- a/agent/transports/anthropic.py +++ b/agent/transports/anthropic.py @@ -84,7 +84,7 @@ class AnthropicTransport(ProviderTransport): to OpenAI finish_reason, and collects reasoning_details in provider_data. """ import json - from agent.anthropic_adapter import _to_plain_data + from agent.anthropic_adapter import _to_plain_data, _sanitize_replay_block from agent.transports.types import ToolCall strip_tool_prefix = kwargs.get("strip_tool_prefix", False) @@ -94,14 +94,40 @@ class AnthropicTransport(ProviderTransport): reasoning_parts = [] reasoning_details = [] tool_calls = [] + # Verbatim, order-preserving copy of every content block in the turn. + # Anthropic signs each thinking block against the turn content that + # PRECEDES it at its position; when a turn interleaves thinking and + # tool_use (adaptive/interleaved thinking, Claude 4.6+), the parallel + # reasoning_details + tool_calls lists below lose that cross-type + # ordering. Replaying the latest assistant message in the wrong order + # invalidates the signatures -> HTTP 400 "thinking ... blocks in the + # latest assistant message cannot be modified". Preserve the exact + # block sequence here so the adapter can replay it unchanged. See + # tests/agent/test_anthropic_thinking_block_order.py. + ordered_blocks = [] for block in response.content: + block_dict = _to_plain_data(block) + clean_block = None + if isinstance(block_dict, dict): + # Sanitize at capture so output-only SDK fields (parsed_output, + # caller, citations=None, …) never persist to state.db and leak + # back as request input on replay → HTTP 400 "Extra inputs are + # not permitted". Defence-in-depth with the replay-side sanitize. + clean_block = _sanitize_replay_block(block_dict) + if clean_block is not None: + ordered_blocks.append(clean_block) if block.type == "text": text_parts.append(block.text) - elif block.type == "thinking": - reasoning_parts.append(block.thinking) - block_dict = _to_plain_data(block) - if isinstance(block_dict, dict): + elif block.type in ("thinking", "redacted_thinking"): + if block.type == "thinking": + reasoning_parts.append(block.thinking) + # Use the sanitized block (clean_block) for reasoning_details too, + # since _extract_preserved_thinking_blocks replays these on the + # non-ordered path. Falls back to raw only if sanitize dropped it. + if isinstance(clean_block, dict): + reasoning_details.append(clean_block) + elif isinstance(block_dict, dict): reasoning_details.append(block_dict) elif block.type == "tool_use": name = block.name @@ -130,6 +156,23 @@ class AnthropicTransport(ProviderTransport): provider_data = {} if reasoning_details: provider_data["reasoning_details"] = reasoning_details + # Only worth carrying the ordered-blocks channel when the turn + # actually interleaves signed thinking with tool_use — that's the + # only shape the parallel lists reconstruct incorrectly. A turn that + # is purely text, or thinking-then-tools with a single leading + # thinking block, replays correctly without it. + _has_signed_thinking = any( + isinstance(b, dict) + and b.get("type") in ("thinking", "redacted_thinking") + and (b.get("signature") or b.get("data")) + for b in ordered_blocks + ) + _has_tool_use = any( + isinstance(b, dict) and b.get("type") == "tool_use" + for b in ordered_blocks + ) + if _has_signed_thinking and _has_tool_use: + provider_data["anthropic_content_blocks"] = ordered_blocks return NormalizedResponse( content="\n".join(text_parts) if text_parts else None, @@ -143,10 +186,21 @@ class AnthropicTransport(ProviderTransport): def validate_response(self, response: Any) -> bool: """Check Anthropic response structure is valid. - An empty content list is legitimate when ``stop_reason == "end_turn"`` - — the model's canonical way of signalling "nothing more to add" after - a tool turn that already delivered the user-facing text. Treating it - as invalid falsely retries a completed response. + An empty content list is legitimate for terminal stop reasons that + carry no text payload: + + - ``end_turn`` — the model's canonical "nothing more to add" after a + tool turn that already delivered the user-facing text. + - ``refusal`` — the model declined to respond (Claude 4.5+). The + Messages API returns an empty ``content`` list with this stop + reason. Treating it as invalid sends a deterministic refusal into + the invalid-response retry loop, which reproduces the refusal on + every attempt and surfaces a misleading "rate limited / invalid + response" error instead of the refusal. ``normalize_response`` maps + ``refusal`` → ``content_filter`` so the agent loop's refusal handler + can surface it. + + Treating either as invalid falsely retries a completed response. """ if response is None: return False @@ -154,7 +208,7 @@ class AnthropicTransport(ProviderTransport): if not isinstance(content_blocks, list): return False if not content_blocks: - return getattr(response, "stop_reason", None) == "end_turn" + return getattr(response, "stop_reason", None) in {"end_turn", "refusal"} return True def extract_cache_stats(self, response: Any) -> Optional[Dict[str, int]]: diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index 0c17e309a8b..49bc91f44d2 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -664,8 +664,42 @@ class ChatCompletionsTransport(ProviderTransport): if rd: provider_data["reasoning_details"] = rd + # OpenAI structured-refusal field. When a model declines, the SDK + # populates ``message.refusal`` with the explanation and leaves + # ``content`` empty. OpenAI-compatible proxies that front Anthropic / + # Bedrock (e.g. Nous Portal) surface a Claude refusal this way — or via + # ``finish_reason="content_filter"`` — instead of the native + # ``stop_reason="refusal"``. Without capturing it the refusal looks + # like an empty response, so the agent loop retries a deterministic + # refusal three times and gives up with "no content after retries". + # Promote it to content + a ``content_filter`` finish reason so the + # loop's refusal handler surfaces it clearly and stops. ``refusal`` is + # ``None`` for normal responses, so this is a no-op in the common case. + content = msg.content + refusal = getattr(msg, "refusal", None) + if refusal is None and hasattr(msg, "model_extra"): + _msg_extra = getattr(msg, "model_extra", None) or {} + if isinstance(_msg_extra, dict): + refusal = _msg_extra.get("refusal") + if isinstance(refusal, str) and refusal.strip(): + # Record the refusal explanation regardless — it's useful provider + # metadata even when the model also returned a usable payload. + provider_data["refusal"] = refusal + _has_text = isinstance(content, str) and content.strip() + _has_tool_calls = bool(tool_calls) + # Only promote to a terminal ``content_filter`` when the refusal is + # the *sole* payload — no visible text and no tool calls. A response + # that carries real content (or tool calls) alongside a refusal note + # is a normal, usable turn: surfacing it as a failed safety refusal + # would discard the model's actual work. In the empty-payload case, + # adopt the refusal as content so the loop has something to show. + if not _has_text and not _has_tool_calls: + content = refusal + if finish_reason in (None, "stop"): + finish_reason = "content_filter" + return NormalizedResponse( - content=msg.content, + content=content, tool_calls=tool_calls, finish_reason=finish_reason, reasoning=reasoning, diff --git a/agent/transports/codex.py b/agent/transports/codex.py index ab82f6202f1..1d24ac3355a 100644 --- a/agent/transports/codex.py +++ b/agent/transports/codex.py @@ -218,22 +218,10 @@ class ResponsesApiTransport(ProviderTransport): kwargs.pop("timeout", None) if is_codex_backend: - prompt_cache_key = kwargs.get("prompt_cache_key") - cache_scope_id = str(prompt_cache_key or session_id or "").strip() - if cache_scope_id: - existing_extra_headers = kwargs.get("extra_headers") - merged_extra_headers: Dict[str, str] = {} - if isinstance(existing_extra_headers, dict): - merged_extra_headers.update( - { - str(key): str(value) - for key, value in existing_extra_headers.items() - if key and value is not None - } - ) - merged_extra_headers["session_id"] = cache_scope_id - merged_extra_headers["x-client-request-id"] = cache_scope_id - kwargs["extra_headers"] = merged_extra_headers + # chatgpt.com/backend-api/codex rejects body-level + # ``extra_headers`` with HTTP 400. Correlation/cache routing for + # this backend must not be sent through the Responses payload. + kwargs.pop("extra_headers", None) max_tokens = params.get("max_tokens") if max_tokens is not None and not is_codex_backend: diff --git a/agent/transports/types.py b/agent/transports/types.py index 2deb157535b..6ad20f2376d 100644 --- a/agent/transports/types.py +++ b/agent/transports/types.py @@ -121,6 +121,18 @@ class NormalizedResponse: pd = self.provider_data or {} return pd.get("reasoning_details") + @property + def anthropic_content_blocks(self): + """Verbatim, order-preserving Anthropic content blocks for a turn. + + Present only when an Anthropic turn interleaves signed thinking with + tool_use — the one shape the parallel reasoning_details + tool_calls + lists reconstruct in the wrong order, invalidating thinking-block + signatures on replay. See agent/transports/anthropic.py. + """ + pd = self.provider_data or {} + return pd.get("anthropic_content_blocks") + @property def codex_reasoning_items(self): pd = self.provider_data or {} diff --git a/apps/bootstrap-installer/package.json b/apps/bootstrap-installer/package.json index 6b7991eafd1..9b3dc46a4a0 100644 --- a/apps/bootstrap-installer/package.json +++ b/apps/bootstrap-installer/package.json @@ -11,7 +11,8 @@ "tauri": "tauri", "tauri:dev": "tauri dev", "tauri:build": "tauri build", - "tauri:build:debug": "tauri build --debug" + "tauri:build:debug": "tauri build --debug", + "typecheck": "tsc -p . --noEmit" }, "dependencies": { "@nous-research/ui": "0.16.0", @@ -40,7 +41,7 @@ "@types/react": "^19.2.14", "@types/react-dom": "^19.2.3", "@vitejs/plugin-react": "^5.2.0", - "typescript": "~5.9.3", + "typescript": "^6.0.3", "vite": "^7.3.1" } } diff --git a/apps/bootstrap-installer/src-tauri/src/update.rs b/apps/bootstrap-installer/src-tauri/src/update.rs index 658bff6c540..40d136f960d 100644 --- a/apps/bootstrap-installer/src-tauri/src/update.rs +++ b/apps/bootstrap-installer/src-tauri/src/update.rs @@ -3,8 +3,9 @@ //! Driven when the installer is launched as `Hermes-Setup.exe --update` (see //! `AppMode` in lib.rs). The desktop app hands off to us — it exits, then we: //! -//! 1. wait for the old Hermes desktop process to fully exit (so the venv -//! shim is free; otherwise `hermes update` aborts with exit code 2), +//! 1. wait for the old Hermes desktop process to fully exit (so both the +//! venv shim and packaged app.asar are free; otherwise `hermes update` +//! or repair bootstrap can race locked files), //! 2. run `hermes update --yes --gateway` (Python/repo update; this does NOT //! rebuild apps/desktop by design — see cmd_update in hermes_cli/main.py), //! 3. run `hermes desktop --build-only` (the rebuild step update skips), @@ -38,8 +39,8 @@ use crate::events::{BootstrapEvent, LogStream, StageInfo, StageState}; /// hermes_cli/main.py (sys.exit(2)). We surface a targeted message for this. const UPDATE_EXIT_CONCURRENT: i32 = 2; -/// How long to wait for the old desktop process to release the venv shim -/// before giving up and letting `hermes update`'s own guard decide. +/// How long to wait for the old desktop process to release files under the +/// install tree before giving up and letting `hermes update`'s own guard decide. const DESKTOP_EXIT_WAIT: Duration = Duration::from_secs(20); const DESKTOP_EXIT_POLL: Duration = Duration::from_millis(500); @@ -150,8 +151,10 @@ async fn run_update(app: AppHandle) -> Result<()> { // ---- pre-step: wait for the old desktop to die ----------------------- // The desktop exec'd us then called app.exit(), but process teardown is // async on Windows. If it still holds the venv shim, `hermes update` - // aborts with exit 2. Give it a bounded window to clear. - wait_for_venv_free(&install_root, &app).await; + // aborts with exit 2. If it still holds the packaged app.asar, + // install.ps1's repair/re-clone path cannot move/remove the install tree. + // Give both handles a bounded window to clear. + wait_for_install_locks_free(&install_root, &app, "update").await; // ---- stage 1: hermes update ----------------------------------------- // Pass --branch so `hermes update` targets the branch this installer was @@ -173,8 +176,8 @@ async fn run_update(app: AppHandle) -> Result<()> { vec!["update".into(), "--yes".into(), "--gateway".into()]; // --force skips `hermes update`'s Windows running-exe guard (which would // `sys.exit(2)` and dead-end the handoff). By contract the desktop has - // already exited and waited for the venv shim to unlock before launching - // us, and wait_for_venv_free below force-kills any straggler — so by the + // already exited and waited for the install locks to clear before launching + // us, and wait_for_install_locks_free below force-kills any straggler — so by the // time `hermes update` runs there is no legitimate hermes.exe to protect, // and the guard would only produce a false "Hermes is still running" stop. update_args.push("--force".into()); @@ -391,48 +394,57 @@ async fn run_update(app: AppHandle) -> Result<()> { Ok(()) } -/// Poll until the venv shim is no longer locked (Windows) or a bounded timeout -/// elapses. On non-Windows this is a short fixed grace since file locking -/// isn't the failure mode there. -async fn wait_for_venv_free(install_root: &Path, app: &AppHandle) { - let shim = venv_hermes(install_root); +/// Poll until the venv shim AND packaged desktop app bundle are no longer locked +/// (Windows) or a bounded timeout elapses. On non-Windows this is a short fixed +/// grace since file locking isn't the failure mode there. +pub(crate) async fn wait_for_install_locks_free(install_root: &Path, app: &AppHandle, stage: &str) { + let lock_targets = install_lock_probe_paths(install_root); let deadline = Instant::now() + DESKTOP_EXIT_WAIT; - emit_log(app, Some("update"), LogStream::Stdout, "[update] waiting for Hermes to exit…"); + emit_log(app, Some(stage), LogStream::Stdout, "[handoff] waiting for Hermes to exit…"); loop { - if !is_locked(&shim) { + let locked = locked_paths(&lock_targets); + if locked.is_empty() { return; } if Instant::now() >= deadline { - // Last resort: a backend hermes.exe (or a grandchild it spawned) - // is still holding the shim. The desktop should have reaped its - // tree before handing off, but SIGTERM races / detached - // grandchildren / AV handles can leave a straggler. Rather than - // "proceed anyway" straight into uv's "Access is denied", force-kill - // every hermes.exe except ourselves, then give the OS a beat to - // unload the image. + // Last resort: a backend hermes.exe (or the desktop Hermes.exe + // itself) is still holding one of the update-sensitive files. The + // desktop should have reaped its tree before handing off, but + // SIGTERM races / detached grandchildren / AV handles can leave a + // straggler. Rather than "proceed anyway" straight into uv's + // "Access is denied" or install.ps1's locked app.asar failure, + // force-kill every Hermes.exe except ourselves, then give the OS a + // beat to unload the image. emit_log( app, - Some("update"), + Some(stage), LogStream::Stdout, - "[update] Hermes still holding the venv shim; force-killing stragglers…", + &format!( + "[handoff] Hermes still holding install files ({}); force-killing stragglers…", + format_locked_paths(&locked) + ), ); force_kill_other_hermes(); tokio::time::sleep(Duration::from_millis(800)).await; - if !is_locked(&shim) { + let locked_after_kill = locked_paths(&lock_targets); + if locked_after_kill.is_empty() { emit_log( app, - Some("update"), + Some(stage), LogStream::Stdout, - "[update] venv shim freed after force-kill", + "[handoff] install files freed after force-kill", ); } else { emit_log( app, - Some("update"), + Some(stage), LogStream::Stdout, - "[update] venv shim still locked; proceeding (--force + quarantine will handle it)", + &format!( + "[handoff] install files still locked ({}); proceeding (--force + quarantine will handle it)", + format_locked_paths(&locked_after_kill) + ), ); } return; @@ -441,13 +453,44 @@ async fn wait_for_venv_free(install_root: &Path, app: &AppHandle) { } } +fn install_lock_probe_paths(install_root: &Path) -> Vec { + let mut paths = vec![venv_hermes(install_root)]; + paths.extend(desktop_app_payload_paths(install_root)); + paths +} + +fn desktop_app_payload_paths(install_root: &Path) -> Vec { + let release = install_root.join("apps").join("desktop").join("release"); + if cfg!(target_os = "windows") { + vec![ + release.join("win-unpacked").join("resources").join("app.asar"), + release.join("win-arm64-unpacked").join("resources").join("app.asar"), + ] + } else if cfg!(target_os = "macos") { + vec![ + release.join("mac").join("Hermes.app").join("Contents").join("Resources").join("app.asar"), + release.join("mac-arm64").join("Hermes.app").join("Contents").join("Resources").join("app.asar"), + ] + } else { + vec![release.join("linux-unpacked").join("resources").join("app.asar")] + } +} + +fn locked_paths(paths: &[PathBuf]) -> Vec { + paths.iter().filter(|p| is_locked(p)).cloned().collect() +} + +fn format_locked_paths(paths: &[PathBuf]) -> String { + paths.iter().map(|p| p.display().to_string()).collect::>().join(", ") +} + /// Force-kill any `hermes.exe` other than this process. Windows-only; a no-op /// elsewhere (POSIX has no mandatory-lock contention). We can't selectively /// target "the backend" by PID here — the desktop already exited and we never /// knew its children — so we kill the whole `hermes.exe` image tree via /// taskkill, excluding our own PID. /// -/// Safe w.r.t. our own update child: this runs inside `wait_for_venv_free`, +/// Safe w.r.t. our own update child: this runs inside the install-lock wait, /// which completes BEFORE we spawn `venv\Scripts\hermes.exe update`. At this /// point no update-driven hermes.exe exists yet, so the only hermes.exe images /// are stragglers from the old desktop — exactly what we want gone. (`/FI PID @@ -891,6 +934,29 @@ mod tests { assert!(!is_locked(Path::new("/nonexistent/does/not/exist/xyz"))); } + #[test] + fn lock_probe_paths_include_desktop_app_payload() { + let root = Path::new("/x/hermes-agent"); + let probes = install_lock_probe_paths(root); + + assert!( + probes.iter().any(|p| p == &venv_hermes(root)), + "venv shim remains part of the update lock probe" + ); + assert!( + probes.iter().any(|p| p.ends_with(Path::new("resources/app.asar"))), + "packaged app.asar must be probed so repair/re-clone waits for the old desktop to exit" + ); + } + + #[test] + fn locked_paths_ignores_missing_payloads() { + let root = Path::new("/nonexistent/hermes-agent"); + let probes = install_lock_probe_paths(root); + + assert!(locked_paths(&probes).is_empty()); + } + #[test] fn parses_update_branch_from_space_or_equals_args() { assert_eq!( diff --git a/apps/bootstrap-installer/tsconfig.json b/apps/bootstrap-installer/tsconfig.json index e2a5f6bbeb6..9227970f066 100644 --- a/apps/bootstrap-installer/tsconfig.json +++ b/apps/bootstrap-installer/tsconfig.json @@ -16,9 +16,8 @@ "noUnusedParameters": true, "esModuleInterop": true, "noFallthroughCasesInSwitch": true, - "baseUrl": ".", "paths": { - "@/*": ["src/*"] + "@/*": ["./src/*"] } }, "include": ["src"], diff --git a/apps/desktop/README.md b/apps/desktop/README.md index f3084a9b647..301b094592f 100644 --- a/apps/desktop/README.md +++ b/apps/desktop/README.md @@ -93,7 +93,7 @@ Run before opening a PR (lint may surface pre-existing warnings but must exit cl ```bash npm run fix -npm run type-check +npm run typecheck npm run lint npm run test:desktop:all ``` diff --git a/apps/desktop/electron/backend-env.cjs b/apps/desktop/electron/backend-env.cjs new file mode 100644 index 00000000000..76329785be4 --- /dev/null +++ b/apps/desktop/electron/backend-env.cjs @@ -0,0 +1,112 @@ +const path = require('node:path') + +// Match the POSIX fallback surface used by the Python terminal environment. +// macOS apps launched from Finder/Dock often inherit only /usr/bin:/bin:/usr/sbin:/sbin, +// which misses Apple Silicon Homebrew and user-installed CLI tools such as codex. +const POSIX_SANE_PATH_ENTRIES = Object.freeze([ + '/opt/homebrew/bin', + '/opt/homebrew/sbin', + '/usr/local/sbin', + '/usr/local/bin', + '/usr/sbin', + '/usr/bin', + '/sbin', + '/bin' +]) + +function delimiterForPlatform(platform = process.platform) { + return platform === 'win32' ? ';' : ':' +} + +function pathModuleForPlatform(platform = process.platform) { + return platform === 'win32' ? path.win32 : path.posix +} + +function pathEnvKey(env = process.env, platform = process.platform) { + if (platform !== 'win32') return 'PATH' + return Object.keys(env || {}).find(key => key.toUpperCase() === 'PATH') || 'PATH' +} + +function currentPathValue(env = process.env, platform = process.platform) { + const key = pathEnvKey(env, platform) + return env?.[key] || '' +} + +function appendUniquePathEntries(entries, { delimiter = path.delimiter } = {}) { + const seen = new Set() + const ordered = [] + + for (const entry of entries) { + if (!entry) continue + const parts = Array.isArray(entry) ? entry : String(entry).split(delimiter) + for (const part of parts) { + if (!part || seen.has(part)) continue + seen.add(part) + ordered.push(part) + } + } + + return ordered.join(delimiter) +} + +function buildDesktopBackendPath({ + hermesHome, + venvRoot, + currentPath = '', + platform = process.platform, + pathModule = pathModuleForPlatform(platform) +} = {}) { + const delimiter = delimiterForPlatform(platform) + const hermesNodeBin = hermesHome ? pathModule.join(hermesHome, 'node', 'bin') : null + const venvBin = venvRoot ? pathModule.join(venvRoot, platform === 'win32' ? 'Scripts' : 'bin') : null + const saneEntries = platform === 'win32' ? [] : POSIX_SANE_PATH_ENTRIES + + return appendUniquePathEntries( + [hermesNodeBin, venvBin, currentPath, saneEntries], + { delimiter } + ) +} + +function normalizeHermesHomeRoot(hermesHome, { pathModule = pathModuleForPlatform(process.platform) } = {}) { + if (!hermesHome) return hermesHome + const resolved = pathModule.resolve(String(hermesHome)) + const parent = pathModule.dirname(resolved) + if (pathModule.basename(parent).toLowerCase() === 'profiles') { + return pathModule.dirname(parent) + } + return resolved +} + +function buildDesktopBackendEnv({ + hermesHome, + pythonPathEntries = [], + venvRoot, + currentEnv = process.env, + platform = process.platform, + pathModule = pathModuleForPlatform(platform) +} = {}) { + const delimiter = delimiterForPlatform(platform) + const currentPythonPath = currentEnv?.PYTHONPATH || '' + const key = pathEnvKey(currentEnv, platform) + + return { + PYTHONPATH: appendUniquePathEntries([...pythonPathEntries, currentPythonPath], { delimiter }), + [key]: buildDesktopBackendPath({ + hermesHome, + venvRoot, + currentPath: currentPathValue(currentEnv, platform), + platform, + pathModule + }) + } +} + +module.exports = { + POSIX_SANE_PATH_ENTRIES, + appendUniquePathEntries, + buildDesktopBackendEnv, + buildDesktopBackendPath, + delimiterForPlatform, + normalizeHermesHomeRoot, + pathEnvKey +} diff --git a/apps/desktop/electron/backend-env.test.cjs b/apps/desktop/electron/backend-env.test.cjs new file mode 100644 index 00000000000..75e0c79d5d6 --- /dev/null +++ b/apps/desktop/electron/backend-env.test.cjs @@ -0,0 +1,111 @@ +const test = require('node:test') +const assert = require('node:assert/strict') +const path = require('node:path') + +const { + POSIX_SANE_PATH_ENTRIES, + appendUniquePathEntries, + buildDesktopBackendEnv, + buildDesktopBackendPath, + normalizeHermesHomeRoot, + pathEnvKey +} = require('./backend-env.cjs') + +test('desktop backend PATH adds Hermes-managed bins and missing POSIX sane entries', () => { + const result = buildDesktopBackendPath({ + hermesHome: '/Users/test/.hermes', + venvRoot: '/Users/test/.hermes/hermes-agent/venv', + currentPath: '/usr/bin:/bin:/usr/sbin:/sbin:/usr/local/bin', + platform: 'darwin', + pathModule: path.posix + }) + + const entries = result.split(':') + assert.equal(entries[0], '/Users/test/.hermes/node/bin') + assert.equal(entries[1], '/Users/test/.hermes/hermes-agent/venv/bin') + assert.ok(entries.includes('/opt/homebrew/bin'), 'Apple Silicon Homebrew bin is added') + assert.ok(entries.includes('/opt/homebrew/sbin'), 'Apple Silicon Homebrew sbin is added') + assert.ok(entries.includes('/usr/local/sbin'), 'missing standard sbin is added') + + for (const expected of POSIX_SANE_PATH_ENTRIES) { + assert.ok(entries.includes(expected), `${expected} should be present`) + } +}) + +test('desktop backend PATH preserves first occurrence and avoids duplicates', () => { + const result = buildDesktopBackendPath({ + hermesHome: '/Users/test/.hermes', + venvRoot: '/Users/test/.hermes/hermes-agent/venv', + currentPath: '/opt/homebrew/bin:/usr/bin:/opt/homebrew/bin:/bin', + platform: 'darwin', + pathModule: path.posix + }) + + const entries = result.split(':') + assert.equal(entries.filter(entry => entry === '/opt/homebrew/bin').length, 1) + assert.ok( + entries.indexOf('/opt/homebrew/bin') < entries.indexOf('/opt/homebrew/sbin'), + 'existing Homebrew bin keeps its precedence over appended missing sane entries' + ) +}) + +test('buildDesktopBackendEnv extends PYTHONPATH and backend PATH together', () => { + const env = buildDesktopBackendEnv({ + hermesHome: '/Users/test/.hermes', + pythonPathEntries: ['/repo/hermes-agent'], + venvRoot: '/Users/test/.hermes/hermes-agent/venv', + currentEnv: { + PATH: '/usr/bin:/bin', + PYTHONPATH: '/existing/pythonpath' + }, + platform: 'darwin', + pathModule: path.posix + }) + + assert.equal(env.PYTHONPATH, '/repo/hermes-agent:/existing/pythonpath') + assert.ok(env.PATH.startsWith('/Users/test/.hermes/node/bin:/Users/test/.hermes/hermes-agent/venv/bin:')) + assert.ok(env.PATH.includes('/opt/homebrew/bin')) +}) + +test('normalizeHermesHomeRoot maps profile homes back to the global Hermes root', () => { + assert.equal( + normalizeHermesHomeRoot('/Users/test/.hermes/profiles/oracle', { pathModule: path.posix }), + '/Users/test/.hermes' + ) + assert.equal( + normalizeHermesHomeRoot('C:\\Users\\test\\AppData\\Local\\hermes\\profiles\\oracle', { pathModule: path.win32 }), + 'C:\\Users\\test\\AppData\\Local\\hermes' + ) + assert.equal( + normalizeHermesHomeRoot('/Users/test/.hermes', { pathModule: path.posix }), + '/Users/test/.hermes' + ) +}) + +test('Windows PATH casing and delimiter are preserved without POSIX sane entries', () => { + const env = buildDesktopBackendEnv({ + hermesHome: 'C:\\Users\\test\\AppData\\Local\\hermes', + pythonPathEntries: ['C:\\repo\\hermes-agent'], + venvRoot: 'C:\\Users\\test\\AppData\\Local\\hermes\\hermes-agent\\venv', + currentEnv: { + Path: 'C:\\Windows\\System32;C:\\Windows', + PYTHONPATH: 'C:\\existing\\pythonpath' + }, + platform: 'win32', + pathModule: path.win32 + }) + + assert.equal(pathEnvKey({ Path: 'x' }, 'win32'), 'Path') + assert.equal(env.PATH, undefined) + assert.ok(env.Path.startsWith('C:\\Users\\test\\AppData\\Local\\hermes\\node\\bin;')) + assert.ok(env.Path.includes('\\venv\\Scripts;')) + assert.ok(env.Path.includes(';C:\\Windows\\System32;C:\\Windows')) + assert.equal(env.Path.includes('/opt/homebrew/bin'), false) +}) + +test('appendUniquePathEntries drops empty entries and keeps first occurrence', () => { + assert.equal( + appendUniquePathEntries([':/a::/b', ['/a', '/c']], { delimiter: ':' }), + '/a:/b:/c' + ) +}) diff --git a/apps/desktop/electron/backend-ready.cjs b/apps/desktop/electron/backend-ready.cjs new file mode 100644 index 00000000000..9af41e549c4 --- /dev/null +++ b/apps/desktop/electron/backend-ready.cjs @@ -0,0 +1,66 @@ +const _READY_RE = /^HERMES_DASHBOARD_READY port=(\d+)/m + +/** + * Watch a child process's stdout for the `HERMES_DASHBOARD_READY port=` + * line that web_server.py prints after uvicorn binds its socket. + * + * Returns the parsed port. Rejects if: + * - the child exits before emitting the line + * - the child emits an `error` event + * - no line arrives within the timeout + * + * A single `cleanup()` tears down every listener (data/exit/error/timeout) + * on every terminal path — resolve, reject, or timeout — so repeated + * backend spawns don't leak listener slots on the child. + */ +function waitForDashboardPort(child, timeoutMs = 45_000) { + return new Promise((resolve, reject) => { + let buf = '' + let done = false + + function cleanup() { + if (done) return + done = true + clearTimeout(timer) + child.stdout.off('data', onData) + child.off('exit', onExit) + child.off('error', onError) + } + + function onData(chunk) { + buf += chunk.toString() + let nl + while ((nl = buf.indexOf('\n')) !== -1) { + const line = buf.slice(0, nl) + buf = buf.slice(nl + 1) + const m = line.match(_READY_RE) + if (m) { + cleanup() + resolve(parseInt(m[1], 10)) + return + } + } + } + + function onExit(code, signal) { + cleanup() + reject(new Error(`Hermes backend: exited before port announcement (${signal || code})`)) + } + + function onError(err) { + cleanup() + reject(err) + } + + const timer = setTimeout(() => { + cleanup() + reject(new Error(`Timed out waiting for Hermes backend port announcement (${timeoutMs}ms)`)) + }, timeoutMs) + + child.stdout.on('data', onData) + child.on('exit', onExit) + child.on('error', onError) + }) +} + +module.exports = { waitForDashboardPort } diff --git a/apps/desktop/electron/dashboard-token.cjs b/apps/desktop/electron/dashboard-token.cjs new file mode 100644 index 00000000000..1a9ca50ad9c --- /dev/null +++ b/apps/desktop/electron/dashboard-token.cjs @@ -0,0 +1,99 @@ +/** + * Helpers for local dashboard session-token discovery. + * + * The desktop main process can pass HERMES_DASHBOARD_SESSION_TOKEN when it + * spawns the local dashboard, but the dashboard is the source of truth for the + * token it actually serves to the renderer. If those drift, HTTP readiness + * probes still pass while /api/ws rejects the renderer's token. + */ + +const DEFAULT_TOKEN_FETCH_TIMEOUT_MS = 3_000 + +async function fetchPublicText(url, options = {}) { + const { protocol } = new URL(url) + if (protocol !== 'http:' && protocol !== 'https:') { + throw new Error(`Unsupported Hermes backend URL protocol: ${protocol}`) + } + + const timeoutMs = options.timeoutMs ?? DEFAULT_TOKEN_FETCH_TIMEOUT_MS + const res = await fetch(url, { signal: AbortSignal.timeout(timeoutMs) }).catch(error => { + if (error.name === 'TimeoutError') { + throw new Error(`Timed out connecting to Hermes backend after ${timeoutMs}ms`) + } + throw error + }) + const text = await res.text() + + if (!res.ok) throw new Error(`${res.status}: ${text || res.statusText}`) + + return text +} + +function extractInjectedDashboardToken(html) { + const match = /window\.__HERMES_SESSION_TOKEN__\s*=\s*("(?:\\.|[^"\\])*")/.exec(String(html || '')) + if (!match) return null + try { + return JSON.parse(match[1]) + } catch { + return null + } +} + +function dashboardIndexUrl(baseUrl) { + return `${String(baseUrl || '').replace(/\/+$/, '')}/` +} + +async function resolveServedDashboardToken(baseUrl, fallbackToken, options = {}) { + const fetchText = options.fetchText || fetchPublicText + const html = await fetchText(dashboardIndexUrl(baseUrl), { + timeoutMs: options.timeoutMs ?? DEFAULT_TOKEN_FETCH_TIMEOUT_MS + }) + const servedToken = extractInjectedDashboardToken(html) + + if (servedToken && servedToken !== fallbackToken && typeof options.rememberLog === 'function') { + options.rememberLog('[boot] dashboard served a different session token; using served token for WebSocket auth') + } + + return servedToken || fallbackToken +} + +/** + * A served token that differs from our spawn token while our child is DEAD + * came from a process we did not spawn (orphan/port squatter that satisfied + * the public /api/status readiness probe). With a live child the mismatch is + * benign: our own backend regenerated the token because the env pin did not + * survive the spawn. + */ +function isForeignBackendToken({ servedToken, spawnToken, childAlive }) { + return Boolean(servedToken) && servedToken !== spawnToken && !childAlive +} + +/** + * Resolve the token the backend actually serves, adopting benign drift and + * failing loudly on a foreign backend. `childAlive` is a thunk so liveness is + * sampled after the fetch, not before. + */ +async function adoptServedDashboardToken(baseUrl, spawnToken, { childAlive, label = 'Hermes backend', ...options }) { + const servedToken = await resolveServedDashboardToken(baseUrl, spawnToken, options).catch(error => { + options.rememberLog?.(`[boot] could not read served dashboard token (${label}): ${error.message}`) + return spawnToken + }) + + if (isForeignBackendToken({ servedToken, spawnToken, childAlive: childAlive() })) { + throw new Error( + `${label} exited and ${dashboardIndexUrl(baseUrl)} is served by a process we did not spawn; refusing its session token.` + ) + } + + return servedToken +} + +module.exports = { + DEFAULT_TOKEN_FETCH_TIMEOUT_MS, + adoptServedDashboardToken, + dashboardIndexUrl, + extractInjectedDashboardToken, + fetchPublicText, + isForeignBackendToken, + resolveServedDashboardToken +} diff --git a/apps/desktop/electron/dashboard-token.test.cjs b/apps/desktop/electron/dashboard-token.test.cjs new file mode 100644 index 00000000000..d598ffc2bc1 --- /dev/null +++ b/apps/desktop/electron/dashboard-token.test.cjs @@ -0,0 +1,142 @@ +/** + * Tests for electron/dashboard-token.cjs. + * + * Run with: node --test electron/dashboard-token.test.cjs + * (Wired into npm test:desktop:platforms in package.json.) + */ + +const test = require('node:test') +const assert = require('node:assert/strict') + +const { + adoptServedDashboardToken, + dashboardIndexUrl, + extractInjectedDashboardToken, + fetchPublicText, + isForeignBackendToken, + resolveServedDashboardToken +} = require('./dashboard-token.cjs') + +test('extractInjectedDashboardToken reads the JSON-encoded dashboard token', () => { + const html = '' + assert.equal(extractInjectedDashboardToken(html), 'served-token') +}) + +test('extractInjectedDashboardToken handles escaped token strings', () => { + const html = '' + assert.equal(extractInjectedDashboardToken(html), 'served\\token"quoted') +}) + +test('extractInjectedDashboardToken returns null for missing or malformed values', () => { + assert.equal(extractInjectedDashboardToken(''), null) + assert.equal(extractInjectedDashboardToken(''), null) +}) + +test('dashboardIndexUrl preserves dashboard path prefixes', () => { + assert.equal(dashboardIndexUrl('http://127.0.0.1:9120'), 'http://127.0.0.1:9120/') + assert.equal(dashboardIndexUrl('https://host.example/hermes/'), 'https://host.example/hermes/') +}) + +test('resolveServedDashboardToken uses the served token and logs when it differs', async () => { + const logs = [] + const token = await resolveServedDashboardToken('http://127.0.0.1:9120', 'spawn-token', { + fetchText: async url => { + assert.equal(url, 'http://127.0.0.1:9120/') + return '' + }, + rememberLog: line => logs.push(line) + }) + + assert.equal(token, 'served-token') + assert.equal(logs.length, 1) + assert.match(logs[0], /served a different session token/) +}) + +test('resolveServedDashboardToken falls back when the served HTML has no token', async () => { + const token = await resolveServedDashboardToken('http://127.0.0.1:9120', 'spawn-token', { + fetchText: async () => '', + rememberLog: () => { + throw new Error('should not log when no served token is present') + } + }) + + assert.equal(token, 'spawn-token') +}) + +test('resolveServedDashboardToken does not log when served token matches fallback', async () => { + const token = await resolveServedDashboardToken('http://127.0.0.1:9120', 'same-token', { + fetchText: async () => '', + rememberLog: () => { + throw new Error('should not log when token already matches') + } + }) + + assert.equal(token, 'same-token') +}) + +test('resolveServedDashboardToken propagates fetch errors so callers can fall back explicitly', async () => { + await assert.rejects( + () => + resolveServedDashboardToken('http://127.0.0.1:9120', 'spawn-token', { + fetchText: async () => { + throw new Error('boom') + } + }), + /boom/ + ) +}) + +test('fetchPublicText rejects unsupported protocols', async () => { + await assert.rejects(() => fetchPublicText('file:///tmp/index.html'), /Unsupported Hermes backend URL protocol/) +}) + +test('isForeignBackendToken only flags a mismatched token from a dead child', () => { + const cases = [ + [{ servedToken: 'other', spawnToken: 'mine', childAlive: false }, true], + // Live child + drift = our backend regenerated the token (env pin lost). + [{ servedToken: 'other', spawnToken: 'mine', childAlive: true }, false], + [{ servedToken: 'mine', spawnToken: 'mine', childAlive: false }, false], + [{ servedToken: 'mine', spawnToken: 'mine', childAlive: true }, false], + [{ servedToken: null, spawnToken: 'mine', childAlive: false }, false], + [{ servedToken: '', spawnToken: 'mine', childAlive: false }, false] + ] + for (const [input, expected] of cases) { + assert.equal(isForeignBackendToken(input), expected, JSON.stringify(input)) + } +}) + +test('adoptServedDashboardToken adopts drift from a live child', async () => { + const token = await adoptServedDashboardToken('http://127.0.0.1:9120', 'spawn-token', { + childAlive: () => true, + fetchText: async () => '' + }) + + assert.equal(token, 'served-token') +}) + +test('adoptServedDashboardToken refuses a foreign token when our child is dead', async () => { + await assert.rejects( + () => + adoptServedDashboardToken('http://127.0.0.1:9120', 'spawn-token', { + childAlive: () => false, + fetchText: async () => '', + label: 'Hermes backend for profile "work"' + }), + /profile "work".*process we did not spawn/ + ) +}) + +test('adoptServedDashboardToken falls back to the spawn token when the fetch fails', async () => { + const logs = [] + const token = await adoptServedDashboardToken('http://127.0.0.1:9120', 'spawn-token', { + childAlive: () => true, + fetchText: async () => { + throw new Error('boom') + }, + rememberLog: line => logs.push(line) + }) + + assert.equal(token, 'spawn-token') + assert.equal(logs.length, 1) + assert.match(logs[0], /could not read served dashboard token \(Hermes backend\): boom/) +}) diff --git a/apps/desktop/electron/fs-read-dir.cjs b/apps/desktop/electron/fs-read-dir.cjs new file mode 100644 index 00000000000..52d182ad567 --- /dev/null +++ b/apps/desktop/electron/fs-read-dir.cjs @@ -0,0 +1,109 @@ +'use strict' + +const fs = require('node:fs') +const path = require('node:path') +const { resolveDirectoryForIpc } = require('./hardening.cjs') + +const FS_READDIR_STAT_CONCURRENCY = 16 + +// Always-hidden noise (covers non-git projects too; gitignore catches many of +// these, but the project tree should keep the same hygiene without one). +const FS_READDIR_HIDDEN = new Set([ + '.git', + '.hg', + '.svn', + '.cache', + '.next', + '.turbo', + '.venv', + '__pycache__', + 'build', + 'dist', + 'node_modules', + 'target', + 'venv' +]) + +function direntIsDirectory(dirent) { + return typeof dirent.isDirectory === 'function' && dirent.isDirectory() +} + +function direntIsFile(dirent) { + return typeof dirent.isFile === 'function' && dirent.isFile() +} + +function direntIsSymbolicLink(dirent) { + return typeof dirent.isSymbolicLink === 'function' && dirent.isSymbolicLink() +} + +function shouldStatDirent(dirent) { + if (direntIsDirectory(dirent)) return false + + return direntIsSymbolicLink(dirent) || !direntIsFile(dirent) +} + +async function entryForDirent(dirent, resolved, fsImpl) { + const fullPath = path.join(resolved, dirent.name) + let isDirectory = direntIsDirectory(dirent) + + if (!isDirectory && shouldStatDirent(dirent)) { + try { + isDirectory = (await fsImpl.promises.stat(fullPath)).isDirectory() + } catch { + isDirectory = false + } + } + + return { name: dirent.name, path: fullPath, isDirectory } +} + +async function mapWithStatConcurrency(items, mapper) { + const results = new Array(items.length) + let nextIndex = 0 + + async function runWorker() { + while (nextIndex < items.length) { + const index = nextIndex + nextIndex += 1 + results[index] = await mapper(items[index]) + } + } + + const workerCount = Math.min(FS_READDIR_STAT_CONCURRENCY, items.length) + const workers = Array.from({ length: workerCount }, () => runWorker()) + await Promise.all(workers) + + return results +} + +async function readDirForIpc(dirPath, options = {}) { + const fsImpl = options.fs || fs + let resolved + + try { + ;({ resolvedPath: resolved } = await resolveDirectoryForIpc(dirPath, { + fs: fsImpl, + purpose: 'Directory read' + })) + } catch (error) { + return { entries: [], error: error?.code || 'read-error' } + } + + try { + const dirents = await fsImpl.promises.readdir(resolved, { withFileTypes: true }) + const visibleDirents = dirents.filter(dirent => !FS_READDIR_HIDDEN.has(dirent.name)) + const entries = await mapWithStatConcurrency(visibleDirents, dirent => + entryForDirent(dirent, resolved, fsImpl) + ) + + entries.sort((a, b) => Number(b.isDirectory) - Number(a.isDirectory) || a.name.localeCompare(b.name)) + + return { entries } + } catch (error) { + return { entries: [], error: error?.code || 'read-error' } + } +} + +module.exports = { + readDirForIpc +} diff --git a/apps/desktop/electron/fs-read-dir.test.cjs b/apps/desktop/electron/fs-read-dir.test.cjs new file mode 100644 index 00000000000..42e80af3489 --- /dev/null +++ b/apps/desktop/electron/fs-read-dir.test.cjs @@ -0,0 +1,364 @@ +'use strict' + +const assert = require('node:assert/strict') +const fs = require('node:fs') +const os = require('node:os') +const path = require('node:path') +const test = require('node:test') +const { pathToFileURL } = require('node:url') + +const { readDirForIpc } = require('./fs-read-dir.cjs') + +function mkTmpDir() { + return fs.mkdtempSync(path.join(os.tmpdir(), 'hermes-fs-read-dir-')) +} + +function fakeDirent(name, flags = {}) { + return { + name, + isDirectory: () => Boolean(flags.directory), + isFile: () => Boolean(flags.file), + isSymbolicLink: () => Boolean(flags.symlink) + } +} + +test('readDirForIpc hides noisy directories and files from the project tree', async () => { + const root = mkTmpDir() + + try { + fs.mkdirSync(path.join(root, 'node_modules')) + fs.mkdirSync(path.join(root, 'src')) + fs.writeFileSync(path.join(root, 'target'), 'hidden file') + fs.writeFileSync(path.join(root, 'README.md'), 'visible file') + + const result = await readDirForIpc(root) + + assert.equal(result.error, undefined) + assert.deepEqual( + result.entries.map(entry => entry.name), + ['src', 'README.md'] + ) + } finally { + fs.rmSync(root, { recursive: true, force: true }) + } +}) + +test('readDirForIpc filters a hidden basename whether it is a file or directory', async () => { + const dirRoot = mkTmpDir() + const fileRoot = mkTmpDir() + + try { + fs.mkdirSync(path.join(dirRoot, 'node_modules')) + fs.writeFileSync(path.join(dirRoot, 'visible.txt'), 'visible') + fs.writeFileSync(path.join(fileRoot, 'node_modules'), 'hidden file') + fs.writeFileSync(path.join(fileRoot, 'visible.txt'), 'visible') + + assert.deepEqual( + (await readDirForIpc(dirRoot)).entries.map(entry => entry.name), + ['visible.txt'] + ) + assert.deepEqual( + (await readDirForIpc(fileRoot)).entries.map(entry => entry.name), + ['visible.txt'] + ) + } finally { + fs.rmSync(dirRoot, { recursive: true, force: true }) + fs.rmSync(fileRoot, { recursive: true, force: true }) + } +}) + +test('readDirForIpc returns directories before files and sorts by name within groups', async () => { + const root = mkTmpDir() + + try { + fs.writeFileSync(path.join(root, 'z.txt'), 'z') + fs.mkdirSync(path.join(root, 'src')) + fs.writeFileSync(path.join(root, 'a.txt'), 'a') + fs.mkdirSync(path.join(root, 'lib')) + + const result = await readDirForIpc(root) + + assert.equal(result.error, undefined) + assert.deepEqual( + result.entries.map(entry => entry.name), + ['lib', 'src', 'a.txt', 'z.txt'] + ) + } finally { + fs.rmSync(root, { recursive: true, force: true }) + } +}) + +test('readDirForIpc accepts file URLs for directories', async () => { + const root = mkTmpDir() + + try { + fs.mkdirSync(path.join(root, 'src')) + fs.writeFileSync(path.join(root, 'README.md'), 'visible file') + + const result = await readDirForIpc(pathToFileURL(root).toString()) + + assert.equal(result.error, undefined) + assert.deepEqual( + result.entries.map(entry => entry.name), + ['src', 'README.md'] + ) + } finally { + fs.rmSync(root, { recursive: true, force: true }) + } +}) + +test('readDirForIpc returns invalid-path for blank or non-string input', async () => { + let readdirCalls = 0 + const fsImpl = { + promises: { + readdir: async () => { + readdirCalls += 1 + return [] + } + } + } + + assert.deepEqual(await readDirForIpc('', { fs: fsImpl }), { entries: [], error: 'invalid-path' }) + assert.deepEqual(await readDirForIpc(' ', { fs: fsImpl }), { entries: [], error: 'invalid-path' }) + assert.deepEqual(await readDirForIpc(null, { fs: fsImpl }), { entries: [], error: 'invalid-path' }) + assert.equal(readdirCalls, 0) +}) + +test('readDirForIpc rejects Windows device paths before readdir', async () => { + let readdirCalls = 0 + const fsImpl = { + promises: { + readdir: async () => { + readdirCalls += 1 + return [] + } + } + } + + assert.deepEqual(await readDirForIpc('\\\\?\\C:\\secret', { fs: fsImpl }), { + entries: [], + error: 'device-path' + }) + assert.equal(readdirCalls, 0) +}) + +test('readDirForIpc returns filesystem error codes instead of throwing', async () => { + const root = mkTmpDir() + + try { + const result = await readDirForIpc(path.join(root, 'missing')) + + assert.deepEqual(result, { entries: [], error: 'ENOENT' }) + } finally { + fs.rmSync(root, { recursive: true, force: true }) + } +}) + +test('readDirForIpc marks a symlink to a directory as a directory', async t => { + const root = mkTmpDir() + + try { + fs.mkdirSync(path.join(root, 'actual-dir')) + + try { + fs.symlinkSync(path.join(root, 'actual-dir'), path.join(root, 'linked-dir'), 'dir') + } catch (error) { + if (error?.code === 'EPERM' || error?.code === 'EACCES') { + t.skip(`symlink creation is not permitted on this platform (${error.code})`) + + return + } + + throw error + } + + const result = await readDirForIpc(root) + const linked = result.entries.find(entry => entry.name === 'linked-dir') + + assert.equal(result.error, undefined) + assert.equal(linked?.isDirectory, true) + } finally { + fs.rmSync(root, { recursive: true, force: true }) + } +}) + +test('readDirForIpc marks a Windows junction to a directory as a directory', async t => { + if (process.platform !== 'win32') { + t.skip('junctions are a Windows-specific symlink type') + + return + } + + const root = mkTmpDir() + + try { + fs.mkdirSync(path.join(root, 'actual-dir')) + + try { + fs.symlinkSync(path.join(root, 'actual-dir'), path.join(root, 'junction-dir'), 'junction') + } catch (error) { + if (error?.code === 'EPERM' || error?.code === 'EACCES') { + t.skip(`junction creation is not permitted on this platform (${error.code})`) + + return + } + + throw error + } + + const result = await readDirForIpc(root) + const junction = result.entries.find(entry => entry.name === 'junction-dir') + + assert.equal(result.error, undefined) + assert.equal(junction?.isDirectory, true) + } finally { + fs.rmSync(root, { recursive: true, force: true }) + } +}) + +test('readDirForIpc allows expanding symlink or junction directories outside the project root', async t => { + const root = mkTmpDir() + const outside = mkTmpDir() + + try { + fs.writeFileSync(path.join(outside, 'outside.txt'), 'ok') + + const linkPath = path.join(root, 'outside-link') + try { + fs.symlinkSync(outside, linkPath, process.platform === 'win32' ? 'junction' : 'dir') + } catch (error) { + if (error?.code === 'EPERM' || error?.code === 'EACCES') { + t.skip(`directory symlink creation is not permitted on this platform (${error.code})`) + + return + } + + throw error + } + + const result = await readDirForIpc(linkPath) + + assert.equal(result.error, undefined) + assert.deepEqual(result.entries, [ + { name: 'outside.txt', path: path.join(linkPath, 'outside.txt'), isDirectory: false } + ]) + } finally { + fs.rmSync(root, { recursive: true, force: true }) + fs.rmSync(outside, { recursive: true, force: true }) + } +}) + +test('readDirForIpc stats symbolic links and unknown entries without dropping the whole listing', async () => { + const input = path.join('virtual-root') + const resolved = path.resolve(input) + const statCalls = [] + const fsImpl = { + promises: { + readdir: async () => [ + fakeDirent('unknown-entry'), + fakeDirent('linked-dir', { symlink: true }), + fakeDirent('broken-link', { symlink: true }), + fakeDirent('plain.txt', { file: true }) + ], + stat: async fullPath => { + if (fullPath === resolved) { + return { isDirectory: () => true } + } + + statCalls.push(fullPath) + if (fullPath.endsWith(`${path.sep}linked-dir`)) { + return { isDirectory: () => true } + } + throw Object.assign(new Error('gone'), { code: 'ENOENT' }) + } + } + } + + const result = await readDirForIpc(input, { fs: fsImpl }) + + assert.equal(result.error, undefined) + assert.deepEqual( + statCalls.sort(), + [path.join(resolved, 'broken-link'), path.join(resolved, 'linked-dir'), path.join(resolved, 'unknown-entry')].sort() + ) + assert.deepEqual(result.entries, [ + { name: 'linked-dir', path: path.join(resolved, 'linked-dir'), isDirectory: true }, + { name: 'broken-link', path: path.join(resolved, 'broken-link'), isDirectory: false }, + { name: 'plain.txt', path: path.join(resolved, 'plain.txt'), isDirectory: false }, + { name: 'unknown-entry', path: path.join(resolved, 'unknown-entry'), isDirectory: false } + ]) +}) + +test('readDirForIpc bounds concurrent stats while preserving complete sorted output', async () => { + const input = path.join('virtual-root') + const resolved = path.resolve(input) + const names = Array.from({ length: 105 }, (_, index) => `entry-${String(104 - index).padStart(3, '0')}`) + const failedName = 'entry-100' + const directoryNames = new Set(names.filter((_, index) => index % 10 === 4)) + const successfulDirectoryNames = new Set([...directoryNames].filter(name => name !== failedName)) + const statCalls = [] + let active = 0 + let peak = 0 + let releaseStats + let markFirstStatStarted + const statsReleased = new Promise(resolve => { + releaseStats = resolve + }) + const firstStatStarted = new Promise(resolve => { + markFirstStatStarted = resolve + }) + const fsImpl = { + promises: { + readdir: async () => [ + fakeDirent('node_modules', { symlink: true }), + ...names.map((name, index) => fakeDirent(name, { symlink: index % 2 === 0 })) + ], + stat: async fullPath => { + if (fullPath === resolved) { + return { isDirectory: () => true } + } + + statCalls.push(fullPath) + active += 1 + peak = Math.max(peak, active) + markFirstStatStarted() + await statsReleased + active -= 1 + + const name = path.basename(fullPath) + if (name === failedName) { + throw Object.assign(new Error('gone'), { code: 'ENOENT' }) + } + + return { isDirectory: () => successfulDirectoryNames.has(name) } + } + } + } + + const resultPromise = readDirForIpc(input, { fs: fsImpl }) + await firstStatStarted + await new Promise(resolve => setImmediate(resolve)) + releaseStats() + const result = await resultPromise + + const expectedNames = [ + ...names.filter(name => successfulDirectoryNames.has(name)).sort(), + ...names.filter(name => !successfulDirectoryNames.has(name)).sort() + ] + + assert.equal(result.error, undefined) + assert.equal(result.entries.length, names.length) + assert.equal(statCalls.length, names.length) + assert.equal(statCalls.some(fullPath => fullPath.endsWith(`${path.sep}node_modules`)), false) + assert.ok(peak > 1, `expected concurrent stats, observed peak ${peak}`) + assert.ok(peak <= 16, `expected at most 16 concurrent stats, observed peak ${peak}`) + assert.deepEqual( + result.entries.map(entry => entry.name), + expectedNames + ) + assert.equal(result.entries.find(entry => entry.name === failedName)?.isDirectory, false) + assert.equal( + result.entries.filter(entry => entry.isDirectory).length, + successfulDirectoryNames.size + ) +}) diff --git a/apps/desktop/electron/git-root.cjs b/apps/desktop/electron/git-root.cjs new file mode 100644 index 00000000000..593d3531ebc --- /dev/null +++ b/apps/desktop/electron/git-root.cjs @@ -0,0 +1,54 @@ +'use strict' + +const fs = require('node:fs') +const path = require('node:path') +const { resolveRequestedPathForIpc } = require('./hardening.cjs') + +function findGitRoot(start, fsImpl = fs) { + let dir = start + + for (let i = 0; i < 50; i += 1) { + try { + if (fsImpl.existsSync(path.join(dir, '.git'))) { + return dir + } + } catch { + return null + } + + const parent = path.dirname(dir) + + if (parent === dir) { + return null + } + + dir = parent + } + + return null +} + +async function gitRootForIpc(startPath, options = {}) { + const fsImpl = options.fs || fs + let resolved + + try { + resolved = resolveRequestedPathForIpc(startPath, { purpose: 'Git root' }) + } catch { + return null + } + + try { + const stat = await fsImpl.promises.stat(resolved) + const start = stat.isDirectory() ? resolved : path.dirname(resolved) + + return findGitRoot(start, fsImpl) + } catch { + return findGitRoot(resolved, fsImpl) + } +} + +module.exports = { + findGitRoot, + gitRootForIpc +} diff --git a/apps/desktop/electron/git-root.test.cjs b/apps/desktop/electron/git-root.test.cjs new file mode 100644 index 00000000000..ba649b259f3 --- /dev/null +++ b/apps/desktop/electron/git-root.test.cjs @@ -0,0 +1,40 @@ +'use strict' + +const assert = require('node:assert/strict') +const fs = require('node:fs') +const os = require('node:os') +const path = require('node:path') +const test = require('node:test') +const { pathToFileURL } = require('node:url') + +const { gitRootForIpc } = require('./git-root.cjs') + +function mkTmpDir() { + return fs.mkdtempSync(path.join(os.tmpdir(), 'hermes-git-root-')) +} + +test('gitRootForIpc returns null for invalid and device paths', async () => { + assert.equal(await gitRootForIpc(''), null) + assert.equal(await gitRootForIpc(' '), null) + assert.equal(await gitRootForIpc(null), null) + assert.equal(await gitRootForIpc('\\\\?\\C:\\secret'), null) + assert.equal(await gitRootForIpc('file:///%E0%A4%A'), null) +}) + +test('gitRootForIpc resolves directories files missing descendants and file URLs', async t => { + const root = mkTmpDir() + t.after(() => fs.rmSync(root, { recursive: true, force: true })) + + const gitDir = path.join(root, '.git') + const srcDir = path.join(root, 'src') + const filePath = path.join(srcDir, 'index.ts') + fs.mkdirSync(gitDir) + fs.mkdirSync(srcDir) + fs.writeFileSync(filePath, 'export {}\n', 'utf8') + + assert.equal(await gitRootForIpc(root), root) + assert.equal(await gitRootForIpc(srcDir), root) + assert.equal(await gitRootForIpc(filePath), root) + assert.equal(await gitRootForIpc(pathToFileURL(filePath).toString()), root) + assert.equal(await gitRootForIpc(path.join(srcDir, 'missing.ts')), root) +}) diff --git a/apps/desktop/electron/git-worktrees.cjs b/apps/desktop/electron/git-worktrees.cjs new file mode 100644 index 00000000000..570397b2c95 --- /dev/null +++ b/apps/desktop/electron/git-worktrees.cjs @@ -0,0 +1,174 @@ +'use strict' + +// Resolve git-worktree relationships for a set of session cwds, reading git's +// on-disk metadata directly (no `git` spawn per path): +// +// - A normal checkout has a `.git` DIRECTORY at its root → it's the main +// worktree; its repo root IS that directory's parent. +// - A linked worktree has a `.git` FILE: `gitdir: /.git/worktrees/`. +// That admin dir's `commondir` points back at the shared `/.git`, whose +// parent is the main repo root. +// +// Grouping by repoRoot therefore clusters a repo's main checkout with all of its +// linked worktrees, regardless of how the worktree directories are named. The +// branch (read from the worktree's own HEAD) gives each worktree a meaningful +// label. + +const fs = require('node:fs') +const path = require('node:path') +const { resolveRequestedPathForIpc } = require('./hardening.cjs') + +// Walk up from `start` to the nearest ancestor that carries a `.git` entry +// (file for a linked worktree, dir for the main checkout). Capped so a stray +// path can't loop forever. +function findGitHost(start, fsImpl) { + let dir = start + + for (let i = 0; i < 64; i += 1) { + const dotgit = path.join(dir, '.git') + + try { + if (fsImpl.existsSync(dotgit)) { + return dir + } + } catch { + return null + } + + const parent = path.dirname(dir) + + if (parent === dir) { + return null + } + + dir = parent + } + + return null +} + +function readBranch(gitDir, fsImpl) { + try { + const head = fsImpl.readFileSync(path.join(gitDir, 'HEAD'), 'utf8').trim() + const ref = head.match(/^ref:\s*refs\/heads\/(.+)$/) + + if (ref) { + return ref[1] + } + + // Detached HEAD: surface a short sha so the worktree still gets a label. + return /^[0-9a-f]{7,40}$/i.test(head) ? head.slice(0, 8) : null + } catch { + return null + } +} + +// Given the directory that owns the `.git` entry, resolve its worktree identity. +function resolveFromHost(host, fsImpl) { + const dotgit = path.join(host, '.git') + let stat + + try { + stat = fsImpl.statSync(dotgit) + } catch { + return null + } + + if (stat.isDirectory()) { + return { + repoRoot: host, + worktreeRoot: host, + isMainWorktree: true, + branch: readBranch(dotgit, fsImpl) + } + } + + // Linked worktree: `.git` is a file pointing at the admin dir. + let contents + + try { + contents = fsImpl.readFileSync(dotgit, 'utf8').trim() + } catch { + return null + } + + const match = contents.match(/^gitdir:\s*(.+)$/m) + + if (!match) { + return null + } + + const adminDir = path.resolve(host, match[1].trim()) + + // `commondir` resolves to the shared `/.git`; fall back to walking two + // levels up from `/.git/worktrees/` if it's missing. + let commonDir + + try { + const rel = fsImpl.readFileSync(path.join(adminDir, 'commondir'), 'utf8').trim() + commonDir = path.resolve(adminDir, rel) + } catch { + commonDir = path.dirname(path.dirname(adminDir)) + } + + return { + repoRoot: path.dirname(commonDir), + worktreeRoot: host, + isMainWorktree: false, + branch: readBranch(adminDir, fsImpl) + } +} + +function resolveWorktree(startPath, fsImpl = fs) { + let resolved + + try { + resolved = resolveRequestedPathForIpc(startPath, { purpose: 'Worktree lookup' }) + } catch { + return null + } + + let start = resolved + + try { + const stat = fsImpl.statSync(resolved) + + if (!stat.isDirectory()) { + start = path.dirname(resolved) + } + } catch { + return null + } + + const host = findGitHost(start, fsImpl) + + if (!host) { + return null + } + + return resolveFromHost(host, fsImpl) +} + +// Batch entry point for the renderer: maps each requested cwd to its worktree +// info (or null when it isn't inside a git checkout / can't be read). Dedupes so +// many sessions sharing a cwd cost one lookup. +async function worktreesForIpc(cwds, options = {}) { + const fsImpl = options.fs || fs + const list = Array.isArray(cwds) ? cwds : [] + const out = {} + + for (const cwd of list) { + if (typeof cwd !== 'string' || !cwd.trim() || cwd in out) { + continue + } + + out[cwd] = resolveWorktree(cwd, fsImpl) + } + + return out +} + +module.exports = { + resolveWorktree, + worktreesForIpc +} diff --git a/apps/desktop/electron/hardening.cjs b/apps/desktop/electron/hardening.cjs index 4ffdea051b5..7b568ec3d11 100644 --- a/apps/desktop/electron/hardening.cjs +++ b/apps/desktop/electron/hardening.cjs @@ -1,4 +1,5 @@ const fs = require('node:fs') +const os = require('node:os') const path = require('node:path') const { fileURLToPath } = require('node:url') @@ -106,71 +107,162 @@ function sensitiveFileBlockReason(filePath) { return null } -function resolveRequestedFilePath(filePath, baseDir = process.cwd(), purpose = 'File read') { - const raw = String(filePath || '').trim() +function ipcPathError(code, message) { + const error = new Error(message) + error.code = code + return error +} + +function rejectUnsafePathSyntax(filePath, purpose = 'File read') { + if (typeof filePath !== 'string') { + throw ipcPathError('invalid-path', `${purpose} failed: file path is required.`) + } + + const raw = filePath.trim() if (!raw) { - throw new Error(`${purpose} failed: file path is required.`) + throw ipcPathError('invalid-path', `${purpose} failed: file path is required.`) } if (raw.includes('\0')) { - throw new Error(`${purpose} failed: file path is invalid.`) + throw ipcPathError('invalid-path', `${purpose} failed: file path is invalid.`) + } + + const normalized = raw.replace(/\\/g, '/').toLowerCase() + if ( + normalized.startsWith('//?/') || + normalized.startsWith('//./') || + normalized.startsWith('globalroot/device/') || + normalized.includes('/globalroot/device/') + ) { + throw ipcPathError('device-path', `${purpose} blocked: Windows device paths are not allowed.`) + } + + return raw +} + +function resolveRequestedPathForIpc(filePath, options = {}) { + const purpose = String(options.purpose || 'File read') + let raw = rejectUnsafePathSyntax(filePath, purpose) + + // Gateway-reported cwds (config `terminal.cwd`, remote sessions) routinely + // arrive as `~/...`. Node's fs has no shell — without expansion the path + // resolves under process.cwd() and every read "ENOENT"s forever. + if (raw === '~' || raw.startsWith('~/') || raw.startsWith('~\\')) { + raw = path.join(os.homedir(), raw.slice(1)) } if (/^file:/i.test(raw)) { + let resolvedPath try { - return fileURLToPath(raw) + const parsed = new URL(raw) + if (parsed.protocol !== 'file:') { + throw new Error('not a file URL') + } + resolvedPath = fileURLToPath(parsed) } catch { - throw new Error(`${purpose} failed: file URL is invalid.`) + throw ipcPathError('invalid-path', `${purpose} failed: file URL is invalid.`) } + + rejectUnsafePathSyntax(resolvedPath, purpose) + return path.resolve(resolvedPath) } - const resolvedBase = path.resolve(String(baseDir || process.cwd())) - return path.resolve(resolvedBase, raw) + const baseInput = typeof options.baseDir === 'string' && options.baseDir.trim() ? options.baseDir : process.cwd() + const safeBaseInput = rejectUnsafePathSyntax(baseInput, purpose) + const resolvedBase = path.resolve(safeBaseInput) + rejectUnsafePathSyntax(resolvedBase, purpose) + const resolvedPath = path.resolve(resolvedBase, raw) + rejectUnsafePathSyntax(resolvedPath, purpose) + + return resolvedPath +} + +async function statForIpc(fsImpl, resolvedPath, purpose, typeLabel) { + try { + return await fsImpl.promises.stat(resolvedPath) + } catch (error) { + const code = error && typeof error === 'object' ? error.code : '' + if (code === 'ENOENT' || code === 'ENOTDIR') { + throw ipcPathError(code || 'ENOENT', `${purpose} failed: ${typeLabel} does not exist.`) + } + throw ipcPathError(code || 'read-error', `${purpose} failed: ${error instanceof Error ? error.message : String(error)}`) + } +} + +async function realpathForIpc(fsImpl, resolvedPath, purpose) { + if (typeof fsImpl.promises.realpath !== 'function') { + return resolvedPath + } + + try { + const realPath = await fsImpl.promises.realpath(resolvedPath) + rejectUnsafePathSyntax(realPath, purpose) + return realPath + } catch (error) { + const code = error && typeof error === 'object' ? error.code : '' + throw ipcPathError(code || 'read-error', `${purpose} failed: ${error instanceof Error ? error.message : String(error)}`) + } +} + +function rejectSensitiveFilePath(filePath, purpose) { + const blockReason = sensitiveFileBlockReason(filePath) + if (blockReason) { + throw ipcPathError('sensitive-file', `${purpose} blocked for sensitive file: ${blockReason}`) + } +} + +async function resolveDirectoryForIpc(dirPath, options = {}) { + const purpose = String(options.purpose || 'Directory read') + const fsImpl = options.fs || fs + const resolvedPath = resolveRequestedPathForIpc(dirPath, { baseDir: options.baseDir, purpose }) + const stat = await statForIpc(fsImpl, resolvedPath, purpose, 'directory') + + if (!stat.isDirectory()) { + throw ipcPathError('ENOTDIR', `${purpose} failed: path is not a directory.`) + } + + const realPath = await realpathForIpc(fsImpl, resolvedPath, purpose) + + return { realPath, resolvedPath, stat } } async function resolveReadableFileForIpc(filePath, options = {}) { const purpose = String(options.purpose || 'File read') - const resolvedPath = resolveRequestedFilePath(filePath, options.baseDir, purpose) + const fsImpl = options.fs || fs + const resolvedPath = resolveRequestedPathForIpc(filePath, { baseDir: options.baseDir, purpose }) if (options.blockSensitive !== false) { - const blockReason = sensitiveFileBlockReason(resolvedPath) - if (blockReason) { - throw new Error(`${purpose} blocked for sensitive file: ${blockReason}`) - } + rejectSensitiveFilePath(resolvedPath, purpose) } - let stat - try { - stat = await fs.promises.stat(resolvedPath) - } catch (error) { - const code = error && typeof error === 'object' ? error.code : '' - if (code === 'ENOENT' || code === 'ENOTDIR') { - throw new Error(`${purpose} failed: file does not exist.`) - } - throw new Error(`${purpose} failed: ${error instanceof Error ? error.message : String(error)}`) - } + const stat = await statForIpc(fsImpl, resolvedPath, purpose, 'file') if (stat.isDirectory()) { - throw new Error(`${purpose} failed: path points to a directory.`) + throw ipcPathError('EISDIR', `${purpose} failed: path points to a directory.`) } if (!stat.isFile()) { - throw new Error(`${purpose} failed: only regular files can be read.`) + throw ipcPathError('EINVAL', `${purpose} failed: only regular files can be read.`) + } + + const realPath = await realpathForIpc(fsImpl, resolvedPath, purpose) + if (options.blockSensitive !== false) { + rejectSensitiveFilePath(realPath, purpose) } const maxBytes = Number.isFinite(options.maxBytes) && Number(options.maxBytes) > 0 ? Number(options.maxBytes) : null if (maxBytes && stat.size > maxBytes) { - throw new Error(`${purpose} failed: file is too large (${stat.size} bytes; limit ${maxBytes} bytes).`) + throw ipcPathError('EFBIG', `${purpose} failed: file is too large (${stat.size} bytes; limit ${maxBytes} bytes).`) } try { - await fs.promises.access(resolvedPath, fs.constants.R_OK) + await fsImpl.promises.access(resolvedPath, fs.constants.R_OK) } catch { - throw new Error(`${purpose} failed: file is not readable.`) + throw ipcPathError('EACCES', `${purpose} failed: file is not readable.`) } - return { resolvedPath, stat } + return { realPath, resolvedPath, stat } } module.exports = { @@ -178,7 +270,10 @@ module.exports = { DEFAULT_FETCH_TIMEOUT_MS, TEXT_PREVIEW_SOURCE_MAX_BYTES, encryptDesktopSecret, + rejectUnsafePathSyntax, + resolveDirectoryForIpc, resolveReadableFileForIpc, + resolveRequestedPathForIpc, resolveTimeoutMs, sensitiveFileBlockReason } diff --git a/apps/desktop/electron/hardening.test.cjs b/apps/desktop/electron/hardening.test.cjs index 865da8fe797..b38a03b0082 100644 --- a/apps/desktop/electron/hardening.test.cjs +++ b/apps/desktop/electron/hardening.test.cjs @@ -8,11 +8,20 @@ const { pathToFileURL } = require('node:url') const { DEFAULT_FETCH_TIMEOUT_MS, encryptDesktopSecret, + resolveDirectoryForIpc, resolveReadableFileForIpc, + resolveRequestedPathForIpc, resolveTimeoutMs, sensitiveFileBlockReason } = require('./hardening.cjs') +async function rejectsWithCode(promise, code) { + await assert.rejects(promise, error => { + assert.equal(error?.code, code) + return true + }) +} + test('resolveTimeoutMs falls back to defaults and accepts overrides', () => { assert.equal(resolveTimeoutMs(undefined), DEFAULT_FETCH_TIMEOUT_MS) assert.equal(resolveTimeoutMs(0), DEFAULT_FETCH_TIMEOUT_MS) @@ -51,6 +60,65 @@ test('sensitiveFileBlockReason blocks obvious secret file patterns', () => { assert.match(String(sensitiveFileBlockReason('/tmp/server-cert.pem')), /\.pem/) }) +test('path helpers reject blank non-string NUL and Windows device syntax', async () => { + await rejectsWithCode(resolveReadableFileForIpc('', { purpose: 'File preview' }), 'invalid-path') + await rejectsWithCode(resolveReadableFileForIpc(' ', { purpose: 'File preview' }), 'invalid-path') + await rejectsWithCode(resolveReadableFileForIpc(null, { purpose: 'File preview' }), 'invalid-path') + await rejectsWithCode(resolveReadableFileForIpc(`safe${String.fromCharCode(0)}name.txt`), 'invalid-path') + + const devicePaths = [ + '\\\\?\\C:\\secret.txt', + '\\\\.\\C:\\secret.txt', + '\\\\?\\UNC\\server\\share\\secret.txt', + 'GLOBALROOT/Device/HarddiskVolumeShadowCopy1/secret.txt' + ] + + for (const devicePath of devicePaths) { + assert.throws( + () => resolveRequestedPathForIpc(devicePath, { purpose: 'File preview' }), + error => { + assert.equal(error?.code, 'device-path') + return true + } + ) + await rejectsWithCode(resolveReadableFileForIpc(devicePath, { purpose: 'File preview' }), 'device-path') + } + + assert.throws( + () => resolveRequestedPathForIpc('file:///%E0%A4%A', { purpose: 'File preview' }), + error => { + assert.equal(error?.code, 'invalid-path') + return true + } + ) + await rejectsWithCode(resolveReadableFileForIpc('file:///%E0%A4%A', { purpose: 'File preview' }), 'invalid-path') +}) + +test('resolveRequestedPathForIpc resolves relative paths from the trimmed base directory', () => { + const baseDir = path.join(os.tmpdir(), 'hermes-desktop-base') + + assert.equal( + resolveRequestedPathForIpc('notes.txt', { + baseDir: ` ${baseDir} `, + purpose: 'File preview' + }), + path.resolve(baseDir, 'notes.txt') + ) +}) + +test('resolveRequestedPathForIpc expands ~ to the home directory', () => { + assert.equal(resolveRequestedPathForIpc('~', { purpose: 'Directory read' }), path.resolve(os.homedir())) + assert.equal( + resolveRequestedPathForIpc('~/www/project', { purpose: 'Directory read' }), + path.resolve(os.homedir(), 'www/project') + ) + // `~user` shorthand is NOT expanded — only the caller's own home. + assert.equal( + resolveRequestedPathForIpc('~other/secret', { baseDir: os.tmpdir(), purpose: 'Directory read' }), + path.resolve(os.tmpdir(), '~other/secret') + ) +}) + test('resolveReadableFileForIpc validates existence type size and sensitivity', async t => { const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'hermes-desktop-hardening-')) t.after(() => fs.rmSync(tempDir, { recursive: true, force: true })) @@ -71,6 +139,13 @@ test('resolveReadableFileForIpc validates existence type size and sensitivity', }) assert.equal(fromFileUrl.resolvedPath, textPath) + const spacedPath = path.join(tempDir, 'notes with spaces.txt') + fs.writeFileSync(spacedPath, 'space ok', 'utf8') + const fromSpacedFileUrl = await resolveReadableFileForIpc(pathToFileURL(spacedPath).toString(), { + purpose: 'File preview' + }) + assert.equal(fromSpacedFileUrl.resolvedPath, spacedPath) + await assert.rejects( resolveReadableFileForIpc('missing.txt', { baseDir: tempDir, @@ -114,3 +189,91 @@ test('resolveReadableFileForIpc validates existence type size and sensitivity', }) assert.equal(envTemplate.resolvedPath, envTemplatePath) }) + +test('resolveReadableFileForIpc blocks common sensitive files', async t => { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'hermes-desktop-sensitive-')) + t.after(() => fs.rmSync(tempDir, { recursive: true, force: true })) + + const sshDir = path.join(tempDir, '.ssh') + fs.mkdirSync(sshDir) + + const blockedFiles = [ + path.join(tempDir, '.env'), + path.join(tempDir, '.npmrc'), + path.join(sshDir, 'id_ed25519'), + path.join(tempDir, 'cert.pem'), + path.join(tempDir, 'cert.p12'), + path.join(tempDir, 'cert.pfx') + ] + + for (const filePath of blockedFiles) { + fs.writeFileSync(filePath, 'secret', 'utf8') + await rejectsWithCode(resolveReadableFileForIpc(filePath, { purpose: 'File preview' }), 'sensitive-file') + } + + const allowed = path.join(tempDir, '.env.example') + fs.writeFileSync(allowed, 'EXAMPLE_TOKEN=value', 'utf8') + assert.equal((await resolveReadableFileForIpc(allowed, { purpose: 'File preview' })).resolvedPath, allowed) +}) + +test('resolveReadableFileForIpc blocks symlinks whose realpath is sensitive', async t => { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'hermes-desktop-realpath-')) + t.after(() => fs.rmSync(tempDir, { recursive: true, force: true })) + + const envPath = path.join(tempDir, '.env') + const linkPath = path.join(tempDir, 'safe-name.txt') + fs.writeFileSync(envPath, 'SECRET_TOKEN=123', 'utf8') + + try { + fs.symlinkSync(envPath, linkPath, 'file') + } catch (error) { + if (error?.code === 'EPERM' || error?.code === 'EACCES') { + t.skip(`symlink creation is not permitted on this platform (${error.code})`) + return + } + throw error + } + + await rejectsWithCode(resolveReadableFileForIpc(linkPath, { purpose: 'File preview' }), 'sensitive-file') +}) + +test('resolveDirectoryForIpc accepts directories and rejects invalid directory targets', async t => { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'hermes-desktop-dir-')) + t.after(() => fs.rmSync(tempDir, { recursive: true, force: true })) + + const directory = path.join(tempDir, 'project') + const filePath = path.join(tempDir, 'file.txt') + fs.mkdirSync(directory) + fs.writeFileSync(filePath, 'not a directory', 'utf8') + + const resolved = await resolveDirectoryForIpc(directory) + assert.equal(resolved.resolvedPath, directory) + assert.equal(resolved.stat.isDirectory(), true) + + await rejectsWithCode(resolveDirectoryForIpc(filePath), 'ENOTDIR') + await rejectsWithCode(resolveDirectoryForIpc(path.join(tempDir, 'missing')), 'ENOENT') + await rejectsWithCode(resolveDirectoryForIpc('\\\\?\\C:\\secret'), 'device-path') +}) + +test('resolveDirectoryForIpc accepts directory symlinks or junctions', async t => { + const tempDir = fs.mkdtempSync(path.join(os.tmpdir(), 'hermes-desktop-dir-link-')) + t.after(() => fs.rmSync(tempDir, { recursive: true, force: true })) + + const directory = path.join(tempDir, 'actual-project') + const linkPath = path.join(tempDir, 'linked-project') + fs.mkdirSync(directory) + + try { + fs.symlinkSync(directory, linkPath, process.platform === 'win32' ? 'junction' : 'dir') + } catch (error) { + if (error?.code === 'EPERM' || error?.code === 'EACCES') { + t.skip(`directory symlink creation is not permitted on this platform (${error.code})`) + return + } + throw error + } + + const resolved = await resolveDirectoryForIpc(linkPath) + assert.equal(resolved.resolvedPath, linkPath) + assert.equal(resolved.stat.isDirectory(), true) +}) diff --git a/apps/desktop/electron/main.cjs b/apps/desktop/electron/main.cjs index 5e128421a83..c714a46ee46 100644 --- a/apps/desktop/electron/main.cjs +++ b/apps/desktop/electron/main.cjs @@ -22,15 +22,27 @@ const http = require('node:http') const https = require('node:https') const net = require('node:net') const path = require('node:path') -const { fileURLToPath, pathToFileURL } = require('node:url') +const { pathToFileURL } = require('node:url') const { execFileSync, spawn } = require('node:child_process') const { detectRemoteDisplay, isWindowsBinaryPathInWsl, isWslEnvironment } = require('./bootstrap-platform.cjs') const { runBootstrap } = require('./bootstrap-runner.cjs') -const { buildSessionWindowUrl, createSessionWindowRegistry } = require('./session-windows.cjs') +const { + buildSessionWindowUrl, + createSessionWindowRegistry, + SESSION_WINDOW_MIN_HEIGHT, + SESSION_WINDOW_MIN_WIDTH +} = require('./session-windows.cjs') const { canImportHermesCli, verifyHermesCli } = require('./backend-probes.cjs') const { probeGatewayWebSocket } = require('./gateway-ws-probe.cjs') +const { adoptServedDashboardToken } = require('./dashboard-token.cjs') +const { waitForDashboardPort } = require('./backend-ready.cjs') const { serializeJsonBody, setJsonRequestHeaders } = require('./oauth-net-request.cjs') const { fetchMarketplaceThemes, searchMarketplaceThemes } = require('./vscode-marketplace.cjs') +const { buildDesktopBackendEnv, normalizeHermesHomeRoot } = require('./backend-env.cjs') +const { readDirForIpc } = require('./fs-read-dir.cjs') +const { gitRootForIpc } = require('./git-root.cjs') +const { worktreesForIpc } = require('./git-worktrees.cjs') +const { OFFICIAL_REPO_HTTPS_URL, isOfficialSshRemote } = require('./update-remote.cjs') const { buildPosixCleanupScript, buildWindowsCleanupScript, @@ -61,6 +73,7 @@ const { TEXT_PREVIEW_SOURCE_MAX_BYTES, encryptDesktopSecret: encryptDesktopSecretStrict, resolveReadableFileForIpc, + resolveRequestedPathForIpc, resolveTimeoutMs } = require('./hardening.cjs') @@ -86,6 +99,7 @@ try { nodePty = require(nodePtyDir) } } catch { + console.log(`[terminal] failed to load node-pty from path ${nodePtyDir}`) nodePty = null nodePtyDir = null } @@ -98,8 +112,6 @@ if (USER_DATA_OVERRIDE) { app.setPath('userData', resolvedUserData) } -const PORT_FLOOR = 9120 -const PORT_CEILING = 9199 const DEV_SERVER = process.env.HERMES_DESKTOP_DEV_SERVER const IS_PACKAGED = app.isPackaged const IS_MAC = process.platform === 'darwin' @@ -228,7 +240,7 @@ if (INSTALL_STAMP) { // HERMES_HOME beneath the throwaway userData dir so a fresh-install run never // touches the user's real ~/.hermes / %LOCALAPPDATA%\hermes. function resolveHermesHome() { - if (process.env.HERMES_HOME) return path.resolve(process.env.HERMES_HOME) + if (process.env.HERMES_HOME) return normalizeHermesHomeRoot(process.env.HERMES_HOME) if (USER_DATA_OVERRIDE) return path.join(path.resolve(USER_DATA_OVERRIDE), 'hermes-home') if (IS_WINDOWS && process.env.LOCALAPPDATA) { const localappdata = path.join(process.env.LOCALAPPDATA, 'hermes') @@ -333,10 +345,110 @@ const APP_ICON_PATHS = [ let rendererTitleBarTheme = null const terminalSessions = new Map() +// Force the NATIVE window appearance (vibrancy material, titlebar, the +// pre-first-paint window background) to follow the APP theme instead of the +// OS appearance. With `vibrancy` set, macOS paints an NSVisualEffectView that +// tracks the window's effective appearance and ignores `backgroundColor` — +// so a dark-themed app on a light-mode Mac flashes a white material on every +// new window until the renderer covers it. The renderer reports its mode via +// 'hermes:native-theme' ('dark' | 'light' | 'system'); we pin +// nativeTheme.themeSource to it and persist the value so cold launches paint +// correctly before the renderer has even loaded. +const NATIVE_THEME_CONFIG_PATH = path.join(app.getPath('userData'), 'native-theme.json') +const THEME_SOURCES = new Set(['dark', 'light', 'system']) + +function readPersistedThemeSource() { + try { + const parsed = JSON.parse(fs.readFileSync(NATIVE_THEME_CONFIG_PATH, 'utf8')) + + if (parsed && THEME_SOURCES.has(parsed.themeSource)) { + return parsed.themeSource + } + } catch { + // Missing / malformed → follow the OS like a fresh install. + } + + return 'system' +} + +function writePersistedThemeSource(mode) { + try { + fs.mkdirSync(path.dirname(NATIVE_THEME_CONFIG_PATH), { recursive: true }) + fs.writeFileSync(NATIVE_THEME_CONFIG_PATH, JSON.stringify({ themeSource: mode }, null, 2), 'utf8') + } catch (error) { + rememberLog(`[theme] write native theme failed: ${error.message}`) + } +} + +nativeTheme.themeSource = readPersistedThemeSource() + +// Window translucency (see-through window). One lever, 0–100; 0 = off (the +// default). Mapped to the native window opacity so the desktop shows through +// the whole window. Persisted so a cold launch applies it at window creation, +// before the renderer reports its value. macOS + Windows only; `setOpacity` is +// a no-op on Linux. See store/translucency. +const TRANSLUCENCY_CONFIG_PATH = path.join(app.getPath('userData'), 'translucency.json') + +function clampIntensity(value) { + const n = Math.round(Number(value)) + + return Number.isFinite(n) ? Math.min(100, Math.max(0, n)) : 0 +} + +function readPersistedTranslucency() { + try { + return clampIntensity(JSON.parse(fs.readFileSync(TRANSLUCENCY_CONFIG_PATH, 'utf8')).intensity) + } catch { + return 0 + } +} + +function writePersistedTranslucency(intensity) { + try { + fs.mkdirSync(path.dirname(TRANSLUCENCY_CONFIG_PATH), { recursive: true }) + fs.writeFileSync(TRANSLUCENCY_CONFIG_PATH, JSON.stringify({ intensity }, null, 2), 'utf8') + } catch (error) { + rememberLog(`[translucency] write failed: ${error.message}`) + } +} + +let translucencyIntensity = readPersistedTranslucency() + +// Map the 0–100 lever to a window opacity. Floor at 0.3 so the most see-through +// setting is still usable rather than nearly invisible. 0 → fully opaque. +function windowOpacity() { + return 1 - (translucencyIntensity / 100) * 0.7 +} + +// Re-apply translucency to a live window (runtime toggle, no recreation). +// `setOpacity` is a no-op on Linux, which is fine — it just stays opaque there. +function applyWindowTranslucency(win) { + if (!win || win.isDestroyed() || typeof win.setOpacity !== 'function') { + return + } + + try { + win.setOpacity(windowOpacity()) + } catch (error) { + rememberLog(`[translucency] apply failed: ${error.message}`) + } +} + function isHexColor(value) { return typeof value === 'string' && /^#[0-9a-f]{6}$/i.test(value) } +// Background color to paint a window with BEFORE its renderer loads, so a new +// (or reopened) window doesn't flash white/light in dark mode. Prefer the theme +// the renderer last reported; fall back to the OS preference on first launch. +function getWindowBackgroundColor() { + if (rendererTitleBarTheme && isHexColor(rendererTitleBarTheme.background)) { + return rendererTitleBarTheme.background + } + + return nativeTheme.shouldUseDarkColors ? '#111111' : '#f7f7f7' +} + function getTitleBarOverlayOptions() { if (IS_MAC) { return { height: TITLEBAR_HEIGHT } @@ -726,7 +838,7 @@ function openExternalUrl(rawUrl) { if (parsed.protocol === 'file:') { let localPath try { - localPath = fileURLToPath(parsed.toString()) + localPath = resolveRequestedPathForIpc(parsed.toString(), { purpose: 'Open external file' }) } catch { return false } @@ -1149,10 +1261,14 @@ function findSystemPython() { if (pyExe) { for (const version of SUPPORTED_VERSIONS) { try { - const out = execFileSync(pyExe, [`-${version}`, '-c', 'import sys; print(sys.executable)'], hiddenWindowsChildOptions({ - encoding: 'utf8', - stdio: ['ignore', 'pipe', 'ignore'] - })) + const out = execFileSync( + pyExe, + [`-${version}`, '-c', 'import sys; print(sys.executable)'], + hiddenWindowsChildOptions({ + encoding: 'utf8', + stdio: ['ignore', 'pipe', 'ignore'] + }) + ) const candidate = out.trim() if (candidate && fileExists(candidate)) return candidate } catch { @@ -1287,11 +1403,15 @@ function resolveUpdateRoot() { function runGit(args, options = {}) { return new Promise((resolve, reject) => { - const child = spawn(resolveGitBinary(), IS_WINDOWS ? ['-c', 'windows.appendAtomically=false', ...args] : args, hiddenWindowsChildOptions({ - cwd: options.cwd, - env: { ...process.env, ...(options.env || {}), GIT_TERMINAL_PROMPT: '0' }, - stdio: ['ignore', 'pipe', 'pipe'] - })) + const child = spawn( + resolveGitBinary(), + IS_WINDOWS ? ['-c', 'windows.appendAtomically=false', ...args] : args, + hiddenWindowsChildOptions({ + cwd: options.cwd, + env: { ...process.env, ...(options.env || {}), GIT_TERMINAL_PROMPT: '0' }, + stdio: ['ignore', 'pipe', 'pipe'] + }) + ) let stdout = '' let stderr = '' @@ -1312,6 +1432,11 @@ function runGit(args, options = {}) { const firstLine = text => (text || '').split('\n').find(Boolean) || '' +async function getOriginUrl(updateRoot) { + const origin = await runGit(['remote', 'get-url', 'origin'], { cwd: updateRoot }) + return origin.code === 0 ? origin.stdout.trim() : '' +} + function emitUpdateProgress(payload) { const merged = { stage: 'idle', message: '', percent: null, error: null, ...payload, at: Date.now() } rememberLog(`[updates] ${merged.stage}: ${merged.message || merged.error || ''}`) @@ -1331,7 +1456,9 @@ async function resolveHealedBranch(updateRoot, branch) { return branch || 'main' } - const probe = await runGit(['ls-remote', '--exit-code', '--heads', 'origin', branch], { cwd: updateRoot }) + const originUrl = await getOriginUrl(updateRoot) + const remote = isOfficialSshRemote(originUrl) ? OFFICIAL_REPO_HTTPS_URL : 'origin' + const probe = await runGit(['ls-remote', '--exit-code', '--heads', remote, branch], { cwd: updateRoot }) if (probe.code !== 2) { return branch } @@ -1359,6 +1486,40 @@ async function checkUpdates() { } branch = await resolveHealedBranch(updateRoot, branch) + const originUrl = await getOriginUrl(updateRoot) + if (isOfficialSshRemote(originUrl)) { + const git = args => runGit(args, { cwd: updateRoot }).then(r => r.stdout.trim()) + const [currentSha, target, dirtyStr, currentBranch] = await Promise.all([ + git(['rev-parse', 'HEAD']), + runGit(['ls-remote', OFFICIAL_REPO_HTTPS_URL, `refs/heads/${branch}`], { cwd: updateRoot }), + git(['status', '--porcelain']), + git(['rev-parse', '--abbrev-ref', 'HEAD']) + ]) + const targetSha = firstLine(target.stdout).split(/\s+/)[0] || '' + if (target.code !== 0 || !targetSha) { + return { + supported: true, + branch, + error: 'fetch-failed', + message: firstLine(target.stderr) || 'git ls-remote failed.', + hermesRoot: updateRoot, + fetchedAt: Date.now() + } + } + return { + supported: true, + branch, + currentBranch, + behind: currentSha && currentSha === targetSha ? 0 : 1, + currentSha, + targetSha, + commits: [], + dirty: dirtyStr.length > 0, + hermesRoot: updateRoot, + fetchedAt: Date.now() + } + } + const fetched = await runGit(['fetch', '--quiet', 'origin', branch], { cwd: updateRoot }) if (fetched.code !== 0) { return { @@ -1674,6 +1835,44 @@ async function applyUpdates(opts = {}) { } } +async function handOffWindowsBootstrapRecovery(reason) { + if (!IS_WINDOWS || !IS_PACKAGED) return false + + const updater = resolveUpdaterBinary() + if (!updater) return false + + const updateRoot = resolveUpdateRoot() + const { branch: configuredBranch } = readDesktopUpdateConfig() + const branch = directoryExists(path.join(updateRoot, '.git')) + ? await resolveHealedBranch(updateRoot, configuredBranch || DEFAULT_UPDATE_BRANCH) + : configuredBranch || DEFAULT_UPDATE_BRANCH + const venvBin = path.join(updateRoot, 'venv', IS_WINDOWS ? 'Scripts' : 'bin') + const venvHermes = path.join(venvBin, IS_WINDOWS ? 'hermes.exe' : 'hermes') + const updaterArgs = fileExists(venvHermes) ? ['--update', '--branch', branch] : ['--repair', '--branch', branch] + + await releaseBackendLockForUpdate(updateRoot) + + const child = spawn(updater, updaterArgs, { + cwd: HERMES_HOME, + env: { + ...process.env, + HERMES_HOME, + PATH: [path.join(HERMES_HOME, 'node', 'bin'), venvBin, process.env.PATH].filter(Boolean).join(path.delimiter) + }, + detached: true, + stdio: 'ignore', + windowsHide: false + }) + child.unref() + + rememberLog(`[bootstrap] handed off ${reason} recovery to updater: ${updater} ${updaterArgs.join(' ')}; exiting desktop to release app.asar`) + setTimeout(() => { + app.quit() + }, 600) + + return true +} + // Resolve the hermes CLI to drive an in-app update: prefer the venv shim in // the install we're updating, fall back to `hermes` on PATH. function resolveHermesCliBinary(updateRoot) { @@ -1687,11 +1886,15 @@ function runStreamedUpdate(command, args, { cwd, env, stage } = {}) { return new Promise(resolve => { let child try { - child = spawn(command, args, hiddenWindowsChildOptions({ - cwd, - env: { ...process.env, ...(env || {}) }, - stdio: ['ignore', 'pipe', 'pipe'] - })) + child = spawn( + command, + args, + hiddenWindowsChildOptions({ + cwd, + env: { ...process.env, ...(env || {}) }, + stdio: ['ignore', 'pipe', 'pipe'] + }) + ) } catch (err) { resolve({ code: 1, error: err.message }) return @@ -2079,9 +2282,11 @@ function createPythonBackend(root, label, dashboardArgs, options = {}) { label, command: python, args: ['-m', 'hermes_cli.main', ...dashboardArgs], - env: { - PYTHONPATH: [root, process.env.PYTHONPATH].filter(Boolean).join(path.delimiter) - }, + env: buildDesktopBackendEnv({ + hermesHome: HERMES_HOME, + pythonPathEntries: [root], + venvRoot: path.join(root, 'venv') + }), root, bootstrap: Boolean(options.bootstrap), shell: false @@ -2100,9 +2305,11 @@ function createActiveBackend(dashboardArgs) { label: `Hermes at ${ACTIVE_HERMES_ROOT}`, command: fileExists(venvPython) ? venvPython : findSystemPython(), args: ['-m', 'hermes_cli.main', ...dashboardArgs], - env: { - PYTHONPATH: [ACTIVE_HERMES_ROOT, process.env.PYTHONPATH].filter(Boolean).join(path.delimiter) - }, + env: buildDesktopBackendEnv({ + hermesHome: HERMES_HOME, + pythonPathEntries: [ACTIVE_HERMES_ROOT], + venvRoot: VENV_ROOT + }), root: ACTIVE_HERMES_ROOT, bootstrap: true, shell: false @@ -2263,6 +2470,14 @@ async function ensureRuntime(backend) { if (backend.kind === 'bootstrap-needed') { rememberLog('[bootstrap] no Hermes install found; starting first-launch bootstrap') + if (await handOffWindowsBootstrapRecovery('bootstrap-needed')) { + const handoffError = new Error('Hermes recovery was handed off to Hermes Setup. The desktop will restart when recovery completes.') + handoffError.isBootstrapFailure = true + handoffError.bootstrapHandedOff = true + bootstrapFailure = handoffError + throw handoffError + } + // Eagerly flip the bootstrap UI state to 'active' so the renderer // shows the install overlay BEFORE the runner finishes fetching the // manifest (which on slow networks can take tens of seconds and would @@ -2392,23 +2607,6 @@ async function ensureRuntime(backend) { return backend } -function isPortAvailable(port) { - return new Promise(resolve => { - const server = net.createServer() - server.once('error', () => resolve(false)) - server.once('listening', () => { - server.close(() => resolve(true)) - }) - server.listen(port, '127.0.0.1') - }) -} - -async function pickPort() { - for (let port = PORT_FLOOR; port <= PORT_CEILING; port += 1) { - if (await isPortAvailable(port)) return port - } - throw new Error(`No free localhost port in ${PORT_FLOOR}-${PORT_CEILING}`) -} function fetchJson(url, token, options = {}) { return new Promise((resolve, reject) => { @@ -2833,10 +3031,10 @@ async function resourceBufferFromUrl(rawUrl) { const buffer = match[2] ? Buffer.from(encoded, 'base64') : Buffer.from(decodeURIComponent(encoded), 'utf8') return { buffer, mimeType } } - if (rawUrl.startsWith('file:')) { - const filePath = fileURLToPath(rawUrl) - const buffer = await fs.promises.readFile(filePath) - return { buffer, mimeType: mimeTypeForPath(filePath) } + if (/^file:/i.test(rawUrl)) { + const { resolvedPath } = await resolveReadableFileForIpc(rawUrl, { purpose: 'Image file' }) + const buffer = await fs.promises.readFile(resolvedPath) + return { buffer, mimeType: mimeTypeForPath(resolvedPath) } } const parsed = new URL(rawUrl) @@ -2914,11 +3112,13 @@ function expandUserPath(filePath) { return value } -function previewFileTarget(rawTarget, baseDir) { +async function previewFileTarget(rawTarget, baseDir) { const raw = String(rawTarget || '').trim() const base = baseDir ? path.resolve(expandUserPath(baseDir)) : resolveHermesCwd() - const filePath = raw.startsWith('file:') ? fileURLToPath(raw) : path.resolve(base, expandUserPath(raw)) - let resolved = filePath + let resolved = resolveRequestedPathForIpc(/^file:/i.test(raw) ? raw : expandUserPath(raw), { + baseDir: base, + purpose: 'Preview target' + }) if (directoryExists(resolved)) { resolved = path.join(resolved, 'index.html') @@ -2929,6 +3129,8 @@ function previewFileTarget(rawTarget, baseDir) { return null } + ;({ resolvedPath: resolved } = await resolveReadableFileForIpc(resolved, { purpose: 'Preview target' })) + const mimeType = mimeTypeForPath(resolved) const metadata = previewFileMetadata(resolved, mimeType) const isHtml = PREVIEW_HTML_EXTENSIONS.has(ext) @@ -2974,7 +3176,7 @@ function previewUrlTarget(rawTarget) { } } -function normalizePreviewTarget(rawTarget, baseDir) { +async function normalizePreviewTarget(rawTarget, baseDir) { const raw = String(rawTarget || '').trim() if (!raw) { @@ -2986,20 +3188,15 @@ function normalizePreviewTarget(rawTarget, baseDir) { return previewUrlTarget(raw) } - return previewFileTarget(raw, baseDir) + return await previewFileTarget(raw, baseDir) } catch { return null } } -function filePathFromPreviewUrl(rawUrl) { - const filePath = fileURLToPath(String(rawUrl || '')) - - if (!fileExists(filePath)) { - throw new Error('Preview file is not readable') - } - - return filePath +async function filePathFromPreviewUrl(rawUrl) { + const { resolvedPath } = await resolveReadableFileForIpc(String(rawUrl || ''), { purpose: 'Preview file' }) + return resolvedPath } function sendPreviewFileChanged(payload) { @@ -3009,8 +3206,8 @@ function sendPreviewFileChanged(payload) { webContents.send('hermes:preview-file-changed', payload) } -function watchPreviewFile(rawUrl) { - const filePath = filePathFromPreviewUrl(rawUrl) +async function watchPreviewFile(rawUrl) { + const filePath = await filePathFromPreviewUrl(rawUrl) const watchDir = path.dirname(filePath) const targetName = path.basename(filePath) const id = crypto.randomBytes(12).toString('base64url') @@ -4487,38 +4684,41 @@ async function spawnPoolBackend(profile, entry) { } } - const port = await pickPort() const token = crypto.randomBytes(32).toString('base64url') // --profile wins over the inherited HERMES_HOME env (see _apply_profile_override // step 3 in hermes_cli/main.py), so the child re-homes to this profile. - const dashboardArgs = ['--profile', profile, 'dashboard', '--no-open', '--host', '127.0.0.1', '--port', String(port)] + // --port 0: the OS assigns an ephemeral port; the child announces it on stdout. + const dashboardArgs = ['--profile', profile, 'dashboard', '--no-open', '--host', '127.0.0.1', '--port', '0'] const backend = await ensureRuntime(resolveHermesBackend(dashboardArgs)) const hermesCwd = resolveHermesCwd() const webDist = resolveWebDist() rememberLog(`Starting Hermes backend for profile "${profile}" via ${backend.label}`) - const child = spawn(backend.command, backend.args, hiddenWindowsChildOptions({ - cwd: hermesCwd, - env: { - ...process.env, - HERMES_HOME, - ...backend.env, - // Pin the gateway's tool/terminal cwd to the same directory we chose for - // the child process. Inherited TERMINAL_CWD (or a stale config bridge) - // can still point at the install dir even when spawn cwd is home. - TERMINAL_CWD: hermesCwd, - HERMES_DASHBOARD_SESSION_TOKEN: token, - // Marks this dashboard backend as desktop-spawned so it runs the cron - // scheduler tick loop (the gateway isn't running under the app). - HERMES_DESKTOP: '1', - HERMES_WEB_DIST: webDist - }, - shell: backend.shell, - stdio: ['ignore', 'pipe', 'pipe'] - })) + const child = spawn( + backend.command, + backend.args, + hiddenWindowsChildOptions({ + cwd: hermesCwd, + env: { + ...process.env, + HERMES_HOME, + ...backend.env, + // Pin the gateway's tool/terminal cwd to the same directory we chose for + // the child process. Inherited TERMINAL_CWD (or a stale config bridge) + // can still point at the install dir even when spawn cwd is home. + TERMINAL_CWD: hermesCwd, + HERMES_DASHBOARD_SESSION_TOKEN: token, + // Marks this dashboard backend as desktop-spawned so it runs the cron + // scheduler tick loop (the gateway isn't running under the app). + HERMES_DESKTOP: '1', + HERMES_WEB_DIST: webDist + }, + shell: backend.shell, + stdio: ['ignore', 'pipe', 'pipe'] + }) + ) entry.process = child - entry.port = port entry.token = token child.stdout.on('data', rememberLog) @@ -4544,18 +4744,28 @@ async function spawnPoolBackend(profile, entry) { } }) + // Discover the ephemeral port the child bound to + const port = await Promise.race([waitForDashboardPort(child), startFailed]) + entry.port = port + const baseUrl = `http://127.0.0.1:${port}` await Promise.race([waitForHermes(baseUrl, token), startFailed]) ready = true + const authToken = await adoptServedDashboardToken(baseUrl, token, { + childAlive: () => child.exitCode === null && !child.killed, + label: `Hermes backend for profile "${profile}"`, + rememberLog + }) + entry.token = authToken return { baseUrl, mode: 'local', source: 'local', authMode: 'token', - token, + token: authToken, profile, - wsUrl: `ws://127.0.0.1:${port}/api/ws?token=${encodeURIComponent(token)}`, + wsUrl: `ws://127.0.0.1:${port}/api/ws?token=${encodeURIComponent(authToken)}`, logs: hermesLog.slice(-80), ...getWindowState() } @@ -4677,10 +4887,9 @@ async function startHermes() { } } - await advanceBootProgress('backend.port', 'Finding an open local port', 16) - const port = await pickPort() const token = crypto.randomBytes(32).toString('base64url') - const dashboardArgs = ['dashboard', '--no-open', '--host', '127.0.0.1', '--port', String(port)] + // --port 0: the OS assigns an ephemeral port; the child announces it on stdout. + const dashboardArgs = ['dashboard', '--no-open', '--host', '127.0.0.1', '--port', '0'] // Pin the desktop's chosen profile via the global --profile flag. This is // deterministic (it wins over the sticky ~/.hermes/active_profile file) and // resolves HERMES_HOME the same way `hermes -p ` does on the CLI. An @@ -4698,30 +4907,34 @@ async function startHermes() { await advanceBootProgress('backend.spawn', `Starting Hermes backend via ${backend.label}`, 84) rememberLog(`Starting Hermes backend via ${backend.label}`) - hermesProcess = spawn(backend.command, backend.args, hiddenWindowsChildOptions({ - cwd: hermesCwd, - env: { - ...process.env, - // Explicitly pin HERMES_HOME for the child so Python's get_hermes_home() - // resolves to the SAME location our resolveHermesHome() picked. Without - // this pin, Python falls back to ~/.hermes on every platform — fine on - // mac/linux (where our default matches), but on Windows our default is - // %LOCALAPPDATA%\hermes, which differs from C:\Users\\.hermes. - // Mismatch would split config / sessions / .env / logs across two - // directories. install.ps1 sets HERMES_HOME via setx; the desktop - // can't reliably do that, so we set it inline for every spawn. - HERMES_HOME, - ...backend.env, - TERMINAL_CWD: hermesCwd, - HERMES_DASHBOARD_SESSION_TOKEN: token, - // Marks this dashboard backend as desktop-spawned so it runs the cron - // scheduler tick loop (the gateway isn't running under the app). - HERMES_DESKTOP: '1', - HERMES_WEB_DIST: webDist - }, - shell: backend.shell, - stdio: ['ignore', 'pipe', 'pipe'] - })) + hermesProcess = spawn( + backend.command, + backend.args, + hiddenWindowsChildOptions({ + cwd: hermesCwd, + env: { + ...process.env, + // Explicitly pin HERMES_HOME for the child so Python's get_hermes_home() + // resolves to the SAME location our resolveHermesHome() picked. Without + // this pin, Python falls back to ~/.hermes on every platform — fine on + // mac/linux (where our default matches), but on Windows our default is + // %LOCALAPPDATA%\hermes, which differs from C:\Users\\.hermes. + // Mismatch would split config / sessions / .env / logs across two + // directories. install.ps1 sets HERMES_HOME via setx; the desktop + // can't reliably do that, so we set it inline for every spawn. + HERMES_HOME, + ...backend.env, + TERMINAL_CWD: hermesCwd, + HERMES_DASHBOARD_SESSION_TOKEN: token, + // Marks this dashboard backend as desktop-spawned so it runs the cron + // scheduler tick loop (the gateway isn't running under the app). + HERMES_DESKTOP: '1', + HERMES_WEB_DIST: webDist + }, + shell: backend.shell, + stdio: ['ignore', 'pipe', 'pipe'] + }) + ) hermesProcess.stdout.on('data', rememberLog) hermesProcess.stderr.on('data', rememberLog) @@ -4770,10 +4983,19 @@ async function startHermes() { } }) + await advanceBootProgress('backend.port', 'Waiting for Hermes backend to launch', 86) + // Discover the ephemeral port the child bound to + const port = await Promise.race([waitForDashboardPort(hermesProcess), backendStartFailed]) + const baseUrl = `http://127.0.0.1:${port}` await advanceBootProgress('backend.wait', 'Waiting for Hermes backend to become ready', 90) await Promise.race([waitForHermes(baseUrl, token), backendStartFailed]) backendReady = true + const authToken = await adoptServedDashboardToken(baseUrl, token, { + // The exit/error handlers null hermesProcess when the child dies. + childAlive: () => hermesProcess !== null && hermesProcess.exitCode === null && !hermesProcess.killed, + rememberLog + }) updateBootProgress({ phase: 'backend.ready', message: 'Hermes backend is ready. Finalizing desktop startup', @@ -4787,8 +5009,8 @@ async function startHermes() { mode: 'local', source: 'local', authMode: 'token', - token, - wsUrl: `ws://127.0.0.1:${port}/api/ws?token=${encodeURIComponent(token)}`, + token: authToken, + wsUrl: `ws://127.0.0.1:${port}/api/ws?token=${encodeURIComponent(authToken)}`, logs: hermesLog.slice(-80), ...getWindowState() } @@ -4851,21 +5073,29 @@ function focusWindow(win) { } // Open (or focus) a standalone window for a single chat session. -function createSessionWindow(sessionId) { +function createSessionWindow(sessionId, { watch = false } = {}) { return sessionWindows.openOrFocus(sessionId, () => { const icon = getAppIconPath() const win = new BrowserWindow({ - width: 480, - height: 800, - minWidth: 420, - minHeight: 620, + width: SESSION_WINDOW_MIN_WIDTH, + height: SESSION_WINDOW_MIN_HEIGHT, + minWidth: SESSION_WINDOW_MIN_WIDTH, + minHeight: SESSION_WINDOW_MIN_HEIGHT, title: 'Hermes', titleBarStyle: 'hidden', titleBarOverlay: getTitleBarOverlayOptions(), trafficLightPosition: IS_MAC ? WINDOW_BUTTON_POSITION : undefined, vibrancy: IS_MAC ? 'sidebar' : undefined, + opacity: windowOpacity(), icon, - backgroundColor: '#f7f7f7', + // Don't show until the renderer's first themed paint is ready. macOS + // `vibrancy` ignores `backgroundColor` and paints a translucent OS + // material (which follows the OS appearance, not the app theme), so a + // dark-themed app on a light-mode Mac flashes white until the renderer + // covers it. ready-to-show fires after the boot-time paint in + // themes/context.tsx, so the window appears already themed. + show: false, + backgroundColor: getWindowBackgroundColor(), webPreferences: { preload: path.join(__dirname, 'preload.cjs'), contextIsolation: true, @@ -4880,6 +5110,10 @@ function createSessionWindow(sessionId) { win.setWindowButtonPosition?.(WINDOW_BUTTON_POSITION) } + win.once('ready-to-show', () => { + if (!win.isDestroyed()) win.show() + }) + win.on('will-enter-full-screen', () => sendWindowStateChanged(true)) win.on('enter-full-screen', () => sendWindowStateChanged(true)) win.on('will-leave-full-screen', () => sendWindowStateChanged(false)) @@ -4890,7 +5124,8 @@ function createSessionWindow(sessionId) { win.loadURL( buildSessionWindowUrl(sessionId, { devServer: DEV_SERVER, - rendererIndexPath: DEV_SERVER ? undefined : resolveRendererIndex() + rendererIndexPath: DEV_SERVER ? undefined : resolveRendererIndex(), + watch }) ) @@ -4916,8 +5151,13 @@ function createWindow() { titleBarOverlay: getTitleBarOverlayOptions(), trafficLightPosition: IS_MAC ? WINDOW_BUTTON_POSITION : undefined, vibrancy: IS_MAC ? 'sidebar' : undefined, + opacity: windowOpacity(), icon, - backgroundColor: '#f7f7f7', + // Hidden until the first themed paint so macOS `vibrancy` (which ignores + // `backgroundColor` and follows the OS appearance) can't flash a light + // material before the renderer paints the app theme. See createSessionWindow. + show: false, + backgroundColor: getWindowBackgroundColor(), webPreferences: { preload: path.join(__dirname, 'preload.cjs'), contextIsolation: true, @@ -4953,6 +5193,10 @@ function createWindow() { } } + mainWindow.once('ready-to-show', () => { + if (mainWindow && !mainWindow.isDestroyed()) mainWindow.show() + }) + mainWindow.on('will-enter-full-screen', () => sendWindowStateChanged(true)) mainWindow.on('enter-full-screen', () => sendWindowStateChanged(true)) mainWindow.on('will-leave-full-screen', () => sendWindowStateChanged(false)) @@ -5064,12 +5308,12 @@ ipcMain.handle('hermes:backend:touch', async (_event, profile) => { return { ok: true } }) ipcMain.handle('hermes:gateway:ws-url', async (_event, profile) => freshGatewayWsUrl(profile)) -ipcMain.handle('hermes:window:openSession', async (_event, sessionId) => { +ipcMain.handle('hermes:window:openSession', async (_event, sessionId, opts) => { if (typeof sessionId !== 'string' || !sessionId.trim()) { return { ok: false, error: 'invalid-session-id' } } - createSessionWindow(sessionId.trim()) + createSessionWindow(sessionId.trim(), { watch: opts?.watch === true }) return { ok: true } }) @@ -5078,8 +5322,8 @@ ipcMain.handle('hermes:bootstrap:reset', async () => { // reset connection state so the next startHermes() call restarts the // full backend flow (including a fresh runBootstrap pass). rememberLog('[bootstrap] reset requested by renderer; clearing latched failure') + await teardownPrimaryBackendAndWait() bootstrapFailure = null - connectionPromise = null bootstrapState = { active: false, manifest: null, @@ -5365,11 +5609,30 @@ ipcMain.handle('hermes:api', async (_event, request) => { ipcMain.handle('hermes:notify', (_event, payload) => { if (!Notification.isSupported()) return false - new Notification({ + // Action buttons render only on signed macOS builds; elsewhere they're dropped + // and the body click still works. + const actions = Array.isArray(payload?.actions) ? payload.actions : [] + const notification = new Notification({ title: payload?.title || 'Hermes', body: payload?.body || '', - silent: Boolean(payload?.silent) - }).show() + silent: Boolean(payload?.silent), + actions: actions.map(action => ({ type: 'button', text: String(action?.text || '') })) + }) + notification.on('click', () => { + if (!mainWindow || mainWindow.isDestroyed()) return + focusWindow(mainWindow) + if (payload?.sessionId) { + mainWindow.webContents.send('hermes:focus-session', payload.sessionId) + } + }) + notification.on('action', (_actionEvent, index) => { + if (!mainWindow || mainWindow.isDestroyed()) return + const action = actions[index] + if (action?.id) { + mainWindow.webContents.send('hermes:notification-action', { sessionId: payload?.sessionId, actionId: action.id }) + } + }) + notification.show() return true }) @@ -5477,6 +5740,35 @@ ipcMain.on('hermes:titlebar-theme', (_event, payload) => { mainWindow?.setTitleBarOverlay?.(getTitleBarOverlayOptions()) }) +// Pin the native appearance to the app theme (see NATIVE_THEME_CONFIG_PATH). +ipcMain.on('hermes:native-theme', (_event, mode) => { + if (!THEME_SOURCES.has(mode)) { + return + } + + if (nativeTheme.themeSource !== mode) { + nativeTheme.themeSource = mode + writePersistedThemeSource(mode) + } +}) + +// See-through window translucency. Persist + re-apply opacity to every open +// window at runtime (no recreation, so caching/sessions are untouched). +ipcMain.on('hermes:translucency', (_event, payload) => { + const next = clampIntensity(payload && payload.intensity) + + if (next === translucencyIntensity) { + return + } + + translucencyIntensity = next + writePersistedTranslucency(next) + + for (const win of BrowserWindow.getAllWindows()) { + applyWindowTranslucency(win) + } +}) + ipcMain.handle('hermes:openExternal', (_event, url) => { if (!openExternalUrl(url)) { throw new Error('Invalid external URL') @@ -5542,48 +5834,6 @@ ipcMain.handle('hermes:logs:reveal', async () => { ipcMain.handle('hermes:logs:recent', async () => ({ path: DESKTOP_LOG_PATH, lines: hermesLog.slice(-200) })) -// Always-hidden noise (covers non-git projects too — gitignore would catch -// these anyway when present, but we want the same hygiene without one). -const FS_READDIR_HIDDEN = new Set([ - '.git', - '.hg', - '.svn', - '.cache', - '.next', - '.turbo', - '.venv', - '__pycache__', - 'build', - 'dist', - 'node_modules', - 'target', - 'venv' -]) - -function findGitRoot(start) { - let dir = start - - for (let i = 0; i < 50; i += 1) { - try { - if (fs.existsSync(path.join(dir, '.git'))) { - return dir - } - } catch { - return null - } - - const parent = path.dirname(dir) - - if (parent === dir) { - return null - } - - dir = parent - } - - return null -} - function isExecutableFile(filePath) { if (!filePath || !path.isAbsolute(filePath)) { return false @@ -5766,46 +6016,11 @@ function disposeTerminalSession(id) { return true } -ipcMain.handle('hermes:fs:readDir', async (_event, dirPath) => { - const resolved = path.resolve(String(dirPath || '')) +ipcMain.handle('hermes:fs:readDir', async (_event, dirPath) => readDirForIpc(dirPath)) - if (!resolved) { - return { entries: [], error: 'invalid-path' } - } +ipcMain.handle('hermes:fs:gitRoot', async (_event, startPath) => gitRootForIpc(startPath)) - try { - const dirents = await fs.promises.readdir(resolved, { withFileTypes: true }) - - const entries = dirents - .filter(d => { - if (FS_READDIR_HIDDEN.has(d.name)) { - return false - } - - return true - }) - .map(d => ({ name: d.name, path: path.join(resolved, d.name), isDirectory: d.isDirectory() })) - .sort((a, b) => Number(b.isDirectory) - Number(a.isDirectory) || a.name.localeCompare(b.name)) - - return { entries } - } catch (error) { - return { entries: [], error: error?.code || 'read-error' } - } -}) - -ipcMain.handle('hermes:fs:gitRoot', async (_event, startPath) => { - const input = String(startPath || '') - const resolved = input.startsWith('file:') ? fileURLToPath(input) : path.resolve(input) - - try { - const stat = await fs.promises.stat(resolved) - const start = stat.isDirectory() ? resolved : path.dirname(resolved) - - return findGitRoot(start) - } catch { - return findGitRoot(resolved) - } -}) +ipcMain.handle('hermes:fs:worktrees', async (_event, cwds) => worktreesForIpc(cwds)) ipcMain.handle('hermes:terminal:start', async (event, payload = {}) => { if (!nodePty) { @@ -5993,11 +6208,15 @@ async function getUninstallSummary() { resolve(value) } try { - const child = spawn(py, ['-m', 'hermes_cli.main', 'uninstall', '--gui-summary'], hiddenWindowsChildOptions({ - cwd: agentRoot, - env: { ...process.env, HERMES_HOME, NO_COLOR: '1' }, - stdio: ['ignore', 'pipe', 'ignore'] - })) + const child = spawn( + py, + ['-m', 'hermes_cli.main', 'uninstall', '--gui-summary'], + hiddenWindowsChildOptions({ + cwd: agentRoot, + env: { ...process.env, HERMES_HOME, NO_COLOR: '1' }, + stdio: ['ignore', 'pipe', 'ignore'] + }) + ) child.stdout.on('data', chunk => { stdout += chunk.toString() }) @@ -6143,6 +6362,106 @@ ipcMain.handle('hermes:vscode-theme:fetch', async (_event, id) => fetchMarketpla // Search the Marketplace for color-theme extensions (empty query = top installs). ipcMain.handle('hermes:vscode-theme:search', async (_event, query) => searchMarketplaceThemes(String(query || ''), 20)) +// --------------------------------------------------------------------------- +// hermes:// deep links (e.g. hermes://blueprint/morning-brief?time=08:00). +// A docs/dashboard "Send to App" button opens this URL; we route it into the +// running app's chat composer. Three delivery paths: macOS 'open-url', +// Win/Linux running-app 'second-instance' (argv), Win/Linux cold-start argv. +// --------------------------------------------------------------------------- +const HERMES_PROTOCOL = 'hermes' +let _pendingDeepLink = null +let _rendererReadyForDeepLink = false + +function _extractDeepLink(argv) { + if (!Array.isArray(argv)) return null + return argv.find(a => typeof a === 'string' && a.startsWith(`${HERMES_PROTOCOL}://`)) || null +} + +function handleDeepLink(url) { + if (!url || typeof url !== 'string') return + let parsed + try { + parsed = new URL(url) + } catch { + rememberLog(`[deeplink] ignoring malformed url: ${url}`) + return + } + // hermes://blueprint/?slot=val -> host="blueprint", path="/" + const kind = parsed.hostname || '' + const name = decodeURIComponent((parsed.pathname || '').replace(/^\//, '')) + const params = {} + parsed.searchParams.forEach((v, k) => { + params[k] = v + }) + const payload = { kind, name, params } + + if (!_rendererReadyForDeepLink || !mainWindow || mainWindow.isDestroyed()) { + _pendingDeepLink = payload + return + } + try { + if (mainWindow.isMinimized()) mainWindow.restore() + mainWindow.focus() + mainWindow.webContents.send('hermes:deep-link', payload) + rememberLog(`[deeplink] delivered ${kind}/${name}`) + } catch (err) { + rememberLog(`[deeplink] delivery failed: ${err.message}`) + } +} + +// Renderer calls this (via IPC) once it has mounted its deep-link listener, so +// a link that arrived during boot/install is flushed exactly once. +ipcMain.handle('hermes:deep-link-ready', () => { + _rendererReadyForDeepLink = true + if (_pendingDeepLink) { + const queued = _pendingDeepLink + _pendingDeepLink = null + handleDeepLink( + `${HERMES_PROTOCOL}://${queued.kind}/${encodeURIComponent(queued.name)}` + + (Object.keys(queued.params).length ? '?' + new URLSearchParams(queued.params).toString() : '') + ) + } + return { ok: true } +}) + +function registerDeepLinkProtocol() { + try { + if (process.defaultApp && process.argv.length >= 2) { + // Dev: register with the electron exec path + entry script so the OS can + // relaunch us with the URL. + app.setAsDefaultProtocolClient(HERMES_PROTOCOL, process.execPath, [path.resolve(process.argv[1])]) + } else { + app.setAsDefaultProtocolClient(HERMES_PROTOCOL) + } + } catch (err) { + rememberLog(`[deeplink] protocol registration failed: ${err.message}`) + } +} + +// Single-instance lock: deep links on a running app (Win/Linux) arrive as a +// second-instance argv. Without the lock a second `hermes://` launch spawns a +// whole new app instead of routing into the running one. +const _gotSingleInstanceLock = app.requestSingleInstanceLock() +if (!_gotSingleInstanceLock) { + app.quit() +} else { + app.on('second-instance', (_event, argv) => { + const url = _extractDeepLink(argv) + if (url) handleDeepLink(url) + else if (mainWindow) { + if (mainWindow.isMinimized()) mainWindow.restore() + mainWindow.focus() + } + }) +} + +// macOS delivers deep links via 'open-url' — register early (can fire before +// whenReady; handleDeepLink queues until the renderer is ready). +app.on('open-url', (event, url) => { + event.preventDefault() + handleDeepLink(url) +}) + app.whenReady().then(() => { if (IS_MAC) { Menu.setApplicationMenu(buildApplicationMenu()) @@ -6151,11 +6470,16 @@ app.whenReady().then(() => { } installMediaPermissions() registerMediaProtocol() + registerDeepLinkProtocol() ensureWslWindowsFonts() configureSpellChecker() registerPowerResumeListeners() createWindow() + // Win/Linux cold start: the launching hermes:// URL is in our own argv. + const _coldStartLink = _extractDeepLink(process.argv) + if (_coldStartLink) handleDeepLink(_coldStartLink) + app.on('activate', () => { // Recreate the primary window if it's gone. Guard on mainWindow directly // (not just total window count) so a dock click still restores the main diff --git a/apps/desktop/electron/preload.cjs b/apps/desktop/electron/preload.cjs index d39bc88fb68..544037e7869 100644 --- a/apps/desktop/electron/preload.cjs +++ b/apps/desktop/electron/preload.cjs @@ -5,7 +5,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', { revalidateConnection: () => ipcRenderer.invoke('hermes:connection:revalidate'), touchBackend: profile => ipcRenderer.invoke('hermes:backend:touch', profile), getGatewayWsUrl: profile => ipcRenderer.invoke('hermes:gateway:ws-url', profile), - openSessionWindow: sessionId => ipcRenderer.invoke('hermes:window:openSession', sessionId), + openSessionWindow: (sessionId, opts) => ipcRenderer.invoke('hermes:window:openSession', sessionId, opts), getBootProgress: () => ipcRenderer.invoke('hermes:boot-progress:get'), getConnectionConfig: profile => ipcRenderer.invoke('hermes:connection-config:get', profile), saveConnectionConfig: payload => ipcRenderer.invoke('hermes:connection-config:save', payload), @@ -39,6 +39,8 @@ contextBridge.exposeInMainWorld('hermesDesktop', { watchPreviewFile: url => ipcRenderer.invoke('hermes:watchPreviewFile', url), stopPreviewFileWatch: id => ipcRenderer.invoke('hermes:stopPreviewFileWatch', id), setTitleBarTheme: payload => ipcRenderer.send('hermes:titlebar-theme', payload), + setNativeTheme: mode => ipcRenderer.send('hermes:native-theme', mode), + setTranslucency: payload => ipcRenderer.send('hermes:translucency', payload), setPreviewShortcutActive: active => ipcRenderer.send('hermes:previewShortcutActive', Boolean(active)), openExternal: url => ipcRenderer.invoke('hermes:openExternal', url), fetchLinkTitle: url => ipcRenderer.invoke('hermes:fetchLinkTitle', url), @@ -52,6 +54,7 @@ contextBridge.exposeInMainWorld('hermesDesktop', { getRecentLogs: () => ipcRenderer.invoke('hermes:logs:recent'), readDir: dirPath => ipcRenderer.invoke('hermes:fs:readDir', dirPath), gitRoot: startPath => ipcRenderer.invoke('hermes:fs:gitRoot', startPath), + worktrees: cwds => ipcRenderer.invoke('hermes:fs:worktrees', cwds), terminal: { dispose: id => ipcRenderer.invoke('hermes:terminal:dispose', id), resize: (id, size) => ipcRenderer.invoke('hermes:terminal:resize', id, size), @@ -80,11 +83,27 @@ contextBridge.exposeInMainWorld('hermesDesktop', { ipcRenderer.on('hermes:open-updates', listener) return () => ipcRenderer.removeListener('hermes:open-updates', listener) }, + onDeepLink: callback => { + const listener = (_event, payload) => callback(payload) + ipcRenderer.on('hermes:deep-link', listener) + return () => ipcRenderer.removeListener('hermes:deep-link', listener) + }, + signalDeepLinkReady: () => ipcRenderer.invoke('hermes:deep-link-ready'), onWindowStateChanged: callback => { const listener = (_event, payload) => callback(payload) ipcRenderer.on('hermes:window-state-changed', listener) return () => ipcRenderer.removeListener('hermes:window-state-changed', listener) }, + onFocusSession: callback => { + const listener = (_event, sessionId) => callback(sessionId) + ipcRenderer.on('hermes:focus-session', listener) + return () => ipcRenderer.removeListener('hermes:focus-session', listener) + }, + onNotificationAction: callback => { + const listener = (_event, payload) => callback(payload) + ipcRenderer.on('hermes:notification-action', listener) + return () => ipcRenderer.removeListener('hermes:notification-action', listener) + }, onPreviewFileChanged: callback => { const listener = (_event, payload) => callback(payload) ipcRenderer.on('hermes:preview-file-changed', listener) diff --git a/apps/desktop/electron/session-windows.cjs b/apps/desktop/electron/session-windows.cjs index 8775feb1bce..172ca16c757 100644 --- a/apps/desktop/electron/session-windows.cjs +++ b/apps/desktop/electron/session-windows.cjs @@ -5,22 +5,30 @@ const { pathToFileURL } = require('node:url') +// Secondary windows open at the minimum usable size — a compact side panel for +// subagent watch / cmd-click session pop-out, not a second full desktop. +const SESSION_WINDOW_MIN_WIDTH = 420 +const SESSION_WINDOW_MIN_HEIGHT = 620 + // Build the renderer URL for a secondary window. The renderer uses a // HashRouter, so the session route lives after the '#'. The `?win=secondary` // flag MUST sit in the query string BEFORE the '#': anything after the '#' is // treated as the route by HashRouter and would break routeSessionId(). The // renderer reads the flag from window.location.search to suppress the install / -// onboarding overlays and the global session sidebar. -function buildSessionWindowUrl(sessionId, { devServer, rendererIndexPath } = {}) { +// onboarding overlays and the global session sidebar. `watch=1` marks a +// spectator window (e.g. a running subagent's session): the renderer resumes +// it lazily so the gateway never builds an agent just to stream into it. +function buildSessionWindowUrl(sessionId, { devServer, rendererIndexPath, watch } = {}) { + const query = `?win=secondary${watch ? '&watch=1' : ''}` const route = `#/${encodeURIComponent(sessionId)}` if (devServer) { const base = devServer.endsWith('/') ? devServer.slice(0, -1) : devServer - return `${base}/?win=secondary${route}` + return `${base}/${query}${route}` } - return `${pathToFileURL(rendererIndexPath).toString()}?win=secondary${route}` + return `${pathToFileURL(rendererIndexPath).toString()}${query}${route}` } // A small registry keyed by sessionId that guarantees one window per chat: @@ -83,4 +91,9 @@ function createSessionWindowRegistry() { } } -module.exports = { buildSessionWindowUrl, createSessionWindowRegistry } +module.exports = { + buildSessionWindowUrl, + createSessionWindowRegistry, + SESSION_WINDOW_MIN_HEIGHT, + SESSION_WINDOW_MIN_WIDTH +} diff --git a/apps/desktop/electron/session-windows.test.cjs b/apps/desktop/electron/session-windows.test.cjs index 3453971eb51..a668b0ac082 100644 --- a/apps/desktop/electron/session-windows.test.cjs +++ b/apps/desktop/electron/session-windows.test.cjs @@ -76,6 +76,12 @@ test('buildSessionWindowUrl builds a packaged file URL with the flag before the assert.match(url, /^file:\/\/.*index\.html\?win=secondary#\/abc$/) }) +test('buildSessionWindowUrl adds the watch flag for spectator windows, before the hash', () => { + const url = buildSessionWindowUrl('abc', { devServer: 'http://localhost:5173', watch: true }) + + assert.equal(url, 'http://localhost:5173/?win=secondary&watch=1#/abc') +}) + test('registry opens one window per session and focuses on re-open', () => { const registry = createSessionWindowRegistry() let built = 0 diff --git a/apps/desktop/electron/update-remote.cjs b/apps/desktop/electron/update-remote.cjs new file mode 100644 index 00000000000..3cb432d1b1e --- /dev/null +++ b/apps/desktop/electron/update-remote.cjs @@ -0,0 +1,56 @@ +/** + * Pure helpers for choosing a remote URL during passive update checks. + * + * A public install can end up with `origin=git@github.com:NousResearch/hermes-agent.git`. + * If the user's GitHub SSH key is FIDO2/passkey-backed, a background `git fetch + * origin` triggers an unexplained hardware-touch prompt. For passive checks + * against the official repo we substitute the public HTTPS `ls-remote` path, + * which needs no auth and cannot prompt. Active update/apply flows are left + * unchanged. + * + * Extracted from main.cjs so the security-critical remote detection is unit + * testable without booting Electron (main.cjs requires('electron') at load). + */ + +const OFFICIAL_REPO_HTTPS_URL = 'https://github.com/NousResearch/hermes-agent.git' +const OFFICIAL_REPO_CANONICAL = 'github.com/nousresearch/hermes-agent' + +// Normalize common GitHub remote URL forms to `host/owner/repo` (lowercased, +// no trailing slash, no .git suffix) so SSH and HTTPS forms of the same repo +// compare equal. +function canonicalGitHubRemote(url) { + if (!url) return '' + let value = String(url).trim() + if (value.startsWith('git@github.com:')) { + value = `github.com/${value.slice('git@github.com:'.length)}` + } else if (value.startsWith('ssh://git@github.com/')) { + value = `github.com/${value.slice('ssh://git@github.com/'.length)}` + } else { + try { + const parsed = new URL(value) + if (parsed.hostname && parsed.pathname) value = `${parsed.hostname}${parsed.pathname}` + } catch { + // Leave non-URL forms unchanged. + } + } + value = value.trim().replace(/\/+$/, '') + if (value.endsWith('.git')) value = value.slice(0, -4) + return value.toLowerCase() +} + +function isSshRemote(url) { + const value = String(url || '').trim().toLowerCase() + return value.startsWith('git@') || value.startsWith('ssh://') +} + +function isOfficialSshRemote(url) { + return isSshRemote(url) && canonicalGitHubRemote(url) === OFFICIAL_REPO_CANONICAL +} + +module.exports = { + OFFICIAL_REPO_HTTPS_URL, + OFFICIAL_REPO_CANONICAL, + canonicalGitHubRemote, + isSshRemote, + isOfficialSshRemote +} diff --git a/apps/desktop/electron/update-remote.test.cjs b/apps/desktop/electron/update-remote.test.cjs new file mode 100644 index 00000000000..0dfba970138 --- /dev/null +++ b/apps/desktop/electron/update-remote.test.cjs @@ -0,0 +1,78 @@ +/** + * Tests for electron/update-remote.cjs — the remote-detection helpers that + * keep passive update checks off the SSH origin for official installs. + * + * Run with: node --test electron/update-remote.test.cjs + * (Wired into npm test:desktop:platforms in package.json.) + * + * Why this matters: a public install can carry + * origin=git@github.com:NousResearch/hermes-agent.git. A background + * `git fetch origin` then authenticates over SSH and, with a FIDO2/passkey + * key, triggers an unexplained hardware-touch prompt. isOfficialSshRemote + * must reliably recognize the official SSH remote (in every URL form, + * case-insensitively) so the caller can swap in the anonymous HTTPS path — + * while NOT misclassifying forks, other hosts, or the HTTPS remote (which + * never prompts and should keep the normal fetch path). + */ + +const test = require('node:test') +const assert = require('node:assert/strict') + +const { + OFFICIAL_REPO_HTTPS_URL, + OFFICIAL_REPO_CANONICAL, + canonicalGitHubRemote, + isSshRemote, + isOfficialSshRemote +} = require('./update-remote.cjs') + +test('canonicalGitHubRemote normalizes SSH and HTTPS forms to the same value', () => { + assert.equal(canonicalGitHubRemote('git@github.com:NousResearch/hermes-agent.git'), OFFICIAL_REPO_CANONICAL) + assert.equal(canonicalGitHubRemote('git@github.com:NousResearch/hermes-agent'), OFFICIAL_REPO_CANONICAL) + assert.equal(canonicalGitHubRemote('ssh://git@github.com/NousResearch/hermes-agent.git'), OFFICIAL_REPO_CANONICAL) + assert.equal(canonicalGitHubRemote('https://github.com/NousResearch/hermes-agent.git'), OFFICIAL_REPO_CANONICAL) + // Case-insensitive: an uppercased owner still canonicalizes to the same repo. + assert.equal(canonicalGitHubRemote('git@github.com:nousresearch/hermes-agent.git'), OFFICIAL_REPO_CANONICAL) + // Trailing slashes are stripped. + assert.equal(canonicalGitHubRemote('https://github.com/NousResearch/hermes-agent/'), OFFICIAL_REPO_CANONICAL) +}) + +test('canonicalGitHubRemote is empty for falsy input', () => { + assert.equal(canonicalGitHubRemote(''), '') + assert.equal(canonicalGitHubRemote(null), '') + assert.equal(canonicalGitHubRemote(undefined), '') +}) + +test('isSshRemote detects scp-like and ssh:// forms only', () => { + assert.equal(isSshRemote('git@github.com:NousResearch/hermes-agent.git'), true) + assert.equal(isSshRemote('ssh://git@github.com/NousResearch/hermes-agent.git'), true) + assert.equal(isSshRemote('https://github.com/NousResearch/hermes-agent.git'), false) + assert.equal(isSshRemote(''), false) + assert.equal(isSshRemote(null), false) +}) + +test('isOfficialSshRemote is true only for the official repo over SSH', () => { + assert.equal(isOfficialSshRemote('git@github.com:NousResearch/hermes-agent.git'), true) + assert.equal(isOfficialSshRemote('git@github.com:NousResearch/hermes-agent'), true) + assert.equal(isOfficialSshRemote('ssh://git@github.com/NousResearch/hermes-agent.git'), true) + // Case-insensitive owner/repo match. + assert.equal(isOfficialSshRemote('git@github.com:nousresearch/hermes-agent.git'), true) +}) + +test('isOfficialSshRemote does NOT match forks, other hosts, or HTTPS', () => { + // A fork over SSH belongs to the user — fetching it is their own remote, + // not the official upstream, so the SSH-avoidance swap must not apply. + assert.equal(isOfficialSshRemote('git@github.com:someuser/hermes-agent.git'), false) + // Same repo name on a different host is not the official repo. + assert.equal(isOfficialSshRemote('git@gitlab.com:NousResearch/hermes-agent.git'), false) + // HTTPS to the official repo never prompts for SSH/FIDO2, so it keeps the + // normal fetch path — must not be flagged as an official SSH remote. + assert.equal(isOfficialSshRemote('https://github.com/NousResearch/hermes-agent.git'), false) + assert.equal(isOfficialSshRemote(''), false) + assert.equal(isOfficialSshRemote(null), false) +}) + +test('OFFICIAL_REPO_HTTPS_URL canonicalizes to OFFICIAL_REPO_CANONICAL', () => { + // Invariant: the URL we substitute in must be the same repo we detect. + assert.equal(canonicalGitHubRemote(OFFICIAL_REPO_HTTPS_URL), OFFICIAL_REPO_CANONICAL) +}) diff --git a/apps/desktop/electron/windows-child-process.test.cjs b/apps/desktop/electron/windows-child-process.test.cjs index 6bcc58a0a33..4239da56e23 100644 --- a/apps/desktop/electron/windows-child-process.test.cjs +++ b/apps/desktop/electron/windows-child-process.test.cjs @@ -8,7 +8,7 @@ const path = require('node:path') const ELECTRON_DIR = __dirname function readElectronFile(name) { - return fs.readFileSync(path.join(ELECTRON_DIR, name), 'utf8') + return fs.readFileSync(path.join(ELECTRON_DIR, name), 'utf8').replace(/\r\n/g, '\n') } function requireHiddenChildOptions(source, needle) { @@ -42,6 +42,9 @@ test('intentional or interactive desktop child processes stay documented', () => const source = readElectronFile('main.cjs') assert.match(source, /windowsHide: false/) + assert.match(source, /handOffWindowsBootstrapRecovery/) + assert.match(source, /'--repair', '--branch'/) + assert.match(source, /'--update', '--branch'/) assert.match(source, /nodePty\.spawn\(command, args/) assert.match(source, /spawn\('cmd\.exe', \['\/c', 'start'/) }) diff --git a/apps/desktop/eslint.config.mjs b/apps/desktop/eslint.config.mjs index 7650c747dbe..069a0056bbb 100644 --- a/apps/desktop/eslint.config.mjs +++ b/apps/desktop/eslint.config.mjs @@ -3,7 +3,6 @@ import typescriptEslint from '@typescript-eslint/eslint-plugin' import typescriptParser from '@typescript-eslint/parser' import perfectionist from 'eslint-plugin-perfectionist' import reactPlugin from 'eslint-plugin-react' -import reactCompiler from 'eslint-plugin-react-compiler' import hooksPlugin from 'eslint-plugin-react-hooks' import unusedImports from 'eslint-plugin-unused-imports' import globals from 'globals' @@ -47,7 +46,6 @@ export default [ 'custom-rules': customRules, perfectionist, react: reactPlugin, - 'react-compiler': reactCompiler, 'react-hooks': hooksPlugin, 'unused-imports': unusedImports }, @@ -98,7 +96,6 @@ export default [ 'perfectionist/sort-jsx-props': ['error', { order: 'asc', type: 'natural' }], 'perfectionist/sort-named-exports': ['error', { order: 'asc', type: 'natural' }], 'perfectionist/sort-named-imports': ['error', { order: 'asc', type: 'natural' }], - 'react-compiler/react-compiler': 'warn', 'react-hooks/exhaustive-deps': 'warn', 'react-hooks/rules-of-hooks': 'error', 'unused-imports/no-unused-imports': 'error' diff --git a/apps/desktop/index.html b/apps/desktop/index.html index 0ef2dcb59ab..831478f6e93 100644 --- a/apps/desktop/index.html +++ b/apps/desktop/index.html @@ -9,6 +9,28 @@ Hermes +
diff --git a/apps/desktop/package.json b/apps/desktop/package.json index b80d44d0435..52be586f013 100644 --- a/apps/desktop/package.json +++ b/apps/desktop/package.json @@ -18,7 +18,8 @@ "profile:main": "wait-on http://127.0.0.1:5174 && cross-env XCURSOR_SIZE=24 HERMES_DESKTOP_DEV_SERVER=http://127.0.0.1:5174 electron --inspect=9229 .", "profile:main:cpu": "wait-on http://127.0.0.1:5174 && cross-env XCURSOR_SIZE=24 NODE_OPTIONS=--cpu-prof HERMES_DESKTOP_DEV_SERVER=http://127.0.0.1:5174 electron .", "start": "npm run build && electron .", - "build": "node scripts/assert-root-install.cjs && node scripts/write-build-stamp.cjs && node scripts/stage-native-deps.cjs && tsc -b && vite build && node scripts/assert-dist-built.cjs", + "build": "node scripts/assert-root-install.cjs && node scripts/write-build-stamp.cjs && node scripts/stage-native-deps.cjs && tsc -b && vite build && npm run postbuild", + "postbuild": "node scripts/assert-dist-built.cjs", "builder": "cross-env NODE_OPTIONS=--max-old-space-size=16384 electron-builder", "pack": "npm run build && npm run builder -- --dir", "dist": "npm run build && npm run builder", @@ -35,8 +36,8 @@ "test:desktop:nsis": "node scripts/test-desktop.mjs nsis", "test:desktop:existing": "node scripts/test-desktop.mjs existing", "test:desktop:fresh": "node scripts/test-desktop.mjs fresh", - "test:desktop:platforms": "node --test electron/bootstrap-platform.test.cjs electron/hardening.test.cjs electron/backend-probes.test.cjs electron/bootstrap-runner.test.cjs electron/connection-config.test.cjs electron/gateway-ws-probe.test.cjs electron/oauth-net-request.test.cjs electron/desktop-uninstall.test.cjs electron/session-windows.test.cjs electron/workspace-cwd.test.cjs electron/windows-child-process.test.cjs", - "type-check": "tsc -b", + "test:desktop:platforms": "node --test electron/bootstrap-platform.test.cjs electron/hardening.test.cjs electron/backend-env.test.cjs electron/backend-probes.test.cjs electron/bootstrap-runner.test.cjs electron/connection-config.test.cjs electron/dashboard-token.test.cjs electron/gateway-ws-probe.test.cjs electron/oauth-net-request.test.cjs electron/desktop-uninstall.test.cjs electron/session-windows.test.cjs electron/workspace-cwd.test.cjs electron/fs-read-dir.test.cjs electron/git-root.test.cjs electron/windows-child-process.test.cjs electron/update-remote.test.cjs", + "typecheck": "tsc -p . --noEmit", "lint": "eslint src/ electron/", "lint:fix": "eslint src/ electron/ --fix", "fmt": "prettier --write 'src/**/*.{ts,tsx}' 'electron/**/*.{js,cjs}' 'vite.config.ts'", @@ -72,6 +73,7 @@ "class-variance-authority": "^0.7.1", "clsx": "^2.1.1", "cmdk": "^1.1.1", + "dnd-core": "^14.0.1", "hast-util-from-html-isomorphic": "^2.0.0", "hast-util-to-text": "^4.0.2", "ignore": "^7.0.5", @@ -83,10 +85,12 @@ "radix-ui": "^1.4.3", "react": "^19.2.5", "react-arborist": "^3.5.0", + "react-dnd-html5-backend": "^14.0.3", "react-dom": "^19.2.5", "react-router-dom": "^7.17.0", "react-shiki": "^0.9.3", "remark-math": "^6.0.0", + "remend": "^1.3.0", "shiki": "^4.0.2", "streamdown": "^2.5.0", "tailwind-merge": "^3.5.0", @@ -95,6 +99,7 @@ "unicode-animations": "^1.0.3", "unified": "^11.0.5", "unist-util-visit-parents": "^6.0.2", + "use-stick-to-bottom": "^1.1.6", "vfile": "^6.0.3", "web-haptics": "^0.0.6" }, @@ -103,20 +108,19 @@ "@testing-library/dom": "^10.4.0", "@testing-library/react": "^16.3.2", "@types/hast": "^3.0.4", - "@types/node": "^24.12.2", + "@types/node": "^24.13.2", "@types/react": "^19.2.14", "@types/react-dom": "^19.2.3", "@typescript-eslint/eslint-plugin": "^8.59.1", "@typescript-eslint/parser": "^8.59.1", "@vitejs/plugin-react": "^6.0.1", - "concurrently": "^9.2.1", + "concurrently": "^10.0.3", "cross-env": "^10.1.0", "electron": "^40.9.3", "electron-builder": "^26.8.1", "eslint": "^9.39.4", "eslint-plugin-perfectionist": "^5.9.0", "eslint-plugin-react": "^7.37.5", - "eslint-plugin-react-compiler": "^19.1.0-rc.2", "eslint-plugin-react-hooks": "^7.1.1", "eslint-plugin-unused-imports": "^4.4.1", "globals": "^16.5.0", @@ -133,6 +137,14 @@ "appId": "com.nousresearch.hermes", "productName": "Hermes", "executableName": "Hermes", + "protocols": [ + { + "name": "Hermes Protocol", + "schemes": [ + "hermes" + ] + } + ], "artifactName": "Hermes-${version}-${os}-${arch}.${ext}", "icon": "assets/icon", "directories": { diff --git a/apps/desktop/src/app/agents/index.tsx b/apps/desktop/src/app/agents/index.tsx index ff0aa8fb654..ec8f186dd1b 100644 --- a/apps/desktop/src/app/agents/index.tsx +++ b/apps/desktop/src/app/agents/index.tsx @@ -3,8 +3,8 @@ import { type ReactNode, useEffect, useMemo, useState } from 'react' import { useElapsedSeconds } from '@/components/chat/activity-timer' import { ActivityTimerText } from '@/components/chat/activity-timer-text' -import { BrailleSpinner } from '@/components/ui/braille-spinner' import { FadeText } from '@/components/ui/fade-text' +import { GlyphSpinner } from '@/components/ui/glyph-spinner' import { type Translations, useI18n } from '@/i18n' import { AlertCircle, CheckCircle2, Sparkles } from '@/lib/icons' import { useEnterAnimation } from '@/lib/use-enter-animation' @@ -25,7 +25,7 @@ import { OverlayView } from '../overlays/overlay-view' function statusGlyph(status: SubagentStatus, a: Translations['agents']): ReactNode { if (status === 'running' || status === 'queued') { return ( - {entry.text} {active ? ( - 0 ? (
-

{t.agents.files}

+

+ {t.agents.files} +

{fileLines.slice(0, 8).map(line => (

{line} diff --git a/apps/desktop/src/app/artifacts/index.tsx b/apps/desktop/src/app/artifacts/index.tsx index fd1569d7caf..8e98dd9d40d 100644 --- a/apps/desktop/src/app/artifacts/index.tsx +++ b/apps/desktop/src/app/artifacts/index.tsx @@ -18,7 +18,7 @@ import { } from '@/components/ui/pagination' import { TextTab, TextTabMeta } from '@/components/ui/text-tab' import { Tip } from '@/components/ui/tooltip' -import { getSessionMessages, listSessions } from '@/hermes' +import { getSessionMessages, listAllProfileSessions } from '@/hermes' import { type Translations, useI18n } from '@/i18n' import { sessionTitle } from '@/lib/chat-runtime' import { ExternalLink, ExternalLinkIcon, hostPathLabel, urlSlugTitleLabel, useLinkTitle } from '@/lib/external-link' @@ -388,8 +388,8 @@ export function ArtifactsView({ setStatusbarItemGroup: _setStatusbarItemGroup, . setRefreshing(true) try { - const sessions = (await listSessions(30, 1)).sessions - const results = await Promise.allSettled(sessions.map(session => getSessionMessages(session.id))) + const sessions = (await listAllProfileSessions(30, 1)).sessions + const results = await Promise.allSettled(sessions.map(session => getSessionMessages(session.id, session.profile))) const nextArtifacts: ArtifactRecord[] = [] results.forEach((result, index) => { diff --git a/apps/desktop/src/app/chat/composer/completion-drawer.tsx b/apps/desktop/src/app/chat/composer/completion-drawer.tsx index 8b23c54f879..021af0bda56 100644 --- a/apps/desktop/src/app/chat/composer/completion-drawer.tsx +++ b/apps/desktop/src/app/chat/composer/completion-drawer.tsx @@ -2,32 +2,21 @@ import type { Unstable_TriggerAdapter } from '@assistant-ui/core' import { ComposerPrimitive } from '@assistant-ui/react' import type { ReactNode } from 'react' -export const COMPLETION_DRAWER_CLASS = [ - 'absolute bottom-[calc(100%+0.25rem)] left-0 z-50', - 'w-60 max-w-[calc(100vw-2rem)]', - 'max-h-[min(23rem,calc(100vh-8rem))] overflow-y-auto overscroll-contain', - 'rounded-lg border border-(--ui-stroke-secondary)', - 'bg-[color-mix(in_srgb,var(--ui-bg-elevated)_96%,transparent)]', - 'p-1 text-xs text-popover-foreground shadow-md', - 'backdrop-blur-md' -].join(' ') +import { composerFusedDockCard } from '@/components/chat/composer-dock' +import { cn } from '@/lib/utils' -export const COMPLETION_DRAWER_BELOW_CLASS = [ - 'absolute left-0 top-[calc(100%+0.25rem)] z-50', - 'w-60 max-w-[calc(100vw-2rem)]', - 'max-h-[min(23rem,calc(100vh-8rem))] overflow-y-auto overscroll-contain', - 'rounded-lg border border-(--ui-stroke-secondary)', - 'bg-[color-mix(in_srgb,var(--ui-bg-elevated)_96%,transparent)]', - 'p-1 text-xs text-popover-foreground shadow-md', - 'backdrop-blur-md' -].join(' ') +// Same docked chrome as the queue/status stack, but its own thing: a narrow, +// left-aligned card (not full width) that fuses to the composer's edge instead +// of floating above it. `left-1` matches the stack's `mx-1` inset; the negative +// margin overlaps the seam so the composer's (now-transparent) edge border reads +// as shared. Fused (opaque) fill — the composer surface swaps to the same fill +// while a drawer is open, so the two paint as one panel. +const DRAWER_SHELL = + 'absolute left-1 z-50 w-80 max-w-[calc(100%-0.5rem)] max-h-[min(22rem,calc(100vh-8rem))] overflow-y-auto overscroll-contain p-1 text-xs text-popover-foreground' -export const COMPLETION_DRAWER_ROW_CLASS = [ - 'relative flex cursor-default select-none items-center gap-2 rounded-md px-2 py-1', - 'w-full min-w-0 text-left text-xs outline-hidden transition-colors', - 'hover:bg-(--ui-bg-tertiary)', - 'data-[highlighted]:bg-(--ui-bg-tertiary) data-[highlighted]:text-foreground' -].join(' ') +export const COMPLETION_DRAWER_CLASS = cn(DRAWER_SHELL, 'bottom-full -mb-[9px]', composerFusedDockCard('top')) + +export const COMPLETION_DRAWER_BELOW_CLASS = cn(DRAWER_SHELL, 'top-full -mt-[9px]', composerFusedDockCard('bottom')) export function ComposerCompletionDrawer({ adapter, diff --git a/apps/desktop/src/app/chat/composer/context-menu.tsx b/apps/desktop/src/app/chat/composer/context-menu.tsx index 3f09ec2fccb..22c10985f82 100644 --- a/apps/desktop/src/app/chat/composer/context-menu.tsx +++ b/apps/desktop/src/app/chat/composer/context-menu.tsx @@ -11,6 +11,7 @@ import { DropdownMenuSeparator, DropdownMenuTrigger } from '@/components/ui/dropdown-menu' +import { Kbd } from '@/components/ui/kbd' import { useI18n } from '@/i18n' import { Clipboard, FileText, FolderOpen, type IconComponent, ImageIcon, Link, MessageSquareText } from '@/lib/icons' import { cn } from '@/lib/utils' @@ -86,7 +87,7 @@ export function ContextMenu({

{c.tipPre} - @ + @ {c.tipPost}
diff --git a/apps/desktop/src/app/chat/composer/controls.tsx b/apps/desktop/src/app/chat/composer/controls.tsx index ed65795d1c4..8bc1a2b7cf9 100644 --- a/apps/desktop/src/app/chat/composer/controls.tsx +++ b/apps/desktop/src/app/chat/composer/controls.tsx @@ -1,5 +1,6 @@ import { Button } from '@/components/ui/button' import { Codicon } from '@/components/ui/codicon' +import { KbdCombo } from '@/components/ui/kbd' import { Tip } from '@/components/ui/tooltip' import { useI18n } from '@/i18n' import { triggerHaptic } from '@/lib/haptics' @@ -63,7 +64,14 @@ export function ComposerControls({ }) { const { t } = useI18n() const c = t.composer - const steerLabel = `${c.steer} (${formatCombo('mod+enter')})` + const steerCombo = formatCombo('mod+enter') + const steerLabel = `${c.steer} (${steerCombo})` + const steerTip = ( + + {c.steer} + + + ) if (conversation.active) { return @@ -75,7 +83,7 @@ export function ComposerControls({
{canSteer && ( - +
) } + +function HotkeyRow({ combos, description }: { combos: string[]; description: string }) { + return ( +
+ + {combos.map(combo => ( + + ))} + + {description} +
+ ) +} diff --git a/apps/desktop/src/app/chat/composer/hooks/use-live-completion-adapter.ts b/apps/desktop/src/app/chat/composer/hooks/use-live-completion-adapter.ts index fbeca7d59ee..6da699b602a 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-live-completion-adapter.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-live-completion-adapter.ts @@ -5,6 +5,13 @@ export interface CompletionEntry { text: string display?: unknown meta?: unknown + /** Optional section label (e.g. "Commands", "Skills"). The popover renders a + * header whenever this changes between consecutive items, so the fetcher must + * emit entries already grouped contiguously. */ + group?: string + /** Optional completion-action id. When set, picking the item runs that action + * (e.g. opening an overlay) instead of inserting a chip + waiting for submit. */ + action?: string } export interface CompletionPayload { diff --git a/apps/desktop/src/app/chat/composer/hooks/use-slash-completions.ts b/apps/desktop/src/app/chat/composer/hooks/use-slash-completions.ts index f3344158097..b0bac82825c 100644 --- a/apps/desktop/src/app/chat/composer/hooks/use-slash-completions.ts +++ b/apps/desktop/src/app/chat/composer/hooks/use-slash-completions.ts @@ -2,12 +2,17 @@ import type { Unstable_TriggerAdapter, Unstable_TriggerItem } from '@assistant-u import { useCallback } from 'react' import type { HermesGateway } from '@/hermes' +import { sessionTitle } from '@/lib/chat-runtime' import { type CommandsCatalogLike, + desktopSkinSlashCompletions, desktopSlashDescription, + type DesktopThemeCommandOption, filterDesktopCommandsCatalog, + isDesktopSlashExtensionCommand, isDesktopSlashSuggestion } from '@/lib/desktop-slash-commands' +import { $sessions } from '@/store/session' import type { CompletionEntry, CompletionPayload } from './use-live-completion-adapter' import { useLiveCompletionAdapter } from './use-live-completion-adapter' @@ -16,7 +21,10 @@ interface SlashItemMetadata extends Record { command: string display: string meta: string + group: string rawText: string + /** Completion-action id; empty for ordinary insert-a-chip completions. */ + action: string } function textValue(value: unknown, fallback = ''): string { @@ -38,12 +46,21 @@ function commandText(value: string): string { return value.startsWith('/') ? value : `/${value}` } +/** How many recent sessions to surface inline before the "Browse all…" entry. */ +const SESSION_INLINE_LIMIT = 7 + /** Live `/` completions backed by the gateway's `complete.slash` RPC. */ -export function useSlashCompletions(options: { gateway: HermesGateway | null }): { +export function useSlashCompletions(options: { + gateway: HermesGateway | null + /** Desktop theme list — `/skin` is owned client-side, so its arg completions + * come from here, not the backend (whose skin list is CLI/TUI-only). */ + skinThemes?: DesktopThemeCommandOption[] + activeSkin?: string +}): { adapter: Unstable_TriggerAdapter loading: boolean } { - const { gateway } = options + const { gateway, skinThemes, activeSkin } = options const enabled = Boolean(gateway) const fetcher = useCallback( @@ -54,34 +71,136 @@ export function useSlashCompletions(options: { gateway: HermesGateway | null }): const text = `/${query}` + // The desktop owns /skin entirely (client-side theme context). Surface its + // theme list inside this single popover instead of a bespoke one, and skip + // the backend skin completions (which describe CLI/TUI skins that don't + // apply here). Matches once we're past `/skin ` into the arg stage. + const skinArg = /^\/skin\s+(.*)$/is.exec(text) + + if (skinArg && skinThemes) { + const items = desktopSkinSlashCompletions(skinThemes, activeSkin ?? '', skinArg[1] ?? '').map(entry => ({ + text: entry.text, + display: entry.display, + meta: entry.meta, + group: 'Themes' + })) + + return { items, query } + } + + // /resume (and its aliases) completes recent sessions inline — the same + // client-side list the picker overlay shows — instead of the backend + // (whose /resume opens an interactive TUI picker we can't render here). + const sessionArg = /^\/(?:resume|sessions|switch)\s+(.*)$/is.exec(text) + + if (sessionArg) { + const needle = (sessionArg[1] ?? '').trim().toLowerCase() + + const matches = ( + needle + ? $sessions.get().filter( + session => + sessionTitle(session).toLowerCase().includes(needle) || + (session.preview ?? '').toLowerCase().includes(needle) || + session.id.toLowerCase().includes(needle) + ) + : $sessions.get() + ).slice(0, SESSION_INLINE_LIMIT) + + const items: CompletionEntry[] = matches.map(session => ({ + text: `/resume ${session.id}`, + display: sessionTitle(session), + meta: (session.preview ?? '').trim(), + group: 'Sessions' + })) + + // Trailing "more" affordance (Cursor-style): picking it opens the full + // session picker overlay directly. `text` stays a bare `/resume` so that + // submitting it (Enter) still opens the overlay if the action is skipped. + items.push({ + text: '/resume', + display: 'Browse all sessions…', + meta: '', + group: 'Sessions', + action: 'session-picker' + }) + + return { items, query } + } + try { if (!query) { const catalog = filterDesktopCommandsCatalog(await gateway.request('commands.catalog')) - const items = (catalog.pairs ?? []).map(([command, meta]) => ({ - text: command, - display: command, - meta - })) + // Prefer the categorized layout so the popover renders section headers + // (Session, Tools & Skills, ...). Fall back to the flat list when the + // backend didn't categorize. + const sections = catalog.categories?.length + ? catalog.categories + : [{ name: '', pairs: catalog.pairs ?? [] }] + + const items = sections.flatMap(section => + section.pairs.map(([command, meta]) => ({ + text: command, + display: command, + group: section.name || undefined, + meta + })) + ) return { items, query } } - const result = await gateway.request<{ items?: CompletionEntry[] }>('complete.slash', { text }) + const result = await gateway.request<{ items?: CompletionEntry[]; replace_from?: number }>( + 'complete.slash', + { text } + ) - const items = (result.items ?? []) - .filter(item => isDesktopSlashSuggestion(item.text)) + // Arg-completion items (replace_from > 1) carry just the arg stub — + // e.g. complete.slash returns `{text: "alice"}` for `/personality alic` + // with replace_from = 14. Rewrite those entries so the popover inserts + // the full `/personality alice` token instead of stranding `/alice`. + const replaceFrom = typeof result.replace_from === 'number' ? result.replace_from : 1 + const isArgCompletion = replaceFrom > 1 + const prefix = isArgCompletion ? text.slice(0, replaceFrom) : '' + + const decorated = (result.items ?? []) + .map(item => { + if (!isArgCompletion) { + return item + } + + const argText = typeof item.text === 'string' ? item.text : '' + + return { ...item, text: `${prefix}${argText}` } + }) + .filter(item => isArgCompletion || isDesktopSlashSuggestion(item.text)) .map(item => ({ ...item, - meta: desktopSlashDescription(item.text, textValue(item.meta)) + // Arg suggestions (e.g. `/handoff `) live under one + // header; otherwise split skills out from built-in commands. + group: isArgCompletion ? 'Options' : isDesktopSlashExtensionCommand(item.text) ? 'Skills' : 'Commands', + // Arg items carry their own meta (the personality/toolset/platform + // blurb). Only command rows get the registry description — looking + // one up for `/personality none` would clobber it with the parent + // command's text. + meta: isArgCompletion ? textValue(item.meta) : desktopSlashDescription(item.text, textValue(item.meta)) })) + // Keep each group contiguous so headers render once: Commands before + // Skills (stable within a group, preserving backend relevance order). + const groupOrder = ['Commands', 'Skills', 'Options'] + + const items = isArgCompletion + ? decorated + : [...decorated].sort((a, b) => groupOrder.indexOf(a.group) - groupOrder.indexOf(b.group)) + return { items, query } } catch { return { items: [], query } } }, - [gateway] + [gateway, skinThemes, activeSkin] ) const toItem = useCallback((entry: CompletionEntry, index: number): Unstable_TriggerItem => { @@ -93,6 +212,8 @@ export function useSlashCompletions(options: { gateway: HermesGateway | null }): command, display, meta, + group: textValue(entry.group), + action: textValue(entry.action), // Provide rawText so hermesDirectiveFormatter.serialize uses the // direct-insertion path instead of the legacy @type:id fallback. // Without this, the item.id (which includes a "|index" suffix for diff --git a/apps/desktop/src/app/chat/composer/index.tsx b/apps/desktop/src/app/chat/composer/index.tsx index d8b06a68d37..dc3f0a490cb 100644 --- a/apps/desktop/src/app/chat/composer/index.tsx +++ b/apps/desktop/src/app/chat/composer/index.tsx @@ -13,17 +13,26 @@ import { useState } from 'react' -import { hermesDirectiveFormatter } from '@/components/assistant-ui/directive-text' +import { hermesDirectiveFormatter, type SlashChipKind } from '@/components/assistant-ui/directive-text' +import { composerFill, composerSurfaceGlass } from '@/components/chat/composer-dock' import { Button } from '@/components/ui/button' import { useMediaQuery } from '@/hooks/use-media-query' import { useResizeObserver } from '@/hooks/use-resize-observer' import { useI18n } from '@/i18n' import { chatMessageText } from '@/lib/chat-messages' import { SLASH_COMMAND_RE } from '@/lib/chat-runtime' +import { desktopSlashCommandTakesArgs } from '@/lib/desktop-slash-commands' import { DATA_IMAGE_URL_RE } from '@/lib/embedded-images' import { triggerHaptic } from '@/lib/haptics' import { cn } from '@/lib/utils' -import { $composerAttachments, clearComposerAttachments, type ComposerAttachment } from '@/store/composer' +import { + $composerAttachments, + clearComposerAttachments, + clearSessionDraft, + type ComposerAttachment, + stashSessionDraft, + takeSessionDraft +} from '@/store/composer' import { browseBackward, browseForward, @@ -34,14 +43,19 @@ import { import { $queuedPromptsBySession, enqueueQueuedPrompt, + MAX_AUTO_DRAIN_ATTEMPTS, + migrateQueuedPrompts, promoteQueuedPrompt, type QueuedPromptEntry, removeQueuedPrompt, - shouldAutoDrainOnSettle, + shouldAutoDrain, updateQueuedPrompt } from '@/store/composer-queue' -import { $gatewayState, $messages } from '@/store/session' +import { $statusItemsBySession } from '@/store/composer-status' +import { notify } from '@/store/notifications' +import { $gatewayState, $messages, setSessionPickerOpen } from '@/store/session' import { $threadScrolledUp } from '@/store/thread-scroll' +import { useTheme } from '@/themes' import { extractDroppedFiles, HERMES_PATHS_MIME, partitionDroppedFiles } from '../hooks/use-composer-actions' @@ -71,12 +85,16 @@ import { import { QueuePanel } from './queue-panel' import { composerPlainText, + deleteSelectionInEditor, + insertPlainTextAtCaret, + normalizeComposerEditorDom, placeCaretEnd, refChipElement, renderComposerContents, - RICH_INPUT_SLOT + RICH_INPUT_SLOT, + slashChipElement } from './rich-editor' -import { SkinSlashPopover } from './skin-slash-popover' +import { ComposerStatusStack } from './status-stack' import { detectTrigger, extractClipboardImageBlobs, textBeforeCaret, type TriggerState } from './text-utils' import { ComposerTriggerPopover } from './trigger-popover' import type { ChatBarProps } from './types' @@ -95,6 +113,36 @@ const COMPOSER_FADE_BACKGROUND = const pickPlaceholder = (pool: readonly string[]) => pool[Math.floor(Math.random() * pool.length)] +/** Completion items can carry an `action` (set in use-slash-completions) that + * runs a side effect on pick instead of inserting a chip — e.g. the session + * picker's "Browse all…" entry opens the overlay. Table-driven so new action + * items are a registry row, not a composer branch. */ +const COMPLETION_ACTIONS: Record void> = { + 'session-picker': () => setSessionPickerOpen(true) +} + +/** Map a picked `/` completion to its pill accent. Driven by the completion + * group set in use-slash-completions (Skills / Themes / Commands|Options). */ +function slashChipKindForItem(item: Unstable_TriggerItem): SlashChipKind { + const group = (item.metadata as { group?: unknown } | undefined)?.group + + if (group === 'Skills') { + return 'skill' + } + + if (group === 'Themes') { + return 'theme' + } + + return 'command' +} + +/** A `/` query is at its arg stage once it's past the command name. */ +const slashArgStage = (query: string) => query.includes(' ') + +/** The `/command` token of a slash query (`personality x` → `/personality`). */ +const slashCommandToken = (query: string) => `/${query.split(/\s+/, 1)[0]?.toLowerCase() ?? ''}` + interface QueueEditState { attachments: ComposerAttachment[] draft: string @@ -104,6 +152,10 @@ interface QueueEditState { const cloneAttachments = (attachments: ComposerAttachment[]) => attachments.map(a => ({ ...a })) +// Quiet period after the last keystroke before persisting the draft; +// unmount/pagehide flushes bypass it. +const DRAFT_PERSIST_DEBOUNCE_MS = 400 + export function ChatBar({ busy, cwd, @@ -131,8 +183,8 @@ export function ChatBar({ const draft = useAuiState(s => s.composer.text) const attachments = useStore($composerAttachments) const queuedPromptsBySession = useStore($queuedPromptsBySession) + const statusItemsBySession = useStore($statusItemsBySession) const scrolledUp = useStore($threadScrolledUp) - const sessionMessages = useStore($messages) const activeQueueSessionKey = queueSessionKey || sessionId || null const queuedPrompts = useMemo( @@ -140,12 +192,29 @@ export function ChatBar({ [activeQueueSessionKey, queuedPromptsBySession] ) + // Status items (subagents, background processes) are keyed by the RUNTIME + // session id — gateway events and process.list both speak that id. Only the + // queue uses the stored-session fallback key (prompts can queue pre-resume). + const statusSessionId = sessionId ?? null + + const statusStackVisible = useMemo( + () => + queuedPrompts.length > 0 || (statusSessionId ? (statusItemsBySession[statusSessionId]?.length ?? 0) > 0 : false), + [queuedPrompts.length, statusItemsBySession, statusSessionId] + ) + const composerRef = useRef(null) const composerSurfaceRef = useRef(null) const editorRef = useRef(null) const draftRef = useRef(draft) - const previousBusyRef = useRef(busy) + const pendingDraftPersistRef = useRef<{ scope: string | null; text: string } | null>(null) + const activeQueueSessionKeyRef = useRef(activeQueueSessionKey) + activeQueueSessionKeyRef.current = activeQueueSessionKey + const prevQueueKeyRef = useRef(activeQueueSessionKey) const drainingQueueRef = useRef(false) + // Per-entry auto-drain failure counts; bounds retries so a persistent 404 + // can't spin-loop. Cleared on success; reset naturally on remount/reconnect. + const drainFailuresRef = useRef(new Map()) const urlInputRef = useRef(null) const [urlOpen, setUrlOpen] = useState(false) @@ -156,14 +225,17 @@ export function ChatBar({ const [dragActive, setDragActive] = useState(false) const [queueEdit, setQueueEdit] = useState(null) const [focusRequestId, setFocusRequestId] = useState(0) + const queueEditRef = useRef(queueEdit) + queueEditRef.current = queueEdit const dragDepthRef = useRef(0) const composingRef = useRef(false) // true during IME composition (CJK input) const lastSpokenIdRef = useRef(null) const narrow = useMediaQuery('(max-width: 30rem)') + const { availableThemes, themeName } = useTheme() const at = useAtCompletions({ gateway: gateway ?? null, sessionId: sessionId ?? null, cwd: cwd ?? null }) - const slash = useSlashCompletions({ gateway: gateway ?? null }) + const slash = useSlashCompletions({ activeSkin: themeName, gateway: gateway ?? null, skinThemes: availableThemes }) const stacked = expanded || narrow || tight const trimmedDraft = draft.trim() @@ -171,16 +243,20 @@ export function ChatBar({ const canSubmit = busy || hasComposerPayload const editingQueuedPrompt = queueEdit ? (queuedPrompts.find(entry => entry.id === queueEdit.entryId) ?? null) : null const busyAction = busy && hasComposerPayload ? 'queue' : 'stop' + // Steer only makes sense mid-turn, text-only (the gateway can't carry images // into a tool result) and never for a slash command (those execute inline). const canSteer = busy && !!onSteer && attachments.length === 0 && trimmedDraft.length > 0 && !SLASH_COMMAND_RE.test(trimmedDraft) + const showHelpHint = draft === '?' const { t } = useI18n() const gatewayState = useStore($gatewayState) const newSessionPlaceholders = t.composer.newSessionPlaceholders const followUpPlaceholders = t.composer.followUpPlaceholders + const reconnecting = gatewayState === 'closed' || gatewayState === 'error' + const inputDisabled = disabled && !reconnecting // Resting placeholder: a starter for brand-new sessions, a continuation for // existing ones. Picked once and only re-rolled when we genuinely move to a @@ -211,11 +287,13 @@ export function ChatBar({ setRestingPlaceholder(pickPlaceholder(sessionId ? followUpPlaceholders : newSessionPlaceholders)) }, [followUpPlaceholders, newSessionPlaceholders, sessionId]) - // When the bar is disabled it's because the gateway isn't open. Distinguish a - // cold start ("Starting Hermes...") from a dropped connection we're trying to - // restore (e.g. after the Mac slept) so the stuck state reads as recoverable. + // When the transport is disabled it's because the gateway isn't open. + // Distinguish a cold start ("Starting Hermes...") from a dropped connection + // we're trying to restore. During reconnect, keep the textbox editable so a + // flaky network doesn't block drafting; only submit/backend actions stay + // disabled until the gateway is open again. const placeholder = disabled - ? gatewayState === 'closed' || gatewayState === 'error' + ? reconnecting ? t.composer.placeholderReconnecting : t.composer.placeholderStarting : restingPlaceholder @@ -257,13 +335,13 @@ export function ChatBar({ ) useEffect(() => { - if (!disabled) { + if (!inputDisabled) { focusInput() } - }, [disabled, focusInput, focusKey, focusRequestId]) + }, [focusInput, focusKey, focusRequestId, inputDisabled]) useEffect(() => { - if (disabled) { + if (inputDisabled) { return undefined } @@ -283,7 +361,7 @@ export function ChatBar({ offFocus() offInsert() } - }, [appendExternalText, disabled]) + }, [appendExternalText, inputDisabled]) // Keep draftRef in sync with the assistant-ui composer state for callers // that read the latest text outside the React render cycle. We don't push @@ -462,10 +540,88 @@ export function ChatBar({ }) }, []) - const selectSkinSlashCommand = (command: string) => { - draftRef.current = command - aui.composer().setText(command) - requestMainFocus() + const [trigger, setTrigger] = useState(null) + const [triggerActive, setTriggerActive] = useState(0) + const [triggerItems, setTriggerItems] = useState([]) + // Set synchronously in keydown when the open trigger popover consumes a + // navigation/control key (Arrow/Enter/Tab/Escape). The subsequent keyup must + // NOT run refreshTrigger for that keypress: it never edits text, and for + // Escape the keydown has already set trigger=null, so a keyup refresh would + // re-detect the still-present `/` and instantly reopen the menu. A ref is + // used instead of reading `trigger` in keyup because by keyup time React has + // re-rendered and the handler closure sees the post-keydown state. + const triggerKeyConsumedRef = useRef(false) + + const refreshTrigger = useCallback(() => { + const editor = editorRef.current + + if (!editor) { + return + } + + // Fast-bail: if neither `@` nor `/` appears in the current draft, there's + // nothing for `detectTrigger` to match. Use `textContent` (cheap browser- + // native walk) for the precondition check rather than `composerPlainText` + // (recursive child walk with chip-aware logic). Only when a trigger char + // is present do we pay the cost of the full walk + DOM range work. + const rawText = editor.textContent ?? '' + + if (!rawText.includes('@') && !rawText.includes('/')) { + if (trigger) { + setTrigger(null) + setTriggerActive(0) + } + + return + } + + const before = textBeforeCaret(editor) + const found = detectTrigger(before ?? composerPlainText(editor)) + + // The arg-stage popover is only useful for commands with an options screen. + // For a no-arg command it would dead-end on "No matches", so drop it — the + // directive is already complete. + const detected = + found?.kind === '/' && slashArgStage(found.query) && !desktopSlashCommandTakesArgs(slashCommandToken(found.query)) + ? null + : found + + setTrigger(detected) + + // Only reset the highlight when the trigger actually changed (opened, or + // the query/kind differs). Re-detecting the *same* trigger — e.g. on a + // caret move (mouseup) or a stray refresh — must preserve the user's + // current selection instead of snapping back to the first item. + if (detected?.kind !== trigger?.kind || detected?.query !== trigger?.query) { + setTriggerActive(0) + } + }, [trigger]) + + // Pull the live contentEditable text into draftRef + the AUI composer state + // (which drives `hasComposerPayload` → the send button). Shared by the input + // and compositionend paths so committed IME text reaches state through either. + const flushEditorToDraft = (editor: HTMLDivElement) => { + normalizeComposerEditorDom(editor) + + const nextDraft = composerPlainText(editor) + + if (nextDraft !== draftRef.current) { + draftRef.current = nextDraft + aui.composer().setText(nextDraft) + } + + window.setTimeout(refreshTrigger, 0) + } + + const handleEditorInput = (event: FormEvent) => { + // During IME composition the DOM contains uncommitted preedit text + // mixed with real content. Skip state writes — compositionend flushes + // the finalized text (see onCompositionEnd). + if (composingRef.current) { + return + } + + flushEditorToDraft(event.currentTarget) } const handlePaste = (event: ClipboardEvent) => { @@ -504,87 +660,7 @@ export function ChatBar({ } event.preventDefault() - document.execCommand('insertText', false, pastedText) - const nextDraft = composerPlainText(event.currentTarget) - draftRef.current = nextDraft - aui.composer().setText(nextDraft) - } - - const [trigger, setTrigger] = useState(null) - const [triggerActive, setTriggerActive] = useState(0) - const [triggerItems, setTriggerItems] = useState([]) - // Set synchronously in keydown when the open trigger popover consumes a - // navigation/control key (Arrow/Enter/Tab/Escape). The subsequent keyup must - // NOT run refreshTrigger for that keypress: it never edits text, and for - // Escape the keydown has already set trigger=null, so a keyup refresh would - // re-detect the still-present `/` and instantly reopen the menu. A ref is - // used instead of reading `trigger` in keyup because by keyup time React has - // re-rendered and the handler closure sees the post-keydown state. - const triggerKeyConsumedRef = useRef(false) - - const refreshTrigger = useCallback(() => { - const editor = editorRef.current - - if (!editor) { - return - } - - // Fast-bail: if neither `@` nor `/` appears in the current draft, there's - // nothing for `detectTrigger` to match. Use `textContent` (cheap browser- - // native walk) for the precondition check rather than `composerPlainText` - // (recursive child walk with chip-aware logic). Only when a trigger char - // is present do we pay the cost of the full walk + DOM range work. - const rawText = editor.textContent ?? '' - - if (!rawText.includes('@') && !rawText.includes('/')) { - if (trigger) { - setTrigger(null) - setTriggerActive(0) - } - - return - } - - const before = textBeforeCaret(editor) - const detected = detectTrigger(before ?? composerPlainText(editor)) - - setTrigger(detected) - - // Only reset the highlight when the trigger actually changed (opened, or - // the query/kind differs). Re-detecting the *same* trigger — e.g. on a - // caret move (mouseup) or a stray refresh — must preserve the user's - // current selection instead of snapping back to the first item. - if (detected?.kind !== trigger?.kind || detected?.query !== trigger?.query) { - setTriggerActive(0) - } - }, [trigger]) - - // Pull the live contentEditable text into draftRef + the AUI composer state - // (which drives `hasComposerPayload` → the send button). Shared by the input - // and compositionend paths so committed IME text reaches state through either. - const flushEditorToDraft = (editor: HTMLDivElement) => { - if (editor.childNodes.length === 1 && editor.firstChild?.nodeName === 'BR') { - editor.replaceChildren() - } - - const nextDraft = composerPlainText(editor) - - if (nextDraft !== draftRef.current) { - draftRef.current = nextDraft - aui.composer().setText(nextDraft) - } - - window.setTimeout(refreshTrigger, 0) - } - - const handleEditorInput = (event: FormEvent) => { - // During IME composition the DOM contains uncommitted preedit text - // mixed with real content. Skip state writes — compositionend flushes - // the finalized text (see onCompositionEnd). - if (composingRef.current) { - return - } - + insertPlainTextAtCaret(event.currentTarget, pastedText) flushEditorToDraft(event.currentTarget) } @@ -603,6 +679,12 @@ export function ChatBar({ const triggerLoading = trigger?.kind === '@' ? at.loading : trigger?.kind === '/' ? slash.loading : false + // Suppress the "No matches" empty state once a slash command is past its name: + // a no-arg command has nothing to offer, and a fully-typed arg commits on + // Space/Tab — neither should dead-end on a popover. + const argStageEmpty = + trigger?.kind === '/' && slashArgStage(trigger.query) && !triggerLoading && !triggerItems.length + const closeTrigger = () => { setTrigger(null) setTriggerItems([]) @@ -613,6 +695,25 @@ export function ChatBar({ setTriggerActive(idx => Math.min(idx, Math.max(0, triggerItems.length - 1))) }, [triggerItems.length]) + // Commit the literally-typed `/command arg` as a directive chip — used when + // the completion list is empty because the arg is already fully typed (the + // backend completer drops exact matches). Reuses the chip path via a + // synthetic item whose serialized form is the verbatim text. + const commitTypedSlashDirective = () => { + if (trigger?.kind !== '/') { + return + } + + const text = `/${trigger.query.trimEnd()}` + + replaceTriggerWithChip({ + id: text, + type: 'slash', + label: text.slice(1), + metadata: { command: slashCommandToken(trigger.query), display: text, meta: '', group: '', action: '', rawText: text } + }) + } + const replaceTriggerWithChip = (item: Unstable_TriggerItem) => { const editor = editorRef.current @@ -620,16 +721,49 @@ export function ChatBar({ return } + // Action items (e.g. "Browse all sessions…") run a side effect instead of + // inserting a chip: strip the typed trigger token, then fire the action. + const completionAction = (item.metadata as { action?: unknown } | undefined)?.action + const runAction = typeof completionAction === 'string' ? COMPLETION_ACTIONS[completionAction] : undefined + + if (runAction) { + const current = composerPlainText(editor) + const prefix = current.slice(0, Math.max(0, current.length - trigger.tokenLength)) + + renderComposerContents(editor, prefix) + placeCaretEnd(editor) + draftRef.current = composerPlainText(editor) + aui.composer().setText(draftRef.current) + closeTrigger() + runAction() + requestMainFocus() + + return + } + const serialized = hermesDirectiveFormatter.serialize(item) const starter = serialized.endsWith(':') + + // Picking a bare arg-taking command (e.g. `/personality`) shouldn't commit + // it — expand to its options step so the popover shows the inline list, just + // as typing `/personality ` by hand would. A serialized value with a space is + // already an arg pick (`/personality alice`), so it commits normally. + const command = (item.metadata as { command?: string } | undefined)?.command ?? '' + + const expandsToArgs = trigger.kind === '/' && !serialized.includes(' ') && desktopSlashCommandTakesArgs(command) + const text = starter || serialized.endsWith(' ') ? serialized : `${serialized} ` const directive = !starter && serialized.match(/^@([^:]+):(.+)$/) + // No pill while expanding — the bare command stays plain text until an arg + // is picked, at which point a single pill is emitted for the full command. + const slashKind = !expandsToArgs && trigger.kind === '/' ? slashChipKindForItem(item) : null + const keepTriggerOpen = starter || expandsToArgs const finish = () => { draftRef.current = composerPlainText(editor) aui.composer().setText(draftRef.current) requestMainFocus() - starter ? window.setTimeout(refreshTrigger, 0) : closeTrigger() + keepTriggerOpen ? window.setTimeout(refreshTrigger, 0) : closeTrigger() } const sel = window.getSelection() @@ -639,7 +773,20 @@ export function ChatBar({ if (!sel || !range || node?.nodeType !== Node.TEXT_NODE || offset < trigger.tokenLength) { const current = composerPlainText(editor) - renderComposerContents(editor, `${current.slice(0, Math.max(0, current.length - trigger.tokenLength))}${text}`) + const prefix = current.slice(0, Math.max(0, current.length - trigger.tokenLength)) + + if (slashKind) { + // Two-step arg picks (e.g. `/handoff` pill already inserted, now picking + // the platform) land here because the caret sits past a contenteditable + // chip. Rebuild the prefix and re-emit a single pill for the full command. + renderComposerContents(editor, prefix) + editor.append(slashChipElement(serialized, slashKind), document.createTextNode(' ')) + placeCaretEnd(editor) + + return finish() + } + + renderComposerContents(editor, `${prefix}${text}`) placeCaretEnd(editor) return finish() @@ -650,8 +797,13 @@ export function ChatBar({ replaceRange.setEnd(node, offset) replaceRange.deleteContents() - if (directive) { - const chip = refChipElement(directive[1], directive[2]) + const chip = slashKind + ? slashChipElement(serialized, slashKind) + : directive + ? refChipElement(directive[1], directive[2]) + : null + + if (chip) { const space = document.createTextNode(' ') const fragment = document.createDocumentFragment() fragment.append(chip, space) @@ -680,6 +832,18 @@ export function ChatBar({ return } + // Non-collapsed Backspace/Delete: native selection-delete is ~O(n²) on large + // drafts (Ctrl+A → Delete froze ~1.3s). Collapsed carets fall through. + if ( + (event.key === 'Backspace' || event.key === 'Delete') && + deleteSelectionInEditor(event.currentTarget) + ) { + event.preventDefault() + flushEditorToDraft(event.currentTarget) + + return + } + // Cmd/Ctrl+Shift+K drains the next queued message. Plain Cmd/Ctrl+K is // reserved for the global command palette. if ((event.metaKey || event.ctrlKey) && !event.altKey && event.shiftKey && event.key.toLowerCase() === 'k') { @@ -709,7 +873,15 @@ export function ChatBar({ return } - if (event.key === 'Enter' || event.key === 'Tab') { + // Enter / Tab / Space all accept the highlighted item: a no-arg command + // commits its directive chip, an arg-taking command expands to its + // options step, and an arg option commits the full `/cmd arg` chip. Space + // is slash-only (an `@` mention takes a literal space) and gated to a + // non-empty query so a bare `/ ` still types a space. + const acceptOnSpace = event.key === ' ' && trigger.kind === '/' && Boolean(trigger.query.trim()) + const accept = event.key === 'Enter' || event.key === 'Tab' || acceptOnSpace + + if (accept) { event.preventDefault() triggerKeyConsumedRef.current = true const item = triggerItems[triggerActive] @@ -730,6 +902,24 @@ export function ChatBar({ } } + // Arg stage with nothing left to suggest — a fully-typed arg the backend + // completer no longer echoes (it drops the exact match), e.g. + // `/personality creative`. Space/Tab still commit what's typed as a single + // directive chip; Enter falls through to submit (send it as-is). + if ( + trigger?.kind === '/' && + !triggerItems.length && + (event.key === ' ' || event.key === 'Tab') && + slashArgStage(trigger.query) && + trigger.query.trim() + ) { + event.preventDefault() + triggerKeyConsumedRef.current = true + commitTypedSlashDirective() + + return + } + // ArrowUp/ArrowDown navigate, in priority order: the queue (edit entries in // place) then sent-message history. The history ring is derived from live // session messages each press — single source of truth, no mirror. @@ -762,7 +952,9 @@ export function ChatBar({ event.preventDefault() triggerKeyConsumedRef.current = true - const history = deriveUserHistory(sessionMessages, chatMessageText) + // $messages is read imperatively (not subscribed) so the composer + // doesn't re-render on every streaming delta flush. + const history = deriveUserHistory($messages.get(), chatMessageText) const entry = browseBackward(sessionId, currentDraft, history) if (entry !== null) { @@ -787,7 +979,7 @@ export function ChatBar({ event.preventDefault() triggerKeyConsumedRef.current = true - const history = deriveUserHistory(sessionMessages, chatMessageText) + const history = deriveUserHistory($messages.get(), chatMessageText) const result = browseForward(sessionId, history) if (result !== null) { @@ -823,6 +1015,10 @@ export function ChatBar({ const editorText = editorRef.current ? composerPlainText(editorRef.current) : draftRef.current const hasLivePayload = editorText.trim().length > 0 || attachments.length > 0 + if (disabled) { + return + } + if (!busy && !hasLivePayload && queuedPrompts.length > 0) { void drainNextQueued() @@ -1022,6 +1218,66 @@ export function ChatBar({ } } + const stashAt = (scope: string | null, text = draftRef.current, attachments = $composerAttachments.get()) => + stashSessionDraft(scope, text, attachments) + + // Per-thread draft swap — the composer's only session coupling. Lifecycle + // never clears composer state; this effect alone stashes on leave, restores + // on enter. Keyed writes are idempotent, so no skip-sentinel. + useEffect(() => { + const { attachments, text } = takeSessionDraft(activeQueueSessionKey) + loadIntoComposer(text, attachments) + + return () => { + const editing = queueEditRef.current + + if (editing?.sessionKey === activeQueueSessionKey) { + stashAt(activeQueueSessionKey, editing.draft, editing.attachments) + } else if (!isBrowsingHistory(sessionId)) { + stashAt(activeQueueSessionKey) + } + } + }, [activeQueueSessionKey]) // eslint-disable-line react-hooks/exhaustive-deps + + // Debounced stash into the active scope. Skipped while browsing history or + // editing a queued prompt — recalled text must not clobber the real draft. + useEffect(() => { + if (isBrowsingHistory(sessionId) || queueEdit) { + return + } + + pendingDraftPersistRef.current = { scope: activeQueueSessionKey, text: draft } + + const handle = window.setTimeout(() => { + pendingDraftPersistRef.current = null + stashAt(activeQueueSessionKey, draft) + }, DRAFT_PERSIST_DEBOUNCE_MS) + + return () => window.clearTimeout(handle) + }, [activeQueueSessionKey, draft, queueEdit, sessionId]) + + // pagehide is load-bearing: React skips effect cleanups on reload, so Cmd+R + // inside the debounce window would drop trailing keystrokes without this. + useEffect(() => { + const flushPendingDraftPersist = () => { + const pending = pendingDraftPersistRef.current + + if (!pending) { + return + } + + pendingDraftPersistRef.current = null + stashAt(pending.scope, pending.text) + } + + window.addEventListener('pagehide', flushPendingDraftPersist) + + return () => { + window.removeEventListener('pagehide', flushPendingDraftPersist) + flushPendingDraftPersist() + } + }, []) + const beginQueuedEdit = (entry: QueuedPromptEntry) => { if (!activeQueueSessionKey || queueEdit) { return @@ -1161,6 +1417,7 @@ export function ChatBar({ return false } + drainFailuresRef.current.delete(entry.id) removeQueuedPrompt(activeQueueSessionKey, entry.id) resetBrowseState(sessionId) @@ -1172,16 +1429,17 @@ export function ChatBar({ [activeQueueSessionKey, onSubmit, queuedPrompts, sessionId] ) - const drainNextQueued = useCallback( - () => - runDrain(entries => { - const skip = queueEdit?.entryId + const pickDrainHead = useCallback( + (entries: QueuedPromptEntry[]) => { + const skip = queueEditRef.current?.entryId - return skip ? entries.find(e => e.id !== skip) : entries[0] - }), - [queueEdit, runDrain] + return skip ? entries.find(e => e.id !== skip) : entries[0] + }, + [] // reads the edit id off a ref so the lock-holder always sees the latest ) + const drainNextQueued = useCallback(() => runDrain(pickDrainHead), [pickDrainHead, runDrain]) + const sendQueuedNow = useCallback( (id: string) => { if (!activeQueueSessionKey || id === queueEdit?.entryId) { @@ -1199,46 +1457,114 @@ export function ChatBar({ return true } + // A manual send clears the auto-drain backoff so a stuck entry the user + // taps gets a fresh attempt (and re-enables auto-retry on success). + drainFailuresRef.current.delete(id) + return runDrain(entries => entries.find(e => e.id === id)) }, [activeQueueSessionKey, busy, onCancel, queueEdit, runDrain] ) - // Auto-drain on busy → false (turn settled). Queued turns always flow once - // the session is idle again — whether the turn finished naturally or the - // user interrupted it. Interrupting to reach a queued message is the whole - // point of the queue, so we never suppress the drain. To cancel queued - // turns, the user deletes them from the panel. - useEffect(() => { - const wasBusy = previousBusyRef.current - previousBusyRef.current = busy - - if ( - shouldAutoDrainOnSettle({ - isBusy: busy, - queueLength: queuedPrompts.length, - wasBusy - }) - ) { - void drainNextQueued() + // Edge-independent auto-drain: send the head whenever the session is idle and + // the queue is non-empty, bounding retries so a thrown/rejected onSubmit (e.g. + // a stale-session 404) can't strand the entry permanently nor spin-loop. The + // drain lock serializes sends; a remount/reconnect resets the failure counts. + const autoDrainNext = useCallback(() => { + if (busy || drainingQueueRef.current || !activeQueueSessionKey) { + return } - }, [busy, drainNextQueued, queuedPrompts.length]) - // Clean up queue edit when its target disappears (session swap or external delete). + const entry = pickDrainHead(queuedPrompts) + + if (!entry || (drainFailuresRef.current.get(entry.id) ?? 0) >= MAX_AUTO_DRAIN_ATTEMPTS) { + return + } + + const onFail = () => { + const fails = (drainFailuresRef.current.get(entry.id) ?? 0) + 1 + drainFailuresRef.current.set(entry.id, fails) + + if (fails >= MAX_AUTO_DRAIN_ATTEMPTS) { + notify({ + id: 'composer-queue-stuck', + kind: 'error', + title: t.composer.queueStuckTitle, + message: t.composer.queueStuckBody + }) + } + } + + void runDrain(() => entry) + .then(sent => { + if (!sent) { + onFail() + } + }) + .catch(onFail) + }, [activeQueueSessionKey, busy, pickDrainHead, queuedPrompts, runDrain, t]) + + // Re-key on a runtime session-id change. A stable stored id (queueSessionKey) + // never churns, so a change there is a real session switch and must NOT + // migrate; only the runtime-derived key (queueSessionKey falsy → key is + // sessionId) churns on a backend bounce/resume of the same conversation. + useEffect(() => { + const prev = prevQueueKeyRef.current + prevQueueKeyRef.current = activeQueueSessionKey + + if (queueSessionKey || !prev || !activeQueueSessionKey || prev === activeQueueSessionKey) { + return + } + + migrateQueuedPrompts(prev, activeQueueSessionKey) + }, [activeQueueSessionKey, queueSessionKey]) + + // Queued turns flow whenever the session is idle — on the busy→false settle + // edge, on mount/reconnect, and after a re-key — so a swallowed edge can't + // strand them. To cancel queued turns, the user deletes them from the panel. + useEffect(() => { + if (shouldAutoDrain({ isBusy: busy, queueLength: queuedPrompts.length })) { + autoDrainNext() + } + }, [autoDrainNext, busy, queuedPrompts.length]) + + // Queue-edit cleanup: on session swap the scope effect already stashed the + // edit snapshot; only restore into the composer when still on the same scope. useEffect(() => { if (!queueEdit) { return } - if (queueEdit.sessionKey === activeQueueSessionKey && editingQueuedPrompt) { - return + if (queueEdit.sessionKey === activeQueueSessionKey) { + if (editingQueuedPrompt) { + return + } + + loadIntoComposer(queueEdit.draft, queueEdit.attachments) } - loadIntoComposer(queueEdit.draft, queueEdit.attachments) setQueueEdit(null) }, [activeQueueSessionKey, editingQueuedPrompt, queueEdit]) // eslint-disable-line react-hooks/exhaustive-deps + const dispatchSubmit = (text: string, attachments?: ComposerAttachment[]) => { + const submittedScope = activeQueueSessionKeyRef.current + const submittedAttachments = attachments ?? [] + + const restore = () => { + loadIntoComposer(text, submittedAttachments) + stashAt(activeQueueSessionKeyRef.current, text, submittedAttachments) + } + + void Promise.resolve(attachments ? onSubmit(text, { attachments }) : onSubmit(text)) + .then(accepted => void (accepted === false ? restore() : clearSessionDraft(submittedScope))) + .catch(restore) + } + const submitDraft = () => { + if (disabled) { + return + } + // Source the text from the DOM editor, not React state. The AUI composer // state (`draft`) and the derived `hasComposerPayload` lag the DOM by a // render, so on fast typing or IME composition the final keystroke(s) may @@ -1248,8 +1574,10 @@ export function ChatBar({ // input event; refresh it from the editor once more to also cover an // in-flight keystroke that hasn't fired its input event yet. const editor = editorRef.current + if (editor) { const domText = composerPlainText(editor) + if (domText !== draftRef.current) { draftRef.current = domText aui.composer().setText(domText) @@ -1270,10 +1598,9 @@ export function ChatBar({ // /send directives). Queuing them would make every slash command wait // for the current turn to finish, which is how the TUI never behaves. if (!attachments.length && SLASH_COMMAND_RE.test(text.trim())) { - const submitted = text triggerHaptic('submit') clearDraft() - void onSubmit(submitted) + dispatchSubmit(text) } else if (payloadPresent) { queueCurrentDraft() } else { @@ -1285,12 +1612,12 @@ export function ChatBar({ } else if (!payloadPresent && queuedPrompts.length > 0) { void drainNextQueued() } else if (payloadPresent) { - const submitted = text + const submittedAttachments = cloneAttachments(attachments) triggerHaptic('submit') resetBrowseState(sessionId) clearDraft() clearComposerAttachments() - void onSubmit(submitted, { attachments }) + dispatchSubmit(text, submittedAttachments) } focusInput() @@ -1418,6 +1745,7 @@ export function ChatBar({ const input = (
window.setTimeout(closeTrigger, 80)} @@ -1457,7 +1785,7 @@ export function ChatBar({ onPaste={handlePaste} ref={editorRef} role="textbox" - spellCheck="true" + spellCheck={false} suppressContentEditableWarning /> {/* assistant-ui requires ComposerPrimitive.Input somewhere in the tree @@ -1476,7 +1804,15 @@ export function ChatBar({ `asChild` swaps TextareaAutosize for a Radix Slot wrapping our plain