diff --git a/hermes_cli/prompt_size.py b/hermes_cli/prompt_size.py
index 3b6530bfd68..9ce06ae5a21 100644
--- a/hermes_cli/prompt_size.py
+++ b/hermes_cli/prompt_size.py
@@ -15,16 +15,35 @@ from __future__ import annotations
import json
import re
-from typing import Any, Dict, List, Tuple
+from pathlib import Path
+from typing import Any, Dict, List, Optional, Tuple
# The skills index is wrapped in this tag pair inside the stable tier.
_SKILLS_BLOCK_RE = re.compile(r".*?", re.DOTALL)
+# A rendered skill entry inside is `` - name: desc`` (or
+# `` - name`` when the skill has no description). Category headers use two
+# leading spaces, so the four-space + ``- `` prefix isolates skill lines.
+_SKILL_LINE_PREFIX = " - "
+
+# Cap the human-readable "Skills by size" table; ``--json`` always has them all.
+_SKILLS_TABLE_LIMIT = 20
+
def _bytes(s: str) -> int:
return len(s.encode("utf-8"))
+def _tool_name(tool: Any) -> str:
+ """Return the callable name of a tool schema (OpenAI ``function`` shape)."""
+ if not isinstance(tool, dict):
+ return ""
+ fn = tool.get("function")
+ if isinstance(fn, dict) and fn.get("name"):
+ return str(fn["name"])
+ return str(tool.get("name", ""))
+
+
def _build_inspection_agent(platform: str) -> Any:
"""Construct an offline AIAgent for prompt inspection.
@@ -57,12 +76,118 @@ def _build_inspection_agent(platform: str) -> Any:
)
+def _skill_md_paths_by_name() -> Dict[str, Path]:
+ """Map each installed skill's name to its ``SKILL.md`` path on disk.
+
+ Keyed by both the frontmatter ``name`` (what the index renders) and the
+ skill directory name, so either resolves. Local skills win over external
+ dirs (``get_all_skills_dirs`` yields local first), matching the index's own
+ precedence. Used to attribute the real on-disk read cost per skill.
+ """
+ from agent.skill_utils import (
+ get_all_skills_dirs,
+ iter_skill_index_files,
+ parse_frontmatter,
+ )
+
+ mapping: Dict[str, Path] = {}
+ for skills_dir in get_all_skills_dirs():
+ if not skills_dir.exists():
+ continue
+ for skill_file in iter_skill_index_files(skills_dir, "SKILL.md"):
+ frontmatter_name = skill_file.parent.name
+ try:
+ frontmatter, _ = parse_frontmatter(
+ skill_file.read_text(encoding="utf-8")
+ )
+ frontmatter_name = str(frontmatter.get("name") or frontmatter_name)
+ except Exception:
+ pass
+ # setdefault keeps the first (local) occurrence on name collisions.
+ mapping.setdefault(frontmatter_name, skill_file)
+ mapping.setdefault(skill_file.parent.name, skill_file)
+ return mapping
+
+
+def _compute_skills_breakdown(skills_block: str) -> List[Dict[str, Any]]:
+ """Per-skill byte breakdown parsed from the rendered ````.
+
+ Two honest, distinct numbers per skill:
+
+ * ``index_line_bytes`` — the skill's actual rendered line in the always-on
+ index (the fixed per-call cost of *listing* the skill).
+ * ``skill_md_bytes`` — the on-disk size of the skill's ``SKILL.md`` (the
+ real token cost paid only when the model loads it via ``skill_view``).
+ ``None`` when the name can't be mapped to a file (e.g. a plugin skill
+ whose source lives outside the scanned skill dirs).
+
+ Sorted largest-first by ``skill_md_bytes`` (the read cost that dominates
+ pruning decisions), tie-broken by name.
+ """
+ name_to_path = _skill_md_paths_by_name()
+ entries: List[Dict[str, Any]] = []
+ for line in skills_block.splitlines():
+ if not line.startswith(_SKILL_LINE_PREFIX):
+ continue
+ rest = line[len(_SKILL_LINE_PREFIX):]
+ # ``name: desc`` — the first ``": "`` separates name from description.
+ # Namespaced names (``codex:rescue``) have no space after their colon,
+ # so partitioning on ``": "`` keeps the full name intact.
+ name = rest.partition(": ")[0].strip()
+ if not name:
+ continue
+ path = name_to_path.get(name)
+ md_bytes: Optional[int] = None
+ if path is not None:
+ try:
+ md_bytes = path.stat().st_size
+ except OSError:
+ md_bytes = None
+ entries.append({
+ "name": name,
+ "index_line_bytes": _bytes(line),
+ "skill_md_bytes": md_bytes,
+ "path": str(path) if path is not None else "",
+ })
+ entries.sort(key=lambda e: (-(e["skill_md_bytes"] or 0), e["name"]))
+ return entries
+
+
+def _compute_toolsets_breakdown(tools: List[Any]) -> List[Dict[str, Any]]:
+ """Per-toolset schema-byte breakdown of the resolved tool list.
+
+ Each tool is attributed to its single canonical toolset from the registry,
+ so ``json_bytes`` sums are fully attributable: the grand total equals the
+ sum of the individual tool serializations (which is the array total from
+ ``tools['json_bytes']`` minus JSON framing of ``2 * count`` bytes). Sorted
+ largest-first by ``json_bytes``, tie-broken by toolset name.
+ """
+ from tools.registry import registry
+
+ tool_to_toolset = registry.get_tool_to_toolset_map()
+ groups: Dict[str, Dict[str, Any]] = {}
+ for tool in tools:
+ name = _tool_name(tool)
+ toolset = tool_to_toolset.get(name) or "(unknown)"
+ group = groups.setdefault(
+ toolset, {"toolset": toolset, "tool_count": 0, "json_bytes": 0}
+ )
+ group["tool_count"] += 1
+ group["json_bytes"] += _bytes(json.dumps(tool, ensure_ascii=False))
+ out = list(groups.values())
+ out.sort(key=lambda g: (-g["json_bytes"], g["toolset"]))
+ return out
+
+
def compute_prompt_breakdown(platform: str = "cli") -> Dict[str, Any]:
"""Return a dict of prompt-size measurements for a fresh session.
Keys: ``system_prompt`` (chars/bytes), ``skills_index``, ``memory``,
- ``user_profile``, ``tools`` (count + json bytes), and ``sections`` (a list
- of (label, chars, bytes) for the three prompt tiers).
+ ``user_profile``, ``tools`` (count + json bytes), ``sections`` (a list of
+ (label, chars, bytes) for the three prompt tiers), ``skills_breakdown``
+ (per-skill index-line + on-disk SKILL.md bytes, largest-first), and
+ ``toolsets_breakdown`` (per-toolset tool count + schema json bytes,
+ largest-first). The last two answer "what should I disable to cut tokens?".
"""
from agent.system_prompt import build_system_prompt, build_system_prompt_parts
@@ -114,6 +239,8 @@ def compute_prompt_breakdown(platform: str = "cli") -> Dict[str, Any]:
"user_profile": {"chars": len(user_block), "bytes": _bytes(user_block)},
"tools": {"count": len(tools), "json_bytes": _bytes(tools_json)},
"sections": sections,
+ "skills_breakdown": _compute_skills_breakdown(skills_index),
+ "toolsets_breakdown": _compute_toolsets_breakdown(tools),
}
@@ -143,6 +270,41 @@ def render_breakdown(data: Dict[str, Any]) -> str:
lines.append("")
tools = data["tools"]
lines.append(f" Tool schemas : {tools['json_bytes']:>8,} B ({_fmt_kb(tools['json_bytes'])}, {tools['count']} tools)")
+
+ # Per-toolset schema cost — which toolset's tools cost the most to ship.
+ toolsets = data.get("toolsets_breakdown") or []
+ if toolsets:
+ lines.append("")
+ lines.append(" Toolsets by size (tool-schema JSON, largest first):")
+ lines.append(f" {'toolset':<22} {'tools':>5} {'schema':>10}")
+ for ts in toolsets:
+ lines.append(
+ f" {ts['toolset']:<22} {ts['tool_count']:>5} "
+ f"{ts['json_bytes']:>8,} B ({_fmt_kb(ts['json_bytes'])})"
+ )
+
+ # Per-skill cost — index line (always shipped) vs SKILL.md (read on load).
+ skills = data.get("skills_breakdown") or []
+ if skills:
+ lines.append("")
+ lines.append(
+ " Skills by size (SKILL.md on-disk = read cost; index line = "
+ "always-on cost, largest first):"
+ )
+ lines.append(f" {'skill':<28} {'SKILL.md':>10} {'index line':>10}")
+ shown = skills[:_SKILLS_TABLE_LIMIT]
+ for sk in shown:
+ md = sk["skill_md_bytes"]
+ md_str = f"{md:>8,} B" if md is not None else f"{'n/a':>10}"
+ name = sk["name"]
+ if len(name) > 28:
+ name = name[:27] + "…"
+ lines.append(
+ f" {name:<28} {md_str} {sk['index_line_bytes']:>8,} B"
+ )
+ remaining = len(skills) - len(shown)
+ if remaining > 0:
+ lines.append(f" … and {remaining} more (use --json for the full list)")
return "\n".join(lines)
diff --git a/tests/hermes_cli/test_prompt_size.py b/tests/hermes_cli/test_prompt_size.py
index e536c971300..c6fb8ed1934 100644
--- a/tests/hermes_cli/test_prompt_size.py
+++ b/tests/hermes_cli/test_prompt_size.py
@@ -2,6 +2,7 @@
import json
import sys
+from pathlib import Path
from types import SimpleNamespace
import pytest
@@ -9,6 +10,7 @@ import pytest
from hermes_cli.prompt_size import (
_SKILLS_BLOCK_RE,
_build_inspection_agent,
+ _compute_skills_breakdown,
compute_prompt_breakdown,
render_breakdown,
)
@@ -155,6 +157,94 @@ def test_skills_block_regex_matches_tagged_block():
assert m.group(0).endswith("")
+def test_toolsets_breakdown_reconciles_and_sorted(isolated_home):
+ """Per-toolset schema bytes attribute every tool exactly once.
+
+ Each resolved tool belongs to one registry toolset, so the grand total of
+ per-toolset json bytes equals the whole-array total minus JSON framing
+ (``2 * count`` bytes: brackets + ``", "`` separators between items).
+ """
+ data = compute_prompt_breakdown("cli")
+ toolsets = data["toolsets_breakdown"]
+ assert toolsets # CLI always resolves at least terminal + file
+ for ts in toolsets:
+ assert set(ts) >= {"toolset", "tool_count", "json_bytes"}
+ assert ts["tool_count"] >= 1
+ assert ts["json_bytes"] > 0
+ # Sorted largest-first.
+ byte_sizes = [ts["json_bytes"] for ts in toolsets]
+ assert byte_sizes == sorted(byte_sizes, reverse=True)
+ # Every tool attributed to exactly one toolset.
+ assert sum(ts["tool_count"] for ts in toolsets) == data["tools"]["count"]
+ # Bytes reconcile to the existing whole-array total.
+ grand = sum(ts["json_bytes"] for ts in toolsets)
+ assert grand == data["tools"]["json_bytes"] - 2 * data["tools"]["count"]
+
+
+def test_skills_breakdown_shape_sorted_and_attributed(isolated_home):
+ """Per-skill breakdown reports index-line + on-disk SKILL.md bytes.
+
+ Seeded before the first build (skills prompt is cached per-process).
+ """
+ _seed_skill(isolated_home, "small-skill", "short desc")
+ _seed_skill(isolated_home, "big-skill", "a much longer description " * 20)
+ data = compute_prompt_breakdown("cli")
+ skills = data["skills_breakdown"]
+ names = {s["name"] for s in skills}
+ assert {"small-skill", "big-skill"} <= names
+ for s in skills:
+ assert set(s) >= {"name", "index_line_bytes", "skill_md_bytes", "path"}
+ assert s["index_line_bytes"] > 0
+ # Sorted largest-first by on-disk SKILL.md size.
+ md_sizes = [s["skill_md_bytes"] or 0 for s in skills]
+ assert md_sizes == sorted(md_sizes, reverse=True)
+ # On-disk bytes match the real file; big-skill's SKILL.md is the larger.
+ by_name = {s["name"]: s for s in skills}
+ big = by_name["big-skill"]
+ assert big["path"] and Path(big["path"]).stat().st_size == big["skill_md_bytes"]
+ assert big["skill_md_bytes"] > by_name["small-skill"]["skill_md_bytes"]
+ # Per-skill index lines are a subset of the whole block,
+ # so they never exceed it (on-disk SKILL.md bytes are separate and don't).
+ assert sum(s["index_line_bytes"] for s in skills) <= data["skills_index"]["bytes"]
+
+
+def test_skills_breakdown_unmapped_name_is_none():
+ """A skill line with no matching SKILL.md on disk reports None, not a crash."""
+ block = (
+ "\n"
+ " demo:\n"
+ " - phantom-skill: not on disk\n"
+ "\n"
+ )
+ entries = _compute_skills_breakdown(block)
+ assert len(entries) == 1
+ assert entries[0]["name"] == "phantom-skill"
+ assert entries[0]["skill_md_bytes"] is None
+ assert entries[0]["path"] == ""
+ assert entries[0]["index_line_bytes"] > 0
+
+
+def test_skills_breakdown_parses_namespaced_names():
+ """Namespaced names (``ns:skill``) survive the ``name: desc`` split."""
+ block = (
+ "\n"
+ " plugins:\n"
+ " - codex:rescue: rescue helper\n"
+ "\n"
+ )
+ entries = _compute_skills_breakdown(block)
+ assert [e["name"] for e in entries] == ["codex:rescue"]
+
+
+def test_render_includes_per_component_tables(isolated_home):
+ """The rendered report gains the two new sorted tables (additive)."""
+ _seed_skill(isolated_home, "demo-skill", "a demo skill")
+ data = compute_prompt_breakdown("cli")
+ out = render_breakdown(data)
+ assert "Toolsets by size" in out
+ assert "Skills by size" in out
+
+
def test_render_breakdown_is_plain_text(isolated_home):
data = compute_prompt_breakdown("cli")
out = render_breakdown(data)