mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-31 19:16:29 +00:00
Second, deeper pass over tools/gateway/hermes_cli plus first pass over the trees wave 1 missed (acp, acp_adapter, skills, computer_use, docker, dashboard, conformance, monitoring, secret_sources, hermes_state, providers). Same rubric as wave 1 (AGENTS.md test policy); security, alternation/caching invariants, issue-number regressions, and E2E kept. Real test-quality fixes found and rooted out along the way: - tests/tools/test_command_guards.py made real auxiliary-LLM HTTPS calls (DEFAULT_CONFIG smart-approval leaked in) — pinned approval mode=manual via autouse fixture: 17.4s → 0.4s. - test_model_switch_custom_providers.py / test_user_providers_model_switch.py silently probed live provider catalogs (~2s/test) — stubbed cached_provider_model_ids/provider_model_ids/fetch_api_models. - test_telegram_noise_filter.py: 15-platform copy-paste matrix over shared gateway.run logic → 3 representative platforms (55s → 3.9s). - test_gateway_shutdown.py: stop()'s 5s interrupt-deadline loop spun on MagicMock agents — interrupt.side_effect now clears _running_agents (22s → 1.0s). - test_gateway_inactivity_timeout.py poll-harness timings shrunk 3-5x (24s → 1.1s); test_mcp_stability.py backoff/SIGTERM-grace sleeps patched (15.4s → 2.5s); test_async_delegation.py negative-drain wait 5s → 0.5s. - test_telegram_init_deadline.py: loop-block margin restored to 1.0s with rationale comment — the watchdog-dump assertion needs the loop blocked well past deadline+grace under parallel load (flaked once in the 40-worker verification run at a 0.2s margin). Verification: full hermetic suite via scripts/run_tests.sh — 2,438 files, 21,718 tests passed, 0 failed, 293.9s wall. Suite totals vs original baseline: 46,820 → 19,757 test functions (−57.8%), wall 583.5s → 293.9s (−50%), subprocess CPU 13,564s → 11,623s.
164 lines
4.9 KiB
Python
164 lines
4.9 KiB
Python
"""Tests for the blueprints layer (skill frontmatter <-> cron automation bridge).
|
|
|
|
A blueprint is a skill with a metadata.hermes.blueprint block. These verify parsing,
|
|
the create-job bridge, and the export round-trip without touching the real
|
|
cron store.
|
|
"""
|
|
|
|
import sys
|
|
from pathlib import Path
|
|
from unittest.mock import patch
|
|
|
|
import pytest
|
|
|
|
from tools.blueprints import (
|
|
BlueprintError,
|
|
BlueprintSpec,
|
|
create_blueprint_job,
|
|
export_blueprint,
|
|
parse_blueprint,
|
|
blueprint_spec_for_installed,
|
|
)
|
|
|
|
|
|
BLUEPRINT_SKILL = """---
|
|
name: morning-brief
|
|
description: Summarize unread email and calendar every morning.
|
|
version: 1.0.0
|
|
metadata:
|
|
hermes:
|
|
tags: [blueprint, email]
|
|
blueprint:
|
|
schedule: "0 8 * * *"
|
|
deliver: telegram
|
|
prompt: "Summarize my unread email and today's calendar."
|
|
---
|
|
|
|
# Morning Brief
|
|
|
|
Every morning, gather unread email and the day's calendar and send a digest.
|
|
"""
|
|
|
|
PLAIN_SKILL = """---
|
|
name: not-a-blueprint
|
|
description: Just a regular skill.
|
|
metadata:
|
|
hermes:
|
|
tags: [misc]
|
|
---
|
|
|
|
# Not a blueprint
|
|
"""
|
|
|
|
MALFORMED_BLUEPRINT = """---
|
|
name: broken
|
|
description: Blueprint with no schedule.
|
|
metadata:
|
|
hermes:
|
|
blueprint:
|
|
deliver: origin
|
|
---
|
|
|
|
# Broken
|
|
"""
|
|
|
|
|
|
class TestParseBlueprint:
|
|
def test_parses_full_blueprint(self):
|
|
spec = parse_blueprint(BLUEPRINT_SKILL)
|
|
assert spec is not None
|
|
assert spec.skill_name == "morning-brief"
|
|
assert spec.schedule == "0 8 * * *"
|
|
assert spec.deliver == "telegram"
|
|
assert spec.prompt is not None and spec.prompt.startswith("Summarize")
|
|
|
|
|
|
def test_deliver_defaults_to_origin(self):
|
|
skill = (
|
|
"---\nname: r\ndescription: d\nmetadata:\n hermes:\n"
|
|
' blueprint:\n schedule: "every 1h"\n---\n\nbody'
|
|
)
|
|
spec = parse_blueprint(skill)
|
|
assert spec is not None
|
|
assert spec.deliver == "origin"
|
|
|
|
|
|
class TestBlueprintSpecForInstalled:
|
|
def test_finds_and_parses_installed_blueprint(self, tmp_path):
|
|
skills_dir = tmp_path / "skills"
|
|
rec_dir = skills_dir / "productivity" / "morning-brief"
|
|
rec_dir.mkdir(parents=True)
|
|
(rec_dir / "SKILL.md").write_text(BLUEPRINT_SKILL, encoding="utf-8")
|
|
|
|
with patch("tools.skills_hub.SKILLS_DIR", skills_dir):
|
|
spec = blueprint_spec_for_installed("morning-brief")
|
|
assert spec is not None
|
|
assert spec.schedule == "0 8 * * *"
|
|
|
|
|
|
def test_plain_skill_returns_none(self, tmp_path):
|
|
skills_dir = tmp_path / "skills"
|
|
d = skills_dir / "misc" / "not-a-blueprint"
|
|
d.mkdir(parents=True)
|
|
(d / "SKILL.md").write_text(PLAIN_SKILL, encoding="utf-8")
|
|
with patch("tools.skills_hub.SKILLS_DIR", skills_dir):
|
|
assert blueprint_spec_for_installed("not-a-blueprint") is None
|
|
|
|
|
|
class TestCreateBlueprintJob:
|
|
def test_bridges_to_create_job(self):
|
|
spec = parse_blueprint(BLUEPRINT_SKILL)
|
|
assert spec is not None
|
|
captured = {}
|
|
|
|
def fake_create_job(**kwargs):
|
|
captured.update(kwargs)
|
|
return {"id": "abc123", **kwargs}
|
|
|
|
with patch("cron.jobs.create_job", fake_create_job):
|
|
job = create_blueprint_job(spec, origin={"platform": "telegram"})
|
|
|
|
assert captured["schedule"] == "0 8 * * *"
|
|
assert captured["skills"] == ["morning-brief"]
|
|
assert captured["deliver"] == "telegram"
|
|
assert captured["prompt"].startswith("Summarize")
|
|
assert job["id"] == "abc123"
|
|
|
|
|
|
class TestExportBlueprint:
|
|
def test_round_trips_job_to_skill_md(self):
|
|
job = {
|
|
"name": "My Morning Brief",
|
|
"schedule_display": "0 8 * * *",
|
|
"skills": ["morning-brief"],
|
|
"deliver": "telegram",
|
|
"prompt": "Summarize my unread email.",
|
|
}
|
|
md = export_blueprint(job, "# Morning Brief\n\nDoes the morning digest.")
|
|
# The exported SKILL.md must itself parse back as a blueprint.
|
|
spec = parse_blueprint(md)
|
|
assert spec is not None
|
|
assert spec.schedule == "0 8 * * *"
|
|
assert spec.deliver == "telegram"
|
|
# Name is sanitized to a valid skill identifier.
|
|
assert spec.skill_name == "my-morning-brief"
|
|
|
|
|
|
def test_export_interval_job_without_display(self):
|
|
# Regression: parse_schedule stores interval periods as "minutes" —
|
|
# exporting a job with only the parsed schedule dict must round-trip
|
|
# the real interval, not fall back to the daily default.
|
|
job = {
|
|
"name": "poller",
|
|
"schedule": {"kind": "interval", "minutes": 30},
|
|
"skills": ["poller"],
|
|
}
|
|
md = export_blueprint(job, "body")
|
|
spec = parse_blueprint(md)
|
|
assert spec is not None
|
|
assert spec.schedule == "every 30m"
|
|
|
|
job["schedule"] = {"kind": "interval", "minutes": 120}
|
|
spec = parse_blueprint(export_blueprint(job, "body"))
|
|
assert spec is not None
|
|
assert spec.schedule == "every 2h"
|