feat(honcho): make latency-adding paths configurable

queryRewrite (default off) gates the latest-message rewrite so the
extra auxiliary LLM call is opt-in. firstTurnBaseWait and
firstTurnDialecticWait expose the turn-1 bounded waits in seconds
(0 disables). All three resolve host-block-first like every other
field. Also pins per-host timeout resolution with tests.
This commit is contained in:
Erosika 2026-07-16 12:59:50 -04:00 committed by Teknium
parent e7fb51d5ac
commit e8957babf4
5 changed files with 113 additions and 2 deletions

View file

@ -100,6 +100,7 @@ def test_long_input_keeps_both_ends_with_a_hard_bound():
def _provider(query_rewriter, *, depth=1):
provider = HonchoMemoryProvider(query_rewriter=query_rewriter)
provider._query_rewrite_enabled = True
provider._manager = MagicMock()
provider._manager.dialectic_query.return_value = "memory synthesis"
provider._session_key = "test-session"
@ -189,6 +190,7 @@ def test_first_user_message_is_not_shadowed_by_generic_dialectic_prewarm():
enabled=True,
recall_mode="hybrid",
timeout=1,
query_rewrite=True,
)
with (
@ -237,3 +239,24 @@ def test_query_rewrite_has_an_independent_auxiliary_model_config():
assert task_config["provider"] == "auto"
assert task_config["timeout"] == 8
assert TASK_KEY in {key for key, _name, _description in _AUX_TASKS}
def test_query_rewrite_disabled_by_default():
"""queryRewrite defaults OFF — the rewriter must not add an LLM call."""
rewriter = MagicMock(return_value="What does the user prefer?")
provider = _provider(rewriter)
provider._query_rewrite_enabled = False
provider._run_dialectic_depth("what did we decide?")
rewriter.assert_not_called()
provider._manager.dialectic_query.assert_called_once()
def test_config_defaults_keep_latency_additions_off():
from plugins.memory.honcho.client import HonchoClientConfig
cfg = HonchoClientConfig(api_key="k", enabled=True)
assert cfg.query_rewrite is False
assert cfg.first_turn_base_wait == 3.0
assert cfg.first_turn_dialectic_wait == 2.0

View file

@ -125,3 +125,50 @@ def test_save_config_sets_owner_only_permissions(tmp_path, monkeypatch):
assert config_file.exists()
mode = stat.S_IMODE(config_file.stat().st_mode)
assert mode == 0o600, f"Expected 0o600 (owner-only), got {oct(mode)}"
class TestLatencyFlagResolution:
def test_defaults(self, tmp_path, monkeypatch):
monkeypatch.delenv('HONCHO_BASE_URL', raising=False)
config_path = tmp_path / 'config.json'
config_path.write_text(json.dumps({'apiKey': 'k'}))
cfg = HonchoClientConfig.from_global_config(config_path=config_path)
assert cfg.query_rewrite is False
assert cfg.first_turn_base_wait == 3.0
assert cfg.first_turn_dialectic_wait == 2.0
def test_host_block_wins(self, tmp_path, monkeypatch):
monkeypatch.delenv('HONCHO_BASE_URL', raising=False)
config_path = tmp_path / 'config.json'
config_path.write_text(json.dumps({
'apiKey': 'k',
'queryRewrite': False,
'firstTurnBaseWait': 3,
'hosts': {'hermes': {
'queryRewrite': True,
'firstTurnBaseWait': 0,
'firstTurnDialecticWait': 0.5,
}},
}))
cfg = HonchoClientConfig.from_global_config(config_path=config_path)
assert cfg.query_rewrite is True
assert cfg.first_turn_base_wait == 0.0
assert cfg.first_turn_dialectic_wait == 0.5
def test_per_host_timeout_wins_over_global(self, tmp_path, monkeypatch):
monkeypatch.delenv('HONCHO_TIMEOUT', raising=False)
config_path = tmp_path / 'config.json'
config_path.write_text(json.dumps({
'apiKey': 'k',
'timeout': 30,
'hosts': {'hermes': {'timeout': 5}},
}))
cfg = HonchoClientConfig.from_global_config(config_path=config_path)
assert cfg.timeout == 5.0
def test_timeout_falls_back_to_global(self, tmp_path, monkeypatch):
monkeypatch.delenv('HONCHO_TIMEOUT', raising=False)
config_path = tmp_path / 'config.json'
config_path.write_text(json.dumps({'apiKey': 'k', 'timeout': 30}))
cfg = HonchoClientConfig.from_global_config(config_path=config_path)
assert cfg.timeout == 30.0

View file

@ -25,6 +25,9 @@ def _configured_hybrid_config() -> _FakeHonchoConfig:
injection_frequency="every-turn",
context_cadence=1,
dialectic_cadence=1,
query_rewrite=False,
first_turn_base_wait=3.0,
first_turn_dialectic_wait=2.0,
dialectic_depth=1,
dialectic_depth_levels=None,
reasoning_heuristic=True,