feat(desktop): first-class local Ollama provider with detection, model management, and capability-aware picking

Promote bare "ollama" from a custom-endpoint alias to a real provider and
build the desktop UX around it. A local server's connection kind is
reachability rather than a credential, so every credential-shaped gate
(provider registry, picker filters, settings surfaces) gets an explicit
path for it.

Backend:
- Provider overlay + registry entry (127.0.0.1:11434/v1 default, keyless
  with a local-only placeholder, base-url normalization for /api and /v1
  forms). Existing provider=custom configs are untouched.
- GET /api/local-servers/detect fingerprints well-known local ports plus a
  local configured base_url; response shape leaves room for a future
  installed/running/managed distinction.
- /api/ollama/* management endpoints: installed+running+recommended models,
  registry pull as a poll-able background job streaming native NDJSON
  progress, delete, and load (warm-up / keep_alive pinning). Pull and
  delete bust the picker's model-id cache.
- Model picker payload: per-model capabilities widened to tools/vision/
  context_length; local Ollama rows enriched from the server's native
  /api/show (authoritative for on-disk tags, where models.dev is sparse),
  backfilled off the request path by a background thread.
- The explicit-only picker filter keeps ollama rows: the row only exists
  when the server answered a probe, which is as explicit as a pasted key.
- Reasoning safety: /api/show thinking capability gates all reasoning
  fields (Ollama 400s reasoning_effort on non-thinking models), and
  OpenAI-only effort levels map to the nearest accepted level
  (xhigh->max, minimal->low).
- model.ollama_keep_alive config: sent per-request as extra_body.keep_alive.
- Latency discipline for a local server that may be down: a 300ms TCP
  pre-check with a short negative cache guards every native-API read; the
  status endpoint probes fresh so a just-started server is noticed
  immediately; localhost is rewritten to 127.0.0.1 (Windows resolves
  localhost to ::1 first and Ollama binds IPv4 loopback — each request
  otherwise pays a ~2s failed IPv6 connect, including chat inference).

Desktop:
- Providers -> Accounts: a "Local servers" card mirroring the OAuth card
  language ("Running · N models" / a start-the-server hint), expanding to
  model management: installed models with size/quant/VRAM, delete, warm-up,
  curated pull recommendations with a progress bar, free-form pull, and a
  KV-cache advisory when a loaded model runs well under its trained window.
  Polls while down so it flips to Running by itself.
- Model picker: "No tools" badge (explicit tools:false only — absence means
  unknown) with demotion, plus context window / parameter size / quant per
  row; refetches once after open so backfilled metadata appears in place.
- Onboarding: detected-server row with a model select, replacing blind
  first-model assignment for detected servers.
- Composer status stack: "Loading <model> into memory" row during cold
  starts, confirmed against /api/ps so ordinary slow generations stay quiet.

Chat inference stays on the OpenAI-compatible /v1 endpoint; the native
/api surface is used read-only for metadata plus explicit management
actions. Lifecycle management (starting or installing Ollama) is not
included.
This commit is contained in:
emozilla 2026-07-09 13:50:40 -04:00
parent 74e28f7d10
commit 83f88909ef
33 changed files with 2161 additions and 39 deletions

View file

@ -2012,6 +2012,22 @@ def init_agent(
agent._ollama_num_ctx,
)
# ── Ollama keep_alive ──
# How long the server keeps the model resident after a request (Ollama
# default: 5 minutes). Set model.ollama_keep_alive in config.yaml to a Go
# duration string ("30m", "2h") or seconds; -1 keeps it loaded until the
# server exits. Sent per-request alongside num_ctx.
agent._ollama_keep_alive = None
if isinstance(_model_cfg, dict):
_keep_alive_raw = _model_cfg.get("ollama_keep_alive")
if _keep_alive_raw is not None:
if isinstance(_keep_alive_raw, (int, float)):
agent._ollama_keep_alive = int(_keep_alive_raw)
else:
_keep_alive_str = str(_keep_alive_raw).strip()
if _keep_alive_str:
agent._ollama_keep_alive = _keep_alive_str
# Codex gpt-5.x autoraise notice: show at most once per profile/config
# state. Without the persisted marker the notice re-fires on every agent
# init — and the gateway rebuilds the agent per inbound message, so Discord

View file

@ -879,6 +879,12 @@ def build_api_kwargs(agent, api_messages: list) -> dict:
session_id=getattr(agent, "session_id", None),
provider_profile=_profile,
ollama_num_ctx=agent._ollama_num_ctx,
ollama_keep_alive=getattr(agent, "_ollama_keep_alive", None),
ollama_supports_thinking=(
agent._ollama_supports_thinking_cached()
if (agent.provider or "").strip().lower() == "ollama"
else None
),
# Context forwarded to profile hooks:
provider_preferences=_prefs or None,
openrouter_min_coding_score=agent.openrouter_min_coding_score,

View file

@ -1435,6 +1435,51 @@ def query_ollama_supports_vision(model: str, base_url: str, api_key: str = "") -
return None
def query_ollama_supports_thinking(model: str, base_url: str, api_key: str = "") -> Optional[bool]:
"""Return True/False when Ollama ``/api/show`` reports thinking support.
Uses the ``capabilities`` field (Ollama 0.6.0+). Returns None when the
server is unreachable, not Ollama, the model is unknown, or the server
predates the capabilities field — callers should treat None as "unknown"
and not gate on it.
"""
import httpx
bare_model = _strip_provider_prefix(model)
if not bare_model or not base_url:
return None
try:
if detect_local_server_type(base_url, api_key=api_key) != "ollama":
return None
except Exception:
return None
server_url = base_url.rstrip("/")
if server_url.endswith("/v1"):
server_url = server_url[:-3]
headers = _auth_headers(api_key)
try:
with httpx.Client(timeout=3.0, headers=headers) as client:
resp = client.post(f"{server_url}/api/show", json={"name": bare_model})
if resp.status_code != 200:
return None
data = resp.json()
except Exception:
return None
caps = data.get("capabilities")
if isinstance(caps, list):
if any(str(cap).lower() == "thinking" for cap in caps):
return True
if caps:
return False
return None
def _query_ollama_api_show(model: str, base_url: str, api_key: str = "") -> Optional[int]:
"""Query an Ollama server's native ``/api/show`` for context length.

View file

@ -256,6 +256,8 @@ class ChatCompletionsTransport(ProviderTransport):
is_lmstudio: bool
is_custom_provider: bool
ollama_num_ctx: int | None
ollama_keep_alive: int | str | None
ollama_supports_thinking: bool | None
# Provider routing
provider_preferences: dict | None
# Qwen-specific
@ -534,6 +536,8 @@ class ChatCompletionsTransport(ProviderTransport):
model=model,
base_url=params.get("base_url"),
ollama_num_ctx=params.get("ollama_num_ctx"),
ollama_keep_alive=params.get("ollama_keep_alive"),
ollama_supports_thinking=params.get("ollama_supports_thinking"),
session_id=params.get("session_id"),
)
)

View file

@ -23,6 +23,7 @@ import { $previewStatusBySession, dismissPreviewArtifact } from '@/store/preview
import { $threadScrolledUp } from '@/store/thread-scroll'
import { openSessionInNewWindow } from '@/store/windows'
import { OllamaColdStartRow, useOllamaColdStart } from './ollama-cold-start'
import { PreviewStatusRow } from './preview-row'
import { StatusItemRow } from './status-row'
@ -173,6 +174,13 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro
sections.push({ key: 'preview', node: previewBlock })
}
// Cold-start feedback while a local Ollama model loads its weights.
const ollamaColdStartModel = useOllamaColdStart()
if (ollamaColdStartModel) {
sections.push({ key: 'ollama-cold-start', node: <OllamaColdStartRow model={ollamaColdStartModel} /> })
}
if (queue) {
sections.push({ key: 'queue', node: queue })
}

View file

@ -0,0 +1,91 @@
import { useStore } from '@nanostores/react'
import { useEffect, useState } from 'react'
import { GlyphSpinner } from '@/components/ui/glyph-spinner'
import { getOllamaModels } from '@/hermes'
import { useI18n } from '@/i18n'
import { $awaitingResponse, $currentModel, $currentProvider } from '@/store/session'
// Wait this long after a turn starts before concluding the silence is a cold
// model load (a warm model answers well inside it), then confirm against the
// server before showing anything.
const COLD_START_GRACE_MS = 4_000
const POLL_INTERVAL_MS = 2_000
/**
* Cold-start detection for local Ollama models: returns the model name while
* a turn has been awaiting its first token past the grace period and the
* target model is not yet in the server's loaded list — i.e. Ollama is
* reading weights into memory. Returns '' for non-Ollama providers and for
* warm models (the loaded check keeps ordinary slow generations quiet).
*
* A hook rather than a self-hiding component so the status stack can count
* it toward its own visibility (the stack collapses when every section is
* empty).
*/
export function useOllamaColdStart(): string {
const awaiting = useStore($awaitingResponse)
const provider = useStore($currentProvider)
const model = useStore($currentModel)
const [loadingModel, setLoadingModel] = useState('')
const active = awaiting && provider === 'ollama' && Boolean(model)
useEffect(() => {
if (!active) {
setLoadingModel('')
return
}
let cancelled = false
let timer: ReturnType<typeof setTimeout>
const check = async () => {
try {
const data = await getOllamaModels()
if (cancelled) {
return
}
const loaded = data.running.some(r => r.name === model || r.name.split(':', 1)[0] === model.split(':', 1)[0])
// Loaded (or server unreachable — nothing useful to say): clear and
// let the normal streaming UI take over.
if (loaded || !data.reachable) {
setLoadingModel('')
return
}
setLoadingModel(model)
timer = setTimeout(() => void check(), POLL_INTERVAL_MS)
} catch {
if (!cancelled) {
setLoadingModel('')
}
}
}
timer = setTimeout(() => void check(), COLD_START_GRACE_MS)
return () => {
cancelled = true
clearTimeout(timer)
}
}, [active, model])
return active ? loadingModel : ''
}
export function OllamaColdStartRow({ model }: { model: string }) {
const { t } = useI18n()
return (
<div className="flex items-center gap-2 px-3 py-1.5 text-xs text-muted-foreground" data-slot="ollama-cold-start">
<GlyphSpinner className="opacity-70" spinner="braille" />
<span>{t.statusStack.ollamaLoading(model)}</span>
</div>
)
}

View file

@ -0,0 +1,364 @@
import { useCallback, useEffect, useRef, useState } from 'react'
import { Button } from '@/components/ui/button'
import { Input } from '@/components/ui/input'
import { RowButton } from '@/components/ui/row-button'
import {
deleteOllamaModel,
getOllamaModels,
loadOllamaModel,
pollOllamaPull,
startOllamaPull
} from '@/hermes'
import { useI18n } from '@/i18n'
import { Check, ChevronDown, Cpu, Download, Loader2, Trash2 } from '@/lib/icons'
import { cn } from '@/lib/utils'
import { notify, notifyError } from '@/store/notifications'
import type { OllamaModelsResponse, OllamaPullStatus } from '@/types/hermes'
import { CONTROL_TEXT } from './constants'
import { SettingsCategoryHeading } from './env-credentials'
import { Pill } from './primitives'
const PULL_POLL_INTERVAL_MS = 750
function formatBytes(bytes?: number): string {
if (!bytes || bytes <= 0) {
return ''
}
const gib = bytes / 1024 ** 3
return gib >= 1 ? `${gib.toFixed(1)} GB` : `${Math.round(bytes / 1024 ** 2)} MB`
}
function formatContext(tokens?: number): string {
if (!tokens || tokens <= 0) {
return ''
}
return tokens >= 1024 ? `${Math.round(tokens / 1024)}k` : String(tokens)
}
/**
* Local Ollama server card for the Providers page. Same row language as the
* OAuth provider cards, but the connection kind is reachability rather than
* a credential: the status tag shows "Running · N models" or a start-the-
* server hint. Expanding a running server reveals model management —
* installed models (size, delete, warm-up), loaded models (VRAM, context),
* curated pull recommendations, and free-form pull.
*/
export function OllamaProviderCard() {
const { t } = useI18n()
const copy = t.settings.ollama
const [data, setData] = useState<null | OllamaModelsResponse>(null)
const [expanded, setExpanded] = useState(false)
const [pull, setPull] = useState<null | OllamaPullStatus>(null)
const [pullModel, setPullModel] = useState('')
const [busyModel, setBusyModel] = useState<null | string>(null)
const pollTimer = useRef<null | ReturnType<typeof setTimeout>>(null)
const refresh = useCallback(async () => {
try {
setData(await getOllamaModels())
} catch {
setData(null)
}
}, [])
useEffect(() => {
void refresh()
return () => {
if (pollTimer.current) {
clearTimeout(pollTimer.current)
}
}
}, [refresh])
// While the server is down (or detection failed), poll so the card flips
// to "Running" by itself when the user starts Ollama — the status endpoint
// answers in ~20ms, so this is cheap. Stops as soon as it's reachable.
const reachableNow = Boolean(data?.reachable)
useEffect(() => {
if (reachableNow) {
return
}
const timer = setInterval(() => void refresh(), 5_000)
return () => clearInterval(timer)
}, [reachableNow, refresh])
const pollUntilDone = useCallback(
(jobId: string) => {
const tick = async () => {
let status: OllamaPullStatus
try {
status = await pollOllamaPull(jobId)
} catch (err) {
setPull(null)
notifyError(err, copy.pullFailed)
return
}
setPull(status)
if (status.status === 'pulling') {
pollTimer.current = setTimeout(() => void tick(), PULL_POLL_INTERVAL_MS)
return
}
if (status.status === 'done') {
notify({ kind: 'success', title: copy.pullDone(status.model), message: '' })
setPull(null)
setPullModel('')
void refresh()
} else {
notifyError(new Error(status.error_message || status.detail || 'pull failed'), copy.pullFailed)
setPull(null)
}
}
void tick()
},
[copy, refresh]
)
const beginPull = useCallback(
async (model: string) => {
const name = model.trim()
if (!name || pull) {
return
}
try {
const { job_id } = await startOllamaPull(name)
setPull({ job_id, model: name, status: 'pulling', detail: 'starting' })
pollUntilDone(job_id)
} catch (err) {
notifyError(err, copy.pullFailed)
}
},
[copy.pullFailed, pollUntilDone, pull]
)
const removeModel = useCallback(
async (model: string) => {
setBusyModel(model)
try {
const result = await deleteOllamaModel(model)
if (!result.ok) {
notifyError(new Error(result.message), copy.deleteFailed)
} else {
void refresh()
}
} catch (err) {
notifyError(err, copy.deleteFailed)
} finally {
setBusyModel(null)
}
},
[copy.deleteFailed, refresh]
)
const warmModel = useCallback(
async (model: string) => {
setBusyModel(model)
try {
const result = await loadOllamaModel(model)
if (!result.ok) {
notifyError(new Error(result.message), copy.loadFailed)
} else {
void refresh()
}
} catch (err) {
notifyError(err, copy.loadFailed)
} finally {
setBusyModel(null)
}
},
[copy.loadFailed, refresh]
)
const reachable = Boolean(data?.reachable)
const installedCount = data?.installed.length ?? 0
const running = new Map((data?.running ?? []).map(r => [r.name, r]))
const pullPercent =
pull?.total_bytes && pull.completed_bytes !== undefined
? Math.min(100, Math.round((pull.completed_bytes / pull.total_bytes) * 100))
: null
// Detection still pending: render nothing rather than flashing the
// not-running hint at every page open on machines without Ollama.
if (data === null) {
return null
}
return (
<section className="mb-5 grid gap-2" data-slot="ollama-provider-card">
<SettingsCategoryHeading icon={Cpu} title={copy.categoryTitle} />
<div className="rounded-[6px] transition-colors">
<RowButton
className="group flex w-full items-center justify-between gap-3 rounded-[6px] px-3 py-2.5 text-left transition-colors hover:bg-(--ui-control-hover-background)"
disabled={!reachable}
onClick={() => setExpanded(open => !open)}
>
<div className="min-w-0">
<div className="flex items-center gap-2">
<span className="text-[length:var(--conversation-text-font-size)] font-semibold">Ollama</span>
{reachable ? (
<span className="inline-flex items-center gap-1 bg-primary/10 px-2 py-0.5 text-xs font-medium text-primary">
<Check className="size-3" />
{copy.running(installedCount)}
</span>
) : null}
</div>
<p className="mt-1 text-xs leading-5 text-muted-foreground">
{reachable ? copy.desc : copy.notRunning}
</p>
</div>
{reachable && (
<ChevronDown
className={cn('size-4 shrink-0 text-muted-foreground transition group-hover:text-foreground', expanded && 'rotate-180')}
/>
)}
</RowButton>
{reachable && expanded && data && (
<div className="px-3 pb-2 pt-1">
{data.kv_cache_advisory && (
<p className="mb-2 rounded-sm bg-amber-500/10 px-2.5 py-1.5 text-xs leading-5 text-amber-700 dark:text-amber-400">
{copy.kvCacheAdvisory(
data.kv_cache_advisory.model,
formatContext(data.kv_cache_advisory.loaded_context),
formatContext(data.kv_cache_advisory.trained_context)
)}
</p>
)}
<div className="grid gap-1">
{data.installed.map(model => {
const live = running.get(model.name)
const busy = busyModel === model.name
const meta = [model.parameter_size, model.quantization, formatBytes(model.size_bytes)]
.filter(Boolean)
.join(' · ')
return (
<div
className="flex items-center justify-between gap-3 rounded-[6px] px-3 py-2 transition-colors hover:bg-(--ui-control-hover-background)"
key={model.name}
>
<div className="min-w-0">
<div className="flex items-center gap-2">
<span className="truncate font-mono text-xs">{model.name}</span>
{live && (
<Pill tone="primary">
{copy.loaded}
{live.context_length ? ` · ${formatContext(live.context_length)}` : ''}
{live.size_vram_bytes ? ` · ${formatBytes(live.size_vram_bytes)} VRAM` : ''}
</Pill>
)}
</div>
{meta && <p className="mt-0.5 text-[0.68rem] text-muted-foreground">{meta}</p>}
</div>
<div className="flex shrink-0 items-center gap-1">
{!live && (
<Button disabled={busy} onClick={() => void warmModel(model.name)} size="sm" variant="text">
{busy ? <Loader2 className="size-3.5 animate-spin" /> : copy.load}
</Button>
)}
<Button
aria-label={copy.deleteLabel(model.name)}
disabled={busy}
onClick={() => void removeModel(model.name)}
size="icon"
variant="ghost"
>
<Trash2 className="size-3.5" />
</Button>
</div>
</div>
)
})}
</div>
<div className="mt-3">
<p className="mb-1.5 text-xs font-medium">{copy.getModels}</p>
{pull ? (
<div className="rounded-[6px] bg-primary/[0.06] px-3 py-2">
<div className="flex items-center gap-2 text-xs">
<Loader2 className="size-3.5 animate-spin text-primary" />
<span className="font-mono">{pull.model}</span>
<span className="text-muted-foreground">
{pullPercent !== null ? `${pullPercent}%` : pull.detail || ''}
</span>
</div>
{pullPercent !== null && (
<div className="mt-1.5 h-1 overflow-hidden rounded-full bg-primary/15">
<div className="h-full bg-primary transition-all" style={{ width: `${pullPercent}%` }} />
</div>
)}
</div>
) : (
<>
<div className="grid gap-1">
{data.recommended.map(rec => (
<div
className="flex items-center justify-between gap-3 rounded-[6px] px-3 py-1.5 transition-colors hover:bg-(--ui-control-hover-background)"
key={rec.model}
>
<div className="min-w-0">
<span className="font-mono text-xs">{rec.model}</span>
<p className="mt-0.5 truncate text-[0.68rem] text-muted-foreground">{rec.description}</p>
</div>
<Button
className="shrink-0"
onClick={() => void beginPull(rec.model)}
size="sm"
variant="text"
>
<Download className="size-3.5" />
{copy.pull}
</Button>
</div>
))}
</div>
<div className="mt-2 flex items-center gap-2">
<Input
className={cn('max-w-72 font-mono', CONTROL_TEXT)}
onChange={event => setPullModel(event.target.value)}
onKeyDown={event => {
if (event.key === 'Enter') {
void beginPull(pullModel)
}
}}
placeholder={copy.pullPlaceholder}
value={pullModel}
/>
<Button disabled={!pullModel.trim()} onClick={() => void beginPull(pullModel)} size="sm" variant="text">
{copy.pull}
</Button>
</div>
</>
)}
</div>
</div>
)}
</div>
</section>
)
}

View file

@ -26,6 +26,7 @@ import type { EnvVarInfo, OAuthProvider } from '@/types/hermes'
import { isKeyVar, ProviderKeyRows } from './credential-key-ui'
import { SettingsCategoryHeading, useEnvCredentials } from './env-credentials'
import { providerGroup, providerMeta, providerPriority } from './helpers'
import { OllamaProviderCard } from './ollama-panel'
import { LoadingState, SettingsContent } from './primitives'
// The embedded terminal (and thus the "run disconnect command" path) only
@ -457,6 +458,10 @@ export function ProvidersSettings({ onClose, onViewChange, view }: ProvidersSett
onWantApiKey={() => onViewChange('keys')}
providers={oauthProviders}
/>
{/* Local Ollama server — a provider whose connection kind is
reachability rather than a credential, so it renders its own card
instead of an OAuth or API-key row. */}
<OllamaProviderCard />
</SettingsContent>
)
}

View file

@ -1,11 +1,11 @@
import { useQuery } from '@tanstack/react-query'
import { useState } from 'react'
import { useEffect, useState } from 'react'
import { useI18n } from '@/i18n'
import { requestModelOptions } from '@/lib/model-options'
import { currentPickerSelection } from '@/lib/model-status-label'
import { normalize } from '@/lib/text'
import type { ModelOptionProvider, ModelPricing } from '@/types/hermes'
import type { ModelCapabilities, ModelOptionProvider, ModelPricing } from '@/types/hermes'
import type { HermesGateway } from '../hermes'
import { cn } from '../lib/utils'
@ -61,6 +61,30 @@ export function ModelPickerDialog({
const providers = modelOptions.data?.providers ?? []
// Local Ollama capability metadata is backfilled server-side off the
// request path (the payload returns immediately; native /api/show data
// lands moments later). When an ollama row is missing its enrichment
// (no tools flag), refetch once shortly after so badges appear in-place
// instead of on the next open.
const ollamaUnenriched = providers.some(
p =>
p.slug === 'ollama' &&
(p.models?.length ?? 0) > 0 &&
p.models?.some(m => p.capabilities?.[m]?.tools === undefined)
)
const refetchOptions = modelOptions.refetch
useEffect(() => {
if (!open || !ollamaUnenriched) {
return
}
const timer = setTimeout(() => void refetchOptions(), 1_500)
return () => clearTimeout(timer)
}, [open, ollamaUnenriched, refetchOptions])
const { model: optionsModel, provider: optionsProvider } = currentPickerSelection(
!!sessionId,
{ model: currentModel, provider: currentProvider },
@ -205,6 +229,10 @@ function ModelResults({
const isCurrent = model === currentModel && provider.slug === currentProvider
const price = provider.pricing?.[model]
const locked = unavailable.has(model)
const caps = provider.capabilities?.[model]
// Only an explicit tools:false demotes — absent means unknown,
// and models.dev gaps must not smear working models.
const noTools = caps?.tools === false
return (
<CommandItem
@ -212,7 +240,8 @@ function ModelResults({
'flex items-center gap-2 pl-6 font-mono',
isCurrent &&
'bg-primary text-primary-foreground data-[selected=true]:bg-primary data-[selected=true]:text-primary-foreground',
locked && 'cursor-not-allowed opacity-45'
locked && 'cursor-not-allowed opacity-45',
noTools && !isCurrent && 'opacity-60'
)}
disabled={locked}
key={`${provider.slug}:${model}`}
@ -224,6 +253,20 @@ function ModelResults({
value={`${provider.slug}:${model}`}
>
<span className="min-w-0 flex-1 truncate">{model}</span>
{noTools && (
<span
className={cn(
'shrink-0 rounded-sm px-1 py-0.5 text-[0.6rem] font-semibold uppercase tracking-wide',
isCurrent
? 'bg-primary-foreground/20'
: 'bg-amber-500/15 text-amber-600 dark:text-amber-400'
)}
title={copy.noToolsTitle}
>
{copy.noTools}
</span>
)}
<ModelMeta caps={caps} isCurrent={isCurrent} />
{locked && (
<span className="shrink-0 text-[0.62rem] uppercase tracking-wide opacity-80">{copy.pro}</span>
)}
@ -243,6 +286,44 @@ function ModelResults({
)
}
// Compact local-model metadata: context window (agent-critical) plus
// parameter size / quantization when the native server reported them.
// Renders nothing for models with no metadata beyond the fast/reasoning flags.
function ModelMeta({ caps, isCurrent }: { caps?: ModelCapabilities; isCurrent: boolean }) {
if (!caps) {
return null
}
const parts: string[] = []
if (caps.context_length) {
parts.push(caps.context_length >= 1024 ? `${Math.round(caps.context_length / 1024)}k` : String(caps.context_length))
}
if (caps.parameter_size) {
parts.push(caps.parameter_size)
}
if (caps.quantization) {
parts.push(caps.quantization)
}
if (parts.length === 0) {
return null
}
return (
<span
className={cn(
'shrink-0 text-[0.66rem] tabular-nums',
isCurrent ? 'text-primary-foreground/80' : 'text-muted-foreground'
)}
>
{parts.join(' · ')}
</span>
)
}
// Compact In/Out $/Mtok price tag, mirroring the CLI picker's price columns.
// Renders nothing when pricing is unavailable for the model.
function ModelPrice({ price, isCurrent }: { price?: ModelPricing; isCurrent: boolean }) {

View file

@ -1,4 +1,6 @@
import { cleanup, fireEvent, render, screen } from '@testing-library/react'
import { QueryClient, QueryClientProvider } from '@tanstack/react-query'
import { cleanup, fireEvent, render as rtlRender, screen } from '@testing-library/react'
import type { ReactElement } from 'react'
import { afterEach, describe, expect, it } from 'vitest'
import { $desktopOnboarding, type DesktopOnboardingState, type OnboardingContext } from '@/store/onboarding'
@ -6,6 +8,15 @@ import type { OAuthProvider } from '@/types/hermes'
import { Picker } from '.'
// The Picker's local-server detection uses react-query; queries stay pending
// in tests (no window.hermesDesktop bridge), which renders as "no detected
// row" — the same degraded state as detection failing in the app.
function render(ui: ReactElement) {
const client = new QueryClient({ defaultOptions: { queries: { retry: false } } })
return rtlRender(<QueryClientProvider client={client}>{ui}</QueryClientProvider>)
}
function provider(id: string, name = id): OAuthProvider {
return {
cli_command: `hermes login ${id}`,

View file

@ -1,10 +1,11 @@
import { useStore } from '@nanostores/react'
import { useQuery } from '@tanstack/react-query'
import { useEffect, useMemo, useRef, useState } from 'react'
import { Button } from '@/components/ui/button'
import { Codicon } from '@/components/ui/codicon'
import { Input } from '@/components/ui/input'
import { getGlobalModelOptions } from '@/hermes'
import { detectLocalServers, getGlobalModelOptions } from '@/hermes'
import { useI18n } from '@/i18n'
import { Check, ChevronDown, ChevronLeft, KeyRound, Loader2 } from '@/lib/icons'
import { isProviderSetupErrorMessage } from '@/lib/provider-setup-errors'
@ -22,13 +23,14 @@ import {
peekPendingProviderOAuth,
refreshOnboarding,
saveOnboardingApiKey,
saveOnboardingDetectedOllama,
setOnboardingMode,
startProviderOAuth
} from '@/store/onboarding'
import type { ModelOptionProvider, OAuthProvider } from '@/types/hermes'
import { DocsLink, FlowPanel, Status } from './flow'
import { FeaturedProviderRow, KeyProviderRow, ProviderRow, sortProviders } from './providers'
import { DetectedLocalServerRow, FeaturedProviderRow, KeyProviderRow, ProviderRow, sortProviders } from './providers'
export { FeaturedProviderRow, KeyProviderRow, ProviderRow, providerTitle, sortProviders } from './providers'
@ -399,6 +401,22 @@ export function Picker({ ctx }: { ctx: OnboardingContext }) {
const hasOauth = ordered.length > 0
const apiKeyOptions = useApiKeyCatalog()
// Detected local model servers (Ollama). Non-blocking: the picker renders
// immediately and the detected row appears when the probe answers. Failures
// degrade to "no row" — the manual local-endpoint path still exists.
const localServers = useQuery({
queryKey: ['local-servers-detect'],
queryFn: () => detectLocalServers(),
staleTime: 60_000,
retry: false
})
const detectedOllama = useMemo(() => {
const servers = localServers.data?.servers ?? []
return servers.find(s => s.type === 'ollama' && s.reachable && s.models.length > 0) ?? null
}, [localServers.data])
// localEndpoint forces the key form regardless of `mode` (which a manual
// provider refresh may flip back to 'oauth'); it preselects the local option
// and hides the "back to sign in" link since the user came specifically to
@ -438,6 +456,12 @@ export function Picker({ ctx }: { ctx: OnboardingContext }) {
<div className="grid gap-2">
<div className="grid max-h-[60dvh] gap-2 overflow-y-auto p-1">
{featured ? <FeaturedProviderRow onSelect={select} provider={featured} /> : null}
{detectedOllama ? (
<DetectedLocalServerRow
onConnect={(baseUrl, model) => saveOnboardingDetectedOllama(baseUrl, model, ctx)}
server={detectedOllama}
/>
) : null}
{showRest ? (
<>
{rest.map(p => (

View file

@ -1,7 +1,10 @@
import { useState } from 'react'
import { RowButton } from '@/components/ui/row-button'
import { useI18n } from '@/i18n'
import { Check, ChevronRight, Terminal } from '@/lib/icons'
import type { OAuthProvider } from '@/types/hermes'
import { Check, ChevronRight, Cpu, Loader2, Terminal } from '@/lib/icons'
import { cn } from '@/lib/utils'
import type { LocalServerInfo, OAuthProvider } from '@/types/hermes'
const PROVIDER_DISPLAY: Record<string, { order: number; title: string }> = {
nous: { order: 0, title: 'Nous Portal' },
@ -116,3 +119,83 @@ export function ProviderRow({
</RowButton>
)
}
/**
* A local model server found by /api/local-servers/detect (currently the
* Ollama row; LM Studio detection reuses the shape later). Collapsed: a
* provider row showing "detected · N models installed". Expanded: the
* installed-model list, each row a one-click connect — no URL typing, no
* blind first-model assignment.
*/
export function DetectedLocalServerRow({
onConnect,
server
}: {
onConnect: (baseUrl: string, model: string) => Promise<{ ok: boolean; message?: string }>
server: LocalServerInfo
}) {
const { t } = useI18n()
const copy = t.onboarding.detectedLocal
const [expanded, setExpanded] = useState(false)
const [busyModel, setBusyModel] = useState<null | string>(null)
const [error, setError] = useState('')
const connect = async (model: string) => {
if (busyModel) {
return
}
setBusyModel(model)
setError('')
const result = await onConnect(server.base_url, model)
if (!result.ok) {
setError(result.message || copy.connectFailed)
setBusyModel(null)
}
// On success the overlay unmounts via completeDesktopOnboarding().
}
return (
<div className="rounded-[6px] transition-colors">
<RowButton className={PROVIDER_ROW_CLASS} onClick={() => setExpanded(open => !open)}>
<div className="min-w-0">
<div className="flex items-center gap-2">
<Cpu className="size-4 shrink-0 text-primary" />
<span className="text-[length:var(--conversation-text-font-size)] font-semibold">Ollama</span>
<span className="inline-flex items-center gap-1 bg-primary/10 px-2 py-0.5 text-xs font-medium text-primary">
<Check className="size-3" />
{copy.detected}
</span>
</div>
<p className="mt-1 text-xs leading-5 text-muted-foreground">{copy.modelsInstalled(server.models.length)}</p>
</div>
<ChevronRight
className={cn('size-4 text-muted-foreground transition group-hover:text-foreground', expanded && 'rotate-90')}
/>
</RowButton>
{expanded ? (
<div className="grid gap-0.5 px-3 pb-2">
{server.models.map(model => (
<RowButton
className="group flex w-full items-center justify-between gap-3 rounded-[6px] px-3 py-1.5 text-left font-mono text-xs transition-colors hover:bg-(--ui-control-hover-background)"
disabled={busyModel !== null}
key={model}
onClick={() => void connect(model)}
>
<span className="truncate">{model}</span>
{busyModel === model ? (
<Loader2 className="size-3.5 shrink-0 animate-spin text-muted-foreground" />
) : (
<span className="shrink-0 text-[0.64rem] uppercase tracking-wide text-muted-foreground opacity-0 transition group-hover:opacity-100">
{copy.use}
</span>
)}
</RowButton>
))}
{error ? <p className="px-3 pt-1 text-xs leading-5 text-destructive">{error}</p> : null}
</div>
) : null}
</div>
)
}

View file

@ -19,6 +19,7 @@ import type {
EnvVarInfo,
HermesConfig,
HermesConfigRecord,
LocalServersResponse,
LogsResponse,
McpCatalogResponse,
McpServerSummary,
@ -37,6 +38,8 @@ import type {
OAuthProvidersResponse,
OAuthStartResponse,
OAuthSubmitResponse,
OllamaModelsResponse,
OllamaPullStatus,
PaginatedSessions,
ProfileCreatePayload,
ProfileSetupCommand,
@ -902,6 +905,62 @@ export interface RecommendedDefaultModel {
free_tier: boolean | null
}
// Detect running local model servers (Ollama, LM Studio) by probing their
// well-known ports plus the configured base_url. Results are cached
// server-side for 60s; refresh=true busts the cache.
export function detectLocalServers(opts?: { refresh?: boolean }): Promise<LocalServersResponse> {
return window.hermesDesktop.api<LocalServersResponse>({
...profileScoped(),
path: opts?.refresh ? '/api/local-servers/detect?refresh=1' : '/api/local-servers/detect'
})
}
// Installed + running + recommended models for the local Ollama server.
export function getOllamaModels(): Promise<OllamaModelsResponse> {
return window.hermesDesktop.api<OllamaModelsResponse>({
...profileScoped(),
path: '/api/ollama/models'
})
}
// Start pulling a model from the Ollama registry; poll with pollOllamaPull.
export function startOllamaPull(model: string): Promise<{ job_id: string }> {
return window.hermesDesktop.api<{ job_id: string }>({
...profileScoped(),
path: '/api/ollama/pull',
method: 'POST',
body: { model }
})
}
export function pollOllamaPull(jobId: string): Promise<OllamaPullStatus> {
return window.hermesDesktop.api<OllamaPullStatus>({
...profileScoped(),
path: `/api/ollama/pull/${encodeURIComponent(jobId)}`
})
}
export function deleteOllamaModel(model: string): Promise<{ ok: boolean; message: string }> {
return window.hermesDesktop.api<{ ok: boolean; message: string }>({
...profileScoped(),
path: '/api/ollama/delete',
method: 'POST',
body: { model }
})
}
// Load a model into memory. keep_alive pins residence ("5m", "2h", "-1" =
// indefinite); Ollama applies keep_alive per-load, so this doubles as the
// way to change it for an already-loaded model.
export function loadOllamaModel(model: string, keepAlive?: string): Promise<{ ok: boolean; message: string }> {
return window.hermesDesktop.api<{ ok: boolean; message: string }>({
...profileScoped(),
path: '/api/ollama/load',
method: 'POST',
body: { model, keep_alive: keepAlive ?? null }
})
}
// Recommended default model for a freshly-authenticated provider. Mirrors the
// curation `hermes model` does — for Nous it honors the free/paid tier so a
// free user gets a free model instead of a paid default.

View file

@ -680,6 +680,24 @@ export const en: Translations = {
curator: { label: 'Curator', hint: 'Skill-usage review' }
}
},
ollama: {
categoryTitle: 'Local servers',
running: (n: number) => (n === 1 ? 'Running · 1 model' : `Running · ${n} models`),
notRunning: 'Not running — start Ollama (localhost:11434) to manage and use local models.',
desc: 'Pull, delete, and warm up models on the local Ollama server.',
loaded: 'Loaded',
load: 'Load',
loadFailed: 'Could not load the model',
deleteLabel: (model: string) => `Delete ${model}`,
deleteFailed: 'Could not delete the model',
getModels: 'Get models',
pull: 'Pull',
pullPlaceholder: 'model:tag (e.g. qwen3:8b)',
pullDone: (model: string) => `${model} is ready`,
pullFailed: 'Model pull failed',
kvCacheAdvisory: (model: string, loaded: string, trained: string) =>
`${model} is running a ${loaded} context but supports ${trained}. Setting OLLAMA_KV_CACHE_TYPE=q8_0 on the Ollama server roughly doubles the context that fits in the same memory.`
},
providers: {
connectAccount: 'Connect an account',
haveApiKey: 'Have an API key instead?',
@ -1737,6 +1755,7 @@ export const en: Translations = {
stop: 'Stop',
dismiss: 'Dismiss',
exit: code => `exit ${code}`,
ollamaLoading: (model: string) => `Loading ${model} into memory…`,
coding: {
title: 'Working tree',
noBranch: 'No branch',
@ -1896,6 +1915,12 @@ export const en: Translations = {
connected: 'Connected',
featuredPitch: 'One subscription, 300+ frontier models — the recommended way to run Hermes',
openRouterPitch: 'One key, hundreds of models — a solid default',
detectedLocal: {
detected: 'Detected',
modelsInstalled: (n: number) => (n === 1 ? '1 model installed — runs on this machine' : `${n} models installed — runs on this machine`),
use: 'Use',
connectFailed: 'Could not connect to the local server.'
},
apiKeyOptions: {
openrouter: {
short: 'one key, many models',
@ -1965,6 +1990,8 @@ export const en: Translations = {
noAuthenticatedProviders: 'No authenticated providers.',
pro: 'Pro',
proNeedsSubscription: 'Pro models need a paid Nous subscription.',
noTools: 'No tools',
noToolsTitle: 'This model does not support tool calling — most agent features will not work with it.',
free: 'Free',
freeTier: 'Free tier',
priceTitle: 'Input / Output price per million tokens'

View file

@ -1874,6 +1874,12 @@ export const ja = defineLocale({
connected: '接続済み',
featuredPitch: '1 つのサブスクリプションで 300 以上の最先端モデル — Hermes を実行するための推奨方法',
openRouterPitch: '1 つのキーで数百のモデル — 堅実なデフォルト',
detectedLocal: {
detected: '検出済み',
modelsInstalled: (n: number) => `${n} 個のモデルがインストール済み — このマシンで実行`,
use: '使用',
connectFailed: 'ローカルサーバーに接続できませんでした。'
},
apiKeyOptions: {
openrouter: {
short: '1 つのキーで多くのモデル',
@ -1943,6 +1949,8 @@ export const ja = defineLocale({
noAuthenticatedProviders: '認証済みプロバイダーがありません。',
pro: 'Pro',
proNeedsSubscription: 'Pro モデルには有料の Nous サブスクリプションが必要です。',
noTools: 'ツール非対応',
noToolsTitle: 'このモデルはツール呼び出しに対応していません — ほとんどのエージェント機能が動作しません。',
free: '無料',
freeTier: '無料プラン',
priceTitle: '100 万トークンあたりの入力/出力価格'

View file

@ -580,6 +580,23 @@ export interface Translations {
providerDefault: string
tasks: Record<string, AuxTaskCopy>
}
ollama: {
categoryTitle: string
running: (n: number) => string
notRunning: string
desc: string
loaded: string
load: string
loadFailed: string
deleteLabel: (model: string) => string
deleteFailed: string
getModels: string
pull: string
pullPlaceholder: string
pullDone: (model: string) => string
pullFailed: string
kvCacheAdvisory: (model: string, loaded: string, trained: string) => string
}
providers: {
connectAccount: string
haveApiKey: string
@ -1419,6 +1436,7 @@ export interface Translations {
stop: string
dismiss: string
exit: (code: number) => string
ollamaLoading: (model: string) => string
coding: {
title: string
noBranch: string
@ -1554,6 +1572,12 @@ export interface Translations {
connected: string
featuredPitch: string
openRouterPitch: string
detectedLocal: {
detected: string
modelsInstalled: (n: number) => string
use: string
connectFailed: string
}
apiKeyOptions: Record<string, { short: string; description: string }>
backToSignIn: string
getKey: string
@ -1605,6 +1629,8 @@ export interface Translations {
noAuthenticatedProviders: string
pro: string
proNeedsSubscription: string
noTools: string
noToolsTitle: string
free: string
freeTier: string
priceTitle: string

View file

@ -1817,6 +1817,12 @@ export const zhHant = defineLocale({
connected: '已連線',
featuredPitch: '一個訂閱,300+ 前沿模型 — 執行 Hermes 的建議方式',
openRouterPitch: '一個金鑰,數百個模型 — 穩定的預設選擇',
detectedLocal: {
detected: '已偵測到',
modelsInstalled: (n: number) => `已安裝 ${n} 個模型 — 在本機執行`,
use: '使用',
connectFailed: '無法連線到本機伺服器。'
},
apiKeyOptions: {
openrouter: { short: '一個金鑰,多個模型', description: '用一個金鑰存取數百個模型。適合新安裝的預設選擇。' },
openai: { short: 'GPT 等級模型', description: '直接存取 OpenAI 模型。' },
@ -1880,6 +1886,8 @@ export const zhHant = defineLocale({
noAuthenticatedProviders: '沒有已驗證的提供方。',
pro: 'Pro',
proNeedsSubscription: 'Pro 模型需要付費 Nous 訂閱。',
noTools: '不支援工具',
noToolsTitle: '該模型不支援工具呼叫 — 大多數智能體功能將無法使用。',
free: '免費',
freeTier: '免費層',
priceTitle: '每百萬 Token 的輸入/輸出價格'

View file

@ -868,6 +868,24 @@ export const zh: Translations = {
curator: { label: '维护器', hint: '技能使用审查' }
}
},
ollama: {
categoryTitle: '本地服务器',
running: (n: number) => `运行中 · ${n} 个模型`,
notRunning: '未运行 — 启动 Ollama(localhost:11434)以管理和使用本地模型。',
desc: '在本地 Ollama 服务器上拉取、删除和预热模型。',
loaded: '已加载',
load: '加载',
loadFailed: '无法加载模型',
deleteLabel: (model: string) => `删除 ${model}`,
deleteFailed: '无法删除模型',
getModels: '获取模型',
pull: '拉取',
pullPlaceholder: 'model:tag(例如 qwen3:8b)',
pullDone: (model: string) => `${model} 已就绪`,
pullFailed: '模型拉取失败',
kvCacheAdvisory: (model: string, loaded: string, trained: string) =>
`${model} 当前以 ${loaded} 上下文运行,但支持 ${trained}。在 Ollama 服务器上设置 OLLAMA_KV_CACHE_TYPE=q8_0 可在相同内存下大约翻倍可用上下文。`
},
providers: {
connectAccount: '连接账号',
haveApiKey: '改用 API 密钥?',
@ -1912,6 +1930,7 @@ export const zh: Translations = {
stop: '停止',
dismiss: '关闭',
exit: code => `退出码 ${code}`,
ollamaLoading: (model: string) => `正在将 ${model} 加载到内存…`,
coding: {
title: '工作区',
noBranch: '无分支',
@ -2068,6 +2087,12 @@ export const zh: Translations = {
connected: '已连接',
featuredPitch: '一个订阅,300+ 前沿模型 — 运行 Hermes 的推荐方式',
openRouterPitch: '一个密钥,数百个模型 — 稳妥的默认选择',
detectedLocal: {
detected: '已检测到',
modelsInstalled: (n: number) => `已安装 ${n} 个模型 — 在本机运行`,
use: '使用',
connectFailed: '无法连接到本地服务器。'
},
apiKeyOptions: {
openrouter: { short: '一个密钥,多个模型', description: '用一个密钥访问数百个模型。适合新安装的默认选择。' },
openai: { short: 'GPT 级模型', description: '直接访问 OpenAI 模型。' },
@ -2132,6 +2157,8 @@ export const zh: Translations = {
noAuthenticatedProviders: '没有已认证的提供方。',
pro: 'Pro',
proNeedsSubscription: 'Pro 模型需要付费 Nous 订阅。',
noTools: '不支持工具',
noToolsTitle: '该模型不支持工具调用 — 大多数智能体功能将无法使用。',
free: '免费',
freeTier: '免费层',
priceTitle: '每百万 token 的输入/输出价格'

View file

@ -782,6 +782,42 @@ export async function saveOnboardingApiKey(
}
}
// Configure a DETECTED local Ollama server. Unlike the generic local-endpoint
// path above, the server identity is already fingerprinted and the user has
// picked a concrete model from its installed list, so there is no probe step —
// persist provider=ollama + base_url + model and verify the runtime.
export async function saveOnboardingDetectedOllama(baseUrl: string, model: string, ctx: OnboardingContext) {
const url = baseUrl.trim()
const chosen = model.trim()
if (!url || !chosen) {
return { ok: false, message: 'Pick a model first.' }
}
try {
await setModelAssignment({ scope: 'main', provider: 'ollama', model: chosen, base_url: url })
await ctx.requestGateway('reload.env').catch(() => undefined)
const runtime = await checkRuntime(ctx, 'ollama')
if (!runtime.ready) {
const detail = (runtime.reason ?? '').trim()
return { ok: false, message: detail || `Saved, but Hermes still cannot reach ${url}.` }
}
notifyReady('Ollama')
completeDesktopOnboarding()
ctx.onCompleted?.()
return { ok: true }
} catch (error) {
notifyError(error, 'Could not save Ollama endpoint')
return { ok: false, message: errMessage(error) }
}
}
// Configure a local / self-hosted OpenAI-compatible endpoint (vLLM, llama.cpp,
// Ollama, …). Unlike API-key providers, a local endpoint is defined by its URL
// and usually needs NO key. The runtime resolver reads model.base_url from

View file

@ -278,6 +278,83 @@ export interface ModelOptionProvider {
export interface ModelCapabilities {
fast: boolean
reasoning: boolean
/** True/false when a metadata source reported tool support; absent = unknown. */
tools?: boolean
vision?: boolean
/** Trained/served context window in tokens, when known. */
context_length?: number
/** Local Ollama models only, from the server's native metadata. */
parameter_size?: string
quantization?: string
}
/** One detected local model server from GET /api/local-servers/detect. */
export interface LocalServerInfo {
/** Fingerprinted server type ("ollama", "lm-studio", ...) or the expected
* type for an unreachable well-known port. */
type: string
/** OpenAI-compatible endpoint (server root + /v1). */
base_url: string
reachable: boolean
models: string[]
/** "well-known" (default port probe) or "configured" (model.base_url). */
source: string
}
export interface LocalServersResponse {
servers: LocalServerInfo[]
}
/** Installed model row from GET /api/ollama/models. */
export interface OllamaInstalledModel {
name: string
size_bytes?: number
parameter_size?: string
quantization?: string
family?: string
modified_at?: string
}
/** Loaded model row from GET /api/ollama/models (native /api/ps). */
export interface OllamaRunningModel {
name: string
size_bytes?: number
size_vram_bytes?: number
/** Window the model was actually loaded with. */
context_length?: number
expires_at?: string
}
export interface OllamaRecommendedModel {
model: string
description: string
approx_vram_gb?: number
}
/** Loaded-vs-trained context gap that q8_0 KV quantization would close. */
export interface OllamaKvCacheAdvisory {
model: string
loaded_context: number
trained_context: number
}
export interface OllamaModelsResponse {
reachable: boolean
installed: OllamaInstalledModel[]
running: OllamaRunningModel[]
recommended: OllamaRecommendedModel[]
kv_cache_advisory?: null | OllamaKvCacheAdvisory
}
export interface OllamaPullStatus {
job_id: string
model: string
/** pulling | done | error */
status: string
detail?: string
error_message?: string
total_bytes?: number
completed_bytes?: number
}
export interface ModelOptionsResponse {

View file

@ -151,6 +151,10 @@ SERVICE_PROVIDER_NAMES: Dict[str, str] = {
# any remote service.
LMSTUDIO_NOAUTH_PLACEHOLDER = "dummy-lm-api-key"
# Same role for local Ollama servers, which are unauthenticated by default.
# Sent only to the local server, never to any remote service.
OLLAMA_NOAUTH_PLACEHOLDER = "dummy-ollama-api-key"
# =============================================================================
# Provider Registry
@ -217,6 +221,16 @@ PROVIDER_REGISTRY: Dict[str, ProviderConfig] = {
api_key_env_vars=("LM_API_KEY",),
base_url_env_var="LM_BASE_URL",
),
"ollama": ProviderConfig(
id="ollama",
name="Ollama (local)",
auth_type="api_key",
# 127.0.0.1, not localhost — see _normalize_ollama_runtime_base_url.
inference_base_url="http://127.0.0.1:11434/v1",
# No key env vars: local Ollama servers are unauthenticated by default.
# A key for a reverse-proxied server can be set via model.api_key.
api_key_env_vars=(),
),
"copilot": ProviderConfig(
id="copilot",
name="GitHub Copilot",
@ -747,6 +761,26 @@ def _normalize_lmstudio_runtime_base_url(base_url: str) -> str:
return (root or "http://127.0.0.1:1234") + "/v1"
def _normalize_ollama_runtime_base_url(base_url: str) -> str:
"""Return the OpenAI-compatible local Ollama runtime base URL.
Ollama's native management API lives under ``/api`` while its
OpenAI-compatible chat endpoint lives under ``/v1``. Users paste either
form (or the bare server root) into ``model.base_url``; normalize before
the OpenAI SDK appends ``/chat/completions``.
"""
root = str(base_url or "").strip().rstrip("/")
for suffix in ("/api", "/v1"):
if root.endswith(suffix):
root = root[: -len(suffix)].rstrip("/")
break
# IPv4 loopback, not localhost: Ollama binds 127.0.0.1 by default and
# Windows resolves localhost to ::1 first — a ~2s failed IPv6 connect on
# every chat request otherwise.
root = root.replace("://localhost", "://127.0.0.1")
return (root or "http://127.0.0.1:11434") + "/v1"
# =============================================================================
# Error Types
# =============================================================================
@ -1675,7 +1709,7 @@ def resolve_provider(
"kilo": "kilocode", "kilo-code": "kilocode", "kilo-gateway": "kilocode",
"lmstudio": "lmstudio", "lm-studio": "lmstudio", "lm_studio": "lmstudio",
# Local server aliases — route through the generic custom provider
"ollama": "custom", "ollama_cloud": "ollama-cloud",
"ollama": "ollama", "ollama_cloud": "ollama-cloud",
"vllm": "custom", "llamacpp": "custom",
"llama.cpp": "custom", "llama-cpp": "custom",
}
@ -6386,6 +6420,11 @@ def resolve_api_key_provider_credentials(provider_id: str) -> Dict[str, Any]:
api_key = LMSTUDIO_NOAUTH_PLACEHOLDER
key_source = key_source or "default"
# Same for local Ollama servers (unauthenticated by default).
if not api_key and provider_id == "ollama":
api_key = OLLAMA_NOAUTH_PLACEHOLDER
key_source = key_source or "default"
env_url = ""
if pconfig.base_url_env_var:
env_url = os.getenv(pconfig.base_url_env_var, "").strip()
@ -6422,6 +6461,9 @@ def resolve_api_key_provider_credentials(provider_id: str) -> Dict[str, Any]:
if provider_id == "lmstudio":
base_url = _normalize_lmstudio_runtime_base_url(base_url)
if provider_id == "ollama":
base_url = _normalize_ollama_runtime_base_url(base_url)
# Last-resort guard: an API-key provider must never hand back an empty
# base URL (a set-but-empty COPILOT_API_BASE_URL or similar env override
# otherwise wedges chat inference — #50252).

View file

@ -33,6 +33,8 @@ Substrate facts (verified May 2026):
from __future__ import annotations
import threading
from dataclasses import dataclass, replace
from typing import Optional
@ -252,13 +254,23 @@ def build_models_payload(
def _apply_capabilities(rows: list[dict]) -> None:
"""Attach a ``{model: {fast, reasoning}}`` map to each provider row.
"""Attach a per-model capabilities map to each provider row.
Each entry: ``{fast, reasoning, tools?, vision?, context_length?,
parameter_size?, quantization?}``. Only ``fast`` and ``reasoning`` are
always present; the rest are included when a metadata source actually
reported them, so the UI can distinguish "no tool support" from "unknown".
`fast` mirrors ``model_supports_fast_mode`` (the same gate the runtime
enforces). `reasoning` comes from the models.dev catalog when known and
defaults to True otherwise — the effort dial is broadly accepted and a
no-op on models that ignore it, whereas hiding it from a capable-but-
uncatalogued model is the worse failure.
Local Ollama rows are enriched from the server's native ``/api/show``
instead of models.dev: local tags are arbitrary and quant-specific
(``qwen3:27b-q4_K_M``, user Modelfile creations), so the catalog rarely
matches, while ``/api/show`` is authoritative for what is on disk.
"""
from hermes_cli.models import model_supports_fast_mode
@ -269,25 +281,118 @@ def _apply_capabilities(rows: list[dict]) -> None:
for row in rows:
slug = row.get("slug") or ""
caps: dict[str, dict[str, bool]] = {}
caps: dict[str, dict] = {}
for model in row.get("models") or []:
reasoning = True
entry: dict = {
"fast": bool(model_supports_fast_mode(model)),
"reasoning": True,
}
if get_model_capabilities is not None and slug:
try:
meta = get_model_capabilities(slug, model)
if meta is not None:
reasoning = bool(meta.supports_reasoning)
entry["reasoning"] = bool(meta.supports_reasoning)
entry["tools"] = bool(meta.supports_tools)
entry["vision"] = bool(meta.supports_vision)
if meta.context_window:
entry["context_length"] = int(meta.context_window)
except Exception:
reasoning = True
pass
caps[model] = {
"fast": bool(model_supports_fast_mode(model)),
"reasoning": reasoning,
}
caps[model] = entry
row["capabilities"] = caps
_apply_ollama_native_capabilities(rows)
def _apply_ollama_native_capabilities(rows: list[dict]) -> None:
"""Override local Ollama rows' capabilities with native ``/api/show`` data.
Cache-only on the request path: the picker/settings payload must never
wait on per-model ``/api/show`` round-trips (they serialize and stall the
settings skeleton). Missing metadata is backfilled by a background thread
and appears on the next payload build.
"""
ollama_rows = [r for r in rows if str(r.get("slug", "")).lower() == "ollama"]
if not ollama_rows:
return
import time as _time
now = _time.time()
missing: list[tuple[str, str]] = []
for row in ollama_rows:
base_url = str(row.get("api_url", "") or "").strip() or None
caps = row.get("capabilities") or {}
for model in row.get("models") or []:
cache_key = (base_url or "", model)
cached = _OLLAMA_META_CACHE.get(cache_key)
if cached is None or (now - cached[0]) >= _OLLAMA_META_CACHE_TTL:
missing.append(cache_key)
continue
meta = cached[1]
if not meta:
continue
entry = caps.setdefault(model, {"fast": False, "reasoning": True})
native = meta.get("capabilities")
if isinstance(native, list):
entry["tools"] = "tools" in native
entry["vision"] = "vision" in native
entry["reasoning"] = "thinking" in native
if meta.get("context_length"):
entry["context_length"] = int(meta["context_length"])
if meta.get("parameter_size"):
entry["parameter_size"] = meta["parameter_size"]
if meta.get("quantization"):
entry["quantization"] = meta["quantization"]
row["capabilities"] = caps
if missing:
_start_ollama_meta_backfill(missing)
def _start_ollama_meta_backfill(keys: list[tuple[str, str]]) -> None:
"""Fetch missing model metadata off the request path."""
import time as _time
with _OLLAMA_META_BACKFILL_LOCK:
todo = [k for k in keys if k not in _OLLAMA_META_IN_FLIGHT]
if not todo:
return
_OLLAMA_META_IN_FLIGHT.update(todo)
def _backfill() -> None:
try:
from hermes_cli.models import fetch_ollama_model_metadata
for base_url, model in todo:
try:
meta = fetch_ollama_model_metadata(
model, base_url=base_url or None, timeout=5.0
)
if meta:
_OLLAMA_META_CACHE[(base_url, model)] = (_time.time(), meta)
except Exception:
pass
finally:
with _OLLAMA_META_BACKFILL_LOCK:
_OLLAMA_META_IN_FLIGHT.difference_update(todo)
threading.Thread(
target=_backfill, daemon=True, name="ollama-meta-backfill"
).start()
# Metadata for an installed model changes only when the user re-pulls or
# edits a Modelfile — an hour of staleness is fine and keeps repeated picker
# opens off the per-model /api/show round-trips.
_OLLAMA_META_CACHE: dict = {}
_OLLAMA_META_CACHE_TTL = 3600.0
_OLLAMA_META_BACKFILL_LOCK = threading.Lock()
_OLLAMA_META_IN_FLIGHT: set = set()
# ─── Internal: row post-processing ──────────────────────────────────────
@ -342,6 +447,14 @@ def _filter_explicit_provider_rows(rows: list[dict], ctx: ConfigContext) -> list
# provider. Hide it from explicit-only pickers unless it is the
# current provider (handled above).
continue
if slug == "ollama":
# Local Ollama servers are keyless by design — reachability is
# the credential, and the row only exists when the server
# answered the probe. A running local server is as explicit as a
# pasted API key; is_provider_explicitly_configured() can't see
# it because there is no env var or auth-store entry to find.
kept.append(row)
continue
if is_provider_explicitly_configured(slug):
kept.append(row)
return kept
@ -378,11 +491,16 @@ def _apply_picker_hints(rows: list[dict]) -> None:
)
row["auth_type"] = auth_type
row["key_env"] = key_env
row["warning"] = (
f"paste {key_env} to activate"
if auth_type == "api_key" and key_env
else f"run `hermes model` to configure ({auth_type})"
)
if row["slug"] == "ollama":
# Keyless local server: the fix for an unconfigured row is
# starting the server, not pasting a key.
row["warning"] = "start Ollama (localhost:11434) to activate"
else:
row["warning"] = (
f"paste {key_env} to activate"
if auth_type == "api_key" and key_env
else f"run `hermes model` to configure ({auth_type})"
)
def _reorder_canonical(rows: list[dict]) -> list[dict]:

View file

@ -1642,6 +1642,21 @@ def list_authenticated_providers(
live = [current_model]
curated["lmstudio"] = live
# Local Ollama server: no static catalog and no key requirement — the row
# exists when the server is reachable (or is the configured provider).
# Localhost connection-refused fails in milliseconds, so this stays cheap
# when no server is running. Base URL precedence mirrors LM Studio:
# active config's base_url (when current provider is ollama) > localhost.
if "ollama" not in curated:
from hermes_cli.models import fetch_ollama_local_models
is_current_ollama = current_provider.strip().lower() == "ollama"
ollama_base = current_base_url if is_current_ollama and current_base_url else None
ollama_live = fetch_ollama_local_models(base_url=ollama_base, timeout=1.5)
if not ollama_live and is_current_ollama and current_model:
ollama_live = [current_model]
if ollama_live or is_current_ollama:
curated["ollama"] = ollama_live
# --- 1. Check Hermes-mapped providers ---
from hermes_cli.models import _AGGREGATOR_PROVIDERS as _AGG_PROVIDERS
from hermes_cli.providers import ALIASES as _PROVIDER_ALIAS_TABLE
@ -1807,6 +1822,11 @@ def list_authenticated_providers(
has_creds = True
except Exception as exc:
logger.debug("Anthropic external creds check failed: %s", exc)
# Local Ollama server: unauthenticated by design, so reachability is
# the credential. The curated probe above only adds the key when the
# server responded (or it is the configured provider).
if not has_creds and hermes_slug == "ollama":
has_creds = "ollama" in curated
if not has_creds:
continue

View file

@ -1036,6 +1036,7 @@ CANONICAL_PROVIDERS: list[ProviderEntry] = [
ProviderEntry("moa", "Mixture of Agents", "Mixture of Agents (named presets; aggregator acts after reference models)"),
ProviderEntry("novita", "NovitaAI", "NovitaAI (Cloud: Model API, Agent Sandbox, GPU Cloud)"),
ProviderEntry("lmstudio", "LM Studio", "LM Studio (Local desktop app with built-in model server)"),
ProviderEntry("ollama", "Ollama (local)", "Ollama (Local model server on localhost:11434)"),
ProviderEntry("anthropic", "Anthropic", "Anthropic (Claude models via API key or Claude Code)"),
ProviderEntry("openai-codex", "OpenAI Codex", "OpenAI Codex (Codex CLI via ChatGPT subscription or API key)"),
ProviderEntry("openai-api", "OpenAI API", "OpenAI API (api.openai.com, API key)"),
@ -1273,7 +1274,7 @@ _PROVIDER_ALIASES = {
"lmstudio": "lmstudio",
"lm-studio": "lmstudio",
"lm_studio": "lmstudio",
"ollama": "custom", # bare "ollama" = local; use "ollama-cloud" for cloud
"ollama": "ollama", # bare "ollama" = local server; use "ollama-cloud" for cloud
"ollama_cloud": "ollama-cloud",
}
@ -2367,6 +2368,17 @@ def provider_model_ids(provider: Optional[str], *, force_refresh: bool = False)
live = fetch_ollama_cloud_models(force_refresh=force_refresh)
if live:
return live
if normalized == "ollama":
# Local Ollama server: list installed models from the native API.
# Base URL precedence: active config's base_url (when the current
# provider is ollama) > localhost default.
model_cfg = _get_model_config_dict()
cfg_provider = normalize_provider(str(model_cfg.get("provider", "") or ""))
cfg_base_url = str(model_cfg.get("base_url", "") or "").strip() if cfg_provider == "ollama" else ""
cfg_api_key = str(model_cfg.get("api_key", "") or "").strip() if cfg_provider == "ollama" else ""
return fetch_ollama_local_models(
api_key=cfg_api_key, base_url=cfg_base_url or None, timeout=3.0
)
if normalized in ("openai", "openai-api"):
api_key = os.getenv("OPENAI_API_KEY", "").strip()
if api_key:
@ -2637,6 +2649,13 @@ def cached_provider_model_ids(
if not normalized:
return []
# Local Ollama: never disk-cache. The installed-model list changes with
# every pull/delete/server restart, the live /api/tags fetch costs
# milliseconds (fast TCP pre-check when down), and a stale cached list
# surfaces models that no longer exist in the picker.
if normalized == "ollama":
return provider_model_ids(normalized, force_refresh=force_refresh)
cache = _load_provider_models_cache()
fp = _credential_fingerprint(normalized)
entry = cache.get(normalized)
@ -3157,6 +3176,388 @@ def _fetch_github_models(api_key: Optional[str] = None, timeout: float = 5.0) ->
return [item.get("id", "") for item in catalog if item.get("id")]
# ── Local Ollama server (native /api metadata) ──────────────────────────────
#
# The OpenAI-compatible /v1/models endpoint returns bare model ids. Ollama's
# native API is much richer: /api/tags lists installed models with size and
# family, /api/show returns capabilities (tools/vision/thinking), parameter
# size, quantization, and context length. The picker uses this to badge
# models honestly (an agent needs tool support) — inference stays on /v1.
def _ollama_server_root(base_url: Optional[str]) -> Optional[str]:
"""Server root for Ollama's native API (strip a trailing /v1 or /api).
``localhost`` is rewritten to ``127.0.0.1``: Ollama binds IPv4 loopback
by default, and on Windows ``localhost`` resolves to ``::1`` first, so
every request pays a ~2s failed IPv6 connect before falling back.
"""
root = str(base_url or "").strip().rstrip("/")
if not root:
return None
for suffix in ("/api", "/v1"):
if root.endswith(suffix):
root = root[: -len(suffix)].rstrip("/")
break
if "://" not in root:
root = "http://" + root
return root.replace("://localhost", "://127.0.0.1")
# One failed reachability probe covers every fetcher call for a while, so an
# offline server costs the model picker a single sub-second check instead of
# an HTTP timeout per call (multiplied per model by capability enrichment).
_OLLAMA_UNREACHABLE_CACHE: dict = {}
_OLLAMA_UNREACHABLE_TTL = 30.0
def _ollama_server_reachable(
server_root: str, timeout: float = 0.3, use_cache: bool = True
) -> bool:
"""Fast TCP pre-check before any native-API read.
The picker and settings pages call the fetchers below on every open, so
an unreachable server must fail in milliseconds — a raw connect either
succeeds or is refused near-instantly on localhost, and the short timeout
caps the silently-dropped case. Failures are cached briefly so bulk
payload builds don't repeat the probe per model.
``use_cache=False`` forces a fresh probe — for status endpoints that must
notice a server coming up immediately (a cached failure would report a
just-started server as down for the rest of the TTL). A successful fresh
probe clears the negative entry for every cached caller too.
"""
import socket
import time as _time
from urllib.parse import urlparse
now = _time.monotonic()
if use_cache:
failed_at = _OLLAMA_UNREACHABLE_CACHE.get(server_root)
if failed_at is not None and (now - failed_at) < _OLLAMA_UNREACHABLE_TTL:
return False
try:
parsed = urlparse(server_root)
host = parsed.hostname or "localhost"
port = parsed.port or (443 if parsed.scheme == "https" else 80)
with socket.create_connection((host, port), timeout=timeout):
pass
_OLLAMA_UNREACHABLE_CACHE.pop(server_root, None)
return True
except Exception:
_OLLAMA_UNREACHABLE_CACHE[server_root] = now
return False
def _ollama_request_headers(api_key: Optional[str] = None) -> dict:
headers = {"Content-Type": "application/json"}
key = str(api_key or "").strip()
if key:
headers["Authorization"] = f"Bearer {key}"
return headers
def fetch_ollama_local_models(
api_key: Optional[str] = None,
base_url: Optional[str] = None,
timeout: float = 5.0,
) -> list[str]:
"""List installed models from a local Ollama server's native ``/api/tags``.
Returns model names (``model:tag``) or an empty list when the server is
unreachable or the response is malformed.
"""
server_root = _ollama_server_root(base_url or "http://localhost:11434")
if not server_root:
return []
if not _ollama_server_reachable(server_root):
return []
request = urllib.request.Request(
server_root + "/api/tags", headers=_ollama_request_headers(api_key)
)
try:
with urllib.request.urlopen(request, timeout=timeout) as resp:
payload = json.loads(resp.read().decode())
except Exception:
return []
models = payload.get("models")
if not isinstance(models, list):
return []
names: list[str] = []
for raw in models:
if not isinstance(raw, dict):
continue
name = str(raw.get("name") or raw.get("model") or "").strip()
if name and name not in names:
names.append(name)
return names
def fetch_ollama_model_metadata(
model: str,
base_url: Optional[str] = None,
api_key: Optional[str] = None,
timeout: float = 5.0,
) -> Optional[dict]:
"""Fetch one model's metadata from a local Ollama server's ``/api/show``.
Returns a dict with (all fields best-effort, absent keys omitted):
- ``capabilities``: list[str] — e.g. ["completion", "tools", "vision", "thinking"]
- ``parameter_size``: str — e.g. "27B"
- ``quantization``: str — e.g. "Q4_K_M"
- ``context_length``: int — trained context from GGUF metadata
- ``family``: str — model family, e.g. "qwen3"
Returns None when the server is unreachable or the model is unknown.
"""
server_root = _ollama_server_root(base_url or "http://localhost:11434")
if not server_root or not str(model or "").strip():
return None
if not _ollama_server_reachable(server_root):
return None
body = json.dumps({"name": str(model).strip()}).encode()
request = urllib.request.Request(
server_root + "/api/show",
data=body,
headers=_ollama_request_headers(api_key),
method="POST",
)
try:
with urllib.request.urlopen(request, timeout=timeout) as resp:
payload = json.loads(resp.read().decode())
except Exception:
return None
if not isinstance(payload, dict):
return None
meta: dict = {}
caps = payload.get("capabilities")
if isinstance(caps, list):
meta["capabilities"] = [str(c).strip().lower() for c in caps if isinstance(c, str)]
details = payload.get("details")
if isinstance(details, dict):
if details.get("parameter_size"):
meta["parameter_size"] = str(details["parameter_size"]).strip()
if details.get("quantization_level"):
meta["quantization"] = str(details["quantization_level"]).strip()
if details.get("family"):
meta["family"] = str(details["family"]).strip()
# Context length lives in model_info under "<architecture>.context_length".
model_info = payload.get("model_info")
if isinstance(model_info, dict):
for key, value in model_info.items():
if str(key).endswith(".context_length") and isinstance(value, (int, float)):
meta["context_length"] = int(value)
break
return meta or None
# Curated models that work well as Hermes agent backends on a local Ollama
# server. Every entry supports tool calling. ``approx_vram_gb`` is the rough
# working-set need at the default quantization — used to annotate the
# recommendation (never to gate it: Ollama falls back to partial offload).
OLLAMA_RECOMMENDED_MODELS: list[dict] = [
{"model": "qwen3:8b", "description": "Qwen3 8B — tools + thinking, good starter", "approx_vram_gb": 8},
{"model": "gpt-oss:20b", "description": "OpenAI open-weight 20B — tools + thinking", "approx_vram_gb": 16},
{"model": "qwen3:32b", "description": "Qwen3 32B — strong agent at 24GB", "approx_vram_gb": 24},
{"model": "mistral-small3.2:24b", "description": "Mistral Small 3.2 — tools + vision", "approx_vram_gb": 20},
{"model": "llama3.3:70b", "description": "Llama 3.3 70B — tools, needs 48GB+", "approx_vram_gb": 48},
{"model": "gpt-oss:120b", "description": "OpenAI open-weight 120B — tools + thinking, needs 80GB", "approx_vram_gb": 80},
]
def fetch_ollama_installed_model_details(
api_key: Optional[str] = None,
base_url: Optional[str] = None,
timeout: float = 5.0,
) -> list[dict]:
"""Installed models with size/details from native ``/api/tags``.
Returns ``[{name, size_bytes, parameter_size, quantization, family,
modified_at}]`` (absent fields omitted), or ``[]`` when unreachable.
"""
server_root = _ollama_server_root(base_url or "http://localhost:11434")
if not server_root:
return []
if not _ollama_server_reachable(server_root):
return []
request = urllib.request.Request(
server_root + "/api/tags", headers=_ollama_request_headers(api_key)
)
try:
with urllib.request.urlopen(request, timeout=timeout) as resp:
payload = json.loads(resp.read().decode())
except Exception:
return []
models = payload.get("models")
if not isinstance(models, list):
return []
out: list[dict] = []
for raw in models:
if not isinstance(raw, dict):
continue
name = str(raw.get("name") or raw.get("model") or "").strip()
if not name:
continue
entry: dict = {"name": name}
if isinstance(raw.get("size"), (int, float)):
entry["size_bytes"] = int(raw["size"])
if raw.get("modified_at"):
entry["modified_at"] = str(raw["modified_at"])
details = raw.get("details")
if isinstance(details, dict):
if details.get("parameter_size"):
entry["parameter_size"] = str(details["parameter_size"]).strip()
if details.get("quantization_level"):
entry["quantization"] = str(details["quantization_level"]).strip()
if details.get("family"):
entry["family"] = str(details["family"]).strip()
out.append(entry)
return out
def fetch_ollama_running_models(
api_key: Optional[str] = None,
base_url: Optional[str] = None,
timeout: float = 5.0,
) -> list[dict]:
"""Loaded models from native ``/api/ps``.
Returns ``[{name, size_bytes, size_vram_bytes, context_length,
expires_at}]`` (absent fields omitted), or ``[]`` when unreachable or
nothing is loaded. ``context_length`` is the window the model was
actually loaded with — the input for the KV-cache advisory.
"""
server_root = _ollama_server_root(base_url or "http://localhost:11434")
if not server_root:
return []
if not _ollama_server_reachable(server_root):
return []
request = urllib.request.Request(
server_root + "/api/ps", headers=_ollama_request_headers(api_key)
)
try:
with urllib.request.urlopen(request, timeout=timeout) as resp:
payload = json.loads(resp.read().decode())
except Exception:
return []
models = payload.get("models")
if not isinstance(models, list):
return []
out: list[dict] = []
for raw in models:
if not isinstance(raw, dict):
continue
name = str(raw.get("name") or raw.get("model") or "").strip()
if not name:
continue
entry: dict = {"name": name}
if isinstance(raw.get("size"), (int, float)):
entry["size_bytes"] = int(raw["size"])
if isinstance(raw.get("size_vram"), (int, float)):
entry["size_vram_bytes"] = int(raw["size_vram"])
if isinstance(raw.get("context_length"), (int, float)):
entry["context_length"] = int(raw["context_length"])
if raw.get("expires_at"):
entry["expires_at"] = str(raw["expires_at"])
out.append(entry)
return out
def delete_ollama_model(
model: str,
api_key: Optional[str] = None,
base_url: Optional[str] = None,
timeout: float = 15.0,
) -> tuple[bool, str]:
"""Delete an installed model via native ``DELETE /api/delete``.
Returns ``(ok, message)`` — message is empty on success.
"""
server_root = _ollama_server_root(base_url or "http://localhost:11434")
name = str(model or "").strip()
if not server_root or not name:
return False, "missing model or server"
body = json.dumps({"model": name}).encode()
request = urllib.request.Request(
server_root + "/api/delete",
data=body,
headers=_ollama_request_headers(api_key),
method="DELETE",
)
try:
with urllib.request.urlopen(request, timeout=timeout) as resp:
resp.read()
return True, ""
except urllib.error.HTTPError as exc:
if exc.code == 404:
return False, f"model {name!r} not found"
return False, f"delete failed with HTTP {exc.code}"
except Exception:
return False, "could not reach the Ollama server"
def preload_ollama_model(
model: str,
keep_alive: Optional[str] = None,
api_key: Optional[str] = None,
base_url: Optional[str] = None,
timeout: float = 300.0,
) -> tuple[bool, str]:
"""Load a model into memory via a prompt-less native ``/api/generate``.
Ollama loads the model and returns without generating. ``keep_alive``
(e.g. ``"5m"``, ``"2h"``, ``"-1"`` for indefinite) pins how long it stays
resident; Ollama applies keep_alive per-load, so re-issue this call to
change it. The generous timeout covers cold weight loads on big models.
Returns ``(ok, message)`` — message is empty on success.
"""
server_root = _ollama_server_root(base_url or "http://localhost:11434")
name = str(model or "").strip()
if not server_root or not name:
return False, "missing model or server"
req_body: dict = {"model": name}
ka = str(keep_alive or "").strip()
if ka:
# Ollama accepts a Go duration string or a number of seconds;
# "-1" means keep loaded indefinitely.
req_body["keep_alive"] = int(ka) if ka.lstrip("-").isdigit() else ka
request = urllib.request.Request(
server_root + "/api/generate",
data=json.dumps(req_body).encode(),
headers=_ollama_request_headers(api_key),
method="POST",
)
try:
with urllib.request.urlopen(request, timeout=timeout) as resp:
resp.read()
return True, ""
except urllib.error.HTTPError as exc:
if exc.code == 404:
return False, f"model {name!r} not found — pull it first"
return False, f"load failed with HTTP {exc.code}"
except Exception:
return False, "could not reach the Ollama server"
_COPILOT_MODEL_ALIASES = {
"openai/gpt-5": "gpt-5-mini",
"openai/gpt-5-chat": "gpt-5-mini",

View file

@ -201,6 +201,14 @@ HERMES_OVERLAYS: Dict[str, HermesOverlay] = {
base_url_override="https://ollama.com/v1",
base_url_env_var="OLLAMA_BASE_URL",
),
# Local Ollama server. Key optional (most local servers are unauthenticated);
# model.base_url in config overrides the default for remote boxes.
# 127.0.0.1, not localhost: Windows resolves localhost to ::1 first and
# Ollama binds IPv4 loopback — dodge the ~2s IPv6 connect penalty.
"ollama": HermesOverlay(
transport="openai_chat",
base_url_override="http://127.0.0.1:11434/v1",
),
# Azure Foundry: supports both OpenAI-style and Anthropic-style endpoints.
# The transport is determined at runtime from config.yaml model.api_mode.
"azure-foundry": HermesOverlay(
@ -347,7 +355,7 @@ ALIASES: Dict[str, str] = {
"lmstudio": "lmstudio",
"lm-studio": "lmstudio",
"lm_studio": "lmstudio",
"ollama": "custom", # bare "ollama" = local; use "ollama-cloud" for cloud
"ollama": "ollama", # bare "ollama" = local server; use "ollama-cloud" for cloud
"vllm": "local",
"llamacpp": "local",
"llama.cpp": "local",
@ -369,6 +377,7 @@ _LABEL_OVERRIDES: Dict[str, str] = {
"gmi": "GMI Cloud",
"tencent-tokenhub": "Tencent TokenHub",
"lmstudio": "LM Studio",
"ollama": "Ollama (local)",
"local": "Local endpoint",
"bedrock": "AWS Bedrock",
"ollama-cloud": "Ollama Cloud",

View file

@ -2034,6 +2034,8 @@ def resolve_runtime_provider(
base_url = normalize_opencode_base_url(provider, api_mode, base_url)
if provider == "lmstudio":
base_url = auth_mod._normalize_lmstudio_runtime_base_url(base_url)
if provider == "ollama":
base_url = auth_mod._normalize_ollama_runtime_base_url(base_url)
return {
"provider": provider,
"api_mode": api_mode,

View file

@ -5345,6 +5345,434 @@ def get_recommended_default_model(provider: str = ""):
return {"provider": slug, "model": "", "free_tier": None}
# Well-known local model-server ports probed by /api/local-servers/detect.
# Each entry: (server type reported by detect_local_server_type, default root).
_LOCAL_SERVER_PROBE_ROOTS: list[tuple[str, str]] = [
("ollama", "http://127.0.0.1:11434"),
("lm-studio", "http://127.0.0.1:1234"),
]
# Session-lived cache: detection results change rarely (a server starting or
# stopping), and onboarding + settings may both request them in quick
# succession. refresh=1 busts it, mirroring /api/model/options.
_local_servers_cache: dict = {"at": 0.0, "servers": None}
_LOCAL_SERVERS_CACHE_TTL = 60.0
def _detect_local_servers_sync(profile: Optional[str]) -> list[dict]:
"""Probe well-known local ports (plus the configured base_url) for
running model servers.
Returns one entry per candidate root:
{type, base_url, reachable, models: [...], source}
``base_url`` is the OpenAI-compatible endpoint (``<root>/v1``).
``type`` is ``detect_local_server_type()``'s fingerprint when reachable,
else the expected type for the well-known port. The response shape keeps
room for a future installed/running/managed lifecycle distinction.
"""
from agent.model_metadata import detect_local_server_type, is_local_endpoint
from hermes_cli.models import (
_ollama_server_reachable,
fetch_ollama_local_models,
probe_lmstudio_models,
)
candidates: list[tuple[str, str, str]] = [
(expected, root, "well-known") for expected, root in _LOCAL_SERVER_PROBE_ROOTS
]
# Also probe the configured base_url when it is local and isn't one of
# the defaults — the remote-Ollama-over-Tailscale case. Remote provider
# URLs (api.anthropic.com, ...) are never local servers; skip them.
try:
with _profile_scope(profile):
cfg = load_config()
model_cfg = cfg.get("model") or {}
cfg_url = str(model_cfg.get("base_url", "") or "").strip().rstrip("/")
if cfg_url and is_local_endpoint(cfg_url):
cfg_root = cfg_url[:-3].rstrip("/") if cfg_url.endswith("/v1") else cfg_url
known_roots = {root for _, root in _LOCAL_SERVER_PROBE_ROOTS}
if cfg_root and cfg_root not in known_roots:
candidates.append(("", cfg_root, "configured"))
except Exception:
pass
servers: list[dict] = []
for expected_type, root, source in candidates:
detected = None
# Fast TCP pre-check before the HTTP fingerprint: a down server must
# cost milliseconds, not detect_local_server_type's sequential HTTP
# timeouts (which stack across candidates and blow the client's
# request timeout).
if _ollama_server_reachable(root):
try:
detected = detect_local_server_type(root)
except Exception:
detected = None
reachable = detected is not None
models: list[str] = []
if reachable:
try:
if detected == "ollama":
models = fetch_ollama_local_models(base_url=root, timeout=2.0)
elif detected == "lm-studio":
models = probe_lmstudio_models(base_url=root, timeout=2.0) or []
except Exception:
models = []
servers.append(
{
"type": detected or expected_type or "unknown",
"base_url": root + "/v1",
"reachable": reachable,
"models": models,
"source": source,
}
)
return servers
@app.get("/api/local-servers/detect")
async def detect_local_servers(profile: Optional[str] = None, refresh: bool = False):
"""Detect running local model servers (Ollama, LM Studio).
Probes well-known localhost ports plus the configured ``model.base_url``,
fingerprinting each with ``detect_local_server_type()`` and listing the
installed models for recognized servers. Onboarding uses this to offer a
detected server as a provider choice instead of asking for a URL.
"""
import time as _time
now = _time.time()
if (
not refresh
and _local_servers_cache["servers"] is not None
and (now - _local_servers_cache["at"]) < _LOCAL_SERVERS_CACHE_TTL
):
return {"servers": _local_servers_cache["servers"]}
try:
servers = await asyncio.to_thread(_detect_local_servers_sync, profile)
except Exception:
_log.exception("GET /api/local-servers/detect failed")
raise HTTPException(status_code=500, detail="Local server detection failed")
_local_servers_cache["servers"] = servers
_local_servers_cache["at"] = now
return {"servers": servers}
# ---------------------------------------------------------------------------
# Local Ollama model management
# ---------------------------------------------------------------------------
#
# Management actions (list/pull/delete/load) go through Ollama's native /api
# surface; chat inference stays on the OpenAI-compatible /v1 endpoint. Pull
# runs as a background job the UI polls — the same worker+poll shape as the
# OAuth device-code sessions above.
_ollama_pull_jobs: dict = {}
_ollama_pull_jobs_lock = threading.Lock()
_OLLAMA_PULL_JOB_TTL = 3600.0
class OllamaModelAction(BaseModel):
model: str = ""
keep_alive: Optional[str] = None
profile: Optional[str] = None
def _on_ollama_models_changed() -> None:
"""Bust the cached ollama model-id list after a pull/delete.
The picker payload reads ``cached_provider_model_ids("ollama")`` (1h disk
TTL); without this a just-pulled model doesn't appear in the model picker
until the cache expires.
"""
try:
from hermes_cli.models import clear_provider_models_cache
clear_provider_models_cache("ollama")
except Exception:
pass
def _ollama_connection(profile: Optional[str]) -> tuple[str, str]:
"""Resolve (base_url, api_key) for the local Ollama server.
Uses model.base_url/api_key from config when the configured provider is
ollama; otherwise the localhost default. Returns the server root form
accepted by the hermes_cli.models helpers (they normalize /v1 away).
"""
base_url = ""
api_key = ""
try:
with _profile_scope(profile):
cfg = load_config()
model_cfg = cfg.get("model") or {}
provider = str(model_cfg.get("provider", "") or "").strip().lower()
if provider == "ollama":
base_url = str(model_cfg.get("base_url", "") or "").strip()
api_key = str(model_cfg.get("api_key", "") or "").strip()
except Exception:
pass
return base_url or "http://127.0.0.1:11434", api_key
@app.get("/api/ollama/models")
async def get_ollama_models(profile: Optional[str] = None):
"""Installed + running models and the curated recommendations.
Single round-trip payload for the desktop management panel:
``{reachable, installed: [...], running: [...], recommended: [...]}``.
``recommended`` excludes already-installed models.
"""
from hermes_cli.models import (
OLLAMA_RECOMMENDED_MODELS,
fetch_ollama_installed_model_details,
fetch_ollama_running_models,
)
base_url, api_key = _ollama_connection(profile)
# Fresh reachability probe first (bypasses the negative cache): this is
# the status endpoint the desktop polls, so it must notice a just-started
# server immediately. Success clears the cached failure, which un-blinds
# the fetchers below in the same request.
from hermes_cli.models import _ollama_server_reachable, _ollama_server_root
root = _ollama_server_root(base_url)
reachable = bool(root) and await asyncio.to_thread(
_ollama_server_reachable, root, 0.3, False
)
installed: list = []
running: list = []
if reachable:
def _collect():
return (
fetch_ollama_installed_model_details(
api_key=api_key, base_url=base_url, timeout=3.0
),
fetch_ollama_running_models(
api_key=api_key, base_url=base_url, timeout=3.0
),
)
try:
installed, running = await asyncio.to_thread(_collect)
except Exception:
_log.exception("GET /api/ollama/models failed")
raise HTTPException(
status_code=500, detail="Failed to query the Ollama server"
)
installed_names = {m["name"] for m in installed}
# A recommendation is "installed" if any tag of the same base model is —
# qwen3:8b covers qwen3:8b-q4_K_M etc.
installed_bases = {n.split(":", 1)[0] for n in installed_names}
recommended = [
r
for r in OLLAMA_RECOMMENDED_MODELS
if r["model"] not in installed_names
and r["model"].split(":", 1)[0] not in installed_bases
]
# KV-cache advisory (see docs/plans/2026-07-08-001): Ollama runs the KV
# cache at f16 unless the operator sets OLLAMA_KV_CACHE_TYPE, and never
# quantizes it to fit a larger window. When a loaded model runs at well
# under half its trained context, q8_0 KV would roughly double the usable
# window in the same memory — surface that once per panel load. Inference:
# we cannot read a remote server's env, only compare loaded vs trained.
kv_cache_advisory = None
try:
from hermes_cli.models import fetch_ollama_model_metadata
for run in running:
loaded_ctx = int(run.get("context_length") or 0)
if loaded_ctx <= 0:
continue
meta = await asyncio.to_thread(
fetch_ollama_model_metadata, run["name"], base_url, api_key, 2.0
)
trained_ctx = int((meta or {}).get("context_length") or 0)
if trained_ctx > 0 and loaded_ctx * 2 <= trained_ctx:
kv_cache_advisory = {
"model": run["name"],
"loaded_context": loaded_ctx,
"trained_context": trained_ctx,
}
break
except Exception:
kv_cache_advisory = None
return {
"reachable": reachable,
"installed": installed,
"running": running,
"recommended": recommended,
"kv_cache_advisory": kv_cache_advisory,
}
def _run_ollama_pull_job(job_id: str, model: str, base_url: str, api_key: str) -> None:
"""Worker: stream native /api/pull NDJSON progress into the job record."""
import urllib.request as _urlreq
root = base_url.rstrip("/")
for suffix in ("/api", "/v1"):
if root.endswith(suffix):
root = root[: -len(suffix)].rstrip("/")
break
headers = {"Content-Type": "application/json"}
if api_key:
headers["Authorization"] = f"Bearer {api_key}"
def _update(**fields):
with _ollama_pull_jobs_lock:
job = _ollama_pull_jobs.get(job_id)
if job is not None:
job.update(fields)
request = _urlreq.Request(
root + "/api/pull",
data=json.dumps({"model": model, "stream": True}).encode(),
headers=headers,
method="POST",
)
try:
# No read timeout beyond connect: a pull legitimately runs for many
# minutes; progress lines keep the connection demonstrably alive.
with _urlreq.urlopen(request, timeout=30.0) as resp:
for raw_line in resp:
line = raw_line.decode("utf-8", errors="replace").strip()
if not line:
continue
try:
event = json.loads(line)
except Exception:
continue
if event.get("error"):
_update(status="error", error_message=str(event["error"]))
return
status = str(event.get("status", "") or "")
fields: dict = {"detail": status}
if isinstance(event.get("total"), (int, float)) and event["total"]:
fields["total_bytes"] = int(event["total"])
fields["completed_bytes"] = int(event.get("completed", 0) or 0)
_update(**fields)
if status == "success":
_on_ollama_models_changed()
_update(status="done", detail="success")
return
# Stream ended without an explicit success line — treat as done only
# if the last event said so; otherwise report the ambiguity.
with _ollama_pull_jobs_lock:
job = _ollama_pull_jobs.get(job_id)
if job is not None and job.get("status") == "pulling":
job["status"] = "error"
job["error_message"] = "pull stream ended unexpectedly"
except Exception as exc:
_update(status="error", error_message=f"pull failed: {exc}")
@app.post("/api/ollama/pull")
async def start_ollama_pull(body: OllamaModelAction, request: Request):
"""Start pulling a model from the Ollama registry. Token-protected.
Returns ``{job_id}``; poll ``GET /api/ollama/pull/{job_id}`` for progress.
"""
_require_token(request)
model = (body.model or "").strip()
if not model:
raise HTTPException(status_code=400, detail="model is required")
base_url, api_key = _ollama_connection(body.profile)
now = time.time()
job_id = secrets.token_hex(8)
with _ollama_pull_jobs_lock:
# Reuse an active job for the same model instead of double-pulling.
for jid, job in _ollama_pull_jobs.items():
if job.get("model") == model and job.get("status") == "pulling":
return {"job_id": jid}
# Expire finished records so the dict can't grow unboundedly.
for jid in [
j
for j, job in _ollama_pull_jobs.items()
if (now - job.get("at", 0)) > _OLLAMA_PULL_JOB_TTL
]:
_ollama_pull_jobs.pop(jid, None)
_ollama_pull_jobs[job_id] = {
"model": model,
"status": "pulling",
"detail": "starting",
"at": now,
}
threading.Thread(
target=_run_ollama_pull_job,
args=(job_id, model, base_url, api_key),
daemon=True,
name=f"ollama-pull-{model}",
).start()
return {"job_id": job_id}
@app.get("/api/ollama/pull/{job_id}")
async def poll_ollama_pull(job_id: str):
"""Poll a pull job: ``{model, status, detail, total_bytes?, completed_bytes?}``.
``status``: pulling | done | error.
"""
with _ollama_pull_jobs_lock:
job = _ollama_pull_jobs.get(job_id)
if job is None:
raise HTTPException(status_code=404, detail="pull job not found or expired")
return {"job_id": job_id, **{k: v for k, v in job.items() if k != "at"}}
@app.post("/api/ollama/delete")
async def delete_ollama_model_endpoint(body: OllamaModelAction, request: Request):
"""Delete an installed model. Token-protected."""
_require_token(request)
model = (body.model or "").strip()
if not model:
raise HTTPException(status_code=400, detail="model is required")
from hermes_cli.models import delete_ollama_model
base_url, api_key = _ollama_connection(body.profile)
ok, message = await asyncio.to_thread(
delete_ollama_model, model, api_key, base_url
)
if ok:
_on_ollama_models_changed()
return {"ok": ok, "message": message}
@app.post("/api/ollama/load")
async def load_ollama_model_endpoint(body: OllamaModelAction, request: Request):
"""Load a model into memory (optionally pinning keep_alive). Token-protected.
Used both for explicit warm-up from the management panel and to apply a
keep_alive change (Ollama applies keep_alive per-load).
"""
_require_token(request)
model = (body.model or "").strip()
if not model:
raise HTTPException(status_code=400, detail="model is required")
from hermes_cli.models import preload_ollama_model
base_url, api_key = _ollama_connection(body.profile)
ok, message = await asyncio.to_thread(
preload_ollama_model, model, body.keep_alive, api_key, base_url
)
return {"ok": ok, "message": message}
@app.get("/api/model/auxiliary")
def get_auxiliary_models(profile: Optional[str] = None):
"""Return current auxiliary task assignments.

View file

@ -1,9 +1,10 @@
"""Custom / Ollama (local) provider profile.
Covers any endpoint registered as provider="custom", including local
Ollama instances and OpenAI-compatible reasoning endpoints (GLM-5.2 on
Volcengine ARK, vLLM, llama.cpp). Key quirks:
Covers any endpoint registered as provider="custom", plus the first-class
"ollama" provider (routed here by alias), and OpenAI-compatible reasoning
endpoints (GLM-5.2 on Volcengine ARK, vLLM, llama.cpp). Key quirks:
- ollama_num_ctx → extra_body.options.num_ctx (local context window)
- ollama_keep_alive → extra_body.keep_alive (model residence time)
- reasoning_config disabled → extra_body.think = False
- reasoning_config enabled + effort → top-level reasoning_effort
(the native OpenAI-compatible format GLM/ARK expect; unset omits it
@ -24,6 +25,8 @@ class CustomProfile(ProviderProfile):
*,
reasoning_config: dict | None = None,
ollama_num_ctx: int | None = None,
ollama_keep_alive: int | str | None = None,
ollama_supports_thinking: bool | None = None,
**ctx: Any,
) -> tuple[dict[str, Any], dict[str, Any]]:
extra_body: dict[str, Any] = {}
@ -35,6 +38,12 @@ class CustomProfile(ProviderProfile):
options["num_ctx"] = ollama_num_ctx
extra_body["options"] = options
# Ollama model residence time after the request (default 5m). Go
# duration string or seconds; -1 = keep loaded until server exit.
# Ignored by non-Ollama OpenAI-compatible servers.
if ollama_keep_alive is not None:
extra_body["keep_alive"] = ollama_keep_alive
# Reasoning / thinking control for custom OpenAI-compatible endpoints
# (GLM-5.2 on Volcengine ARK, vLLM, Ollama, llama.cpp, …).
#
@ -45,6 +54,10 @@ class CustomProfile(ProviderProfile):
# - enabled + no effort → omit both, so the endpoint applies its own
# server-side default (do NOT force a level the user didn't pick).
#
# Effort levels that only exist on the OpenAI/Codex scale are mapped
# to the nearest level these endpoints accept — Ollama validates
# against high/medium/low/max/none and 400s on "xhigh"/"minimal".
#
# We deliberately do NOT emit ``think=True`` on enable: it is an
# Ollama-only flag and thinking is already server-default-on for these
# backends, so forcing it risks a 400 on GLM/vLLM endpoints that don't
@ -52,10 +65,19 @@ class CustomProfile(ProviderProfile):
if reasoning_config and isinstance(reasoning_config, dict):
_effort = (reasoning_config.get("effort") or "").strip().lower()
_enabled = reasoning_config.get("enabled", True)
if _effort == "none" or _enabled is False:
if ollama_supports_thinking is False:
# The Ollama server reports this model cannot think: any
# reasoning_effort 400s ('"hermes3:8b" does not support
# thinking') and a think flag is at best a no-op. Emit
# nothing — a session's effort dial carried over from a
# thinking model must not brick the chat. None means
# unknown/non-Ollama and changes nothing.
pass
elif _effort == "none" or _enabled is False:
extra_body["think"] = False
elif _effort:
top_level["reasoning_effort"] = _effort
_aliases = {"xhigh": "max", "minimal": "low"}
top_level["reasoning_effort"] = _aliases.get(_effort, _effort)
return extra_body, top_level

View file

@ -5326,6 +5326,15 @@ class AIAgent:
opts = self._lmstudio_reasoning_options_cached()
# "off-only" (or absent) means no real reasoning capability.
return any(opt and opt != "off" for opt in opts)
if (self.provider or "").strip().lower() == "ollama":
# Ollama rejects reasoning_effort outright on models without the
# thinking capability ('"hermes3:8b" does not support thinking'),
# so gate on the server's own /api/show capabilities. Unknown
# (old server, probe failure) errs on sending — a thinking model
# silently losing its effort dial is the worse failure, and the
# capabilities field has shipped since 0.6.0.
supported = self._ollama_supports_thinking_cached()
return supported is not False
if "openrouter" not in self._base_url_lower:
return False
if "api.mistral.ai" in self._base_url_lower:
@ -5379,6 +5388,37 @@ class AIAgent:
cache[key] = (opts, _time.monotonic())
return opts
def _ollama_supports_thinking_cached(self) -> Optional[bool]:
"""Probe Ollama's /api/show thinking capability once per (model, base_url).
True/False are cached permanently (a model's capabilities don't
change); None (probe failure / old server without the capabilities
field) is cached with a 60-second TTL so a transient failure retries
soon without an HTTP round-trip on every turn.
"""
import time as _time
cache = getattr(self, "_ollama_thinking_cache", None)
if cache is None:
cache = self._ollama_thinking_cache = {}
key = (self.model, self.base_url)
cached = cache.get(key)
if cached is not None:
supported, ts = cached
if supported is not None or (_time.monotonic() - ts) < 60:
return supported
try:
from agent.model_metadata import query_ollama_supports_thinking
api_key = self.api_key if isinstance(self.api_key, str) else ""
supported = query_ollama_supports_thinking(
self.model, self.base_url, api_key=api_key or ""
)
except Exception:
supported = None
cache[key] = (supported, _time.monotonic())
return supported
def _resolve_lmstudio_summary_reasoning_effort(self) -> Optional[str]:
"""Resolve a safe top-level ``reasoning_effort`` for LM Studio.

View file

@ -264,6 +264,12 @@ def test_current_custom_endpoint_passthrough_marks_current_row(monkeypatch):
monkeypatch.setattr("hermes_cli.providers.HERMES_OVERLAYS", {})
monkeypatch.setattr("hermes_cli.models.fetch_openrouter_models",
lambda *a, **kw: [])
# The endpoint is offline in this scenario: the live /v1/models probe must
# fail so the declared model list wins. Without this mock a REAL Ollama
# server on localhost:11434 (common on dev machines) answers the probe and
# its installed models replace the declared ones.
monkeypatch.setattr("hermes_cli.models.fetch_api_models",
lambda *a, **kw: [])
result = model_switch.list_picker_providers(
current_provider="custom:ollama",

View file

@ -437,6 +437,10 @@ def test_list_authenticated_providers_groups_same_endpoint(monkeypatch):
returned as a single picker row with all their models merged."""
monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {})
monkeypatch.setattr(providers_mod, "HERMES_OVERLAYS", {})
# Offline endpoint: the live /v1/models probe must fail so the declared
# models win. Unmocked, a real Ollama server on localhost:11434 (common
# on dev machines) answers and its installed models replace the fixtures.
monkeypatch.setattr("hermes_cli.models.fetch_api_models", lambda *a, **kw: [])
providers = list_authenticated_providers(
current_provider="custom",
@ -523,6 +527,8 @@ def test_list_authenticated_providers_distinct_endpoints_stay_separate(monkeypat
even if some display names happen to be similar."""
monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {})
monkeypatch.setattr(providers_mod, "HERMES_OVERLAYS", {})
# Offline endpoints — see test_list_authenticated_providers_groups_same_endpoint.
monkeypatch.setattr("hermes_cli.models.fetch_api_models", lambda *a, **kw: [])
providers = list_authenticated_providers(
user_providers={},
@ -618,6 +624,8 @@ def test_list_authenticated_providers_total_models_reflects_grouped_count(monkey
the full count, and every grouped model appears in the list."""
monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {})
monkeypatch.setattr(providers_mod, "HERMES_OVERLAYS", {})
# Offline endpoint — see test_list_authenticated_providers_groups_same_endpoint.
monkeypatch.setattr("hermes_cli.models.fetch_api_models", lambda *a, **kw: [])
entries = [
{"name": f"Ollama \u2014 Model {i}", "base_url": "http://localhost:11434/v1",

View file

@ -55,14 +55,14 @@ class TestOllamaCloudAliases:
"""ollama_cloud (underscore) is the unambiguous cloud alias."""
assert resolve_provider("ollama_cloud") == "ollama-cloud"
def test_bare_ollama_stays_local(self):
"""Bare 'ollama' alias routes to 'custom' (local) — not cloud."""
assert resolve_provider("ollama") == "custom"
def test_bare_ollama_is_local_provider(self):
"""Bare 'ollama' resolves to the local ollama provider — not cloud."""
assert resolve_provider("ollama") == "ollama"
def test_models_py_aliases(self):
assert _PROVIDER_ALIASES.get("ollama_cloud") == "ollama-cloud"
# bare "ollama" stays local
assert _PROVIDER_ALIASES.get("ollama") == "custom"
# bare "ollama" is the local provider
assert _PROVIDER_ALIASES.get("ollama") == "ollama"
def test_normalize_provider(self):
assert normalize_provider("ollama-cloud") == "ollama-cloud"
@ -381,7 +381,7 @@ class TestOllamaCloudProvidersNew:
def test_alias_resolves(self):
from hermes_cli.providers import normalize_provider as np
assert np("ollama") == "custom" # bare "ollama" = local
assert np("ollama") == "ollama" # bare "ollama" = local ollama provider
assert np("ollama-cloud") == "ollama-cloud"
def test_label_override(self):

View file

@ -9,8 +9,8 @@ was silently dropped for every custom endpoint.
These tests pin the wire-shape contract:
- disabled → extra_body.think = False
- enabled + effort → top-level reasoning_effort (native OpenAI-compat
format GLM/ARK expect), passed through verbatim
including ``max``/``xhigh``
format GLM/ARK expect); OpenAI-only levels
(xhigh/minimal) map to the nearest accepted level
- enabled + no effort → nothing emitted (endpoint's server default applies)
- ollama_num_ctx → extra_body.options.num_ctx, orthogonal to reasoning
"""
@ -65,7 +65,7 @@ class TestCustomReasoningWireShape:
assert tl == {}
@pytest.mark.parametrize(
"effort", ["minimal", "low", "medium", "high", "xhigh", "max"]
"effort", ["low", "medium", "high", "max"]
)
def test_enabled_effort_goes_top_level(self, custom_profile, effort):
"""enabled + effort → TOP-LEVEL reasoning_effort, passed through verbatim.
@ -81,6 +81,26 @@ class TestCustomReasoningWireShape:
assert "reasoning_effort" not in eb
assert "think" not in eb
@pytest.mark.parametrize(
("effort", "expected"), [("xhigh", "max"), ("minimal", "low")]
)
def test_openai_only_efforts_map_to_endpoint_levels(
self, custom_profile, effort, expected
):
"""Efforts that only exist on the OpenAI/Codex scale map to the
nearest level these endpoints accept.
Ollama validates reasoning_effort against high/medium/low/max/none
and rejects "xhigh"/"minimal" with HTTP 400; GLM documents "high" and
"max". Carrying a session's effort dial across a provider switch to a
local model must not brick the chat.
"""
eb, tl = custom_profile.build_api_kwargs_extras(
reasoning_config={"enabled": True, "effort": effort}, model="glm-5.2"
)
assert tl == {"reasoning_effort": expected}
assert "think" not in eb
def test_enabled_without_effort_emits_nothing(self, custom_profile):
"""enabled but no effort → omit; do NOT force a level the user didn't pick."""
eb, tl = custom_profile.build_api_kwargs_extras(