mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-07-31 19:16:29 +00:00
feat(desktop): first-class local Ollama provider with detection, model management, and capability-aware picking
Promote bare "ollama" from a custom-endpoint alias to a real provider and
build the desktop UX around it. A local server's connection kind is
reachability rather than a credential, so every credential-shaped gate
(provider registry, picker filters, settings surfaces) gets an explicit
path for it.
Backend:
- Provider overlay + registry entry (127.0.0.1:11434/v1 default, keyless
with a local-only placeholder, base-url normalization for /api and /v1
forms). Existing provider=custom configs are untouched.
- GET /api/local-servers/detect fingerprints well-known local ports plus a
local configured base_url; response shape leaves room for a future
installed/running/managed distinction.
- /api/ollama/* management endpoints: installed+running+recommended models,
registry pull as a poll-able background job streaming native NDJSON
progress, delete, and load (warm-up / keep_alive pinning). Pull and
delete bust the picker's model-id cache.
- Model picker payload: per-model capabilities widened to tools/vision/
context_length; local Ollama rows enriched from the server's native
/api/show (authoritative for on-disk tags, where models.dev is sparse),
backfilled off the request path by a background thread.
- The explicit-only picker filter keeps ollama rows: the row only exists
when the server answered a probe, which is as explicit as a pasted key.
- Reasoning safety: /api/show thinking capability gates all reasoning
fields (Ollama 400s reasoning_effort on non-thinking models), and
OpenAI-only effort levels map to the nearest accepted level
(xhigh->max, minimal->low).
- model.ollama_keep_alive config: sent per-request as extra_body.keep_alive.
- Latency discipline for a local server that may be down: a 300ms TCP
pre-check with a short negative cache guards every native-API read; the
status endpoint probes fresh so a just-started server is noticed
immediately; localhost is rewritten to 127.0.0.1 (Windows resolves
localhost to ::1 first and Ollama binds IPv4 loopback — each request
otherwise pays a ~2s failed IPv6 connect, including chat inference).
Desktop:
- Providers -> Accounts: a "Local servers" card mirroring the OAuth card
language ("Running · N models" / a start-the-server hint), expanding to
model management: installed models with size/quant/VRAM, delete, warm-up,
curated pull recommendations with a progress bar, free-form pull, and a
KV-cache advisory when a loaded model runs well under its trained window.
Polls while down so it flips to Running by itself.
- Model picker: "No tools" badge (explicit tools:false only — absence means
unknown) with demotion, plus context window / parameter size / quant per
row; refetches once after open so backfilled metadata appears in place.
- Onboarding: detected-server row with a model select, replacing blind
first-model assignment for detected servers.
- Composer status stack: "Loading <model> into memory" row during cold
starts, confirmed against /api/ps so ordinary slow generations stay quiet.
Chat inference stays on the OpenAI-compatible /v1 endpoint; the native
/api surface is used read-only for metadata plus explicit management
actions. Lifecycle management (starting or installing Ollama) is not
included.
This commit is contained in:
parent
74e28f7d10
commit
83f88909ef
33 changed files with 2161 additions and 39 deletions
|
|
@ -2012,6 +2012,22 @@ def init_agent(
|
|||
agent._ollama_num_ctx,
|
||||
)
|
||||
|
||||
# ── Ollama keep_alive ──
|
||||
# How long the server keeps the model resident after a request (Ollama
|
||||
# default: 5 minutes). Set model.ollama_keep_alive in config.yaml to a Go
|
||||
# duration string ("30m", "2h") or seconds; -1 keeps it loaded until the
|
||||
# server exits. Sent per-request alongside num_ctx.
|
||||
agent._ollama_keep_alive = None
|
||||
if isinstance(_model_cfg, dict):
|
||||
_keep_alive_raw = _model_cfg.get("ollama_keep_alive")
|
||||
if _keep_alive_raw is not None:
|
||||
if isinstance(_keep_alive_raw, (int, float)):
|
||||
agent._ollama_keep_alive = int(_keep_alive_raw)
|
||||
else:
|
||||
_keep_alive_str = str(_keep_alive_raw).strip()
|
||||
if _keep_alive_str:
|
||||
agent._ollama_keep_alive = _keep_alive_str
|
||||
|
||||
# Codex gpt-5.x autoraise notice: show at most once per profile/config
|
||||
# state. Without the persisted marker the notice re-fires on every agent
|
||||
# init — and the gateway rebuilds the agent per inbound message, so Discord
|
||||
|
|
|
|||
|
|
@ -879,6 +879,12 @@ def build_api_kwargs(agent, api_messages: list) -> dict:
|
|||
session_id=getattr(agent, "session_id", None),
|
||||
provider_profile=_profile,
|
||||
ollama_num_ctx=agent._ollama_num_ctx,
|
||||
ollama_keep_alive=getattr(agent, "_ollama_keep_alive", None),
|
||||
ollama_supports_thinking=(
|
||||
agent._ollama_supports_thinking_cached()
|
||||
if (agent.provider or "").strip().lower() == "ollama"
|
||||
else None
|
||||
),
|
||||
# Context forwarded to profile hooks:
|
||||
provider_preferences=_prefs or None,
|
||||
openrouter_min_coding_score=agent.openrouter_min_coding_score,
|
||||
|
|
|
|||
|
|
@ -1435,6 +1435,51 @@ def query_ollama_supports_vision(model: str, base_url: str, api_key: str = "") -
|
|||
return None
|
||||
|
||||
|
||||
def query_ollama_supports_thinking(model: str, base_url: str, api_key: str = "") -> Optional[bool]:
|
||||
"""Return True/False when Ollama ``/api/show`` reports thinking support.
|
||||
|
||||
Uses the ``capabilities`` field (Ollama 0.6.0+). Returns None when the
|
||||
server is unreachable, not Ollama, the model is unknown, or the server
|
||||
predates the capabilities field — callers should treat None as "unknown"
|
||||
and not gate on it.
|
||||
"""
|
||||
import httpx
|
||||
|
||||
bare_model = _strip_provider_prefix(model)
|
||||
if not bare_model or not base_url:
|
||||
return None
|
||||
|
||||
try:
|
||||
if detect_local_server_type(base_url, api_key=api_key) != "ollama":
|
||||
return None
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
server_url = base_url.rstrip("/")
|
||||
if server_url.endswith("/v1"):
|
||||
server_url = server_url[:-3]
|
||||
|
||||
headers = _auth_headers(api_key)
|
||||
|
||||
try:
|
||||
with httpx.Client(timeout=3.0, headers=headers) as client:
|
||||
resp = client.post(f"{server_url}/api/show", json={"name": bare_model})
|
||||
if resp.status_code != 200:
|
||||
return None
|
||||
data = resp.json()
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
caps = data.get("capabilities")
|
||||
if isinstance(caps, list):
|
||||
if any(str(cap).lower() == "thinking" for cap in caps):
|
||||
return True
|
||||
if caps:
|
||||
return False
|
||||
|
||||
return None
|
||||
|
||||
|
||||
def _query_ollama_api_show(model: str, base_url: str, api_key: str = "") -> Optional[int]:
|
||||
"""Query an Ollama server's native ``/api/show`` for context length.
|
||||
|
||||
|
|
|
|||
|
|
@ -256,6 +256,8 @@ class ChatCompletionsTransport(ProviderTransport):
|
|||
is_lmstudio: bool
|
||||
is_custom_provider: bool
|
||||
ollama_num_ctx: int | None
|
||||
ollama_keep_alive: int | str | None
|
||||
ollama_supports_thinking: bool | None
|
||||
# Provider routing
|
||||
provider_preferences: dict | None
|
||||
# Qwen-specific
|
||||
|
|
@ -534,6 +536,8 @@ class ChatCompletionsTransport(ProviderTransport):
|
|||
model=model,
|
||||
base_url=params.get("base_url"),
|
||||
ollama_num_ctx=params.get("ollama_num_ctx"),
|
||||
ollama_keep_alive=params.get("ollama_keep_alive"),
|
||||
ollama_supports_thinking=params.get("ollama_supports_thinking"),
|
||||
session_id=params.get("session_id"),
|
||||
)
|
||||
)
|
||||
|
|
|
|||
|
|
@ -23,6 +23,7 @@ import { $previewStatusBySession, dismissPreviewArtifact } from '@/store/preview
|
|||
import { $threadScrolledUp } from '@/store/thread-scroll'
|
||||
import { openSessionInNewWindow } from '@/store/windows'
|
||||
|
||||
import { OllamaColdStartRow, useOllamaColdStart } from './ollama-cold-start'
|
||||
import { PreviewStatusRow } from './preview-row'
|
||||
import { StatusItemRow } from './status-row'
|
||||
|
||||
|
|
@ -173,6 +174,13 @@ export function ComposerStatusStack({ queue, sessionId }: ComposerStatusStackPro
|
|||
sections.push({ key: 'preview', node: previewBlock })
|
||||
}
|
||||
|
||||
// Cold-start feedback while a local Ollama model loads its weights.
|
||||
const ollamaColdStartModel = useOllamaColdStart()
|
||||
|
||||
if (ollamaColdStartModel) {
|
||||
sections.push({ key: 'ollama-cold-start', node: <OllamaColdStartRow model={ollamaColdStartModel} /> })
|
||||
}
|
||||
|
||||
if (queue) {
|
||||
sections.push({ key: 'queue', node: queue })
|
||||
}
|
||||
|
|
|
|||
|
|
@ -0,0 +1,91 @@
|
|||
import { useStore } from '@nanostores/react'
|
||||
import { useEffect, useState } from 'react'
|
||||
|
||||
import { GlyphSpinner } from '@/components/ui/glyph-spinner'
|
||||
import { getOllamaModels } from '@/hermes'
|
||||
import { useI18n } from '@/i18n'
|
||||
import { $awaitingResponse, $currentModel, $currentProvider } from '@/store/session'
|
||||
|
||||
// Wait this long after a turn starts before concluding the silence is a cold
|
||||
// model load (a warm model answers well inside it), then confirm against the
|
||||
// server before showing anything.
|
||||
const COLD_START_GRACE_MS = 4_000
|
||||
const POLL_INTERVAL_MS = 2_000
|
||||
|
||||
/**
|
||||
* Cold-start detection for local Ollama models: returns the model name while
|
||||
* a turn has been awaiting its first token past the grace period and the
|
||||
* target model is not yet in the server's loaded list — i.e. Ollama is
|
||||
* reading weights into memory. Returns '' for non-Ollama providers and for
|
||||
* warm models (the loaded check keeps ordinary slow generations quiet).
|
||||
*
|
||||
* A hook rather than a self-hiding component so the status stack can count
|
||||
* it toward its own visibility (the stack collapses when every section is
|
||||
* empty).
|
||||
*/
|
||||
export function useOllamaColdStart(): string {
|
||||
const awaiting = useStore($awaitingResponse)
|
||||
const provider = useStore($currentProvider)
|
||||
const model = useStore($currentModel)
|
||||
const [loadingModel, setLoadingModel] = useState('')
|
||||
|
||||
const active = awaiting && provider === 'ollama' && Boolean(model)
|
||||
|
||||
useEffect(() => {
|
||||
if (!active) {
|
||||
setLoadingModel('')
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
let cancelled = false
|
||||
let timer: ReturnType<typeof setTimeout>
|
||||
|
||||
const check = async () => {
|
||||
try {
|
||||
const data = await getOllamaModels()
|
||||
|
||||
if (cancelled) {
|
||||
return
|
||||
}
|
||||
|
||||
const loaded = data.running.some(r => r.name === model || r.name.split(':', 1)[0] === model.split(':', 1)[0])
|
||||
|
||||
// Loaded (or server unreachable — nothing useful to say): clear and
|
||||
// let the normal streaming UI take over.
|
||||
if (loaded || !data.reachable) {
|
||||
setLoadingModel('')
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
setLoadingModel(model)
|
||||
timer = setTimeout(() => void check(), POLL_INTERVAL_MS)
|
||||
} catch {
|
||||
if (!cancelled) {
|
||||
setLoadingModel('')
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
timer = setTimeout(() => void check(), COLD_START_GRACE_MS)
|
||||
|
||||
return () => {
|
||||
cancelled = true
|
||||
clearTimeout(timer)
|
||||
}
|
||||
}, [active, model])
|
||||
|
||||
return active ? loadingModel : ''
|
||||
}
|
||||
|
||||
export function OllamaColdStartRow({ model }: { model: string }) {
|
||||
const { t } = useI18n()
|
||||
|
||||
return (
|
||||
<div className="flex items-center gap-2 px-3 py-1.5 text-xs text-muted-foreground" data-slot="ollama-cold-start">
|
||||
<GlyphSpinner className="opacity-70" spinner="braille" />
|
||||
<span>{t.statusStack.ollamaLoading(model)}</span>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
364
apps/desktop/src/app/settings/ollama-panel.tsx
Normal file
364
apps/desktop/src/app/settings/ollama-panel.tsx
Normal file
|
|
@ -0,0 +1,364 @@
|
|||
import { useCallback, useEffect, useRef, useState } from 'react'
|
||||
|
||||
import { Button } from '@/components/ui/button'
|
||||
import { Input } from '@/components/ui/input'
|
||||
import { RowButton } from '@/components/ui/row-button'
|
||||
import {
|
||||
deleteOllamaModel,
|
||||
getOllamaModels,
|
||||
loadOllamaModel,
|
||||
pollOllamaPull,
|
||||
startOllamaPull
|
||||
} from '@/hermes'
|
||||
import { useI18n } from '@/i18n'
|
||||
import { Check, ChevronDown, Cpu, Download, Loader2, Trash2 } from '@/lib/icons'
|
||||
import { cn } from '@/lib/utils'
|
||||
import { notify, notifyError } from '@/store/notifications'
|
||||
import type { OllamaModelsResponse, OllamaPullStatus } from '@/types/hermes'
|
||||
|
||||
import { CONTROL_TEXT } from './constants'
|
||||
import { SettingsCategoryHeading } from './env-credentials'
|
||||
import { Pill } from './primitives'
|
||||
|
||||
const PULL_POLL_INTERVAL_MS = 750
|
||||
|
||||
function formatBytes(bytes?: number): string {
|
||||
if (!bytes || bytes <= 0) {
|
||||
return ''
|
||||
}
|
||||
|
||||
const gib = bytes / 1024 ** 3
|
||||
|
||||
return gib >= 1 ? `${gib.toFixed(1)} GB` : `${Math.round(bytes / 1024 ** 2)} MB`
|
||||
}
|
||||
|
||||
function formatContext(tokens?: number): string {
|
||||
if (!tokens || tokens <= 0) {
|
||||
return ''
|
||||
}
|
||||
|
||||
return tokens >= 1024 ? `${Math.round(tokens / 1024)}k` : String(tokens)
|
||||
}
|
||||
|
||||
/**
|
||||
* Local Ollama server card for the Providers page. Same row language as the
|
||||
* OAuth provider cards, but the connection kind is reachability rather than
|
||||
* a credential: the status tag shows "Running · N models" or a start-the-
|
||||
* server hint. Expanding a running server reveals model management —
|
||||
* installed models (size, delete, warm-up), loaded models (VRAM, context),
|
||||
* curated pull recommendations, and free-form pull.
|
||||
*/
|
||||
export function OllamaProviderCard() {
|
||||
const { t } = useI18n()
|
||||
const copy = t.settings.ollama
|
||||
const [data, setData] = useState<null | OllamaModelsResponse>(null)
|
||||
const [expanded, setExpanded] = useState(false)
|
||||
const [pull, setPull] = useState<null | OllamaPullStatus>(null)
|
||||
const [pullModel, setPullModel] = useState('')
|
||||
const [busyModel, setBusyModel] = useState<null | string>(null)
|
||||
const pollTimer = useRef<null | ReturnType<typeof setTimeout>>(null)
|
||||
|
||||
const refresh = useCallback(async () => {
|
||||
try {
|
||||
setData(await getOllamaModels())
|
||||
} catch {
|
||||
setData(null)
|
||||
}
|
||||
}, [])
|
||||
|
||||
useEffect(() => {
|
||||
void refresh()
|
||||
|
||||
return () => {
|
||||
if (pollTimer.current) {
|
||||
clearTimeout(pollTimer.current)
|
||||
}
|
||||
}
|
||||
}, [refresh])
|
||||
|
||||
// While the server is down (or detection failed), poll so the card flips
|
||||
// to "Running" by itself when the user starts Ollama — the status endpoint
|
||||
// answers in ~20ms, so this is cheap. Stops as soon as it's reachable.
|
||||
const reachableNow = Boolean(data?.reachable)
|
||||
|
||||
useEffect(() => {
|
||||
if (reachableNow) {
|
||||
return
|
||||
}
|
||||
|
||||
const timer = setInterval(() => void refresh(), 5_000)
|
||||
|
||||
return () => clearInterval(timer)
|
||||
}, [reachableNow, refresh])
|
||||
|
||||
const pollUntilDone = useCallback(
|
||||
(jobId: string) => {
|
||||
const tick = async () => {
|
||||
let status: OllamaPullStatus
|
||||
|
||||
try {
|
||||
status = await pollOllamaPull(jobId)
|
||||
} catch (err) {
|
||||
setPull(null)
|
||||
notifyError(err, copy.pullFailed)
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
setPull(status)
|
||||
|
||||
if (status.status === 'pulling') {
|
||||
pollTimer.current = setTimeout(() => void tick(), PULL_POLL_INTERVAL_MS)
|
||||
|
||||
return
|
||||
}
|
||||
|
||||
if (status.status === 'done') {
|
||||
notify({ kind: 'success', title: copy.pullDone(status.model), message: '' })
|
||||
setPull(null)
|
||||
setPullModel('')
|
||||
void refresh()
|
||||
} else {
|
||||
notifyError(new Error(status.error_message || status.detail || 'pull failed'), copy.pullFailed)
|
||||
setPull(null)
|
||||
}
|
||||
}
|
||||
|
||||
void tick()
|
||||
},
|
||||
[copy, refresh]
|
||||
)
|
||||
|
||||
const beginPull = useCallback(
|
||||
async (model: string) => {
|
||||
const name = model.trim()
|
||||
|
||||
if (!name || pull) {
|
||||
return
|
||||
}
|
||||
|
||||
try {
|
||||
const { job_id } = await startOllamaPull(name)
|
||||
|
||||
setPull({ job_id, model: name, status: 'pulling', detail: 'starting' })
|
||||
pollUntilDone(job_id)
|
||||
} catch (err) {
|
||||
notifyError(err, copy.pullFailed)
|
||||
}
|
||||
},
|
||||
[copy.pullFailed, pollUntilDone, pull]
|
||||
)
|
||||
|
||||
const removeModel = useCallback(
|
||||
async (model: string) => {
|
||||
setBusyModel(model)
|
||||
|
||||
try {
|
||||
const result = await deleteOllamaModel(model)
|
||||
|
||||
if (!result.ok) {
|
||||
notifyError(new Error(result.message), copy.deleteFailed)
|
||||
} else {
|
||||
void refresh()
|
||||
}
|
||||
} catch (err) {
|
||||
notifyError(err, copy.deleteFailed)
|
||||
} finally {
|
||||
setBusyModel(null)
|
||||
}
|
||||
},
|
||||
[copy.deleteFailed, refresh]
|
||||
)
|
||||
|
||||
const warmModel = useCallback(
|
||||
async (model: string) => {
|
||||
setBusyModel(model)
|
||||
|
||||
try {
|
||||
const result = await loadOllamaModel(model)
|
||||
|
||||
if (!result.ok) {
|
||||
notifyError(new Error(result.message), copy.loadFailed)
|
||||
} else {
|
||||
void refresh()
|
||||
}
|
||||
} catch (err) {
|
||||
notifyError(err, copy.loadFailed)
|
||||
} finally {
|
||||
setBusyModel(null)
|
||||
}
|
||||
},
|
||||
[copy.loadFailed, refresh]
|
||||
)
|
||||
|
||||
const reachable = Boolean(data?.reachable)
|
||||
const installedCount = data?.installed.length ?? 0
|
||||
const running = new Map((data?.running ?? []).map(r => [r.name, r]))
|
||||
|
||||
const pullPercent =
|
||||
pull?.total_bytes && pull.completed_bytes !== undefined
|
||||
? Math.min(100, Math.round((pull.completed_bytes / pull.total_bytes) * 100))
|
||||
: null
|
||||
|
||||
// Detection still pending: render nothing rather than flashing the
|
||||
// not-running hint at every page open on machines without Ollama.
|
||||
if (data === null) {
|
||||
return null
|
||||
}
|
||||
|
||||
return (
|
||||
<section className="mb-5 grid gap-2" data-slot="ollama-provider-card">
|
||||
<SettingsCategoryHeading icon={Cpu} title={copy.categoryTitle} />
|
||||
<div className="rounded-[6px] transition-colors">
|
||||
<RowButton
|
||||
className="group flex w-full items-center justify-between gap-3 rounded-[6px] px-3 py-2.5 text-left transition-colors hover:bg-(--ui-control-hover-background)"
|
||||
disabled={!reachable}
|
||||
onClick={() => setExpanded(open => !open)}
|
||||
>
|
||||
<div className="min-w-0">
|
||||
<div className="flex items-center gap-2">
|
||||
<span className="text-[length:var(--conversation-text-font-size)] font-semibold">Ollama</span>
|
||||
{reachable ? (
|
||||
<span className="inline-flex items-center gap-1 bg-primary/10 px-2 py-0.5 text-xs font-medium text-primary">
|
||||
<Check className="size-3" />
|
||||
{copy.running(installedCount)}
|
||||
</span>
|
||||
) : null}
|
||||
</div>
|
||||
<p className="mt-1 text-xs leading-5 text-muted-foreground">
|
||||
{reachable ? copy.desc : copy.notRunning}
|
||||
</p>
|
||||
</div>
|
||||
{reachable && (
|
||||
<ChevronDown
|
||||
className={cn('size-4 shrink-0 text-muted-foreground transition group-hover:text-foreground', expanded && 'rotate-180')}
|
||||
/>
|
||||
)}
|
||||
</RowButton>
|
||||
|
||||
{reachable && expanded && data && (
|
||||
<div className="px-3 pb-2 pt-1">
|
||||
{data.kv_cache_advisory && (
|
||||
<p className="mb-2 rounded-sm bg-amber-500/10 px-2.5 py-1.5 text-xs leading-5 text-amber-700 dark:text-amber-400">
|
||||
{copy.kvCacheAdvisory(
|
||||
data.kv_cache_advisory.model,
|
||||
formatContext(data.kv_cache_advisory.loaded_context),
|
||||
formatContext(data.kv_cache_advisory.trained_context)
|
||||
)}
|
||||
</p>
|
||||
)}
|
||||
|
||||
<div className="grid gap-1">
|
||||
{data.installed.map(model => {
|
||||
const live = running.get(model.name)
|
||||
const busy = busyModel === model.name
|
||||
|
||||
const meta = [model.parameter_size, model.quantization, formatBytes(model.size_bytes)]
|
||||
.filter(Boolean)
|
||||
.join(' · ')
|
||||
|
||||
return (
|
||||
<div
|
||||
className="flex items-center justify-between gap-3 rounded-[6px] px-3 py-2 transition-colors hover:bg-(--ui-control-hover-background)"
|
||||
key={model.name}
|
||||
>
|
||||
<div className="min-w-0">
|
||||
<div className="flex items-center gap-2">
|
||||
<span className="truncate font-mono text-xs">{model.name}</span>
|
||||
{live && (
|
||||
<Pill tone="primary">
|
||||
{copy.loaded}
|
||||
{live.context_length ? ` · ${formatContext(live.context_length)}` : ''}
|
||||
{live.size_vram_bytes ? ` · ${formatBytes(live.size_vram_bytes)} VRAM` : ''}
|
||||
</Pill>
|
||||
)}
|
||||
</div>
|
||||
{meta && <p className="mt-0.5 text-[0.68rem] text-muted-foreground">{meta}</p>}
|
||||
</div>
|
||||
<div className="flex shrink-0 items-center gap-1">
|
||||
{!live && (
|
||||
<Button disabled={busy} onClick={() => void warmModel(model.name)} size="sm" variant="text">
|
||||
{busy ? <Loader2 className="size-3.5 animate-spin" /> : copy.load}
|
||||
</Button>
|
||||
)}
|
||||
<Button
|
||||
aria-label={copy.deleteLabel(model.name)}
|
||||
disabled={busy}
|
||||
onClick={() => void removeModel(model.name)}
|
||||
size="icon"
|
||||
variant="ghost"
|
||||
>
|
||||
<Trash2 className="size-3.5" />
|
||||
</Button>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
})}
|
||||
</div>
|
||||
|
||||
<div className="mt-3">
|
||||
<p className="mb-1.5 text-xs font-medium">{copy.getModels}</p>
|
||||
{pull ? (
|
||||
<div className="rounded-[6px] bg-primary/[0.06] px-3 py-2">
|
||||
<div className="flex items-center gap-2 text-xs">
|
||||
<Loader2 className="size-3.5 animate-spin text-primary" />
|
||||
<span className="font-mono">{pull.model}</span>
|
||||
<span className="text-muted-foreground">
|
||||
{pullPercent !== null ? `${pullPercent}%` : pull.detail || ''}
|
||||
</span>
|
||||
</div>
|
||||
{pullPercent !== null && (
|
||||
<div className="mt-1.5 h-1 overflow-hidden rounded-full bg-primary/15">
|
||||
<div className="h-full bg-primary transition-all" style={{ width: `${pullPercent}%` }} />
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
) : (
|
||||
<>
|
||||
<div className="grid gap-1">
|
||||
{data.recommended.map(rec => (
|
||||
<div
|
||||
className="flex items-center justify-between gap-3 rounded-[6px] px-3 py-1.5 transition-colors hover:bg-(--ui-control-hover-background)"
|
||||
key={rec.model}
|
||||
>
|
||||
<div className="min-w-0">
|
||||
<span className="font-mono text-xs">{rec.model}</span>
|
||||
<p className="mt-0.5 truncate text-[0.68rem] text-muted-foreground">{rec.description}</p>
|
||||
</div>
|
||||
<Button
|
||||
className="shrink-0"
|
||||
onClick={() => void beginPull(rec.model)}
|
||||
size="sm"
|
||||
variant="text"
|
||||
>
|
||||
<Download className="size-3.5" />
|
||||
{copy.pull}
|
||||
</Button>
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
<div className="mt-2 flex items-center gap-2">
|
||||
<Input
|
||||
className={cn('max-w-72 font-mono', CONTROL_TEXT)}
|
||||
onChange={event => setPullModel(event.target.value)}
|
||||
onKeyDown={event => {
|
||||
if (event.key === 'Enter') {
|
||||
void beginPull(pullModel)
|
||||
}
|
||||
}}
|
||||
placeholder={copy.pullPlaceholder}
|
||||
value={pullModel}
|
||||
/>
|
||||
<Button disabled={!pullModel.trim()} onClick={() => void beginPull(pullModel)} size="sm" variant="text">
|
||||
{copy.pull}
|
||||
</Button>
|
||||
</div>
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
</section>
|
||||
)
|
||||
}
|
||||
|
|
@ -26,6 +26,7 @@ import type { EnvVarInfo, OAuthProvider } from '@/types/hermes'
|
|||
import { isKeyVar, ProviderKeyRows } from './credential-key-ui'
|
||||
import { SettingsCategoryHeading, useEnvCredentials } from './env-credentials'
|
||||
import { providerGroup, providerMeta, providerPriority } from './helpers'
|
||||
import { OllamaProviderCard } from './ollama-panel'
|
||||
import { LoadingState, SettingsContent } from './primitives'
|
||||
|
||||
// The embedded terminal (and thus the "run disconnect command" path) only
|
||||
|
|
@ -457,6 +458,10 @@ export function ProvidersSettings({ onClose, onViewChange, view }: ProvidersSett
|
|||
onWantApiKey={() => onViewChange('keys')}
|
||||
providers={oauthProviders}
|
||||
/>
|
||||
{/* Local Ollama server — a provider whose connection kind is
|
||||
reachability rather than a credential, so it renders its own card
|
||||
instead of an OAuth or API-key row. */}
|
||||
<OllamaProviderCard />
|
||||
</SettingsContent>
|
||||
)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -1,11 +1,11 @@
|
|||
import { useQuery } from '@tanstack/react-query'
|
||||
import { useState } from 'react'
|
||||
import { useEffect, useState } from 'react'
|
||||
|
||||
import { useI18n } from '@/i18n'
|
||||
import { requestModelOptions } from '@/lib/model-options'
|
||||
import { currentPickerSelection } from '@/lib/model-status-label'
|
||||
import { normalize } from '@/lib/text'
|
||||
import type { ModelOptionProvider, ModelPricing } from '@/types/hermes'
|
||||
import type { ModelCapabilities, ModelOptionProvider, ModelPricing } from '@/types/hermes'
|
||||
|
||||
import type { HermesGateway } from '../hermes'
|
||||
import { cn } from '../lib/utils'
|
||||
|
|
@ -61,6 +61,30 @@ export function ModelPickerDialog({
|
|||
|
||||
const providers = modelOptions.data?.providers ?? []
|
||||
|
||||
// Local Ollama capability metadata is backfilled server-side off the
|
||||
// request path (the payload returns immediately; native /api/show data
|
||||
// lands moments later). When an ollama row is missing its enrichment
|
||||
// (no tools flag), refetch once shortly after so badges appear in-place
|
||||
// instead of on the next open.
|
||||
const ollamaUnenriched = providers.some(
|
||||
p =>
|
||||
p.slug === 'ollama' &&
|
||||
(p.models?.length ?? 0) > 0 &&
|
||||
p.models?.some(m => p.capabilities?.[m]?.tools === undefined)
|
||||
)
|
||||
|
||||
const refetchOptions = modelOptions.refetch
|
||||
|
||||
useEffect(() => {
|
||||
if (!open || !ollamaUnenriched) {
|
||||
return
|
||||
}
|
||||
|
||||
const timer = setTimeout(() => void refetchOptions(), 1_500)
|
||||
|
||||
return () => clearTimeout(timer)
|
||||
}, [open, ollamaUnenriched, refetchOptions])
|
||||
|
||||
const { model: optionsModel, provider: optionsProvider } = currentPickerSelection(
|
||||
!!sessionId,
|
||||
{ model: currentModel, provider: currentProvider },
|
||||
|
|
@ -205,6 +229,10 @@ function ModelResults({
|
|||
const isCurrent = model === currentModel && provider.slug === currentProvider
|
||||
const price = provider.pricing?.[model]
|
||||
const locked = unavailable.has(model)
|
||||
const caps = provider.capabilities?.[model]
|
||||
// Only an explicit tools:false demotes — absent means unknown,
|
||||
// and models.dev gaps must not smear working models.
|
||||
const noTools = caps?.tools === false
|
||||
|
||||
return (
|
||||
<CommandItem
|
||||
|
|
@ -212,7 +240,8 @@ function ModelResults({
|
|||
'flex items-center gap-2 pl-6 font-mono',
|
||||
isCurrent &&
|
||||
'bg-primary text-primary-foreground data-[selected=true]:bg-primary data-[selected=true]:text-primary-foreground',
|
||||
locked && 'cursor-not-allowed opacity-45'
|
||||
locked && 'cursor-not-allowed opacity-45',
|
||||
noTools && !isCurrent && 'opacity-60'
|
||||
)}
|
||||
disabled={locked}
|
||||
key={`${provider.slug}:${model}`}
|
||||
|
|
@ -224,6 +253,20 @@ function ModelResults({
|
|||
value={`${provider.slug}:${model}`}
|
||||
>
|
||||
<span className="min-w-0 flex-1 truncate">{model}</span>
|
||||
{noTools && (
|
||||
<span
|
||||
className={cn(
|
||||
'shrink-0 rounded-sm px-1 py-0.5 text-[0.6rem] font-semibold uppercase tracking-wide',
|
||||
isCurrent
|
||||
? 'bg-primary-foreground/20'
|
||||
: 'bg-amber-500/15 text-amber-600 dark:text-amber-400'
|
||||
)}
|
||||
title={copy.noToolsTitle}
|
||||
>
|
||||
{copy.noTools}
|
||||
</span>
|
||||
)}
|
||||
<ModelMeta caps={caps} isCurrent={isCurrent} />
|
||||
{locked && (
|
||||
<span className="shrink-0 text-[0.62rem] uppercase tracking-wide opacity-80">{copy.pro}</span>
|
||||
)}
|
||||
|
|
@ -243,6 +286,44 @@ function ModelResults({
|
|||
)
|
||||
}
|
||||
|
||||
// Compact local-model metadata: context window (agent-critical) plus
|
||||
// parameter size / quantization when the native server reported them.
|
||||
// Renders nothing for models with no metadata beyond the fast/reasoning flags.
|
||||
function ModelMeta({ caps, isCurrent }: { caps?: ModelCapabilities; isCurrent: boolean }) {
|
||||
if (!caps) {
|
||||
return null
|
||||
}
|
||||
|
||||
const parts: string[] = []
|
||||
|
||||
if (caps.context_length) {
|
||||
parts.push(caps.context_length >= 1024 ? `${Math.round(caps.context_length / 1024)}k` : String(caps.context_length))
|
||||
}
|
||||
|
||||
if (caps.parameter_size) {
|
||||
parts.push(caps.parameter_size)
|
||||
}
|
||||
|
||||
if (caps.quantization) {
|
||||
parts.push(caps.quantization)
|
||||
}
|
||||
|
||||
if (parts.length === 0) {
|
||||
return null
|
||||
}
|
||||
|
||||
return (
|
||||
<span
|
||||
className={cn(
|
||||
'shrink-0 text-[0.66rem] tabular-nums',
|
||||
isCurrent ? 'text-primary-foreground/80' : 'text-muted-foreground'
|
||||
)}
|
||||
>
|
||||
{parts.join(' · ')}
|
||||
</span>
|
||||
)
|
||||
}
|
||||
|
||||
// Compact In/Out $/Mtok price tag, mirroring the CLI picker's price columns.
|
||||
// Renders nothing when pricing is unavailable for the model.
|
||||
function ModelPrice({ price, isCurrent }: { price?: ModelPricing; isCurrent: boolean }) {
|
||||
|
|
|
|||
|
|
@ -1,4 +1,6 @@
|
|||
import { cleanup, fireEvent, render, screen } from '@testing-library/react'
|
||||
import { QueryClient, QueryClientProvider } from '@tanstack/react-query'
|
||||
import { cleanup, fireEvent, render as rtlRender, screen } from '@testing-library/react'
|
||||
import type { ReactElement } from 'react'
|
||||
import { afterEach, describe, expect, it } from 'vitest'
|
||||
|
||||
import { $desktopOnboarding, type DesktopOnboardingState, type OnboardingContext } from '@/store/onboarding'
|
||||
|
|
@ -6,6 +8,15 @@ import type { OAuthProvider } from '@/types/hermes'
|
|||
|
||||
import { Picker } from '.'
|
||||
|
||||
// The Picker's local-server detection uses react-query; queries stay pending
|
||||
// in tests (no window.hermesDesktop bridge), which renders as "no detected
|
||||
// row" — the same degraded state as detection failing in the app.
|
||||
function render(ui: ReactElement) {
|
||||
const client = new QueryClient({ defaultOptions: { queries: { retry: false } } })
|
||||
|
||||
return rtlRender(<QueryClientProvider client={client}>{ui}</QueryClientProvider>)
|
||||
}
|
||||
|
||||
function provider(id: string, name = id): OAuthProvider {
|
||||
return {
|
||||
cli_command: `hermes login ${id}`,
|
||||
|
|
|
|||
|
|
@ -1,10 +1,11 @@
|
|||
import { useStore } from '@nanostores/react'
|
||||
import { useQuery } from '@tanstack/react-query'
|
||||
import { useEffect, useMemo, useRef, useState } from 'react'
|
||||
|
||||
import { Button } from '@/components/ui/button'
|
||||
import { Codicon } from '@/components/ui/codicon'
|
||||
import { Input } from '@/components/ui/input'
|
||||
import { getGlobalModelOptions } from '@/hermes'
|
||||
import { detectLocalServers, getGlobalModelOptions } from '@/hermes'
|
||||
import { useI18n } from '@/i18n'
|
||||
import { Check, ChevronDown, ChevronLeft, KeyRound, Loader2 } from '@/lib/icons'
|
||||
import { isProviderSetupErrorMessage } from '@/lib/provider-setup-errors'
|
||||
|
|
@ -22,13 +23,14 @@ import {
|
|||
peekPendingProviderOAuth,
|
||||
refreshOnboarding,
|
||||
saveOnboardingApiKey,
|
||||
saveOnboardingDetectedOllama,
|
||||
setOnboardingMode,
|
||||
startProviderOAuth
|
||||
} from '@/store/onboarding'
|
||||
import type { ModelOptionProvider, OAuthProvider } from '@/types/hermes'
|
||||
|
||||
import { DocsLink, FlowPanel, Status } from './flow'
|
||||
import { FeaturedProviderRow, KeyProviderRow, ProviderRow, sortProviders } from './providers'
|
||||
import { DetectedLocalServerRow, FeaturedProviderRow, KeyProviderRow, ProviderRow, sortProviders } from './providers'
|
||||
|
||||
export { FeaturedProviderRow, KeyProviderRow, ProviderRow, providerTitle, sortProviders } from './providers'
|
||||
|
||||
|
|
@ -399,6 +401,22 @@ export function Picker({ ctx }: { ctx: OnboardingContext }) {
|
|||
const hasOauth = ordered.length > 0
|
||||
const apiKeyOptions = useApiKeyCatalog()
|
||||
|
||||
// Detected local model servers (Ollama). Non-blocking: the picker renders
|
||||
// immediately and the detected row appears when the probe answers. Failures
|
||||
// degrade to "no row" — the manual local-endpoint path still exists.
|
||||
const localServers = useQuery({
|
||||
queryKey: ['local-servers-detect'],
|
||||
queryFn: () => detectLocalServers(),
|
||||
staleTime: 60_000,
|
||||
retry: false
|
||||
})
|
||||
|
||||
const detectedOllama = useMemo(() => {
|
||||
const servers = localServers.data?.servers ?? []
|
||||
|
||||
return servers.find(s => s.type === 'ollama' && s.reachable && s.models.length > 0) ?? null
|
||||
}, [localServers.data])
|
||||
|
||||
// localEndpoint forces the key form regardless of `mode` (which a manual
|
||||
// provider refresh may flip back to 'oauth'); it preselects the local option
|
||||
// and hides the "back to sign in" link since the user came specifically to
|
||||
|
|
@ -438,6 +456,12 @@ export function Picker({ ctx }: { ctx: OnboardingContext }) {
|
|||
<div className="grid gap-2">
|
||||
<div className="grid max-h-[60dvh] gap-2 overflow-y-auto p-1">
|
||||
{featured ? <FeaturedProviderRow onSelect={select} provider={featured} /> : null}
|
||||
{detectedOllama ? (
|
||||
<DetectedLocalServerRow
|
||||
onConnect={(baseUrl, model) => saveOnboardingDetectedOllama(baseUrl, model, ctx)}
|
||||
server={detectedOllama}
|
||||
/>
|
||||
) : null}
|
||||
{showRest ? (
|
||||
<>
|
||||
{rest.map(p => (
|
||||
|
|
|
|||
|
|
@ -1,7 +1,10 @@
|
|||
import { useState } from 'react'
|
||||
|
||||
import { RowButton } from '@/components/ui/row-button'
|
||||
import { useI18n } from '@/i18n'
|
||||
import { Check, ChevronRight, Terminal } from '@/lib/icons'
|
||||
import type { OAuthProvider } from '@/types/hermes'
|
||||
import { Check, ChevronRight, Cpu, Loader2, Terminal } from '@/lib/icons'
|
||||
import { cn } from '@/lib/utils'
|
||||
import type { LocalServerInfo, OAuthProvider } from '@/types/hermes'
|
||||
|
||||
const PROVIDER_DISPLAY: Record<string, { order: number; title: string }> = {
|
||||
nous: { order: 0, title: 'Nous Portal' },
|
||||
|
|
@ -116,3 +119,83 @@ export function ProviderRow({
|
|||
</RowButton>
|
||||
)
|
||||
}
|
||||
|
||||
/**
|
||||
* A local model server found by /api/local-servers/detect (currently the
|
||||
* Ollama row; LM Studio detection reuses the shape later). Collapsed: a
|
||||
* provider row showing "detected · N models installed". Expanded: the
|
||||
* installed-model list, each row a one-click connect — no URL typing, no
|
||||
* blind first-model assignment.
|
||||
*/
|
||||
export function DetectedLocalServerRow({
|
||||
onConnect,
|
||||
server
|
||||
}: {
|
||||
onConnect: (baseUrl: string, model: string) => Promise<{ ok: boolean; message?: string }>
|
||||
server: LocalServerInfo
|
||||
}) {
|
||||
const { t } = useI18n()
|
||||
const copy = t.onboarding.detectedLocal
|
||||
const [expanded, setExpanded] = useState(false)
|
||||
const [busyModel, setBusyModel] = useState<null | string>(null)
|
||||
const [error, setError] = useState('')
|
||||
|
||||
const connect = async (model: string) => {
|
||||
if (busyModel) {
|
||||
return
|
||||
}
|
||||
|
||||
setBusyModel(model)
|
||||
setError('')
|
||||
const result = await onConnect(server.base_url, model)
|
||||
|
||||
if (!result.ok) {
|
||||
setError(result.message || copy.connectFailed)
|
||||
setBusyModel(null)
|
||||
}
|
||||
// On success the overlay unmounts via completeDesktopOnboarding().
|
||||
}
|
||||
|
||||
return (
|
||||
<div className="rounded-[6px] transition-colors">
|
||||
<RowButton className={PROVIDER_ROW_CLASS} onClick={() => setExpanded(open => !open)}>
|
||||
<div className="min-w-0">
|
||||
<div className="flex items-center gap-2">
|
||||
<Cpu className="size-4 shrink-0 text-primary" />
|
||||
<span className="text-[length:var(--conversation-text-font-size)] font-semibold">Ollama</span>
|
||||
<span className="inline-flex items-center gap-1 bg-primary/10 px-2 py-0.5 text-xs font-medium text-primary">
|
||||
<Check className="size-3" />
|
||||
{copy.detected}
|
||||
</span>
|
||||
</div>
|
||||
<p className="mt-1 text-xs leading-5 text-muted-foreground">{copy.modelsInstalled(server.models.length)}</p>
|
||||
</div>
|
||||
<ChevronRight
|
||||
className={cn('size-4 text-muted-foreground transition group-hover:text-foreground', expanded && 'rotate-90')}
|
||||
/>
|
||||
</RowButton>
|
||||
{expanded ? (
|
||||
<div className="grid gap-0.5 px-3 pb-2">
|
||||
{server.models.map(model => (
|
||||
<RowButton
|
||||
className="group flex w-full items-center justify-between gap-3 rounded-[6px] px-3 py-1.5 text-left font-mono text-xs transition-colors hover:bg-(--ui-control-hover-background)"
|
||||
disabled={busyModel !== null}
|
||||
key={model}
|
||||
onClick={() => void connect(model)}
|
||||
>
|
||||
<span className="truncate">{model}</span>
|
||||
{busyModel === model ? (
|
||||
<Loader2 className="size-3.5 shrink-0 animate-spin text-muted-foreground" />
|
||||
) : (
|
||||
<span className="shrink-0 text-[0.64rem] uppercase tracking-wide text-muted-foreground opacity-0 transition group-hover:opacity-100">
|
||||
{copy.use}
|
||||
</span>
|
||||
)}
|
||||
</RowButton>
|
||||
))}
|
||||
{error ? <p className="px-3 pt-1 text-xs leading-5 text-destructive">{error}</p> : null}
|
||||
</div>
|
||||
) : null}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
|
|
|||
|
|
@ -19,6 +19,7 @@ import type {
|
|||
EnvVarInfo,
|
||||
HermesConfig,
|
||||
HermesConfigRecord,
|
||||
LocalServersResponse,
|
||||
LogsResponse,
|
||||
McpCatalogResponse,
|
||||
McpServerSummary,
|
||||
|
|
@ -37,6 +38,8 @@ import type {
|
|||
OAuthProvidersResponse,
|
||||
OAuthStartResponse,
|
||||
OAuthSubmitResponse,
|
||||
OllamaModelsResponse,
|
||||
OllamaPullStatus,
|
||||
PaginatedSessions,
|
||||
ProfileCreatePayload,
|
||||
ProfileSetupCommand,
|
||||
|
|
@ -902,6 +905,62 @@ export interface RecommendedDefaultModel {
|
|||
free_tier: boolean | null
|
||||
}
|
||||
|
||||
// Detect running local model servers (Ollama, LM Studio) by probing their
|
||||
// well-known ports plus the configured base_url. Results are cached
|
||||
// server-side for 60s; refresh=true busts the cache.
|
||||
export function detectLocalServers(opts?: { refresh?: boolean }): Promise<LocalServersResponse> {
|
||||
return window.hermesDesktop.api<LocalServersResponse>({
|
||||
...profileScoped(),
|
||||
path: opts?.refresh ? '/api/local-servers/detect?refresh=1' : '/api/local-servers/detect'
|
||||
})
|
||||
}
|
||||
|
||||
// Installed + running + recommended models for the local Ollama server.
|
||||
export function getOllamaModels(): Promise<OllamaModelsResponse> {
|
||||
return window.hermesDesktop.api<OllamaModelsResponse>({
|
||||
...profileScoped(),
|
||||
path: '/api/ollama/models'
|
||||
})
|
||||
}
|
||||
|
||||
// Start pulling a model from the Ollama registry; poll with pollOllamaPull.
|
||||
export function startOllamaPull(model: string): Promise<{ job_id: string }> {
|
||||
return window.hermesDesktop.api<{ job_id: string }>({
|
||||
...profileScoped(),
|
||||
path: '/api/ollama/pull',
|
||||
method: 'POST',
|
||||
body: { model }
|
||||
})
|
||||
}
|
||||
|
||||
export function pollOllamaPull(jobId: string): Promise<OllamaPullStatus> {
|
||||
return window.hermesDesktop.api<OllamaPullStatus>({
|
||||
...profileScoped(),
|
||||
path: `/api/ollama/pull/${encodeURIComponent(jobId)}`
|
||||
})
|
||||
}
|
||||
|
||||
export function deleteOllamaModel(model: string): Promise<{ ok: boolean; message: string }> {
|
||||
return window.hermesDesktop.api<{ ok: boolean; message: string }>({
|
||||
...profileScoped(),
|
||||
path: '/api/ollama/delete',
|
||||
method: 'POST',
|
||||
body: { model }
|
||||
})
|
||||
}
|
||||
|
||||
// Load a model into memory. keep_alive pins residence ("5m", "2h", "-1" =
|
||||
// indefinite); Ollama applies keep_alive per-load, so this doubles as the
|
||||
// way to change it for an already-loaded model.
|
||||
export function loadOllamaModel(model: string, keepAlive?: string): Promise<{ ok: boolean; message: string }> {
|
||||
return window.hermesDesktop.api<{ ok: boolean; message: string }>({
|
||||
...profileScoped(),
|
||||
path: '/api/ollama/load',
|
||||
method: 'POST',
|
||||
body: { model, keep_alive: keepAlive ?? null }
|
||||
})
|
||||
}
|
||||
|
||||
// Recommended default model for a freshly-authenticated provider. Mirrors the
|
||||
// curation `hermes model` does — for Nous it honors the free/paid tier so a
|
||||
// free user gets a free model instead of a paid default.
|
||||
|
|
|
|||
|
|
@ -680,6 +680,24 @@ export const en: Translations = {
|
|||
curator: { label: 'Curator', hint: 'Skill-usage review' }
|
||||
}
|
||||
},
|
||||
ollama: {
|
||||
categoryTitle: 'Local servers',
|
||||
running: (n: number) => (n === 1 ? 'Running · 1 model' : `Running · ${n} models`),
|
||||
notRunning: 'Not running — start Ollama (localhost:11434) to manage and use local models.',
|
||||
desc: 'Pull, delete, and warm up models on the local Ollama server.',
|
||||
loaded: 'Loaded',
|
||||
load: 'Load',
|
||||
loadFailed: 'Could not load the model',
|
||||
deleteLabel: (model: string) => `Delete ${model}`,
|
||||
deleteFailed: 'Could not delete the model',
|
||||
getModels: 'Get models',
|
||||
pull: 'Pull',
|
||||
pullPlaceholder: 'model:tag (e.g. qwen3:8b)',
|
||||
pullDone: (model: string) => `${model} is ready`,
|
||||
pullFailed: 'Model pull failed',
|
||||
kvCacheAdvisory: (model: string, loaded: string, trained: string) =>
|
||||
`${model} is running a ${loaded} context but supports ${trained}. Setting OLLAMA_KV_CACHE_TYPE=q8_0 on the Ollama server roughly doubles the context that fits in the same memory.`
|
||||
},
|
||||
providers: {
|
||||
connectAccount: 'Connect an account',
|
||||
haveApiKey: 'Have an API key instead?',
|
||||
|
|
@ -1737,6 +1755,7 @@ export const en: Translations = {
|
|||
stop: 'Stop',
|
||||
dismiss: 'Dismiss',
|
||||
exit: code => `exit ${code}`,
|
||||
ollamaLoading: (model: string) => `Loading ${model} into memory…`,
|
||||
coding: {
|
||||
title: 'Working tree',
|
||||
noBranch: 'No branch',
|
||||
|
|
@ -1896,6 +1915,12 @@ export const en: Translations = {
|
|||
connected: 'Connected',
|
||||
featuredPitch: 'One subscription, 300+ frontier models — the recommended way to run Hermes',
|
||||
openRouterPitch: 'One key, hundreds of models — a solid default',
|
||||
detectedLocal: {
|
||||
detected: 'Detected',
|
||||
modelsInstalled: (n: number) => (n === 1 ? '1 model installed — runs on this machine' : `${n} models installed — runs on this machine`),
|
||||
use: 'Use',
|
||||
connectFailed: 'Could not connect to the local server.'
|
||||
},
|
||||
apiKeyOptions: {
|
||||
openrouter: {
|
||||
short: 'one key, many models',
|
||||
|
|
@ -1965,6 +1990,8 @@ export const en: Translations = {
|
|||
noAuthenticatedProviders: 'No authenticated providers.',
|
||||
pro: 'Pro',
|
||||
proNeedsSubscription: 'Pro models need a paid Nous subscription.',
|
||||
noTools: 'No tools',
|
||||
noToolsTitle: 'This model does not support tool calling — most agent features will not work with it.',
|
||||
free: 'Free',
|
||||
freeTier: 'Free tier',
|
||||
priceTitle: 'Input / Output price per million tokens'
|
||||
|
|
|
|||
|
|
@ -1874,6 +1874,12 @@ export const ja = defineLocale({
|
|||
connected: '接続済み',
|
||||
featuredPitch: '1 つのサブスクリプションで 300 以上の最先端モデル — Hermes を実行するための推奨方法',
|
||||
openRouterPitch: '1 つのキーで数百のモデル — 堅実なデフォルト',
|
||||
detectedLocal: {
|
||||
detected: '検出済み',
|
||||
modelsInstalled: (n: number) => `${n} 個のモデルがインストール済み — このマシンで実行`,
|
||||
use: '使用',
|
||||
connectFailed: 'ローカルサーバーに接続できませんでした。'
|
||||
},
|
||||
apiKeyOptions: {
|
||||
openrouter: {
|
||||
short: '1 つのキーで多くのモデル',
|
||||
|
|
@ -1943,6 +1949,8 @@ export const ja = defineLocale({
|
|||
noAuthenticatedProviders: '認証済みプロバイダーがありません。',
|
||||
pro: 'Pro',
|
||||
proNeedsSubscription: 'Pro モデルには有料の Nous サブスクリプションが必要です。',
|
||||
noTools: 'ツール非対応',
|
||||
noToolsTitle: 'このモデルはツール呼び出しに対応していません — ほとんどのエージェント機能が動作しません。',
|
||||
free: '無料',
|
||||
freeTier: '無料プラン',
|
||||
priceTitle: '100 万トークンあたりの入力/出力価格'
|
||||
|
|
|
|||
|
|
@ -580,6 +580,23 @@ export interface Translations {
|
|||
providerDefault: string
|
||||
tasks: Record<string, AuxTaskCopy>
|
||||
}
|
||||
ollama: {
|
||||
categoryTitle: string
|
||||
running: (n: number) => string
|
||||
notRunning: string
|
||||
desc: string
|
||||
loaded: string
|
||||
load: string
|
||||
loadFailed: string
|
||||
deleteLabel: (model: string) => string
|
||||
deleteFailed: string
|
||||
getModels: string
|
||||
pull: string
|
||||
pullPlaceholder: string
|
||||
pullDone: (model: string) => string
|
||||
pullFailed: string
|
||||
kvCacheAdvisory: (model: string, loaded: string, trained: string) => string
|
||||
}
|
||||
providers: {
|
||||
connectAccount: string
|
||||
haveApiKey: string
|
||||
|
|
@ -1419,6 +1436,7 @@ export interface Translations {
|
|||
stop: string
|
||||
dismiss: string
|
||||
exit: (code: number) => string
|
||||
ollamaLoading: (model: string) => string
|
||||
coding: {
|
||||
title: string
|
||||
noBranch: string
|
||||
|
|
@ -1554,6 +1572,12 @@ export interface Translations {
|
|||
connected: string
|
||||
featuredPitch: string
|
||||
openRouterPitch: string
|
||||
detectedLocal: {
|
||||
detected: string
|
||||
modelsInstalled: (n: number) => string
|
||||
use: string
|
||||
connectFailed: string
|
||||
}
|
||||
apiKeyOptions: Record<string, { short: string; description: string }>
|
||||
backToSignIn: string
|
||||
getKey: string
|
||||
|
|
@ -1605,6 +1629,8 @@ export interface Translations {
|
|||
noAuthenticatedProviders: string
|
||||
pro: string
|
||||
proNeedsSubscription: string
|
||||
noTools: string
|
||||
noToolsTitle: string
|
||||
free: string
|
||||
freeTier: string
|
||||
priceTitle: string
|
||||
|
|
|
|||
|
|
@ -1817,6 +1817,12 @@ export const zhHant = defineLocale({
|
|||
connected: '已連線',
|
||||
featuredPitch: '一個訂閱,300+ 前沿模型 — 執行 Hermes 的建議方式',
|
||||
openRouterPitch: '一個金鑰,數百個模型 — 穩定的預設選擇',
|
||||
detectedLocal: {
|
||||
detected: '已偵測到',
|
||||
modelsInstalled: (n: number) => `已安裝 ${n} 個模型 — 在本機執行`,
|
||||
use: '使用',
|
||||
connectFailed: '無法連線到本機伺服器。'
|
||||
},
|
||||
apiKeyOptions: {
|
||||
openrouter: { short: '一個金鑰,多個模型', description: '用一個金鑰存取數百個模型。適合新安裝的預設選擇。' },
|
||||
openai: { short: 'GPT 等級模型', description: '直接存取 OpenAI 模型。' },
|
||||
|
|
@ -1880,6 +1886,8 @@ export const zhHant = defineLocale({
|
|||
noAuthenticatedProviders: '沒有已驗證的提供方。',
|
||||
pro: 'Pro',
|
||||
proNeedsSubscription: 'Pro 模型需要付費 Nous 訂閱。',
|
||||
noTools: '不支援工具',
|
||||
noToolsTitle: '該模型不支援工具呼叫 — 大多數智能體功能將無法使用。',
|
||||
free: '免費',
|
||||
freeTier: '免費層',
|
||||
priceTitle: '每百萬 Token 的輸入/輸出價格'
|
||||
|
|
|
|||
|
|
@ -868,6 +868,24 @@ export const zh: Translations = {
|
|||
curator: { label: '维护器', hint: '技能使用审查' }
|
||||
}
|
||||
},
|
||||
ollama: {
|
||||
categoryTitle: '本地服务器',
|
||||
running: (n: number) => `运行中 · ${n} 个模型`,
|
||||
notRunning: '未运行 — 启动 Ollama(localhost:11434)以管理和使用本地模型。',
|
||||
desc: '在本地 Ollama 服务器上拉取、删除和预热模型。',
|
||||
loaded: '已加载',
|
||||
load: '加载',
|
||||
loadFailed: '无法加载模型',
|
||||
deleteLabel: (model: string) => `删除 ${model}`,
|
||||
deleteFailed: '无法删除模型',
|
||||
getModels: '获取模型',
|
||||
pull: '拉取',
|
||||
pullPlaceholder: 'model:tag(例如 qwen3:8b)',
|
||||
pullDone: (model: string) => `${model} 已就绪`,
|
||||
pullFailed: '模型拉取失败',
|
||||
kvCacheAdvisory: (model: string, loaded: string, trained: string) =>
|
||||
`${model} 当前以 ${loaded} 上下文运行,但支持 ${trained}。在 Ollama 服务器上设置 OLLAMA_KV_CACHE_TYPE=q8_0 可在相同内存下大约翻倍可用上下文。`
|
||||
},
|
||||
providers: {
|
||||
connectAccount: '连接账号',
|
||||
haveApiKey: '改用 API 密钥?',
|
||||
|
|
@ -1912,6 +1930,7 @@ export const zh: Translations = {
|
|||
stop: '停止',
|
||||
dismiss: '关闭',
|
||||
exit: code => `退出码 ${code}`,
|
||||
ollamaLoading: (model: string) => `正在将 ${model} 加载到内存…`,
|
||||
coding: {
|
||||
title: '工作区',
|
||||
noBranch: '无分支',
|
||||
|
|
@ -2068,6 +2087,12 @@ export const zh: Translations = {
|
|||
connected: '已连接',
|
||||
featuredPitch: '一个订阅,300+ 前沿模型 — 运行 Hermes 的推荐方式',
|
||||
openRouterPitch: '一个密钥,数百个模型 — 稳妥的默认选择',
|
||||
detectedLocal: {
|
||||
detected: '已检测到',
|
||||
modelsInstalled: (n: number) => `已安装 ${n} 个模型 — 在本机运行`,
|
||||
use: '使用',
|
||||
connectFailed: '无法连接到本地服务器。'
|
||||
},
|
||||
apiKeyOptions: {
|
||||
openrouter: { short: '一个密钥,多个模型', description: '用一个密钥访问数百个模型。适合新安装的默认选择。' },
|
||||
openai: { short: 'GPT 级模型', description: '直接访问 OpenAI 模型。' },
|
||||
|
|
@ -2132,6 +2157,8 @@ export const zh: Translations = {
|
|||
noAuthenticatedProviders: '没有已认证的提供方。',
|
||||
pro: 'Pro',
|
||||
proNeedsSubscription: 'Pro 模型需要付费 Nous 订阅。',
|
||||
noTools: '不支持工具',
|
||||
noToolsTitle: '该模型不支持工具调用 — 大多数智能体功能将无法使用。',
|
||||
free: '免费',
|
||||
freeTier: '免费层',
|
||||
priceTitle: '每百万 token 的输入/输出价格'
|
||||
|
|
|
|||
|
|
@ -782,6 +782,42 @@ export async function saveOnboardingApiKey(
|
|||
}
|
||||
}
|
||||
|
||||
// Configure a DETECTED local Ollama server. Unlike the generic local-endpoint
|
||||
// path above, the server identity is already fingerprinted and the user has
|
||||
// picked a concrete model from its installed list, so there is no probe step —
|
||||
// persist provider=ollama + base_url + model and verify the runtime.
|
||||
export async function saveOnboardingDetectedOllama(baseUrl: string, model: string, ctx: OnboardingContext) {
|
||||
const url = baseUrl.trim()
|
||||
const chosen = model.trim()
|
||||
|
||||
if (!url || !chosen) {
|
||||
return { ok: false, message: 'Pick a model first.' }
|
||||
}
|
||||
|
||||
try {
|
||||
await setModelAssignment({ scope: 'main', provider: 'ollama', model: chosen, base_url: url })
|
||||
await ctx.requestGateway('reload.env').catch(() => undefined)
|
||||
|
||||
const runtime = await checkRuntime(ctx, 'ollama')
|
||||
|
||||
if (!runtime.ready) {
|
||||
const detail = (runtime.reason ?? '').trim()
|
||||
|
||||
return { ok: false, message: detail || `Saved, but Hermes still cannot reach ${url}.` }
|
||||
}
|
||||
|
||||
notifyReady('Ollama')
|
||||
completeDesktopOnboarding()
|
||||
ctx.onCompleted?.()
|
||||
|
||||
return { ok: true }
|
||||
} catch (error) {
|
||||
notifyError(error, 'Could not save Ollama endpoint')
|
||||
|
||||
return { ok: false, message: errMessage(error) }
|
||||
}
|
||||
}
|
||||
|
||||
// Configure a local / self-hosted OpenAI-compatible endpoint (vLLM, llama.cpp,
|
||||
// Ollama, …). Unlike API-key providers, a local endpoint is defined by its URL
|
||||
// and usually needs NO key. The runtime resolver reads model.base_url from
|
||||
|
|
|
|||
|
|
@ -278,6 +278,83 @@ export interface ModelOptionProvider {
|
|||
export interface ModelCapabilities {
|
||||
fast: boolean
|
||||
reasoning: boolean
|
||||
/** True/false when a metadata source reported tool support; absent = unknown. */
|
||||
tools?: boolean
|
||||
vision?: boolean
|
||||
/** Trained/served context window in tokens, when known. */
|
||||
context_length?: number
|
||||
/** Local Ollama models only, from the server's native metadata. */
|
||||
parameter_size?: string
|
||||
quantization?: string
|
||||
}
|
||||
|
||||
/** One detected local model server from GET /api/local-servers/detect. */
|
||||
export interface LocalServerInfo {
|
||||
/** Fingerprinted server type ("ollama", "lm-studio", ...) or the expected
|
||||
* type for an unreachable well-known port. */
|
||||
type: string
|
||||
/** OpenAI-compatible endpoint (server root + /v1). */
|
||||
base_url: string
|
||||
reachable: boolean
|
||||
models: string[]
|
||||
/** "well-known" (default port probe) or "configured" (model.base_url). */
|
||||
source: string
|
||||
}
|
||||
|
||||
export interface LocalServersResponse {
|
||||
servers: LocalServerInfo[]
|
||||
}
|
||||
|
||||
/** Installed model row from GET /api/ollama/models. */
|
||||
export interface OllamaInstalledModel {
|
||||
name: string
|
||||
size_bytes?: number
|
||||
parameter_size?: string
|
||||
quantization?: string
|
||||
family?: string
|
||||
modified_at?: string
|
||||
}
|
||||
|
||||
/** Loaded model row from GET /api/ollama/models (native /api/ps). */
|
||||
export interface OllamaRunningModel {
|
||||
name: string
|
||||
size_bytes?: number
|
||||
size_vram_bytes?: number
|
||||
/** Window the model was actually loaded with. */
|
||||
context_length?: number
|
||||
expires_at?: string
|
||||
}
|
||||
|
||||
export interface OllamaRecommendedModel {
|
||||
model: string
|
||||
description: string
|
||||
approx_vram_gb?: number
|
||||
}
|
||||
|
||||
/** Loaded-vs-trained context gap that q8_0 KV quantization would close. */
|
||||
export interface OllamaKvCacheAdvisory {
|
||||
model: string
|
||||
loaded_context: number
|
||||
trained_context: number
|
||||
}
|
||||
|
||||
export interface OllamaModelsResponse {
|
||||
reachable: boolean
|
||||
installed: OllamaInstalledModel[]
|
||||
running: OllamaRunningModel[]
|
||||
recommended: OllamaRecommendedModel[]
|
||||
kv_cache_advisory?: null | OllamaKvCacheAdvisory
|
||||
}
|
||||
|
||||
export interface OllamaPullStatus {
|
||||
job_id: string
|
||||
model: string
|
||||
/** pulling | done | error */
|
||||
status: string
|
||||
detail?: string
|
||||
error_message?: string
|
||||
total_bytes?: number
|
||||
completed_bytes?: number
|
||||
}
|
||||
|
||||
export interface ModelOptionsResponse {
|
||||
|
|
|
|||
|
|
@ -151,6 +151,10 @@ SERVICE_PROVIDER_NAMES: Dict[str, str] = {
|
|||
# any remote service.
|
||||
LMSTUDIO_NOAUTH_PLACEHOLDER = "dummy-lm-api-key"
|
||||
|
||||
# Same role for local Ollama servers, which are unauthenticated by default.
|
||||
# Sent only to the local server, never to any remote service.
|
||||
OLLAMA_NOAUTH_PLACEHOLDER = "dummy-ollama-api-key"
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Provider Registry
|
||||
|
|
@ -217,6 +221,16 @@ PROVIDER_REGISTRY: Dict[str, ProviderConfig] = {
|
|||
api_key_env_vars=("LM_API_KEY",),
|
||||
base_url_env_var="LM_BASE_URL",
|
||||
),
|
||||
"ollama": ProviderConfig(
|
||||
id="ollama",
|
||||
name="Ollama (local)",
|
||||
auth_type="api_key",
|
||||
# 127.0.0.1, not localhost — see _normalize_ollama_runtime_base_url.
|
||||
inference_base_url="http://127.0.0.1:11434/v1",
|
||||
# No key env vars: local Ollama servers are unauthenticated by default.
|
||||
# A key for a reverse-proxied server can be set via model.api_key.
|
||||
api_key_env_vars=(),
|
||||
),
|
||||
"copilot": ProviderConfig(
|
||||
id="copilot",
|
||||
name="GitHub Copilot",
|
||||
|
|
@ -747,6 +761,26 @@ def _normalize_lmstudio_runtime_base_url(base_url: str) -> str:
|
|||
return (root or "http://127.0.0.1:1234") + "/v1"
|
||||
|
||||
|
||||
def _normalize_ollama_runtime_base_url(base_url: str) -> str:
|
||||
"""Return the OpenAI-compatible local Ollama runtime base URL.
|
||||
|
||||
Ollama's native management API lives under ``/api`` while its
|
||||
OpenAI-compatible chat endpoint lives under ``/v1``. Users paste either
|
||||
form (or the bare server root) into ``model.base_url``; normalize before
|
||||
the OpenAI SDK appends ``/chat/completions``.
|
||||
"""
|
||||
root = str(base_url or "").strip().rstrip("/")
|
||||
for suffix in ("/api", "/v1"):
|
||||
if root.endswith(suffix):
|
||||
root = root[: -len(suffix)].rstrip("/")
|
||||
break
|
||||
# IPv4 loopback, not localhost: Ollama binds 127.0.0.1 by default and
|
||||
# Windows resolves localhost to ::1 first — a ~2s failed IPv6 connect on
|
||||
# every chat request otherwise.
|
||||
root = root.replace("://localhost", "://127.0.0.1")
|
||||
return (root or "http://127.0.0.1:11434") + "/v1"
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Error Types
|
||||
# =============================================================================
|
||||
|
|
@ -1675,7 +1709,7 @@ def resolve_provider(
|
|||
"kilo": "kilocode", "kilo-code": "kilocode", "kilo-gateway": "kilocode",
|
||||
"lmstudio": "lmstudio", "lm-studio": "lmstudio", "lm_studio": "lmstudio",
|
||||
# Local server aliases — route through the generic custom provider
|
||||
"ollama": "custom", "ollama_cloud": "ollama-cloud",
|
||||
"ollama": "ollama", "ollama_cloud": "ollama-cloud",
|
||||
"vllm": "custom", "llamacpp": "custom",
|
||||
"llama.cpp": "custom", "llama-cpp": "custom",
|
||||
}
|
||||
|
|
@ -6386,6 +6420,11 @@ def resolve_api_key_provider_credentials(provider_id: str) -> Dict[str, Any]:
|
|||
api_key = LMSTUDIO_NOAUTH_PLACEHOLDER
|
||||
key_source = key_source or "default"
|
||||
|
||||
# Same for local Ollama servers (unauthenticated by default).
|
||||
if not api_key and provider_id == "ollama":
|
||||
api_key = OLLAMA_NOAUTH_PLACEHOLDER
|
||||
key_source = key_source or "default"
|
||||
|
||||
env_url = ""
|
||||
if pconfig.base_url_env_var:
|
||||
env_url = os.getenv(pconfig.base_url_env_var, "").strip()
|
||||
|
|
@ -6422,6 +6461,9 @@ def resolve_api_key_provider_credentials(provider_id: str) -> Dict[str, Any]:
|
|||
if provider_id == "lmstudio":
|
||||
base_url = _normalize_lmstudio_runtime_base_url(base_url)
|
||||
|
||||
if provider_id == "ollama":
|
||||
base_url = _normalize_ollama_runtime_base_url(base_url)
|
||||
|
||||
# Last-resort guard: an API-key provider must never hand back an empty
|
||||
# base URL (a set-but-empty COPILOT_API_BASE_URL or similar env override
|
||||
# otherwise wedges chat inference — #50252).
|
||||
|
|
|
|||
|
|
@ -33,6 +33,8 @@ Substrate facts (verified May 2026):
|
|||
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
|
||||
from dataclasses import dataclass, replace
|
||||
from typing import Optional
|
||||
|
||||
|
|
@ -252,13 +254,23 @@ def build_models_payload(
|
|||
|
||||
|
||||
def _apply_capabilities(rows: list[dict]) -> None:
|
||||
"""Attach a ``{model: {fast, reasoning}}`` map to each provider row.
|
||||
"""Attach a per-model capabilities map to each provider row.
|
||||
|
||||
Each entry: ``{fast, reasoning, tools?, vision?, context_length?,
|
||||
parameter_size?, quantization?}``. Only ``fast`` and ``reasoning`` are
|
||||
always present; the rest are included when a metadata source actually
|
||||
reported them, so the UI can distinguish "no tool support" from "unknown".
|
||||
|
||||
`fast` mirrors ``model_supports_fast_mode`` (the same gate the runtime
|
||||
enforces). `reasoning` comes from the models.dev catalog when known and
|
||||
defaults to True otherwise — the effort dial is broadly accepted and a
|
||||
no-op on models that ignore it, whereas hiding it from a capable-but-
|
||||
uncatalogued model is the worse failure.
|
||||
|
||||
Local Ollama rows are enriched from the server's native ``/api/show``
|
||||
instead of models.dev: local tags are arbitrary and quant-specific
|
||||
(``qwen3:27b-q4_K_M``, user Modelfile creations), so the catalog rarely
|
||||
matches, while ``/api/show`` is authoritative for what is on disk.
|
||||
"""
|
||||
from hermes_cli.models import model_supports_fast_mode
|
||||
|
||||
|
|
@ -269,25 +281,118 @@ def _apply_capabilities(rows: list[dict]) -> None:
|
|||
|
||||
for row in rows:
|
||||
slug = row.get("slug") or ""
|
||||
caps: dict[str, dict[str, bool]] = {}
|
||||
caps: dict[str, dict] = {}
|
||||
|
||||
for model in row.get("models") or []:
|
||||
reasoning = True
|
||||
entry: dict = {
|
||||
"fast": bool(model_supports_fast_mode(model)),
|
||||
"reasoning": True,
|
||||
}
|
||||
if get_model_capabilities is not None and slug:
|
||||
try:
|
||||
meta = get_model_capabilities(slug, model)
|
||||
if meta is not None:
|
||||
reasoning = bool(meta.supports_reasoning)
|
||||
entry["reasoning"] = bool(meta.supports_reasoning)
|
||||
entry["tools"] = bool(meta.supports_tools)
|
||||
entry["vision"] = bool(meta.supports_vision)
|
||||
if meta.context_window:
|
||||
entry["context_length"] = int(meta.context_window)
|
||||
except Exception:
|
||||
reasoning = True
|
||||
pass
|
||||
|
||||
caps[model] = {
|
||||
"fast": bool(model_supports_fast_mode(model)),
|
||||
"reasoning": reasoning,
|
||||
}
|
||||
caps[model] = entry
|
||||
|
||||
row["capabilities"] = caps
|
||||
|
||||
_apply_ollama_native_capabilities(rows)
|
||||
|
||||
|
||||
def _apply_ollama_native_capabilities(rows: list[dict]) -> None:
|
||||
"""Override local Ollama rows' capabilities with native ``/api/show`` data.
|
||||
|
||||
Cache-only on the request path: the picker/settings payload must never
|
||||
wait on per-model ``/api/show`` round-trips (they serialize and stall the
|
||||
settings skeleton). Missing metadata is backfilled by a background thread
|
||||
and appears on the next payload build.
|
||||
"""
|
||||
ollama_rows = [r for r in rows if str(r.get("slug", "")).lower() == "ollama"]
|
||||
if not ollama_rows:
|
||||
return
|
||||
|
||||
import time as _time
|
||||
|
||||
now = _time.time()
|
||||
missing: list[tuple[str, str]] = []
|
||||
for row in ollama_rows:
|
||||
base_url = str(row.get("api_url", "") or "").strip() or None
|
||||
caps = row.get("capabilities") or {}
|
||||
for model in row.get("models") or []:
|
||||
cache_key = (base_url or "", model)
|
||||
cached = _OLLAMA_META_CACHE.get(cache_key)
|
||||
if cached is None or (now - cached[0]) >= _OLLAMA_META_CACHE_TTL:
|
||||
missing.append(cache_key)
|
||||
continue
|
||||
meta = cached[1]
|
||||
if not meta:
|
||||
continue
|
||||
entry = caps.setdefault(model, {"fast": False, "reasoning": True})
|
||||
native = meta.get("capabilities")
|
||||
if isinstance(native, list):
|
||||
entry["tools"] = "tools" in native
|
||||
entry["vision"] = "vision" in native
|
||||
entry["reasoning"] = "thinking" in native
|
||||
if meta.get("context_length"):
|
||||
entry["context_length"] = int(meta["context_length"])
|
||||
if meta.get("parameter_size"):
|
||||
entry["parameter_size"] = meta["parameter_size"]
|
||||
if meta.get("quantization"):
|
||||
entry["quantization"] = meta["quantization"]
|
||||
row["capabilities"] = caps
|
||||
|
||||
if missing:
|
||||
_start_ollama_meta_backfill(missing)
|
||||
|
||||
|
||||
def _start_ollama_meta_backfill(keys: list[tuple[str, str]]) -> None:
|
||||
"""Fetch missing model metadata off the request path."""
|
||||
import time as _time
|
||||
|
||||
with _OLLAMA_META_BACKFILL_LOCK:
|
||||
todo = [k for k in keys if k not in _OLLAMA_META_IN_FLIGHT]
|
||||
if not todo:
|
||||
return
|
||||
_OLLAMA_META_IN_FLIGHT.update(todo)
|
||||
|
||||
def _backfill() -> None:
|
||||
try:
|
||||
from hermes_cli.models import fetch_ollama_model_metadata
|
||||
|
||||
for base_url, model in todo:
|
||||
try:
|
||||
meta = fetch_ollama_model_metadata(
|
||||
model, base_url=base_url or None, timeout=5.0
|
||||
)
|
||||
if meta:
|
||||
_OLLAMA_META_CACHE[(base_url, model)] = (_time.time(), meta)
|
||||
except Exception:
|
||||
pass
|
||||
finally:
|
||||
with _OLLAMA_META_BACKFILL_LOCK:
|
||||
_OLLAMA_META_IN_FLIGHT.difference_update(todo)
|
||||
|
||||
threading.Thread(
|
||||
target=_backfill, daemon=True, name="ollama-meta-backfill"
|
||||
).start()
|
||||
|
||||
|
||||
# Metadata for an installed model changes only when the user re-pulls or
|
||||
# edits a Modelfile — an hour of staleness is fine and keeps repeated picker
|
||||
# opens off the per-model /api/show round-trips.
|
||||
_OLLAMA_META_CACHE: dict = {}
|
||||
_OLLAMA_META_CACHE_TTL = 3600.0
|
||||
_OLLAMA_META_BACKFILL_LOCK = threading.Lock()
|
||||
_OLLAMA_META_IN_FLIGHT: set = set()
|
||||
|
||||
|
||||
# ─── Internal: row post-processing ──────────────────────────────────────
|
||||
|
||||
|
|
@ -342,6 +447,14 @@ def _filter_explicit_provider_rows(rows: list[dict], ctx: ConfigContext) -> list
|
|||
# provider. Hide it from explicit-only pickers unless it is the
|
||||
# current provider (handled above).
|
||||
continue
|
||||
if slug == "ollama":
|
||||
# Local Ollama servers are keyless by design — reachability is
|
||||
# the credential, and the row only exists when the server
|
||||
# answered the probe. A running local server is as explicit as a
|
||||
# pasted API key; is_provider_explicitly_configured() can't see
|
||||
# it because there is no env var or auth-store entry to find.
|
||||
kept.append(row)
|
||||
continue
|
||||
if is_provider_explicitly_configured(slug):
|
||||
kept.append(row)
|
||||
return kept
|
||||
|
|
@ -378,11 +491,16 @@ def _apply_picker_hints(rows: list[dict]) -> None:
|
|||
)
|
||||
row["auth_type"] = auth_type
|
||||
row["key_env"] = key_env
|
||||
row["warning"] = (
|
||||
f"paste {key_env} to activate"
|
||||
if auth_type == "api_key" and key_env
|
||||
else f"run `hermes model` to configure ({auth_type})"
|
||||
)
|
||||
if row["slug"] == "ollama":
|
||||
# Keyless local server: the fix for an unconfigured row is
|
||||
# starting the server, not pasting a key.
|
||||
row["warning"] = "start Ollama (localhost:11434) to activate"
|
||||
else:
|
||||
row["warning"] = (
|
||||
f"paste {key_env} to activate"
|
||||
if auth_type == "api_key" and key_env
|
||||
else f"run `hermes model` to configure ({auth_type})"
|
||||
)
|
||||
|
||||
|
||||
def _reorder_canonical(rows: list[dict]) -> list[dict]:
|
||||
|
|
|
|||
|
|
@ -1642,6 +1642,21 @@ def list_authenticated_providers(
|
|||
live = [current_model]
|
||||
curated["lmstudio"] = live
|
||||
|
||||
# Local Ollama server: no static catalog and no key requirement — the row
|
||||
# exists when the server is reachable (or is the configured provider).
|
||||
# Localhost connection-refused fails in milliseconds, so this stays cheap
|
||||
# when no server is running. Base URL precedence mirrors LM Studio:
|
||||
# active config's base_url (when current provider is ollama) > localhost.
|
||||
if "ollama" not in curated:
|
||||
from hermes_cli.models import fetch_ollama_local_models
|
||||
is_current_ollama = current_provider.strip().lower() == "ollama"
|
||||
ollama_base = current_base_url if is_current_ollama and current_base_url else None
|
||||
ollama_live = fetch_ollama_local_models(base_url=ollama_base, timeout=1.5)
|
||||
if not ollama_live and is_current_ollama and current_model:
|
||||
ollama_live = [current_model]
|
||||
if ollama_live or is_current_ollama:
|
||||
curated["ollama"] = ollama_live
|
||||
|
||||
# --- 1. Check Hermes-mapped providers ---
|
||||
from hermes_cli.models import _AGGREGATOR_PROVIDERS as _AGG_PROVIDERS
|
||||
from hermes_cli.providers import ALIASES as _PROVIDER_ALIAS_TABLE
|
||||
|
|
@ -1807,6 +1822,11 @@ def list_authenticated_providers(
|
|||
has_creds = True
|
||||
except Exception as exc:
|
||||
logger.debug("Anthropic external creds check failed: %s", exc)
|
||||
# Local Ollama server: unauthenticated by design, so reachability is
|
||||
# the credential. The curated probe above only adds the key when the
|
||||
# server responded (or it is the configured provider).
|
||||
if not has_creds and hermes_slug == "ollama":
|
||||
has_creds = "ollama" in curated
|
||||
if not has_creds:
|
||||
continue
|
||||
|
||||
|
|
|
|||
|
|
@ -1036,6 +1036,7 @@ CANONICAL_PROVIDERS: list[ProviderEntry] = [
|
|||
ProviderEntry("moa", "Mixture of Agents", "Mixture of Agents (named presets; aggregator acts after reference models)"),
|
||||
ProviderEntry("novita", "NovitaAI", "NovitaAI (Cloud: Model API, Agent Sandbox, GPU Cloud)"),
|
||||
ProviderEntry("lmstudio", "LM Studio", "LM Studio (Local desktop app with built-in model server)"),
|
||||
ProviderEntry("ollama", "Ollama (local)", "Ollama (Local model server on localhost:11434)"),
|
||||
ProviderEntry("anthropic", "Anthropic", "Anthropic (Claude models via API key or Claude Code)"),
|
||||
ProviderEntry("openai-codex", "OpenAI Codex", "OpenAI Codex (Codex CLI via ChatGPT subscription or API key)"),
|
||||
ProviderEntry("openai-api", "OpenAI API", "OpenAI API (api.openai.com, API key)"),
|
||||
|
|
@ -1273,7 +1274,7 @@ _PROVIDER_ALIASES = {
|
|||
"lmstudio": "lmstudio",
|
||||
"lm-studio": "lmstudio",
|
||||
"lm_studio": "lmstudio",
|
||||
"ollama": "custom", # bare "ollama" = local; use "ollama-cloud" for cloud
|
||||
"ollama": "ollama", # bare "ollama" = local server; use "ollama-cloud" for cloud
|
||||
"ollama_cloud": "ollama-cloud",
|
||||
}
|
||||
|
||||
|
|
@ -2367,6 +2368,17 @@ def provider_model_ids(provider: Optional[str], *, force_refresh: bool = False)
|
|||
live = fetch_ollama_cloud_models(force_refresh=force_refresh)
|
||||
if live:
|
||||
return live
|
||||
if normalized == "ollama":
|
||||
# Local Ollama server: list installed models from the native API.
|
||||
# Base URL precedence: active config's base_url (when the current
|
||||
# provider is ollama) > localhost default.
|
||||
model_cfg = _get_model_config_dict()
|
||||
cfg_provider = normalize_provider(str(model_cfg.get("provider", "") or ""))
|
||||
cfg_base_url = str(model_cfg.get("base_url", "") or "").strip() if cfg_provider == "ollama" else ""
|
||||
cfg_api_key = str(model_cfg.get("api_key", "") or "").strip() if cfg_provider == "ollama" else ""
|
||||
return fetch_ollama_local_models(
|
||||
api_key=cfg_api_key, base_url=cfg_base_url or None, timeout=3.0
|
||||
)
|
||||
if normalized in ("openai", "openai-api"):
|
||||
api_key = os.getenv("OPENAI_API_KEY", "").strip()
|
||||
if api_key:
|
||||
|
|
@ -2637,6 +2649,13 @@ def cached_provider_model_ids(
|
|||
if not normalized:
|
||||
return []
|
||||
|
||||
# Local Ollama: never disk-cache. The installed-model list changes with
|
||||
# every pull/delete/server restart, the live /api/tags fetch costs
|
||||
# milliseconds (fast TCP pre-check when down), and a stale cached list
|
||||
# surfaces models that no longer exist in the picker.
|
||||
if normalized == "ollama":
|
||||
return provider_model_ids(normalized, force_refresh=force_refresh)
|
||||
|
||||
cache = _load_provider_models_cache()
|
||||
fp = _credential_fingerprint(normalized)
|
||||
entry = cache.get(normalized)
|
||||
|
|
@ -3157,6 +3176,388 @@ def _fetch_github_models(api_key: Optional[str] = None, timeout: float = 5.0) ->
|
|||
return [item.get("id", "") for item in catalog if item.get("id")]
|
||||
|
||||
|
||||
# ── Local Ollama server (native /api metadata) ──────────────────────────────
|
||||
#
|
||||
# The OpenAI-compatible /v1/models endpoint returns bare model ids. Ollama's
|
||||
# native API is much richer: /api/tags lists installed models with size and
|
||||
# family, /api/show returns capabilities (tools/vision/thinking), parameter
|
||||
# size, quantization, and context length. The picker uses this to badge
|
||||
# models honestly (an agent needs tool support) — inference stays on /v1.
|
||||
|
||||
def _ollama_server_root(base_url: Optional[str]) -> Optional[str]:
|
||||
"""Server root for Ollama's native API (strip a trailing /v1 or /api).
|
||||
|
||||
``localhost`` is rewritten to ``127.0.0.1``: Ollama binds IPv4 loopback
|
||||
by default, and on Windows ``localhost`` resolves to ``::1`` first, so
|
||||
every request pays a ~2s failed IPv6 connect before falling back.
|
||||
"""
|
||||
root = str(base_url or "").strip().rstrip("/")
|
||||
if not root:
|
||||
return None
|
||||
for suffix in ("/api", "/v1"):
|
||||
if root.endswith(suffix):
|
||||
root = root[: -len(suffix)].rstrip("/")
|
||||
break
|
||||
if "://" not in root:
|
||||
root = "http://" + root
|
||||
return root.replace("://localhost", "://127.0.0.1")
|
||||
|
||||
|
||||
# One failed reachability probe covers every fetcher call for a while, so an
|
||||
# offline server costs the model picker a single sub-second check instead of
|
||||
# an HTTP timeout per call (multiplied per model by capability enrichment).
|
||||
_OLLAMA_UNREACHABLE_CACHE: dict = {}
|
||||
_OLLAMA_UNREACHABLE_TTL = 30.0
|
||||
|
||||
|
||||
def _ollama_server_reachable(
|
||||
server_root: str, timeout: float = 0.3, use_cache: bool = True
|
||||
) -> bool:
|
||||
"""Fast TCP pre-check before any native-API read.
|
||||
|
||||
The picker and settings pages call the fetchers below on every open, so
|
||||
an unreachable server must fail in milliseconds — a raw connect either
|
||||
succeeds or is refused near-instantly on localhost, and the short timeout
|
||||
caps the silently-dropped case. Failures are cached briefly so bulk
|
||||
payload builds don't repeat the probe per model.
|
||||
|
||||
``use_cache=False`` forces a fresh probe — for status endpoints that must
|
||||
notice a server coming up immediately (a cached failure would report a
|
||||
just-started server as down for the rest of the TTL). A successful fresh
|
||||
probe clears the negative entry for every cached caller too.
|
||||
"""
|
||||
import socket
|
||||
import time as _time
|
||||
from urllib.parse import urlparse
|
||||
|
||||
now = _time.monotonic()
|
||||
if use_cache:
|
||||
failed_at = _OLLAMA_UNREACHABLE_CACHE.get(server_root)
|
||||
if failed_at is not None and (now - failed_at) < _OLLAMA_UNREACHABLE_TTL:
|
||||
return False
|
||||
|
||||
try:
|
||||
parsed = urlparse(server_root)
|
||||
host = parsed.hostname or "localhost"
|
||||
port = parsed.port or (443 if parsed.scheme == "https" else 80)
|
||||
with socket.create_connection((host, port), timeout=timeout):
|
||||
pass
|
||||
_OLLAMA_UNREACHABLE_CACHE.pop(server_root, None)
|
||||
return True
|
||||
except Exception:
|
||||
_OLLAMA_UNREACHABLE_CACHE[server_root] = now
|
||||
return False
|
||||
|
||||
|
||||
def _ollama_request_headers(api_key: Optional[str] = None) -> dict:
|
||||
headers = {"Content-Type": "application/json"}
|
||||
key = str(api_key or "").strip()
|
||||
if key:
|
||||
headers["Authorization"] = f"Bearer {key}"
|
||||
return headers
|
||||
|
||||
|
||||
def fetch_ollama_local_models(
|
||||
api_key: Optional[str] = None,
|
||||
base_url: Optional[str] = None,
|
||||
timeout: float = 5.0,
|
||||
) -> list[str]:
|
||||
"""List installed models from a local Ollama server's native ``/api/tags``.
|
||||
|
||||
Returns model names (``model:tag``) or an empty list when the server is
|
||||
unreachable or the response is malformed.
|
||||
"""
|
||||
server_root = _ollama_server_root(base_url or "http://localhost:11434")
|
||||
if not server_root:
|
||||
return []
|
||||
if not _ollama_server_reachable(server_root):
|
||||
return []
|
||||
|
||||
request = urllib.request.Request(
|
||||
server_root + "/api/tags", headers=_ollama_request_headers(api_key)
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=timeout) as resp:
|
||||
payload = json.loads(resp.read().decode())
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
models = payload.get("models")
|
||||
if not isinstance(models, list):
|
||||
return []
|
||||
|
||||
names: list[str] = []
|
||||
for raw in models:
|
||||
if not isinstance(raw, dict):
|
||||
continue
|
||||
name = str(raw.get("name") or raw.get("model") or "").strip()
|
||||
if name and name not in names:
|
||||
names.append(name)
|
||||
return names
|
||||
|
||||
|
||||
def fetch_ollama_model_metadata(
|
||||
model: str,
|
||||
base_url: Optional[str] = None,
|
||||
api_key: Optional[str] = None,
|
||||
timeout: float = 5.0,
|
||||
) -> Optional[dict]:
|
||||
"""Fetch one model's metadata from a local Ollama server's ``/api/show``.
|
||||
|
||||
Returns a dict with (all fields best-effort, absent keys omitted):
|
||||
- ``capabilities``: list[str] — e.g. ["completion", "tools", "vision", "thinking"]
|
||||
- ``parameter_size``: str — e.g. "27B"
|
||||
- ``quantization``: str — e.g. "Q4_K_M"
|
||||
- ``context_length``: int — trained context from GGUF metadata
|
||||
- ``family``: str — model family, e.g. "qwen3"
|
||||
|
||||
Returns None when the server is unreachable or the model is unknown.
|
||||
"""
|
||||
server_root = _ollama_server_root(base_url or "http://localhost:11434")
|
||||
if not server_root or not str(model or "").strip():
|
||||
return None
|
||||
if not _ollama_server_reachable(server_root):
|
||||
return None
|
||||
|
||||
body = json.dumps({"name": str(model).strip()}).encode()
|
||||
request = urllib.request.Request(
|
||||
server_root + "/api/show",
|
||||
data=body,
|
||||
headers=_ollama_request_headers(api_key),
|
||||
method="POST",
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=timeout) as resp:
|
||||
payload = json.loads(resp.read().decode())
|
||||
except Exception:
|
||||
return None
|
||||
if not isinstance(payload, dict):
|
||||
return None
|
||||
|
||||
meta: dict = {}
|
||||
|
||||
caps = payload.get("capabilities")
|
||||
if isinstance(caps, list):
|
||||
meta["capabilities"] = [str(c).strip().lower() for c in caps if isinstance(c, str)]
|
||||
|
||||
details = payload.get("details")
|
||||
if isinstance(details, dict):
|
||||
if details.get("parameter_size"):
|
||||
meta["parameter_size"] = str(details["parameter_size"]).strip()
|
||||
if details.get("quantization_level"):
|
||||
meta["quantization"] = str(details["quantization_level"]).strip()
|
||||
if details.get("family"):
|
||||
meta["family"] = str(details["family"]).strip()
|
||||
|
||||
# Context length lives in model_info under "<architecture>.context_length".
|
||||
model_info = payload.get("model_info")
|
||||
if isinstance(model_info, dict):
|
||||
for key, value in model_info.items():
|
||||
if str(key).endswith(".context_length") and isinstance(value, (int, float)):
|
||||
meta["context_length"] = int(value)
|
||||
break
|
||||
|
||||
return meta or None
|
||||
|
||||
|
||||
# Curated models that work well as Hermes agent backends on a local Ollama
|
||||
# server. Every entry supports tool calling. ``approx_vram_gb`` is the rough
|
||||
# working-set need at the default quantization — used to annotate the
|
||||
# recommendation (never to gate it: Ollama falls back to partial offload).
|
||||
OLLAMA_RECOMMENDED_MODELS: list[dict] = [
|
||||
{"model": "qwen3:8b", "description": "Qwen3 8B — tools + thinking, good starter", "approx_vram_gb": 8},
|
||||
{"model": "gpt-oss:20b", "description": "OpenAI open-weight 20B — tools + thinking", "approx_vram_gb": 16},
|
||||
{"model": "qwen3:32b", "description": "Qwen3 32B — strong agent at 24GB", "approx_vram_gb": 24},
|
||||
{"model": "mistral-small3.2:24b", "description": "Mistral Small 3.2 — tools + vision", "approx_vram_gb": 20},
|
||||
{"model": "llama3.3:70b", "description": "Llama 3.3 70B — tools, needs 48GB+", "approx_vram_gb": 48},
|
||||
{"model": "gpt-oss:120b", "description": "OpenAI open-weight 120B — tools + thinking, needs 80GB", "approx_vram_gb": 80},
|
||||
]
|
||||
|
||||
|
||||
def fetch_ollama_installed_model_details(
|
||||
api_key: Optional[str] = None,
|
||||
base_url: Optional[str] = None,
|
||||
timeout: float = 5.0,
|
||||
) -> list[dict]:
|
||||
"""Installed models with size/details from native ``/api/tags``.
|
||||
|
||||
Returns ``[{name, size_bytes, parameter_size, quantization, family,
|
||||
modified_at}]`` (absent fields omitted), or ``[]`` when unreachable.
|
||||
"""
|
||||
server_root = _ollama_server_root(base_url or "http://localhost:11434")
|
||||
if not server_root:
|
||||
return []
|
||||
if not _ollama_server_reachable(server_root):
|
||||
return []
|
||||
|
||||
request = urllib.request.Request(
|
||||
server_root + "/api/tags", headers=_ollama_request_headers(api_key)
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=timeout) as resp:
|
||||
payload = json.loads(resp.read().decode())
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
models = payload.get("models")
|
||||
if not isinstance(models, list):
|
||||
return []
|
||||
|
||||
out: list[dict] = []
|
||||
for raw in models:
|
||||
if not isinstance(raw, dict):
|
||||
continue
|
||||
name = str(raw.get("name") or raw.get("model") or "").strip()
|
||||
if not name:
|
||||
continue
|
||||
entry: dict = {"name": name}
|
||||
if isinstance(raw.get("size"), (int, float)):
|
||||
entry["size_bytes"] = int(raw["size"])
|
||||
if raw.get("modified_at"):
|
||||
entry["modified_at"] = str(raw["modified_at"])
|
||||
details = raw.get("details")
|
||||
if isinstance(details, dict):
|
||||
if details.get("parameter_size"):
|
||||
entry["parameter_size"] = str(details["parameter_size"]).strip()
|
||||
if details.get("quantization_level"):
|
||||
entry["quantization"] = str(details["quantization_level"]).strip()
|
||||
if details.get("family"):
|
||||
entry["family"] = str(details["family"]).strip()
|
||||
out.append(entry)
|
||||
return out
|
||||
|
||||
|
||||
def fetch_ollama_running_models(
|
||||
api_key: Optional[str] = None,
|
||||
base_url: Optional[str] = None,
|
||||
timeout: float = 5.0,
|
||||
) -> list[dict]:
|
||||
"""Loaded models from native ``/api/ps``.
|
||||
|
||||
Returns ``[{name, size_bytes, size_vram_bytes, context_length,
|
||||
expires_at}]`` (absent fields omitted), or ``[]`` when unreachable or
|
||||
nothing is loaded. ``context_length`` is the window the model was
|
||||
actually loaded with — the input for the KV-cache advisory.
|
||||
"""
|
||||
server_root = _ollama_server_root(base_url or "http://localhost:11434")
|
||||
if not server_root:
|
||||
return []
|
||||
if not _ollama_server_reachable(server_root):
|
||||
return []
|
||||
|
||||
request = urllib.request.Request(
|
||||
server_root + "/api/ps", headers=_ollama_request_headers(api_key)
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=timeout) as resp:
|
||||
payload = json.loads(resp.read().decode())
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
models = payload.get("models")
|
||||
if not isinstance(models, list):
|
||||
return []
|
||||
|
||||
out: list[dict] = []
|
||||
for raw in models:
|
||||
if not isinstance(raw, dict):
|
||||
continue
|
||||
name = str(raw.get("name") or raw.get("model") or "").strip()
|
||||
if not name:
|
||||
continue
|
||||
entry: dict = {"name": name}
|
||||
if isinstance(raw.get("size"), (int, float)):
|
||||
entry["size_bytes"] = int(raw["size"])
|
||||
if isinstance(raw.get("size_vram"), (int, float)):
|
||||
entry["size_vram_bytes"] = int(raw["size_vram"])
|
||||
if isinstance(raw.get("context_length"), (int, float)):
|
||||
entry["context_length"] = int(raw["context_length"])
|
||||
if raw.get("expires_at"):
|
||||
entry["expires_at"] = str(raw["expires_at"])
|
||||
out.append(entry)
|
||||
return out
|
||||
|
||||
|
||||
def delete_ollama_model(
|
||||
model: str,
|
||||
api_key: Optional[str] = None,
|
||||
base_url: Optional[str] = None,
|
||||
timeout: float = 15.0,
|
||||
) -> tuple[bool, str]:
|
||||
"""Delete an installed model via native ``DELETE /api/delete``.
|
||||
|
||||
Returns ``(ok, message)`` — message is empty on success.
|
||||
"""
|
||||
server_root = _ollama_server_root(base_url or "http://localhost:11434")
|
||||
name = str(model or "").strip()
|
||||
if not server_root or not name:
|
||||
return False, "missing model or server"
|
||||
|
||||
body = json.dumps({"model": name}).encode()
|
||||
request = urllib.request.Request(
|
||||
server_root + "/api/delete",
|
||||
data=body,
|
||||
headers=_ollama_request_headers(api_key),
|
||||
method="DELETE",
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=timeout) as resp:
|
||||
resp.read()
|
||||
return True, ""
|
||||
except urllib.error.HTTPError as exc:
|
||||
if exc.code == 404:
|
||||
return False, f"model {name!r} not found"
|
||||
return False, f"delete failed with HTTP {exc.code}"
|
||||
except Exception:
|
||||
return False, "could not reach the Ollama server"
|
||||
|
||||
|
||||
def preload_ollama_model(
|
||||
model: str,
|
||||
keep_alive: Optional[str] = None,
|
||||
api_key: Optional[str] = None,
|
||||
base_url: Optional[str] = None,
|
||||
timeout: float = 300.0,
|
||||
) -> tuple[bool, str]:
|
||||
"""Load a model into memory via a prompt-less native ``/api/generate``.
|
||||
|
||||
Ollama loads the model and returns without generating. ``keep_alive``
|
||||
(e.g. ``"5m"``, ``"2h"``, ``"-1"`` for indefinite) pins how long it stays
|
||||
resident; Ollama applies keep_alive per-load, so re-issue this call to
|
||||
change it. The generous timeout covers cold weight loads on big models.
|
||||
|
||||
Returns ``(ok, message)`` — message is empty on success.
|
||||
"""
|
||||
server_root = _ollama_server_root(base_url or "http://localhost:11434")
|
||||
name = str(model or "").strip()
|
||||
if not server_root or not name:
|
||||
return False, "missing model or server"
|
||||
|
||||
req_body: dict = {"model": name}
|
||||
ka = str(keep_alive or "").strip()
|
||||
if ka:
|
||||
# Ollama accepts a Go duration string or a number of seconds;
|
||||
# "-1" means keep loaded indefinitely.
|
||||
req_body["keep_alive"] = int(ka) if ka.lstrip("-").isdigit() else ka
|
||||
|
||||
request = urllib.request.Request(
|
||||
server_root + "/api/generate",
|
||||
data=json.dumps(req_body).encode(),
|
||||
headers=_ollama_request_headers(api_key),
|
||||
method="POST",
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=timeout) as resp:
|
||||
resp.read()
|
||||
return True, ""
|
||||
except urllib.error.HTTPError as exc:
|
||||
if exc.code == 404:
|
||||
return False, f"model {name!r} not found — pull it first"
|
||||
return False, f"load failed with HTTP {exc.code}"
|
||||
except Exception:
|
||||
return False, "could not reach the Ollama server"
|
||||
|
||||
|
||||
_COPILOT_MODEL_ALIASES = {
|
||||
"openai/gpt-5": "gpt-5-mini",
|
||||
"openai/gpt-5-chat": "gpt-5-mini",
|
||||
|
|
|
|||
|
|
@ -201,6 +201,14 @@ HERMES_OVERLAYS: Dict[str, HermesOverlay] = {
|
|||
base_url_override="https://ollama.com/v1",
|
||||
base_url_env_var="OLLAMA_BASE_URL",
|
||||
),
|
||||
# Local Ollama server. Key optional (most local servers are unauthenticated);
|
||||
# model.base_url in config overrides the default for remote boxes.
|
||||
# 127.0.0.1, not localhost: Windows resolves localhost to ::1 first and
|
||||
# Ollama binds IPv4 loopback — dodge the ~2s IPv6 connect penalty.
|
||||
"ollama": HermesOverlay(
|
||||
transport="openai_chat",
|
||||
base_url_override="http://127.0.0.1:11434/v1",
|
||||
),
|
||||
# Azure Foundry: supports both OpenAI-style and Anthropic-style endpoints.
|
||||
# The transport is determined at runtime from config.yaml model.api_mode.
|
||||
"azure-foundry": HermesOverlay(
|
||||
|
|
@ -347,7 +355,7 @@ ALIASES: Dict[str, str] = {
|
|||
"lmstudio": "lmstudio",
|
||||
"lm-studio": "lmstudio",
|
||||
"lm_studio": "lmstudio",
|
||||
"ollama": "custom", # bare "ollama" = local; use "ollama-cloud" for cloud
|
||||
"ollama": "ollama", # bare "ollama" = local server; use "ollama-cloud" for cloud
|
||||
"vllm": "local",
|
||||
"llamacpp": "local",
|
||||
"llama.cpp": "local",
|
||||
|
|
@ -369,6 +377,7 @@ _LABEL_OVERRIDES: Dict[str, str] = {
|
|||
"gmi": "GMI Cloud",
|
||||
"tencent-tokenhub": "Tencent TokenHub",
|
||||
"lmstudio": "LM Studio",
|
||||
"ollama": "Ollama (local)",
|
||||
"local": "Local endpoint",
|
||||
"bedrock": "AWS Bedrock",
|
||||
"ollama-cloud": "Ollama Cloud",
|
||||
|
|
|
|||
|
|
@ -2034,6 +2034,8 @@ def resolve_runtime_provider(
|
|||
base_url = normalize_opencode_base_url(provider, api_mode, base_url)
|
||||
if provider == "lmstudio":
|
||||
base_url = auth_mod._normalize_lmstudio_runtime_base_url(base_url)
|
||||
if provider == "ollama":
|
||||
base_url = auth_mod._normalize_ollama_runtime_base_url(base_url)
|
||||
return {
|
||||
"provider": provider,
|
||||
"api_mode": api_mode,
|
||||
|
|
|
|||
|
|
@ -5345,6 +5345,434 @@ def get_recommended_default_model(provider: str = ""):
|
|||
return {"provider": slug, "model": "", "free_tier": None}
|
||||
|
||||
|
||||
# Well-known local model-server ports probed by /api/local-servers/detect.
|
||||
# Each entry: (server type reported by detect_local_server_type, default root).
|
||||
_LOCAL_SERVER_PROBE_ROOTS: list[tuple[str, str]] = [
|
||||
("ollama", "http://127.0.0.1:11434"),
|
||||
("lm-studio", "http://127.0.0.1:1234"),
|
||||
]
|
||||
|
||||
# Session-lived cache: detection results change rarely (a server starting or
|
||||
# stopping), and onboarding + settings may both request them in quick
|
||||
# succession. refresh=1 busts it, mirroring /api/model/options.
|
||||
_local_servers_cache: dict = {"at": 0.0, "servers": None}
|
||||
_LOCAL_SERVERS_CACHE_TTL = 60.0
|
||||
|
||||
|
||||
def _detect_local_servers_sync(profile: Optional[str]) -> list[dict]:
|
||||
"""Probe well-known local ports (plus the configured base_url) for
|
||||
running model servers.
|
||||
|
||||
Returns one entry per candidate root:
|
||||
{type, base_url, reachable, models: [...], source}
|
||||
``base_url`` is the OpenAI-compatible endpoint (``<root>/v1``).
|
||||
``type`` is ``detect_local_server_type()``'s fingerprint when reachable,
|
||||
else the expected type for the well-known port. The response shape keeps
|
||||
room for a future installed/running/managed lifecycle distinction.
|
||||
"""
|
||||
from agent.model_metadata import detect_local_server_type, is_local_endpoint
|
||||
from hermes_cli.models import (
|
||||
_ollama_server_reachable,
|
||||
fetch_ollama_local_models,
|
||||
probe_lmstudio_models,
|
||||
)
|
||||
|
||||
candidates: list[tuple[str, str, str]] = [
|
||||
(expected, root, "well-known") for expected, root in _LOCAL_SERVER_PROBE_ROOTS
|
||||
]
|
||||
|
||||
# Also probe the configured base_url when it is local and isn't one of
|
||||
# the defaults — the remote-Ollama-over-Tailscale case. Remote provider
|
||||
# URLs (api.anthropic.com, ...) are never local servers; skip them.
|
||||
try:
|
||||
with _profile_scope(profile):
|
||||
cfg = load_config()
|
||||
model_cfg = cfg.get("model") or {}
|
||||
cfg_url = str(model_cfg.get("base_url", "") or "").strip().rstrip("/")
|
||||
if cfg_url and is_local_endpoint(cfg_url):
|
||||
cfg_root = cfg_url[:-3].rstrip("/") if cfg_url.endswith("/v1") else cfg_url
|
||||
known_roots = {root for _, root in _LOCAL_SERVER_PROBE_ROOTS}
|
||||
if cfg_root and cfg_root not in known_roots:
|
||||
candidates.append(("", cfg_root, "configured"))
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
servers: list[dict] = []
|
||||
for expected_type, root, source in candidates:
|
||||
detected = None
|
||||
# Fast TCP pre-check before the HTTP fingerprint: a down server must
|
||||
# cost milliseconds, not detect_local_server_type's sequential HTTP
|
||||
# timeouts (which stack across candidates and blow the client's
|
||||
# request timeout).
|
||||
if _ollama_server_reachable(root):
|
||||
try:
|
||||
detected = detect_local_server_type(root)
|
||||
except Exception:
|
||||
detected = None
|
||||
|
||||
reachable = detected is not None
|
||||
models: list[str] = []
|
||||
if reachable:
|
||||
try:
|
||||
if detected == "ollama":
|
||||
models = fetch_ollama_local_models(base_url=root, timeout=2.0)
|
||||
elif detected == "lm-studio":
|
||||
models = probe_lmstudio_models(base_url=root, timeout=2.0) or []
|
||||
except Exception:
|
||||
models = []
|
||||
|
||||
servers.append(
|
||||
{
|
||||
"type": detected or expected_type or "unknown",
|
||||
"base_url": root + "/v1",
|
||||
"reachable": reachable,
|
||||
"models": models,
|
||||
"source": source,
|
||||
}
|
||||
)
|
||||
return servers
|
||||
|
||||
|
||||
@app.get("/api/local-servers/detect")
|
||||
async def detect_local_servers(profile: Optional[str] = None, refresh: bool = False):
|
||||
"""Detect running local model servers (Ollama, LM Studio).
|
||||
|
||||
Probes well-known localhost ports plus the configured ``model.base_url``,
|
||||
fingerprinting each with ``detect_local_server_type()`` and listing the
|
||||
installed models for recognized servers. Onboarding uses this to offer a
|
||||
detected server as a provider choice instead of asking for a URL.
|
||||
"""
|
||||
import time as _time
|
||||
|
||||
now = _time.time()
|
||||
if (
|
||||
not refresh
|
||||
and _local_servers_cache["servers"] is not None
|
||||
and (now - _local_servers_cache["at"]) < _LOCAL_SERVERS_CACHE_TTL
|
||||
):
|
||||
return {"servers": _local_servers_cache["servers"]}
|
||||
|
||||
try:
|
||||
servers = await asyncio.to_thread(_detect_local_servers_sync, profile)
|
||||
except Exception:
|
||||
_log.exception("GET /api/local-servers/detect failed")
|
||||
raise HTTPException(status_code=500, detail="Local server detection failed")
|
||||
|
||||
_local_servers_cache["servers"] = servers
|
||||
_local_servers_cache["at"] = now
|
||||
return {"servers": servers}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Local Ollama model management
|
||||
# ---------------------------------------------------------------------------
|
||||
#
|
||||
# Management actions (list/pull/delete/load) go through Ollama's native /api
|
||||
# surface; chat inference stays on the OpenAI-compatible /v1 endpoint. Pull
|
||||
# runs as a background job the UI polls — the same worker+poll shape as the
|
||||
# OAuth device-code sessions above.
|
||||
|
||||
_ollama_pull_jobs: dict = {}
|
||||
_ollama_pull_jobs_lock = threading.Lock()
|
||||
_OLLAMA_PULL_JOB_TTL = 3600.0
|
||||
|
||||
|
||||
class OllamaModelAction(BaseModel):
|
||||
model: str = ""
|
||||
keep_alive: Optional[str] = None
|
||||
profile: Optional[str] = None
|
||||
|
||||
|
||||
def _on_ollama_models_changed() -> None:
|
||||
"""Bust the cached ollama model-id list after a pull/delete.
|
||||
|
||||
The picker payload reads ``cached_provider_model_ids("ollama")`` (1h disk
|
||||
TTL); without this a just-pulled model doesn't appear in the model picker
|
||||
until the cache expires.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.models import clear_provider_models_cache
|
||||
|
||||
clear_provider_models_cache("ollama")
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def _ollama_connection(profile: Optional[str]) -> tuple[str, str]:
|
||||
"""Resolve (base_url, api_key) for the local Ollama server.
|
||||
|
||||
Uses model.base_url/api_key from config when the configured provider is
|
||||
ollama; otherwise the localhost default. Returns the server root form
|
||||
accepted by the hermes_cli.models helpers (they normalize /v1 away).
|
||||
"""
|
||||
base_url = ""
|
||||
api_key = ""
|
||||
try:
|
||||
with _profile_scope(profile):
|
||||
cfg = load_config()
|
||||
model_cfg = cfg.get("model") or {}
|
||||
provider = str(model_cfg.get("provider", "") or "").strip().lower()
|
||||
if provider == "ollama":
|
||||
base_url = str(model_cfg.get("base_url", "") or "").strip()
|
||||
api_key = str(model_cfg.get("api_key", "") or "").strip()
|
||||
except Exception:
|
||||
pass
|
||||
return base_url or "http://127.0.0.1:11434", api_key
|
||||
|
||||
|
||||
@app.get("/api/ollama/models")
|
||||
async def get_ollama_models(profile: Optional[str] = None):
|
||||
"""Installed + running models and the curated recommendations.
|
||||
|
||||
Single round-trip payload for the desktop management panel:
|
||||
``{reachable, installed: [...], running: [...], recommended: [...]}``.
|
||||
``recommended`` excludes already-installed models.
|
||||
"""
|
||||
from hermes_cli.models import (
|
||||
OLLAMA_RECOMMENDED_MODELS,
|
||||
fetch_ollama_installed_model_details,
|
||||
fetch_ollama_running_models,
|
||||
)
|
||||
|
||||
base_url, api_key = _ollama_connection(profile)
|
||||
|
||||
# Fresh reachability probe first (bypasses the negative cache): this is
|
||||
# the status endpoint the desktop polls, so it must notice a just-started
|
||||
# server immediately. Success clears the cached failure, which un-blinds
|
||||
# the fetchers below in the same request.
|
||||
from hermes_cli.models import _ollama_server_reachable, _ollama_server_root
|
||||
|
||||
root = _ollama_server_root(base_url)
|
||||
reachable = bool(root) and await asyncio.to_thread(
|
||||
_ollama_server_reachable, root, 0.3, False
|
||||
)
|
||||
|
||||
installed: list = []
|
||||
running: list = []
|
||||
if reachable:
|
||||
def _collect():
|
||||
return (
|
||||
fetch_ollama_installed_model_details(
|
||||
api_key=api_key, base_url=base_url, timeout=3.0
|
||||
),
|
||||
fetch_ollama_running_models(
|
||||
api_key=api_key, base_url=base_url, timeout=3.0
|
||||
),
|
||||
)
|
||||
|
||||
try:
|
||||
installed, running = await asyncio.to_thread(_collect)
|
||||
except Exception:
|
||||
_log.exception("GET /api/ollama/models failed")
|
||||
raise HTTPException(
|
||||
status_code=500, detail="Failed to query the Ollama server"
|
||||
)
|
||||
|
||||
installed_names = {m["name"] for m in installed}
|
||||
# A recommendation is "installed" if any tag of the same base model is —
|
||||
# qwen3:8b covers qwen3:8b-q4_K_M etc.
|
||||
installed_bases = {n.split(":", 1)[0] for n in installed_names}
|
||||
recommended = [
|
||||
r
|
||||
for r in OLLAMA_RECOMMENDED_MODELS
|
||||
if r["model"] not in installed_names
|
||||
and r["model"].split(":", 1)[0] not in installed_bases
|
||||
]
|
||||
|
||||
# KV-cache advisory (see docs/plans/2026-07-08-001): Ollama runs the KV
|
||||
# cache at f16 unless the operator sets OLLAMA_KV_CACHE_TYPE, and never
|
||||
# quantizes it to fit a larger window. When a loaded model runs at well
|
||||
# under half its trained context, q8_0 KV would roughly double the usable
|
||||
# window in the same memory — surface that once per panel load. Inference:
|
||||
# we cannot read a remote server's env, only compare loaded vs trained.
|
||||
kv_cache_advisory = None
|
||||
try:
|
||||
from hermes_cli.models import fetch_ollama_model_metadata
|
||||
|
||||
for run in running:
|
||||
loaded_ctx = int(run.get("context_length") or 0)
|
||||
if loaded_ctx <= 0:
|
||||
continue
|
||||
meta = await asyncio.to_thread(
|
||||
fetch_ollama_model_metadata, run["name"], base_url, api_key, 2.0
|
||||
)
|
||||
trained_ctx = int((meta or {}).get("context_length") or 0)
|
||||
if trained_ctx > 0 and loaded_ctx * 2 <= trained_ctx:
|
||||
kv_cache_advisory = {
|
||||
"model": run["name"],
|
||||
"loaded_context": loaded_ctx,
|
||||
"trained_context": trained_ctx,
|
||||
}
|
||||
break
|
||||
except Exception:
|
||||
kv_cache_advisory = None
|
||||
|
||||
return {
|
||||
"reachable": reachable,
|
||||
"installed": installed,
|
||||
"running": running,
|
||||
"recommended": recommended,
|
||||
"kv_cache_advisory": kv_cache_advisory,
|
||||
}
|
||||
|
||||
|
||||
def _run_ollama_pull_job(job_id: str, model: str, base_url: str, api_key: str) -> None:
|
||||
"""Worker: stream native /api/pull NDJSON progress into the job record."""
|
||||
import urllib.request as _urlreq
|
||||
|
||||
root = base_url.rstrip("/")
|
||||
for suffix in ("/api", "/v1"):
|
||||
if root.endswith(suffix):
|
||||
root = root[: -len(suffix)].rstrip("/")
|
||||
break
|
||||
headers = {"Content-Type": "application/json"}
|
||||
if api_key:
|
||||
headers["Authorization"] = f"Bearer {api_key}"
|
||||
|
||||
def _update(**fields):
|
||||
with _ollama_pull_jobs_lock:
|
||||
job = _ollama_pull_jobs.get(job_id)
|
||||
if job is not None:
|
||||
job.update(fields)
|
||||
|
||||
request = _urlreq.Request(
|
||||
root + "/api/pull",
|
||||
data=json.dumps({"model": model, "stream": True}).encode(),
|
||||
headers=headers,
|
||||
method="POST",
|
||||
)
|
||||
try:
|
||||
# No read timeout beyond connect: a pull legitimately runs for many
|
||||
# minutes; progress lines keep the connection demonstrably alive.
|
||||
with _urlreq.urlopen(request, timeout=30.0) as resp:
|
||||
for raw_line in resp:
|
||||
line = raw_line.decode("utf-8", errors="replace").strip()
|
||||
if not line:
|
||||
continue
|
||||
try:
|
||||
event = json.loads(line)
|
||||
except Exception:
|
||||
continue
|
||||
if event.get("error"):
|
||||
_update(status="error", error_message=str(event["error"]))
|
||||
return
|
||||
status = str(event.get("status", "") or "")
|
||||
fields: dict = {"detail": status}
|
||||
if isinstance(event.get("total"), (int, float)) and event["total"]:
|
||||
fields["total_bytes"] = int(event["total"])
|
||||
fields["completed_bytes"] = int(event.get("completed", 0) or 0)
|
||||
_update(**fields)
|
||||
if status == "success":
|
||||
_on_ollama_models_changed()
|
||||
_update(status="done", detail="success")
|
||||
return
|
||||
# Stream ended without an explicit success line — treat as done only
|
||||
# if the last event said so; otherwise report the ambiguity.
|
||||
with _ollama_pull_jobs_lock:
|
||||
job = _ollama_pull_jobs.get(job_id)
|
||||
if job is not None and job.get("status") == "pulling":
|
||||
job["status"] = "error"
|
||||
job["error_message"] = "pull stream ended unexpectedly"
|
||||
except Exception as exc:
|
||||
_update(status="error", error_message=f"pull failed: {exc}")
|
||||
|
||||
|
||||
@app.post("/api/ollama/pull")
|
||||
async def start_ollama_pull(body: OllamaModelAction, request: Request):
|
||||
"""Start pulling a model from the Ollama registry. Token-protected.
|
||||
|
||||
Returns ``{job_id}``; poll ``GET /api/ollama/pull/{job_id}`` for progress.
|
||||
"""
|
||||
_require_token(request)
|
||||
model = (body.model or "").strip()
|
||||
if not model:
|
||||
raise HTTPException(status_code=400, detail="model is required")
|
||||
|
||||
base_url, api_key = _ollama_connection(body.profile)
|
||||
|
||||
now = time.time()
|
||||
job_id = secrets.token_hex(8)
|
||||
with _ollama_pull_jobs_lock:
|
||||
# Reuse an active job for the same model instead of double-pulling.
|
||||
for jid, job in _ollama_pull_jobs.items():
|
||||
if job.get("model") == model and job.get("status") == "pulling":
|
||||
return {"job_id": jid}
|
||||
# Expire finished records so the dict can't grow unboundedly.
|
||||
for jid in [
|
||||
j
|
||||
for j, job in _ollama_pull_jobs.items()
|
||||
if (now - job.get("at", 0)) > _OLLAMA_PULL_JOB_TTL
|
||||
]:
|
||||
_ollama_pull_jobs.pop(jid, None)
|
||||
_ollama_pull_jobs[job_id] = {
|
||||
"model": model,
|
||||
"status": "pulling",
|
||||
"detail": "starting",
|
||||
"at": now,
|
||||
}
|
||||
|
||||
threading.Thread(
|
||||
target=_run_ollama_pull_job,
|
||||
args=(job_id, model, base_url, api_key),
|
||||
daemon=True,
|
||||
name=f"ollama-pull-{model}",
|
||||
).start()
|
||||
return {"job_id": job_id}
|
||||
|
||||
|
||||
@app.get("/api/ollama/pull/{job_id}")
|
||||
async def poll_ollama_pull(job_id: str):
|
||||
"""Poll a pull job: ``{model, status, detail, total_bytes?, completed_bytes?}``.
|
||||
|
||||
``status``: pulling | done | error.
|
||||
"""
|
||||
with _ollama_pull_jobs_lock:
|
||||
job = _ollama_pull_jobs.get(job_id)
|
||||
if job is None:
|
||||
raise HTTPException(status_code=404, detail="pull job not found or expired")
|
||||
return {"job_id": job_id, **{k: v for k, v in job.items() if k != "at"}}
|
||||
|
||||
|
||||
@app.post("/api/ollama/delete")
|
||||
async def delete_ollama_model_endpoint(body: OllamaModelAction, request: Request):
|
||||
"""Delete an installed model. Token-protected."""
|
||||
_require_token(request)
|
||||
model = (body.model or "").strip()
|
||||
if not model:
|
||||
raise HTTPException(status_code=400, detail="model is required")
|
||||
|
||||
from hermes_cli.models import delete_ollama_model
|
||||
|
||||
base_url, api_key = _ollama_connection(body.profile)
|
||||
ok, message = await asyncio.to_thread(
|
||||
delete_ollama_model, model, api_key, base_url
|
||||
)
|
||||
if ok:
|
||||
_on_ollama_models_changed()
|
||||
return {"ok": ok, "message": message}
|
||||
|
||||
|
||||
@app.post("/api/ollama/load")
|
||||
async def load_ollama_model_endpoint(body: OllamaModelAction, request: Request):
|
||||
"""Load a model into memory (optionally pinning keep_alive). Token-protected.
|
||||
|
||||
Used both for explicit warm-up from the management panel and to apply a
|
||||
keep_alive change (Ollama applies keep_alive per-load).
|
||||
"""
|
||||
_require_token(request)
|
||||
model = (body.model or "").strip()
|
||||
if not model:
|
||||
raise HTTPException(status_code=400, detail="model is required")
|
||||
|
||||
from hermes_cli.models import preload_ollama_model
|
||||
|
||||
base_url, api_key = _ollama_connection(body.profile)
|
||||
ok, message = await asyncio.to_thread(
|
||||
preload_ollama_model, model, body.keep_alive, api_key, base_url
|
||||
)
|
||||
return {"ok": ok, "message": message}
|
||||
|
||||
|
||||
@app.get("/api/model/auxiliary")
|
||||
def get_auxiliary_models(profile: Optional[str] = None):
|
||||
"""Return current auxiliary task assignments.
|
||||
|
|
|
|||
|
|
@ -1,9 +1,10 @@
|
|||
"""Custom / Ollama (local) provider profile.
|
||||
|
||||
Covers any endpoint registered as provider="custom", including local
|
||||
Ollama instances and OpenAI-compatible reasoning endpoints (GLM-5.2 on
|
||||
Volcengine ARK, vLLM, llama.cpp). Key quirks:
|
||||
Covers any endpoint registered as provider="custom", plus the first-class
|
||||
"ollama" provider (routed here by alias), and OpenAI-compatible reasoning
|
||||
endpoints (GLM-5.2 on Volcengine ARK, vLLM, llama.cpp). Key quirks:
|
||||
- ollama_num_ctx → extra_body.options.num_ctx (local context window)
|
||||
- ollama_keep_alive → extra_body.keep_alive (model residence time)
|
||||
- reasoning_config disabled → extra_body.think = False
|
||||
- reasoning_config enabled + effort → top-level reasoning_effort
|
||||
(the native OpenAI-compatible format GLM/ARK expect; unset omits it
|
||||
|
|
@ -24,6 +25,8 @@ class CustomProfile(ProviderProfile):
|
|||
*,
|
||||
reasoning_config: dict | None = None,
|
||||
ollama_num_ctx: int | None = None,
|
||||
ollama_keep_alive: int | str | None = None,
|
||||
ollama_supports_thinking: bool | None = None,
|
||||
**ctx: Any,
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
extra_body: dict[str, Any] = {}
|
||||
|
|
@ -35,6 +38,12 @@ class CustomProfile(ProviderProfile):
|
|||
options["num_ctx"] = ollama_num_ctx
|
||||
extra_body["options"] = options
|
||||
|
||||
# Ollama model residence time after the request (default 5m). Go
|
||||
# duration string or seconds; -1 = keep loaded until server exit.
|
||||
# Ignored by non-Ollama OpenAI-compatible servers.
|
||||
if ollama_keep_alive is not None:
|
||||
extra_body["keep_alive"] = ollama_keep_alive
|
||||
|
||||
# Reasoning / thinking control for custom OpenAI-compatible endpoints
|
||||
# (GLM-5.2 on Volcengine ARK, vLLM, Ollama, llama.cpp, …).
|
||||
#
|
||||
|
|
@ -45,6 +54,10 @@ class CustomProfile(ProviderProfile):
|
|||
# - enabled + no effort → omit both, so the endpoint applies its own
|
||||
# server-side default (do NOT force a level the user didn't pick).
|
||||
#
|
||||
# Effort levels that only exist on the OpenAI/Codex scale are mapped
|
||||
# to the nearest level these endpoints accept — Ollama validates
|
||||
# against high/medium/low/max/none and 400s on "xhigh"/"minimal".
|
||||
#
|
||||
# We deliberately do NOT emit ``think=True`` on enable: it is an
|
||||
# Ollama-only flag and thinking is already server-default-on for these
|
||||
# backends, so forcing it risks a 400 on GLM/vLLM endpoints that don't
|
||||
|
|
@ -52,10 +65,19 @@ class CustomProfile(ProviderProfile):
|
|||
if reasoning_config and isinstance(reasoning_config, dict):
|
||||
_effort = (reasoning_config.get("effort") or "").strip().lower()
|
||||
_enabled = reasoning_config.get("enabled", True)
|
||||
if _effort == "none" or _enabled is False:
|
||||
if ollama_supports_thinking is False:
|
||||
# The Ollama server reports this model cannot think: any
|
||||
# reasoning_effort 400s ('"hermes3:8b" does not support
|
||||
# thinking') and a think flag is at best a no-op. Emit
|
||||
# nothing — a session's effort dial carried over from a
|
||||
# thinking model must not brick the chat. None means
|
||||
# unknown/non-Ollama and changes nothing.
|
||||
pass
|
||||
elif _effort == "none" or _enabled is False:
|
||||
extra_body["think"] = False
|
||||
elif _effort:
|
||||
top_level["reasoning_effort"] = _effort
|
||||
_aliases = {"xhigh": "max", "minimal": "low"}
|
||||
top_level["reasoning_effort"] = _aliases.get(_effort, _effort)
|
||||
|
||||
return extra_body, top_level
|
||||
|
||||
|
|
|
|||
40
run_agent.py
40
run_agent.py
|
|
@ -5326,6 +5326,15 @@ class AIAgent:
|
|||
opts = self._lmstudio_reasoning_options_cached()
|
||||
# "off-only" (or absent) means no real reasoning capability.
|
||||
return any(opt and opt != "off" for opt in opts)
|
||||
if (self.provider or "").strip().lower() == "ollama":
|
||||
# Ollama rejects reasoning_effort outright on models without the
|
||||
# thinking capability ('"hermes3:8b" does not support thinking'),
|
||||
# so gate on the server's own /api/show capabilities. Unknown
|
||||
# (old server, probe failure) errs on sending — a thinking model
|
||||
# silently losing its effort dial is the worse failure, and the
|
||||
# capabilities field has shipped since 0.6.0.
|
||||
supported = self._ollama_supports_thinking_cached()
|
||||
return supported is not False
|
||||
if "openrouter" not in self._base_url_lower:
|
||||
return False
|
||||
if "api.mistral.ai" in self._base_url_lower:
|
||||
|
|
@ -5379,6 +5388,37 @@ class AIAgent:
|
|||
cache[key] = (opts, _time.monotonic())
|
||||
return opts
|
||||
|
||||
def _ollama_supports_thinking_cached(self) -> Optional[bool]:
|
||||
"""Probe Ollama's /api/show thinking capability once per (model, base_url).
|
||||
|
||||
True/False are cached permanently (a model's capabilities don't
|
||||
change); None (probe failure / old server without the capabilities
|
||||
field) is cached with a 60-second TTL so a transient failure retries
|
||||
soon without an HTTP round-trip on every turn.
|
||||
"""
|
||||
import time as _time
|
||||
|
||||
cache = getattr(self, "_ollama_thinking_cache", None)
|
||||
if cache is None:
|
||||
cache = self._ollama_thinking_cache = {}
|
||||
key = (self.model, self.base_url)
|
||||
cached = cache.get(key)
|
||||
if cached is not None:
|
||||
supported, ts = cached
|
||||
if supported is not None or (_time.monotonic() - ts) < 60:
|
||||
return supported
|
||||
try:
|
||||
from agent.model_metadata import query_ollama_supports_thinking
|
||||
|
||||
api_key = self.api_key if isinstance(self.api_key, str) else ""
|
||||
supported = query_ollama_supports_thinking(
|
||||
self.model, self.base_url, api_key=api_key or ""
|
||||
)
|
||||
except Exception:
|
||||
supported = None
|
||||
cache[key] = (supported, _time.monotonic())
|
||||
return supported
|
||||
|
||||
def _resolve_lmstudio_summary_reasoning_effort(self) -> Optional[str]:
|
||||
"""Resolve a safe top-level ``reasoning_effort`` for LM Studio.
|
||||
|
||||
|
|
|
|||
|
|
@ -264,6 +264,12 @@ def test_current_custom_endpoint_passthrough_marks_current_row(monkeypatch):
|
|||
monkeypatch.setattr("hermes_cli.providers.HERMES_OVERLAYS", {})
|
||||
monkeypatch.setattr("hermes_cli.models.fetch_openrouter_models",
|
||||
lambda *a, **kw: [])
|
||||
# The endpoint is offline in this scenario: the live /v1/models probe must
|
||||
# fail so the declared model list wins. Without this mock a REAL Ollama
|
||||
# server on localhost:11434 (common on dev machines) answers the probe and
|
||||
# its installed models replace the declared ones.
|
||||
monkeypatch.setattr("hermes_cli.models.fetch_api_models",
|
||||
lambda *a, **kw: [])
|
||||
|
||||
result = model_switch.list_picker_providers(
|
||||
current_provider="custom:ollama",
|
||||
|
|
|
|||
|
|
@ -437,6 +437,10 @@ def test_list_authenticated_providers_groups_same_endpoint(monkeypatch):
|
|||
returned as a single picker row with all their models merged."""
|
||||
monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {})
|
||||
monkeypatch.setattr(providers_mod, "HERMES_OVERLAYS", {})
|
||||
# Offline endpoint: the live /v1/models probe must fail so the declared
|
||||
# models win. Unmocked, a real Ollama server on localhost:11434 (common
|
||||
# on dev machines) answers and its installed models replace the fixtures.
|
||||
monkeypatch.setattr("hermes_cli.models.fetch_api_models", lambda *a, **kw: [])
|
||||
|
||||
providers = list_authenticated_providers(
|
||||
current_provider="custom",
|
||||
|
|
@ -523,6 +527,8 @@ def test_list_authenticated_providers_distinct_endpoints_stay_separate(monkeypat
|
|||
even if some display names happen to be similar."""
|
||||
monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {})
|
||||
monkeypatch.setattr(providers_mod, "HERMES_OVERLAYS", {})
|
||||
# Offline endpoints — see test_list_authenticated_providers_groups_same_endpoint.
|
||||
monkeypatch.setattr("hermes_cli.models.fetch_api_models", lambda *a, **kw: [])
|
||||
|
||||
providers = list_authenticated_providers(
|
||||
user_providers={},
|
||||
|
|
@ -618,6 +624,8 @@ def test_list_authenticated_providers_total_models_reflects_grouped_count(monkey
|
|||
the full count, and every grouped model appears in the list."""
|
||||
monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {})
|
||||
monkeypatch.setattr(providers_mod, "HERMES_OVERLAYS", {})
|
||||
# Offline endpoint — see test_list_authenticated_providers_groups_same_endpoint.
|
||||
monkeypatch.setattr("hermes_cli.models.fetch_api_models", lambda *a, **kw: [])
|
||||
|
||||
entries = [
|
||||
{"name": f"Ollama \u2014 Model {i}", "base_url": "http://localhost:11434/v1",
|
||||
|
|
|
|||
|
|
@ -55,14 +55,14 @@ class TestOllamaCloudAliases:
|
|||
"""ollama_cloud (underscore) is the unambiguous cloud alias."""
|
||||
assert resolve_provider("ollama_cloud") == "ollama-cloud"
|
||||
|
||||
def test_bare_ollama_stays_local(self):
|
||||
"""Bare 'ollama' alias routes to 'custom' (local) — not cloud."""
|
||||
assert resolve_provider("ollama") == "custom"
|
||||
def test_bare_ollama_is_local_provider(self):
|
||||
"""Bare 'ollama' resolves to the local ollama provider — not cloud."""
|
||||
assert resolve_provider("ollama") == "ollama"
|
||||
|
||||
def test_models_py_aliases(self):
|
||||
assert _PROVIDER_ALIASES.get("ollama_cloud") == "ollama-cloud"
|
||||
# bare "ollama" stays local
|
||||
assert _PROVIDER_ALIASES.get("ollama") == "custom"
|
||||
# bare "ollama" is the local provider
|
||||
assert _PROVIDER_ALIASES.get("ollama") == "ollama"
|
||||
|
||||
def test_normalize_provider(self):
|
||||
assert normalize_provider("ollama-cloud") == "ollama-cloud"
|
||||
|
|
@ -381,7 +381,7 @@ class TestOllamaCloudProvidersNew:
|
|||
|
||||
def test_alias_resolves(self):
|
||||
from hermes_cli.providers import normalize_provider as np
|
||||
assert np("ollama") == "custom" # bare "ollama" = local
|
||||
assert np("ollama") == "ollama" # bare "ollama" = local ollama provider
|
||||
assert np("ollama-cloud") == "ollama-cloud"
|
||||
|
||||
def test_label_override(self):
|
||||
|
|
|
|||
|
|
@ -9,8 +9,8 @@ was silently dropped for every custom endpoint.
|
|||
These tests pin the wire-shape contract:
|
||||
- disabled → extra_body.think = False
|
||||
- enabled + effort → top-level reasoning_effort (native OpenAI-compat
|
||||
format GLM/ARK expect), passed through verbatim
|
||||
including ``max``/``xhigh``
|
||||
format GLM/ARK expect); OpenAI-only levels
|
||||
(xhigh/minimal) map to the nearest accepted level
|
||||
- enabled + no effort → nothing emitted (endpoint's server default applies)
|
||||
- ollama_num_ctx → extra_body.options.num_ctx, orthogonal to reasoning
|
||||
"""
|
||||
|
|
@ -65,7 +65,7 @@ class TestCustomReasoningWireShape:
|
|||
assert tl == {}
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"effort", ["minimal", "low", "medium", "high", "xhigh", "max"]
|
||||
"effort", ["low", "medium", "high", "max"]
|
||||
)
|
||||
def test_enabled_effort_goes_top_level(self, custom_profile, effort):
|
||||
"""enabled + effort → TOP-LEVEL reasoning_effort, passed through verbatim.
|
||||
|
|
@ -81,6 +81,26 @@ class TestCustomReasoningWireShape:
|
|||
assert "reasoning_effort" not in eb
|
||||
assert "think" not in eb
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("effort", "expected"), [("xhigh", "max"), ("minimal", "low")]
|
||||
)
|
||||
def test_openai_only_efforts_map_to_endpoint_levels(
|
||||
self, custom_profile, effort, expected
|
||||
):
|
||||
"""Efforts that only exist on the OpenAI/Codex scale map to the
|
||||
nearest level these endpoints accept.
|
||||
|
||||
Ollama validates reasoning_effort against high/medium/low/max/none
|
||||
and rejects "xhigh"/"minimal" with HTTP 400; GLM documents "high" and
|
||||
"max". Carrying a session's effort dial across a provider switch to a
|
||||
local model must not brick the chat.
|
||||
"""
|
||||
eb, tl = custom_profile.build_api_kwargs_extras(
|
||||
reasoning_config={"enabled": True, "effort": effort}, model="glm-5.2"
|
||||
)
|
||||
assert tl == {"reasoning_effort": expected}
|
||||
assert "think" not in eb
|
||||
|
||||
def test_enabled_without_effort_emits_nothing(self, custom_profile):
|
||||
"""enabled but no effort → omit; do NOT force a level the user didn't pick."""
|
||||
eb, tl = custom_profile.build_api_kwargs_extras(
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue