diff --git a/README.md b/README.md index 28c70b4..383ba24 100644 --- a/README.md +++ b/README.md @@ -17,9 +17,9 @@ Native-like Hermes dashboard plugin for local Ollama model management and chat. - Paste images directly into the composer and drag/drop images, PDFs, and text files - Streamed Ollama responses with a real Stop action that cancels the active request - Minimized-by-default expandable thinking/progress details with live stage, elapsed time, event, and character counters -- Validation harness mode: choose one primary model and two or more independent validator models; validators review the primary draft and the primary model compiles one final answer +- Validation harness mode: choose one primary model and one or more independent validator models; validators review the primary draft and the primary model compiles one final answer -The chat supports two modes. With one selected model, it sends a normal direct request. With one primary model and at least two validator models selected, the plugin runs a validation harness: the primary creates a draft, validators independently review the request and draft in parallel, and the primary compiles one final user-facing answer from the draft and validation reports. Validator reports are returned as supporting evidence, while only the compiled primary response is persisted and displayed as the answer. +The chat supports two modes. With one selected model, it sends a normal direct request. With one primary model and at least one validator model selected, the plugin runs a validation harness: the primary creates a draft, validators independently review the request and draft in parallel, and the primary compiles one final user-facing answer from the draft and validation reports. Validator reports are returned as supporting evidence, while only the compiled primary response is persisted and displayed as the answer. ## Performance metrics @@ -95,7 +95,7 @@ When Ollama accepts a load request but evicts one model while starting another, While Ollama is starting a runner, the highlighted runtime chart displays an animated **Loading into Ollama memory** state with the selected model names, current stage, and elapsed time. The runtime panel separately reports Ollama resident model-weight bytes and the estimated target weight bytes. This is separate from host `MemAvailable`: CPU-mapped model files may appear as Linux file cache rather than ordinary process RAM usage. ## Multi-model loading and resident state -The model pool now verifies every load request against Ollama `/api/ps` before reporting success. The UI shows an in-progress loading message, then reports which models are actually resident and which Ollama evicted. Resident models are highlighted in the pool with a green loaded state. After a browser refresh, resident models repopulate the pool selection and the answer-harness controls. Select one primary loaded model and two or more validator models; validators review the primary draft and the primary compiles one final answer. +The model pool now verifies every load request against Ollama `/api/ps` before reporting success. The UI shows an in-progress loading message, then reports which models are actually resident and which Ollama evicted. Resident models are highlighted in the pool with a green loaded state. After a browser refresh, resident models repopulate the pool selection and the answer-harness controls. Select one primary loaded model and one or more validator models; validators review the primary draft and the primary compiles one final answer. Ollama still controls the physical resident-model limit. If it cannot keep all requested models at once because of its scheduler, GPU policy, context allocation, or available memory, the plugin reports the non-resident names instead of claiming they were permanently loaded. Increasing that limit requires changing the Ollama service configuration; the plugin does not silently alter or restart the Ollama service. diff --git a/dashboard/dist/index.js b/dashboard/dist/index.js index 2a35841..2abf770 100644 --- a/dashboard/dist/index.js +++ b/dashboard/dist/index.js @@ -232,10 +232,10 @@ h("div", { className: "ollama-pool-grid" }, models.map(function (item) { var loaded = !!item.loaded, selected = props.poolSelection.indexOf(item.name) >= 0, placement = props.placements[item.name] || "gpu_ram"; return h("label", { className: "ollama-pool-item" + (loaded ? " loaded" : "") + (selected ? " selected" : ""), key: item.name, title: loaded ? "Loaded and resident in Ollama" : "Installed but not resident" }, h("input", { type: "checkbox", checked: selected, onChange: function () { props.onTogglePool(item.name); } }), h("span", null, h("strong", null, item.name), h("small", null, loaded ? "Loaded and resident · keep-alive active" : "Installed · not loaded", " · ", (item.capabilities || []).join(", ") || "capabilities unknown"), h("span", { className: "ollama-placement-control" }, h("small", null, "Placement"), h("select", { value: placement, onClick: function (event) { event.stopPropagation(); }, onChange: function (event) { event.stopPropagation(); props.onPlacementChange(item.name, event.target.value); } }, h("option", { value: "gpu_ram" }, "GPU + RAM (automatic offload)"), h("option", { value: "ram_only" }, "RAM only (CPU)"))))); })), h("div", { className: "ollama-pool-actions" }, h(Button, { disabled: !props.poolSelection.length || !!props.busy, onClick: props.onLoad }, props.busy === "/models/load" ? "Loading " + props.poolSelection.length + " model" + (props.poolSelection.length === 1 ? "" : "s") + "…" : "Load selected permanently"), h(Button, { className: "secondary", disabled: !props.poolSelection.length || !!props.busy, onClick: props.onUnload }, props.busy === "/models/unload" ? "Unloading…" : "Unload selected")), h("div", { className: "ollama-chat-model-selection" }, - h("div", null, h("strong", null, "Answer harness"), h("small", null, "Choose one primary model. Add two or more validators to review its draft before the primary compiles the final answer.")), + h("div", null, h("strong", null, "Answer harness"), h("small", null, "Choose one primary model. Add one or more validators to review its draft before the primary compiles the final answer.")), loaded.length ? h("div", { className: "ollama-harness-primary" }, h("label", null, "Primary model", h("select", { value: primary, onChange: function (event) { props.onPrimaryChange(event.target.value); } }, loaded.map(function (item) { return h("option", { key: item.name, value: item.name }, item.name); })))) : h("span", null, "Load one or more models above first."), loaded.length > 1 && h("div", { className: "ollama-harness-validators" }, h("strong", null, "Validator models"), loaded.filter(function (item) { return item.name !== primary; }).map(function (item) { return h("label", { className: "loaded", key: item.name }, h("input", { type: "checkbox", checked: validators.indexOf(item.name) >= 0, onChange: function () { props.onToggleChat(item.name); } }), item.name, " · ", (item.capabilities || []).join(", ")); })), - loaded.length > 1 && h("small", { className: validators.length >= 2 ? "ollama-harness-ready" : "ollama-harness-warning" }, validators.length >= 2 ? "Validation harness ready: the primary will compile one final answer after independent checks." : "Select at least two validator models to enable the validation harness.")) + loaded.length > 1 && h("small", { className: validators.length >= 1 ? "ollama-harness-ready" : "ollama-harness-warning" }, validators.length >= 1 ? "Validation harness ready: the primary will compile one final answer after independent checks." : "Select at least one validator model to enable the validation harness.")) ); } @@ -391,7 +391,7 @@ }); } function send() { - if (busy === "send" || busy === "stop" || !selectedModels.length || selectedModels.length === 2 || (!message.trim() && !attachments.length)) { if (selectedModels.length === 2 && busy !== "send") setNotice({ warning: "Select at least two validator models, or remove the validator to use direct chat." }); return; } + if (busy === "send" || busy === "stop" || !selectedModels.length || (!message.trim() && !attachments.length)) return; var requestId = makeRequestId(); var controller = typeof AbortController === "function" ? new AbortController() : null; var current = { id: requestId, controller: controller, stopped: false }; @@ -402,7 +402,7 @@ fetchJSON(API + "/chat", requestOptions).then(function (result) { var answer = result.message && result.message.content ? result.message.content : "(No response text returned.)"; setConversationId(result.conversation_id || conversationId); setHistory(function (old) { return old.concat([{ role: "assistant", content: answer }]); }); setMetrics(result.metrics || []); setValidationReports(result.mode === "harness" ? (result.validation_reports || []) : null); setAttachments([]); setRuntime(result.runtime || runtime); setNotice({ ok: result.mode === "harness" ? "One final answer compiled by " + result.primary_model + " after validation by " + (result.validator_models || []).join(", ") + "." : "Response complete. Shared conversation and performance metrics saved." }); pollRuntime(); refreshConversations(); }).catch(function (err) { if (!current.stopped) setNotice({ error: err.message || String(err) }); }).finally(function () { if (!current.stopped) { setThinking(null); setActiveRequest(null); } setBusy(""); }); } return h("section", { className: "ollama-chat" }, - h("div", { className: "ollama-chat-header" }, h("div", null, h("div", { className: "ollama-eyebrow" }, "LOCAL OLLAMA CHAT"), h("h2", null, "Chat with a validated model harness"), h("p", null, "Choose one primary model and at least two loaded validator models. The primary produces one final answer after reviewing the independent validation reports.")), h(Button, { className: "secondary", disabled: !history.length || busy === "send", onClick: clearChat }, "Clear chat")), + h("div", { className: "ollama-chat-header" }, h("div", null, h("div", { className: "ollama-eyebrow" }, "LOCAL OLLAMA CHAT"), h("h2", null, "Chat with a validated model harness"), h("p", null, "Choose one primary model and at least one loaded validator model. The primary produces one final answer after reviewing the independent validation reports.")), h(Button, { className: "secondary", disabled: !history.length || busy === "send", onClick: clearChat }, "Clear chat")), h("section", { className: "ollama-persistence-panel" }, h("div", { className: "ollama-persistence-heading" }, h("div", null, h("h3", null, "Shared conversations"), h("p", null, "Saved on this Hermes server; any browser can resume them.")), h(Button, { className: "secondary", onClick: newConversation }, "New conversation")), h("div", { className: "ollama-conversation-list" }, conversations.length ? conversations.map(function (item) { return h(Button, { key: item.id, className: item.id === conversationId ? "selected" : "", onClick: function () { openConversation(item.id); } }, (item.title || "New conversation").slice(0, 70), " · ", item.message_count || 0, " messages"); }) : h("small", null, "No saved conversations yet.")), @@ -417,7 +417,7 @@ h("div", { className: "ollama-chat-layout" }, h("div", { className: "ollama-conversation" }, history.length ? history.map(function (item, index) { return h("div", { className: "ollama-message " + item.role, key: index }, h("small", null, item.role === "assistant" ? "Ollama" : "You"), h("div", null, item.content)); }) : h(Empty, null, "Start a conversation. The selected model will be loaded into Ollama memory when you load it or send the first message.")), h("div", { className: "ollama-composer" + (dragging ? " drop-active" : ""), onDragOver: onDragOver, onDragLeave: onDragLeave, onDrop: onDrop }, dragging && h("div", { className: "ollama-drop-hint" }, "Drop files here to attach"), h("textarea", { value: message, placeholder: selectedModels.length ? "Ask " + selectedModels.length + " loaded model" + (selectedModels.length === 1 ? "" : "s") + "… Press Enter to send; Shift+Enter for a new line." : "Load and select at least one model above…", onPaste: onPaste, onChange: function (event) { setMessage(event.target.value); }, onKeyDown: function (event) { if (event.key === "Enter" && !event.shiftKey) { event.preventDefault(); send(); } } }), - h("div", { className: "ollama-attachment-actions" }, h("label", { className: "ollama-file-button" }, "Attach image / PDF / file", h("input", { type: "file", multiple: true, accept: "image/*,application/pdf,text/*,.txt,.md,.csv,.json,.log,.xml,.yaml,.yml", onChange: onFiles })), h("input", { className: "ollama-url-input", value: url, placeholder: "https://example.com/document", onChange: function (event) { setUrl(event.target.value); }, onKeyDown: function (event) { if (event.key === "Enter") addUrl(); } }), h(Button, { onClick: addUrl, disabled: !url.trim() }, "Add URL"), h(Button, { onClick: send, disabled: busy === "send" || busy === "stop" || !selectedModels.length || selectedModels.length === 2 || (!message.trim() && !attachments.length) }, busy === "send" ? "Sending…" : "Send (Enter)")), + h("div", { className: "ollama-attachment-actions" }, h("label", { className: "ollama-file-button" }, "Attach image / PDF / file", h("input", { type: "file", multiple: true, accept: "image/*,application/pdf,text/*,.txt,.md,.csv,.json,.log,.xml,.yaml,.yml", onChange: onFiles })), h("input", { className: "ollama-url-input", value: url, placeholder: "https://example.com/document", onChange: function (event) { setUrl(event.target.value); }, onKeyDown: function (event) { if (event.key === "Enter") addUrl(); } }), h(Button, { onClick: addUrl, disabled: !url.trim() }, "Add URL"), h(Button, { onClick: send, disabled: busy === "send" || busy === "stop" || !selectedModels.length || (!message.trim() && !attachments.length) }, busy === "send" ? "Sending…" : "Send (Enter)")), attachments.length > 0 && h("div", { className: "ollama-attachments" }, attachments.map(function (item, index) { return h("span", { className: "ollama-attachment", key: index }, item.name || item.url, h("button", { type: "button", onClick: function () { setAttachments(function (old) { return old.filter(function (_, i) { return i !== index; }); }); } }, "×")); })), h("p", { className: "ollama-chat-footnote" }, "Limits: 20 MiB per uploaded file, 15 MiB per fetched URL. Private/local URL targets are blocked. Remote content is treated as untrusted text.")) ) diff --git a/dashboard/plugin_api.py b/dashboard/plugin_api.py index f11cefb..f414710 100644 --- a/dashboard/plugin_api.py +++ b/dashboard/plugin_api.py @@ -69,7 +69,7 @@ MAX_ATTACHMENT_BYTES = 20 * 1024 * 1024 MAX_ATTACHMENT_TEXT = 80_000 MAX_URL_BYTES = 15 * 1024 * 1024 CHAT_KEEP_ALIVE = -1 -HARNESS_MIN_VALIDATORS = 2 +HARNESS_MIN_VALIDATORS = 1 HARNESS_MAX_DRAFT_CHARS = 24_000 HARNESS_MAX_VALIDATION_CHARS = 8_000 diff --git a/tests/test_validation_harness.py b/tests/test_validation_harness.py index 4b30dd3..6c8e516 100644 --- a/tests/test_validation_harness.py +++ b/tests/test_validation_harness.py @@ -14,13 +14,21 @@ class ValidationHarnessTests(unittest.TestCase): self.assertEqual(validators, ["validator-a", "validator-b"]) self.assertTrue(harness) - def test_harness_requires_two_distinct_validators(self): - body = api.ChatRequest(primary_model="primary", validator_models=["validator-a"], harness=True) + def test_harness_requires_one_distinct_validator(self): + body = api.ChatRequest(primary_model="primary", validator_models=[], harness=True) with patch.object(api, "_require_installed_model", side_effect=lambda name: name): with self.assertRaises(api.HTTPException) as context: api._harness_models(body) self.assertEqual(context.exception.status_code, 400) + def test_harness_accepts_one_validator(self): + body = api.ChatRequest(primary_model="primary", validator_models=["validator-a"], harness=True) + with patch.object(api, "_require_installed_model", side_effect=lambda name: name): + primary, validators, harness = api._harness_models(body) + self.assertEqual(primary, "primary") + self.assertEqual(validators, ["validator-a"]) + self.assertTrue(harness) + def test_chat_route_returns_one_compiled_answer(self): body = api.ChatRequest( primary_model="primary",