From 7655720e11a45a6d75aafbf4d299addfa1b5c0e6 Mon Sep 17 00:00:00 2001 From: Local Dev Date: Sun, 20 Sep 2026 15:22:34 +0200 Subject: [PATCH] feat(studio): AI assistant that runs on the user's own device An Assistant drawer in Sirius Studio with two providers and no key, no server and no spend. "This browser" runs an open model on the user's GPU through WebGPU with WebLLM (vendored, Apache-2.0); weights download once from the MLC mirror and stay in the browser cache. "Local endpoint" talks to an OpenAI-compatible runtime on the user's machine (Ollama, LM Studio), which unlocks larger models on a real GPU. Nothing the user writes or builds leaves their device in either mode. Three verbs: make a section, rewrite the selected text, restyle the selection. The model returns HTML and CSS as data; Studio sanitises it (no scripts, frames, handlers or imports) and inserts it through the editor, with Undo. Model output is never executed. The quantisation is chosen per GPU: q4f16 when the adapter exposes shader-f16, q4f32 otherwise (Pascal-era cards lack it). Small models often answer with bare HTML instead of JSON, so the parser accepts both, prompts avoid literal placeholders one model echoed back, and an out-of-memory or disposed runtime is reported as "pick a smaller model" with the engine reset. Switching models starts a fresh worker. Verified on an NVIDIA Pascal card: SmolLM2 360M rewrites text, Qwen2.5 Coder 0.5B builds a section; the 1.5B f32 build exceeded that card's memory and now fails gracefully. --- docs/index.html | 2 +- i18n/de.json | 19 +++- i18n/el.json | 19 +++- i18n/es.json | 19 +++- i18n/fr.json | 19 +++- i18n/pt.json | 19 +++- i18n/ru.json | 19 +++- index.html | 2 +- js/ai-worker.js | 7 ++ js/i18n.js | 2 +- js/studio-ai.js | 223 +++++++++++++++++++++++++++++++++++++ js/studio.js | 2 + market.html | 2 +- portal.html | 2 +- studio.html | 64 ++++++++++- tld.html | 2 +- vendor/web-llm/LICENSE.txt | 211 +++++++++++++++++++++++++++++++++++ vendor/web-llm/index.js | 93 ++++++++++++++++ 18 files changed, 709 insertions(+), 17 deletions(-) create mode 100644 js/ai-worker.js create mode 100644 js/studio-ai.js create mode 100644 vendor/web-llm/LICENSE.txt create mode 100644 vendor/web-llm/index.js diff --git a/docs/index.html b/docs/index.html index d7993e4..c39a1c0 100644 --- a/docs/index.html +++ b/docs/index.html @@ -92,7 +92,7 @@ - + diff --git a/i18n/de.json b/i18n/de.json index 966cde0..deff46c 100644 --- a/i18n/de.json +++ b/i18n/de.json @@ -591,5 +591,22 @@ "Sell a name from your dashboard →": "Einen Namen im Dashboard verkaufen →", "Register a new name": "Neuen Namen registrieren", "↓ Import live site": "↓ Live-Seite importieren", - "Import the live site": "Live-Seite importieren" + "Import the live site": "Live-Seite importieren", + "✨ Assistant": "✨ Assistent", + "Runs on your own device. Nothing you write or build is sent to Sirius.X or to any AI vendor.": "Läuft auf deinem eigenen Gerät. Nichts, was du schreibst oder baust, wird an Sirius.X oder einen KI-Anbieter gesendet.", + "Runs on": "Läuft auf", + "This browser — your GPU, private": "Dieser Browser — deine GPU, privat", + "Local endpoint — Ollama, LM Studio": "Lokaler Endpunkt — Ollama, LM Studio", + "Model": "Modell", + "Downloads once, then stays in the browser cache. Needs WebGPU (current Chrome, Edge, Firefox, Safari).": "Wird einmal geladen und bleibt im Browser-Cache. Braucht WebGPU (aktuelles Chrome, Edge, Firefox, Safari).", + "Load model": "Modell laden", + "Endpoint": "Endpunkt", + "Connect": "Verbinden", + "What do you want?": "Was möchtest du?", + "Make a section": "Abschnitt erstellen", + "Rewrite selected text": "Markierten Text umschreiben", + "Restyle selection": "Auswahl umgestalten", + "Nothing selected": "Nichts ausgewählt", + "Undo last change": "Letzte Änderung rückgängig", + "Stop": "Stopp" } diff --git a/i18n/el.json b/i18n/el.json index 796eab9..56e9a59 100644 --- a/i18n/el.json +++ b/i18n/el.json @@ -591,5 +591,22 @@ "Sell a name from your dashboard →": "Πούλησε ένα όνομα από τον πίνακα ελέγχου →", "Register a new name": "Καταχώρηση νέου ονόματος", "↓ Import live site": "↓ Εισαγωγή ζωντανής σελίδας", - "Import the live site": "Εισαγωγή ζωντανής σελίδας" + "Import the live site": "Εισαγωγή ζωντανής σελίδας", + "✨ Assistant": "✨ Βοηθός", + "Runs on your own device. Nothing you write or build is sent to Sirius.X or to any AI vendor.": "Τρέχει στη δική σου συσκευή. Τίποτα από όσα γράφεις ή φτιάχνεις δεν στέλνεται στο Sirius.X ή σε πάροχο AI.", + "Runs on": "Τρέχει σε", + "This browser — your GPU, private": "Αυτός ο browser — η GPU σου, ιδιωτικά", + "Local endpoint — Ollama, LM Studio": "Τοπικό endpoint — Ollama, LM Studio", + "Model": "Μοντέλο", + "Downloads once, then stays in the browser cache. Needs WebGPU (current Chrome, Edge, Firefox, Safari).": "Κατεβαίνει μία φορά και μένει στην cache του browser. Χρειάζεται WebGPU (σύγχρονο Chrome, Edge, Firefox, Safari).", + "Load model": "Φόρτωση μοντέλου", + "Endpoint": "Endpoint", + "Connect": "Σύνδεση", + "What do you want?": "Τι θέλεις;", + "Make a section": "Φτιάξε ενότητα", + "Rewrite selected text": "Ξαναγράψε το επιλεγμένο κείμενο", + "Restyle selection": "Άλλαξε στυλ στην επιλογή", + "Nothing selected": "Τίποτα επιλεγμένο", + "Undo last change": "Αναίρεση τελευταίας αλλαγής", + "Stop": "Στοπ" } diff --git a/i18n/es.json b/i18n/es.json index 52f9d55..bc592dd 100644 --- a/i18n/es.json +++ b/i18n/es.json @@ -591,5 +591,22 @@ "Sell a name from your dashboard →": "Vender un nombre desde tu panel →", "Register a new name": "Registrar un nombre nuevo", "↓ Import live site": "↓ Importar el sitio en vivo", - "Import the live site": "Importar el sitio en vivo" + "Import the live site": "Importar el sitio en vivo", + "✨ Assistant": "✨ Asistente", + "Runs on your own device. Nothing you write or build is sent to Sirius.X or to any AI vendor.": "Funciona en tu propio dispositivo. Nada de lo que escribes o construyes se envía a Sirius.X ni a ningún proveedor de IA.", + "Runs on": "Se ejecuta en", + "This browser — your GPU, private": "Este navegador — tu GPU, privado", + "Local endpoint — Ollama, LM Studio": "Servidor local — Ollama, LM Studio", + "Model": "Modelo", + "Downloads once, then stays in the browser cache. Needs WebGPU (current Chrome, Edge, Firefox, Safari).": "Se descarga una vez y queda en la caché del navegador. Requiere WebGPU (Chrome, Edge, Firefox, Safari actuales).", + "Load model": "Cargar modelo", + "Endpoint": "Endpoint", + "Connect": "Conectar", + "What do you want?": "¿Qué quieres?", + "Make a section": "Crear una sección", + "Rewrite selected text": "Reescribir el texto seleccionado", + "Restyle selection": "Cambiar estilo de la selección", + "Nothing selected": "Nada seleccionado", + "Undo last change": "Deshacer el último cambio", + "Stop": "Detener" } diff --git a/i18n/fr.json b/i18n/fr.json index a3d75c1..b56dbb7 100644 --- a/i18n/fr.json +++ b/i18n/fr.json @@ -591,5 +591,22 @@ "Sell a name from your dashboard →": "Vendre un nom depuis le tableau de bord →", "Register a new name": "Enregistrer un nouveau nom", "↓ Import live site": "↓ Importer le site en ligne", - "Import the live site": "Importer le site en ligne" + "Import the live site": "Importer le site en ligne", + "✨ Assistant": "✨ Assistant", + "Runs on your own device. Nothing you write or build is sent to Sirius.X or to any AI vendor.": "Fonctionne sur votre propre appareil. Rien de ce que vous écrivez ou construisez n'est envoyé à Sirius.X ni à un fournisseur d'IA.", + "Runs on": "S'exécute sur", + "This browser — your GPU, private": "Ce navigateur — votre GPU, privé", + "Local endpoint — Ollama, LM Studio": "Point local — Ollama, LM Studio", + "Model": "Modèle", + "Downloads once, then stays in the browser cache. Needs WebGPU (current Chrome, Edge, Firefox, Safari).": "Téléchargé une fois, puis conservé dans le cache du navigateur. Nécessite WebGPU (Chrome, Edge, Firefox, Safari récents).", + "Load model": "Charger le modèle", + "Endpoint": "Point d'accès", + "Connect": "Connecter", + "What do you want?": "Que voulez-vous ?", + "Make a section": "Créer une section", + "Rewrite selected text": "Réécrire le texte sélectionné", + "Restyle selection": "Restyler la sélection", + "Nothing selected": "Rien de sélectionné", + "Undo last change": "Annuler la dernière modification", + "Stop": "Arrêter" } diff --git a/i18n/pt.json b/i18n/pt.json index bb3740c..98cb500 100644 --- a/i18n/pt.json +++ b/i18n/pt.json @@ -591,5 +591,22 @@ "Sell a name from your dashboard →": "Vender um nome a partir do painel →", "Register a new name": "Registar um nome novo", "↓ Import live site": "↓ Importar o site em produção", - "Import the live site": "Importar o site em produção" + "Import the live site": "Importar o site em produção", + "✨ Assistant": "✨ Assistente", + "Runs on your own device. Nothing you write or build is sent to Sirius.X or to any AI vendor.": "Corre no seu próprio dispositivo. Nada do que escreve ou constrói é enviado para o Sirius.X ou para qualquer fornecedor de IA.", + "Runs on": "Corre em", + "This browser — your GPU, private": "Este navegador — a sua GPU, privado", + "Local endpoint — Ollama, LM Studio": "Servidor local — Ollama, LM Studio", + "Model": "Modelo", + "Downloads once, then stays in the browser cache. Needs WebGPU (current Chrome, Edge, Firefox, Safari).": "Descarrega uma vez e fica na cache do navegador. Precisa de WebGPU (Chrome, Edge, Firefox, Safari atuais).", + "Load model": "Carregar modelo", + "Endpoint": "Endpoint", + "Connect": "Ligar", + "What do you want?": "O que pretende?", + "Make a section": "Criar uma secção", + "Rewrite selected text": "Reescrever o texto selecionado", + "Restyle selection": "Restilizar a seleção", + "Nothing selected": "Nada selecionado", + "Undo last change": "Anular a última alteração", + "Stop": "Parar" } diff --git a/i18n/ru.json b/i18n/ru.json index 0ae07d9..20da5d0 100644 --- a/i18n/ru.json +++ b/i18n/ru.json @@ -591,5 +591,22 @@ "Sell a name from your dashboard →": "Продать имя из панели →", "Register a new name": "Зарегистрировать новое имя", "↓ Import live site": "↓ Импортировать текущий сайт", - "Import the live site": "Импортировать текущий сайт" + "Import the live site": "Импортировать текущий сайт", + "✨ Assistant": "✨ Ассистент", + "Runs on your own device. Nothing you write or build is sent to Sirius.X or to any AI vendor.": "Работает на вашем устройстве. Ничего из того, что вы пишете или строите, не отправляется в Sirius.X или поставщику ИИ.", + "Runs on": "Работает на", + "This browser — your GPU, private": "Этот браузер — ваша GPU, приватно", + "Local endpoint — Ollama, LM Studio": "Локальный сервер — Ollama, LM Studio", + "Model": "Модель", + "Downloads once, then stays in the browser cache. Needs WebGPU (current Chrome, Edge, Firefox, Safari).": "Скачивается один раз и остаётся в кэше браузера. Нужен WebGPU (актуальные Chrome, Edge, Firefox, Safari).", + "Load model": "Загрузить модель", + "Endpoint": "Адрес сервера", + "Connect": "Подключить", + "What do you want?": "Что нужно сделать?", + "Make a section": "Создать секцию", + "Rewrite selected text": "Переписать выбранный текст", + "Restyle selection": "Изменить стиль выбранного", + "Nothing selected": "Ничего не выбрано", + "Undo last change": "Отменить последнее изменение", + "Stop": "Стоп" } diff --git a/index.html b/index.html index 45a637d..a20bf99 100644 --- a/index.html +++ b/index.html @@ -130,7 +130,7 @@ .verify dd{margin:2px 0 0;font-family:ui-monospace,monospace;font-size:12.5px;word-break:break-all;color:var(--ink)} footer{border-top:1px solid var(--line);padding:2.4rem 0 3.5rem;color:var(--dim);font-size:13px;text-align:center;margin-top:2.5rem} - + diff --git a/js/ai-worker.js b/js/ai-worker.js new file mode 100644 index 0000000..628d6f4 --- /dev/null +++ b/js/ai-worker.js @@ -0,0 +1,7 @@ +// Web Worker for the in-browser assistant: runs the WebLLM engine off the UI +// thread. Weights and the model library are fetched by WebLLM itself and +// cached by the browser; nothing is sent anywhere. +import { WebWorkerMLCEngineHandler } from "../vendor/web-llm/index.js?v=0.2.85"; + +const handler = new WebWorkerMLCEngineHandler(); +self.onmessage = (msg) => handler.onmessage(msg); diff --git a/js/i18n.js b/js/i18n.js index b7d6bb4..2455d3d 100644 --- a/js/i18n.js +++ b/js/i18n.js @@ -22,7 +22,7 @@ (function () { var STORE = "sirius-lang"; var CACHE = "sirius-i18n:"; - var VERSION = "20260920b"; // bump when dictionaries change + var VERSION = "20260920c"; // bump when dictionaries change var LANGS = { en: "English", de: "Deutsch", diff --git a/js/studio-ai.js b/js/studio-ai.js new file mode 100644 index 0000000..cbc7881 --- /dev/null +++ b/js/studio-ai.js @@ -0,0 +1,223 @@ +// Sirius Studio — Assistant panel. +// +// Two providers, both on the user's own hardware, no key and no spend: +// webllm — the model runs inside this tab on the GPU through WebGPU +// (vendor/web-llm, Apache-2.0). Weights download once from the +// MLC mirror and stay in the browser cache. +// local — any OpenAI-compatible endpoint on the user's machine (Ollama, +// LM Studio, llama.cpp). The page talks to localhost directly; the +// runtime has to allow this origin (OLLAMA_ORIGINS=https://silentmode.st). +// +// The model returns data (HTML + CSS or text as JSON); Studio inserts it +// through the editor API. Model output is never executed as code. + +const LS_KEY = "siriusAi"; +// Base ids; the quantisation suffix is chosen at load time: q4f16 when the +// GPU exposes the shader-f16 feature (half the memory), q4f32 otherwise. +const WEBLLM_MODELS = [ + { id: "Qwen2.5-Coder-1.5B-Instruct", label: "Qwen2.5 Coder 1.5B · best HTML/CSS (≈1 GB download)" }, + { id: "Qwen2.5-1.5B-Instruct", label: "Qwen2.5 1.5B · balanced copy + layout (≈1 GB)" }, + { id: "Qwen2.5-Coder-0.5B-Instruct", label: "Qwen2.5 Coder 0.5B · light, still builds sections (≈400 MB, 1 GB GPU)" }, + { id: "SmolLM2-360M-Instruct", label: "SmolLM2 360M · weakest machines; rewrites text, rarely builds sections (≈250 MB)" }, + { id: "Qwen2.5-Coder-3B-Instruct", label: "Qwen2.5 Coder 3B · strongest here (≈2 GB, needs 4 GB GPU)" }, + { id: "Llama-3.2-3B-Instruct", label: "Llama 3.2 3B · better prose (≈2 GB, needs 4 GB GPU)" }, +]; +let gpuF16 = null; +async function gpuSupportsF16() { + if (gpuF16 != null) return gpuF16; + try { const a = await navigator.gpu.requestAdapter(); gpuF16 = !!a?.features?.has("shader-f16"); } catch { gpuF16 = false; } + return gpuF16; +} +const fullModelId = async (base) => `${base}-q4${(await gpuSupportsF16()) ? "f16" : "f32"}_1-MLC`; + +export function initAssistant(editor, { esc, name }) { + const $ = (id) => document.getElementById(id); + const panel = $("ai"); if (!panel) return; + const settings = Object.assign({ provider: "webllm", model: WEBLLM_MODELS[0].id, url: "http://localhost:11434/v1", lmodel: "" }, load()); + let engine = null, engineModel = null, worker = null, webllm = null, busy = false, abort = null, lastApplied = false; + + // ---------- ui ---------- + $("ai-model").innerHTML = WEBLLM_MODELS.map((m) => ``).join(""); + $("ai-model").value = settings.model; + if (!$("ai-model").value) { $("ai-model").value = WEBLLM_MODELS[0].id; settings.model = WEBLLM_MODELS[0].id; } + $("ai-provider").value = settings.provider; + $("ai-url").value = settings.url; + $("ai-lmodel").value = settings.lmodel; + const showProvider = () => { const w = $("ai-provider").value === "webllm"; $("ai-webllm").hidden = !w; $("ai-local").hidden = w; }; + showProvider(); + const st = (text, cls = "") => { const el = $("ai-status"); el.textContent = text; el.className = "ai-status " + cls; }; + const out = (text) => { const el = $("ai-out"); el.textContent = text; el.scrollTop = 1e6; }; + const persist = () => { settings.provider = $("ai-provider").value; settings.model = $("ai-model").value; settings.url = $("ai-url").value.trim().replace(/\/+$/, ""); settings.lmodel = $("ai-lmodel").value.trim(); save(settings); }; + + $("btn-ai").addEventListener("click", () => { panel.hidden = !panel.hidden; $("btn-ai").classList.toggle("on", !panel.hidden); setTimeout(() => editor.refresh(), 60); if (!panel.hidden) paintTarget(); }); + $("ai-close").addEventListener("click", () => { panel.hidden = true; $("btn-ai").classList.remove("on"); setTimeout(() => editor.refresh(), 60); }); + $("ai-provider").addEventListener("change", () => { showProvider(); persist(); st(readyText(), ""); }); + $("ai-model").addEventListener("change", persist); + $("ai-url").addEventListener("change", persist); + $("ai-lmodel").addEventListener("change", persist); + $("ai-load").addEventListener("click", () => ensureWebLLM().catch((e) => st(e.message || String(e), "err"))); + $("ai-connect").addEventListener("click", () => probeLocal().catch((e) => st(e.message || String(e), "err"))); + $("ai-undo").addEventListener("click", () => { if (lastApplied) { editor.UndoManager.undo(); lastApplied = false; $("ai-undo").hidden = true; } }); + $("ai-stop").addEventListener("click", () => { try { abort?.abort(); engine?.interruptGenerate?.(); } catch {} }); + panel.querySelectorAll("[data-verb]").forEach((b) => b.addEventListener("click", () => run(b.dataset.verb).catch((e) => { st(friendly(e), "err"); }))); + function friendly(e) { + const m = e?.message || String(e); + if (/disposed|out of memory|OutOfMemory|device lost|DeviceLost/i.test(m)) { + try { worker?.terminate(); } catch {} engine = null; engineModel = null; + return "Your GPU ran out of memory for this model. Pick a smaller one (Qwen2.5 Coder 0.5B or SmolLM2) and press Load model again."; + } + return m; + } + editor.on("component:selected component:deselected component:toggled", paintTarget); + if (!("gpu" in navigator)) { const o = $("ai-provider").querySelector('option[value="webllm"]'); if (o) o.textContent += " — no WebGPU here"; if (settings.provider === "webllm") st("This browser has no WebGPU. Use a local endpoint, or a browser with WebGPU.", "warn"); else st(readyText()); } + else st(readyText()); + + function readyText() { return $("ai-provider").value === "webllm" ? (engine && engineModel === $("ai-model").value ? "Model loaded — ask away" : "Model not loaded yet — press Load model (once)") : "Press Connect to check the local endpoint"; } + function selected() { return editor.getSelected(); } + function isText(c) { return !!c && (c.get("type") === "text" || c.is?.("text")); } + function paintTarget() { + const c = selected(); + const t = $("ai-target"); + if (!c) { t.textContent = "Nothing selected — “Make a section” adds at the end of the page."; return; } + const tag = c.get("tagName") || "element", cls = (c.getClasses?.() || []).slice(0, 2).join("."), text = (c.getEl?.()?.innerText || "").trim().slice(0, 60); + t.textContent = `Selected: <${tag}${cls ? "." + cls : ""}>${text ? " “" + text + (text.length >= 60 ? "…" : "") + "”" : ""}`; + } + + // ---------- providers ---------- + async function ensureWebLLM() { + if (!("gpu" in navigator)) throw new Error("WebGPU is not available in this browser."); + const base = $("ai-model").value; + if (engine && engineModel === base) return engine; + const modelId = await fullModelId(base); + if (!webllm) { st("Loading the engine…", "busy"); webllm = await import("../vendor/web-llm/index.js?v=0.2.85"); } + $("ai-load").disabled = true; + try { + const progress = (r) => st((r.text || "Loading…").replace(/\[.*?\]\s*/g, "").slice(0, 140), "busy"); + // Switching models: a fresh worker every time. Reloading inside the same + // engine left the runtime with disposed objects ("Object has already + // been disposed" on the next generation). + if (engine) { try { await engine.unload?.(); } catch {} try { worker?.terminate(); } catch {} engine = null; engineModel = null; } + worker = new Worker(new URL("./ai-worker.js?v=20260920ai", import.meta.url), { type: "module" }); + engine = await webllm.CreateWebWorkerMLCEngine(worker, modelId, { initProgressCallback: progress }); + engineModel = base; + st(`Model loaded (${(await gpuSupportsF16()) ? "f16" : "f32 — this GPU has no f16 shaders, so the slightly larger build"}) — runs on this device, nothing leaves it`, "ok"); + return engine; + } finally { $("ai-load").disabled = false; } + } + async function probeLocal() { + persist(); + st("Checking " + settings.url + "…", "busy"); + const r = await fetch(settings.url + "/models", { signal: AbortSignal.timeout(6000) }).catch((e) => { throw new Error(`Cannot reach ${settings.url}: ${e.message}. Is the runtime running and is this origin allowed (e.g. OLLAMA_ORIGINS=${location.origin})?`); }); + if (!r.ok) throw new Error(`Endpoint answered ${r.status}`); + const j = await r.json().catch(() => ({})); + const ids = (j.data || []).map((m) => m.id).filter(Boolean); + $("ai-lmodels").innerHTML = ids.map((id) => `