From a63721f1e5bc7dcc5ca18603e6431a8327edb8a3 Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 3 Oct 2026 01:51:36 +0000 Subject: [PATCH] fix: local model switching, mid-run settings, deterministic chess endings; release 0.0.11 - useWebLLMModel tracks the current modelId: switch it and model / ready / loading / progress / error describe the new one (not loaded yet) instead of the old model, which kept `ready` true and hid the load button. The previous model stays in memory (loadedModelId; switching back is instant) until the next load frees its GPU memory first, or unload(). A load superseded by another one frees its engine; a second load() of the same id shares the download. - useAgent: a rebuild while a run is in flight (a setting changed mid-run) no longer flips the status out of 'running'; the run keeps its agent, the next run gets the new one, and the run's end restores the build's own status. - Thinking: a 'none' level ("No thinking") in the composer chip and labels; with agent-web 0.0.21 it turns a local Qwen3's thinking off. - Demo: the model card reopens for a model that needs a key or a load, names the local model still in GPU memory, and can unload it. Chess outcomes come from the rules, not the model: a mate / stalemate / draw ends the game on the spot (a mating move is never answered by the agent), with the result over the board and in the chat; a turn the agent ends without moving (API error, quota, stop, limit) is reported with its reason and the board stays locked until it moves (Retry or the engine). - e2e: chess endings, a failed agent turn, a mid-run settings change. - Depends on @dudko.dev/agent-web ^0.0.21 (dev + demo); peer stays >=0.0.20. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01A7jHa5Pim4ufrdgxg561G5 --- README.md | 6 + demo/README.md | 13 +- demo/package-lock.json | 8 +- demo/package.json | 2 +- demo/src/App.tsx | 68 ++++++++- demo/src/app.css | 47 ++++++ demo/src/chess/ChessBoard.tsx | 7 +- demo/src/chess/ChessPanel.tsx | 46 +++--- demo/src/chess/game.ts | 67 ++++++++- demo/src/components/AgentSettingsPanel.tsx | 1 + demo/src/components/Settings.tsx | 37 +++-- demo/src/i18n.ts | 8 +- demo/src/settings.ts | 3 +- docs/chat.md | 2 +- e2e/agent.spec.ts | 100 ++++++++++++ e2e/gemini.ts | 26 +++- package-lock.json | 12 +- package.json | 4 +- src/components/AgentComposer.tsx | 1 + src/hooks/use-agent.ts | 21 ++- src/hooks/use-webllm-model.ts | 167 +++++++++++++++------ src/labels.tsx | 11 +- 22 files changed, 542 insertions(+), 115 deletions(-) diff --git a/README.md b/README.md index a1603b8..5e304b1 100644 --- a/README.md +++ b/README.md @@ -223,6 +223,12 @@ function LocalAgent() { > not silently on the first message) and `ready` flips only once the model can > actually answer. The `create` option is what makes it work under a bundler — > see below. +> +> Change the `modelId` and everything the hook reports — `model`, `ready`, +> `loading`, `progress`, `error` — is about the new one, which is not loaded +> yet. The previous model stays in memory (`loadedModelId`; switching back is +> instant) until the new one is loaded — that frees it first, so two models +> never share the GPU — or until `unload()`. ## Models in a bundler (Vite, Next, CRA) diff --git a/demo/README.md b/demo/README.md index 4f0382f..41d586b 100644 --- a/demo/README.md +++ b/demo/README.md @@ -25,11 +25,15 @@ Three tabs, sharing one "Agent settings" panel: - **Chess vs agent** — you play White; **your move is the trigger**: the agent reads the position with `get_position`, weighs candidates (`evaluate_moves`, or analyst **subagents** — each in its own **Web Worker**, or in-process), - and plays with `make_move`. No chat message needed. + and plays with `make_move`. No chat message needed. The rules, not the + model, decide the outcome: a mate, stalemate or draw ends the game on the + spot (a mating move is never answered), and a turn the agent ends without + moving — an API error, a spent quota, a stop — is shown as such, with the + board locked until it moves (Retry, or the built-in engine). The **Agent settings** panel drives the core's features live: tool consent (⚡ autopilot / ask before changes / ask everything / read-only), thinking -level, token / tool-call / step limits, context window and auto-compaction, +level ("none" turns a local Qwen's thinking off), token / tool-call / step limits, context window and auto-compaction, skills (each tab lists its own built-in examples; your own or imported `SKILL.md` ones apply to every tab), the chess analysts — and the **chat UI**: tool calls, token usage, saved chats (IndexedDB), the files panel, theme, and the labels (English / Russian — every @@ -40,6 +44,11 @@ and in total, thoughts, subagents and consent prompts; the composer has attachments, "/" commands, a run timer, the running-agents count, the model + thinking chip, the consent chip and a speech-to-text mic. +Local models: pick one and press "Download & load". Switching to another one +shows it as not loaded (the loaded one stays in memory, so switching back is +instant); loading it frees the previous model's GPU memory first, and +"Unload" frees it on demand. + **Live:** https://dudko-dev.github.io/agent-web-react/ ## Run locally diff --git a/demo/package-lock.json b/demo/package-lock.json index 4bbb7c5..e2341b8 100644 --- a/demo/package-lock.json +++ b/demo/package-lock.json @@ -12,7 +12,7 @@ "@ai-sdk/google": "^4.0.87", "@ai-sdk/openai": "^4.0.83", "@browser-ai/web-llm": "^3.0.4", - "@dudko.dev/agent-web": "^0.0.20", + "@dudko.dev/agent-web": "^0.0.21", "@dudko.dev/pdf-to-md-core": "^0.3.5", "@mlc-ai/web-llm": "^0.2.85", "@modelcontextprotocol/sdk": "^1.32.0", @@ -137,9 +137,9 @@ } }, "node_modules/@dudko.dev/agent-web": { - "version": "0.0.20", - "resolved": "https://registry.npmjs.org/@dudko.dev/agent-web/-/agent-web-0.0.20.tgz", - "integrity": "sha512-p2Ot3CxkZgSxrHVPm4hhsUZ0YWo6KOgmev6tpLSdkPpDLdh03y7GtO9st4LtWiJ9YEpJyN++nLYCmQ3r8kv9Zg==", + "version": "0.0.21", + "resolved": "https://registry.npmjs.org/@dudko.dev/agent-web/-/agent-web-0.0.21.tgz", + "integrity": "sha512-UDcJia0LXbZD3LIFwNICvdc/zdFMRN87YNZly/Ir4NAtUstK2xF9D7y/gwS9pd0jF3AyyB+EsZDzVO0BPF1j0w==", "funding": [ { "type": "individual", diff --git a/demo/package.json b/demo/package.json index e93e406..2cea388 100644 --- a/demo/package.json +++ b/demo/package.json @@ -14,7 +14,7 @@ "@ai-sdk/google": "^4.0.87", "@ai-sdk/openai": "^4.0.83", "@browser-ai/web-llm": "^3.0.4", - "@dudko.dev/agent-web": "^0.0.20", + "@dudko.dev/agent-web": "^0.0.21", "@dudko.dev/pdf-to-md-core": "^0.3.5", "@mlc-ai/web-llm": "^0.2.85", "@modelcontextprotocol/sdk": "^1.32.0", diff --git a/demo/src/App.tsx b/demo/src/App.tsx index b7fce8e..4c9d985 100644 --- a/demo/src/App.tsx +++ b/demo/src/App.tsx @@ -1,4 +1,4 @@ -import { useCallback, useEffect, useMemo, useState } from 'react' +import { useCallback, useEffect, useMemo, useRef, useState } from 'react' import type { Square } from 'chess.js' import type { BrowserAgentConfig, ModelInput } from '@dudko.dev/agent-web' import { @@ -18,7 +18,7 @@ import { import type { LanguageModel } from 'ai' import { ANALYST_PROMPT, engineTools } from './chess/analyst-tools' import { ChessPanel } from './chess/ChessPanel' -import { useChessGame } from './chess/game' +import { outcomeText, useChessGame } from './chess/game' import { AgentSettingsPanel } from './components/AgentSettingsPanel' import { McpPanel } from './components/McpPanel' import { NotesBoard } from './components/NotesBoard' @@ -190,6 +190,8 @@ export const App = () => { // ── Tab 3: chess against the agent, with analyst subagents ───────────────── const game = useChessGame() + // The last step error of the agent's turn (see playAgentTurn). + const chessErrorRef = useRef(undefined) const workerAnalystsBlocked = settings.analysts === 'worker' && local const analysts = useMemo(() => { if (settings.analysts === 'off' || !resolvedModel) return {} as AgentToolSet @@ -252,22 +254,59 @@ export const App = () => { maxPlanSteps: 1, replan: false, }, - { deps: [...deps, analysts, skillsByTab.chess] }, + { + deps: [...deps, analysts, skillsByTab.chess], + // A step error (quota, network) doesn't end the run; keep it for the board. + onEvent: (e) => { + if (e.type === 'error') chessErrorRef.current = e.error + }, + }, ) + // Whether the agent moved is read off the board, never off its words: a turn + // that ends with Black still to move is a failed turn, and the board stays + // locked until it does move (Retry, or the engine). + const [chessIssue, setChessIssue] = useState() const chessRun = chessAgent.run + const { current: chessNow, agentColor } = game + const playAgentTurn = useCallback( + async (goal: string) => { + setChessIssue(undefined) + chessErrorRef.current = undefined + const result = await chessRun(goal) + const now = chessNow() + if (now.outcome || now.turn !== agentColor) return // it moved + const error = chessErrorRef.current + setChessIssue( + result === undefined && !error + ? 'The agent could not start its turn (still loading, or busy).' + : result?.stopped + ? 'The agent’s turn was stopped before it moved.' + : result?.budgetExceeded + ? `The agent hit its ${result.budgetExceeded.kind} limit before moving.` + : error + ? `The agent failed: ${error}` + : 'The agent ended its turn without making a move.', + ) + }, + [chessRun, chessNow, agentColor], + ) const onUserMove = useCallback( (from: Square, to: Square) => { const san = game.userMove(from, to) if (!san) return + // The rules decide the end: a mating (or drawing) move ends the game here, + // and the agent isn't asked to play on. + if (game.current().outcome) return // The move IS the trigger: no chat message, the agent just answers it. - void chessRun(`White played ${san}. Your move.`) + void playAgentTurn(`White played ${san}. Your move.`) }, - [game, chessRun], + [game, playAgentTurn], ) const onNewGame = () => { game.newGame() chessAgent.reset() + setChessIssue(undefined) } // ── Routing & one-model-at-a-time ────────────────────────────────────────── @@ -430,6 +469,18 @@ export const App = () => { composer={composer('Chat with your opponent — or just make a move on the board')} {...chatCommon} files={undefined} + // The game's own verdict, after whatever the model said. + slots={{ + afterMessages: game.outcome ? ( +
+ 🏁 {outcomeText(game.outcome)} +
+ ) : chessIssue && !chessAgent.isRunning ? ( +
+ ⚠ {chessIssue} The board waits for its move — Retry or Engine move. +
+ ) : undefined, + }} /> )} @@ -462,8 +513,11 @@ export const App = () => { } onAnalysts={(analysts) => update({ analysts })} onUserMove={onUserMove} - onAskAgent={() => void chessRun("It's your move.")} - onEngineMove={() => void game.engineMove()} + agentIssue={chessIssue} + onAskAgent={() => void playAgentTurn("It's your move.")} + onEngineMove={() => { + if (game.engineMove()) setChessIssue(undefined) + }} onNewGame={onNewGame} /> )} diff --git a/demo/src/app.css b/demo/src/app.css index 39c79f9..2786be5 100644 --- a/demo/src/app.css +++ b/demo/src/app.css @@ -276,6 +276,12 @@ a { color: #16a34a; font-size: 13px; } +.settings__row { + display: flex; + align-items: center; + justify-content: space-between; + gap: 8px; +} /* ── Board ────────────────────────────────────────────────────────────────── */ .board { @@ -803,6 +809,7 @@ a { overflow: hidden; border: 1px solid var(--border); user-select: none; + position: relative; } .cboard__sq { position: relative; @@ -889,6 +896,46 @@ a { gap: 8px; flex-wrap: wrap; } +.chess__agent-turn--issue > [role='alert'] { + color: #b45309; +} +/* The game's verdict in the chat, after the model's last words. */ +/* Inside the chat, so it follows the chat's theme tokens (light / dark). */ +.chess-note { + margin: 4px 0 8px; + padding: 8px 12px; + border-radius: 10px; + font-size: 13px; + color: var(--awr-fg); + border: 1px solid var(--awr-border); + background: var(--awr-subtle); +} +.chess-note--result { + border-color: var(--awr-accent); + font-weight: 600; +} +.chess-note--issue { + border-color: var(--awr-warn); +} +/* The game result covers the board: the rules decided it, the board is done. */ +.cboard__result { + position: absolute; + inset: 0; + display: flex; + flex-direction: column; + align-items: center; + justify-content: center; + gap: 14px; + padding: 16px; + text-align: center; + background: rgba(15, 23, 42, 0.55); + color: #fff; + font-size: clamp(16px, 4.5cqi, 26px); + backdrop-filter: blur(1px); +} +.cboard__result .settings__btn { + font-size: 15px; +} .agentset__peek { font-weight: 400; font-size: 12px; diff --git a/demo/src/chess/ChessBoard.tsx b/demo/src/chess/ChessBoard.tsx index b0ad109..df7560b 100644 --- a/demo/src/chess/ChessBoard.tsx +++ b/demo/src/chess/ChessBoard.tsx @@ -1,4 +1,4 @@ -import { useState } from 'react' +import { useState, type ReactNode } from 'react' import type { Square } from 'chess.js' import type { ChessGame } from './game' @@ -24,10 +24,12 @@ export interface ChessBoardProps { /** The user can move (their turn, agent idle). */ interactive: boolean onMove: (from: Square, to: Square) => void + /** Covers the board (the game result). */ + overlay?: ReactNode } /** A click-to-move board, White at the bottom. Click a piece, then a highlighted square. */ -export const ChessBoard = ({ game, interactive, onMove }: ChessBoardProps) => { +export const ChessBoard = ({ game, interactive, onMove, overlay }: ChessBoardProps) => { const [selected, setSelected] = useState() const targets = selected ? new Set(game.targets(selected)) : new Set() @@ -81,6 +83,7 @@ export const ChessBoard = ({ game, interactive, onMove }: ChessBoardProps) => { ) }), )} + {overlay} ) } diff --git a/demo/src/chess/ChessPanel.tsx b/demo/src/chess/ChessPanel.tsx index ba3ef1c..554f65e 100644 --- a/demo/src/chess/ChessPanel.tsx +++ b/demo/src/chess/ChessPanel.tsx @@ -1,13 +1,15 @@ import type { Square } from 'chess.js' import type { AnalystMode } from '../settings' import { ChessBoard } from './ChessBoard' -import type { ChessGame } from './game' +import { outcomeText, type ChessGame } from './game' export interface ChessPanelProps { game: ChessGame /** The agent is thinking about / playing its move. */ agentBusy: boolean agentReady: boolean + /** Why the agent's last turn ended without a move (an error, a refusal, a stop). */ + agentIssue?: string analysts: AnalystMode /** Why worker analysts are unavailable (e.g. a local model), if they are. */ analystsNote?: string @@ -20,21 +22,20 @@ export interface ChessPanelProps { onNewGame: () => void } -const RESULT: Record = { - checkmate: 'Checkmate', - stalemate: 'Stalemate', - draw: 'Draw', -} - /** * The board, the move list and the controls. Every user move TRIGGERS the * agent: the app calls `agent.run("White played …")`, and the agent answers by * calling `make_move` — no chat message needed. + * + * Nothing here trusts the model's words: the game ends when the rules say so + * (the result covers the board), and a turn the agent ended without moving is + * shown as such — the board stays locked until it moves (retry, or the engine). */ export const ChessPanel = ({ game, agentBusy, agentReady, + agentIssue, analysts, analystsNote, onAnalysts, @@ -43,7 +44,7 @@ export const ChessPanel = ({ onEngineMove, onNewGame, }: ChessPanelProps) => { - const over = game.status !== 'playing' + const over = game.outcome !== undefined const agentTurn = game.turn === game.agentColor && !over const pairs: string[] = [] for (let i = 0; i < game.history.length; i += 2) { @@ -55,24 +56,21 @@ export const ChessPanel = ({ return (
- {over ? ( - - {RESULT[game.status]} - {game.winner ? ` — ${game.winner === 'w' ? 'you win' : 'the agent wins'}` : ''} - + {game.outcome ? ( + {outcomeText(game.outcome)} ) : agentTurn ? ( agentBusy ? ( The agent is choosing its move… ) : ( - - Agent to move. + + {agentIssue ? ⚠ {agentIssue} : 'Agent to move.'}
- + + {outcomeText(game.outcome)} + +
+ ) + } + />
+
) : ( - + <> + {held && ( +

+ {held} is in GPU memory — loading this one frees it first. +

+ )} + + )} ) : ( diff --git a/demo/src/i18n.ts b/demo/src/i18n.ts index ec5e160..3951497 100644 --- a/demo/src/i18n.ts +++ b/demo/src/i18n.ts @@ -92,7 +92,13 @@ export const RU_LABELS: AgentLabelsOverride = { readOnly: 'Только чтение, изменения — отказ', think: (levels) => `Уровень размышлений: ${levels}`, }, - thinkingLevels: { off: 'По умолчанию', low: 'Низкий', medium: 'Средний', high: 'Высокий' }, + thinkingLevels: { + off: 'По умолчанию', + none: 'Без размышлений', + low: 'Низкий', + medium: 'Средний', + high: 'Высокий', + }, filesEmpty: 'Пока пусто. Здесь появятся вложения и файлы, которые пишет агент.', upload: 'Загрузить', download: 'Скачать', diff --git a/demo/src/settings.ts b/demo/src/settings.ts index 963520d..64c1f4f 100644 --- a/demo/src/settings.ts +++ b/demo/src/settings.ts @@ -4,7 +4,8 @@ import { BUILTIN_SKILLS, SKILL_TABS } from './skills' export type View = 'notes' | 'mcp' | 'chess' -export type ThinkingChoice = 'off' | 'low' | 'medium' | 'high' +/** 'off' leaves the provider default (a local Qwen3 thinks); 'none' turns it off. */ +export type ThinkingChoice = 'off' | 'none' | 'low' | 'medium' | 'high' export type AnalystMode = 'off' | 'worker' | 'in-process' /** Everything the "Agent settings" panel controls, shared by every tab. */ diff --git a/docs/chat.md b/docs/chat.md index 11cbc52..d1e6d50 100644 --- a/docs/chat.md +++ b/docs/chat.md @@ -103,7 +103,7 @@ full size) and file chips. The header shows the conversation's total. | Prop | | | --- | --- | | `model` | `{ label, options?, value?, onSelect? }` — chip and switcher | -| `thinking` | `{ value, options?, onChange }` — shown next to the model | +| `thinking` | `{ value, options?, onChange }` — shown next to the model; default levels `off` (the provider's default), `none` (thinking off), `low`, `medium`, `high` | | `commands` / `builtinCommands` / `skillCommands` | slash commands | | `attachments` | `false`, or `{ accept, maxBytes, imageMaxDimension, imageMaxPixels }` | | `convertFile` / `convert` | file → text (e.g. PDF → Markdown), `'when-needed'` or `'always'` | diff --git a/e2e/agent.spec.ts b/e2e/agent.spec.ts index 33efc1c..4444ee0 100644 --- a/e2e/agent.spec.ts +++ b/e2e/agent.spec.ts @@ -154,6 +154,106 @@ test('chess: the user’s move triggers the agent, which answers with its own', await expect(page.getByRole('button', { name: 'e5 black p' })).toBeVisible() }) +/** Black answers the user's moves with these, one per turn. */ +const blackPlays = (moves: string[]) => { + let next = 0 + return (c: GeminiCall): GeminiReply => { + if (c.stage === 'planner') return plan('Answer the move') + if (c.stage === 'executor') { + return c.functionResponses.length === 0 + ? { call: { name: 'make_move', args: { move: moves[next++] } } } + : { text: 'Played.' } + } + return { text: 'Your move.' } + } +} + +const playWhite = async (page: Page, from: string, to: string) => { + await page.locator(`[data-square="${from}"]`).click() + await page.locator(`[data-square="${to}"]`).click() +} + +test('chess: the rules end the game — a mating move is not answered by the agent', async ({ + page, +}) => { + const calls = await mockGemini(page, blackPlays(['e5', 'Nc6', 'Nf6'])) + await openWithKey(page, 'Chess vs agent') + // Scholar's mate: 1. e4 e5 2. Bc4 Nc6 3. Qh5 Nf6 4. Qxf7# + for (const [from, to, reply] of [ + ['e2', 'e4', 'e5'], + ['f1', 'c4', 'c6'], + ['d1', 'h5', 'f6'], + ]) { + await playWhite(page, from, to) + await expect(page.locator(`[data-square="${reply}"]`)).toHaveAttribute('aria-label', / black /) + await expect(page.getByText('Your move (White).')).toBeVisible() + } + const planned = calls.filter((c) => c.stage === 'planner').length + await playWhite(page, 'h5', 'f7') + + await expect(page.locator('.cboard__result')).toContainText('Checkmate — you win (1-0)') + await expect(page.locator('.chess-note--result')).toContainText('Checkmate — you win') + await expect(page.locator('.cboard__sq.is-movable')).toHaveCount(0) + expect(calls.filter((c) => c.stage === 'planner').length).toBe(planned) + + await page.locator('.cboard__result').getByRole('button', { name: 'New game' }).click() + await expect(page.locator('.cboard__result')).toHaveCount(0) + await expect(page.getByText('Your move (White).')).toBeVisible() +}) + +test('chess: a turn the agent fails keeps the board locked until it moves', async ({ page }) => { + let quota = false + const black = blackPlays(['e5']) + await mockGemini(page, (c) => + c.stage === 'executor' && !quota + ? { error: { status: 429, message: 'Resource has been exhausted (e.g. check quota).' } } + : black(c), + ) + await openWithKey(page, 'Chess vs agent') + await playWhite(page, 'e2', 'e4') + + const status = page.locator('.chess__status') + await expect(status).toContainText('The agent failed:') + await expect(status).toContainText('exhausted') + await expect(page.locator('.chess-note--issue')).toBeVisible() + await expect(page.locator('.cboard__sq.is-movable')).toHaveCount(0) + + quota = true + await status.getByRole('button', { name: 'Retry' }).click() + await expect(page.locator('[data-square="e5"]')).toHaveAttribute('aria-label', / black /) + await expect(page.getByText('Your move (White).')).toBeVisible() + await expect(page.locator('.chess-note--issue')).toHaveCount(0) +}) + +test('a setting changed mid-run keeps the run going and applies to the next one', async ({ + page, +}) => { + const calls = await mockGemini(page, (c) => + c.stage === 'planner' + ? plan('Say hi') + : c.stage === 'executor' + ? { text: 'Hi.', delayMs: 1500 } + : { text: 'Hi there.' }, + ) + await openWithKey(page) + await send(page, 'hello') + await expect(page.locator('.awr-sendbtn--stop')).toBeVisible() + await page.locator('.agentset__title').click() + await page.locator('.agentset select').first().selectOption('high') + // The rebuild doesn't flip the panel out of "running". + await expect(page.locator('.awr-sendbtn--stop')).toBeVisible() + await expect(page.locator('.awr-msg--assistant .awr-msg__bubble')).toHaveText('Hi there.') + + // The run in flight kept the agent it started with… + const first = calls.splice(0) + expect(first.map((c) => c.thinkingLevel)).toEqual([undefined, undefined, undefined]) + await send(page, 'again') + await expect(page.locator('.awr-msg--assistant .awr-msg__bubble').nth(1)).toHaveText('Hi there.') + // …and the next run is built with the new setting. + expect(calls.length).toBe(3) + expect(calls.every((c) => c.thinkingLevel === 'high')).toBe(true) +}) + test('labels: every text of the chat can be swapped (Russian)', async ({ page }) => { await mockGemini(page, () => ({ text: 'ok' })) await openWithKey(page) diff --git a/e2e/gemini.ts b/e2e/gemini.ts index 5436f4b..605af44 100644 --- a/e2e/gemini.ts +++ b/e2e/gemini.ts @@ -16,11 +16,20 @@ export interface GeminiCall { functionResponses: { name: string; response: unknown }[] /** Inline files sent (images, PDFs). */ inlineData: { mimeType: string; bytes: number }[] + /** generationConfig.thinkingConfig.thinkingLevel, when the call asked for thinking. */ + thinkingLevel?: string streaming: boolean } -export type GeminiReply = - { text: string } | { call: { name: string; args: Record } } +export type GeminiReply = ( + | { text: string } + | { call: { name: string; args: Record } } + /** An HTTP error from the API (e.g. 429 when the quota is spent). */ + | { error: { status: number; message: string } } +) & { + /** Hold the answer this long (a slow model, to act mid-run). */ + delayMs?: number +} type Part = { text?: string @@ -31,6 +40,7 @@ type Part = { type Body = { systemInstruction?: { parts?: Part[] } contents?: { role: string; parts?: Part[] }[] + generationConfig?: { thinkingConfig?: { thinkingLevel?: string } } } const stageOf = (system: string): GeminiCall['stage'] => { @@ -74,10 +84,22 @@ export const mockGemini = async ( inlineData: parts .filter((p) => p.inlineData) .map((p) => ({ mimeType: p.inlineData!.mimeType, bytes: p.inlineData!.data.length })), + thinkingLevel: body.generationConfig?.thinkingConfig?.thinkingLevel, streaming: req.url().includes(':streamGenerateContent'), } calls.push(call) const reply = script(call) + if (reply.delayMs) await new Promise((r) => setTimeout(r, reply.delayMs)) + if ('error' in reply) { + await route.fulfill({ + status: reply.error.status, + headers: { 'content-type': 'application/json', 'access-control-allow-origin': '*' }, + body: JSON.stringify({ + error: { code: reply.error.status, message: reply.error.message, status: 'ERROR' }, + }), + }) + return + } const candidate = { content: { role: 'model', diff --git a/package-lock.json b/package-lock.json index e2f48c2..fb67e14 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,12 +1,12 @@ { "name": "@dudko.dev/agent-web-react", - "version": "0.0.10", + "version": "0.0.11", "lockfileVersion": 3, "requires": true, "packages": { "": { "name": "@dudko.dev/agent-web-react", - "version": "0.0.10", + "version": "0.0.11", "funding": [ { "type": "individual", @@ -27,7 +27,7 @@ ], "license": "MIT", "devDependencies": { - "@dudko.dev/agent-web": "^0.0.20", + "@dudko.dev/agent-web": "^0.0.21", "@modelcontextprotocol/sdk": "^1.30.1", "@playwright/test": "^1.63.0", "@types/node": "^22.9.0", @@ -100,9 +100,9 @@ } }, "node_modules/@dudko.dev/agent-web": { - "version": "0.0.20", - "resolved": "https://registry.npmjs.org/@dudko.dev/agent-web/-/agent-web-0.0.20.tgz", - "integrity": "sha512-p2Ot3CxkZgSxrHVPm4hhsUZ0YWo6KOgmev6tpLSdkPpDLdh03y7GtO9st4LtWiJ9YEpJyN++nLYCmQ3r8kv9Zg==", + "version": "0.0.21", + "resolved": "https://registry.npmjs.org/@dudko.dev/agent-web/-/agent-web-0.0.21.tgz", + "integrity": "sha512-UDcJia0LXbZD3LIFwNICvdc/zdFMRN87YNZly/Ir4NAtUstK2xF9D7y/gwS9pd0jF3AyyB+EsZDzVO0BPF1j0w==", "dev": true, "funding": [ { diff --git a/package.json b/package.json index b351371..1baecbd 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "@dudko.dev/agent-web-react", - "version": "0.0.10", + "version": "0.0.11", "description": "React bindings for @dudko.dev/agent-web: a headless useAgent hook, an AgentProvider context, and optional pre-styled components (chat panel, plan/step view, BYOK key form, WebLLM load bar) that connect the in-browser LLM agent to any React site. UI you can drop in — or a headless reducer you can build your own around.", "type": "module", "sideEffects": [ @@ -110,7 +110,7 @@ "react-dom": ">=18" }, "devDependencies": { - "@dudko.dev/agent-web": "^0.0.20", + "@dudko.dev/agent-web": "^0.0.21", "@modelcontextprotocol/sdk": "^1.30.1", "@playwright/test": "^1.63.0", "@types/node": "^22.9.0", diff --git a/src/components/AgentComposer.tsx b/src/components/AgentComposer.tsx index 7717016..3e51093 100644 --- a/src/components/AgentComposer.tsx +++ b/src/components/AgentComposer.tsx @@ -133,6 +133,7 @@ export interface AgentComposerProps { const thinkingDefaults = (l: AgentLabels) => [ { id: 'off', label: l.thinkingLevels.off }, + { id: 'none', label: l.thinkingLevels.none }, { id: 'low', label: l.thinkingLevels.low }, { id: 'medium', label: l.thinkingLevels.medium }, { id: 'high', label: l.thinkingLevels.high }, diff --git a/src/hooks/use-agent.ts b/src/hooks/use-agent.ts index db0947f..aa61d5e 100644 --- a/src/hooks/use-agent.ts +++ b/src/hooks/use-agent.ts @@ -164,6 +164,16 @@ export const useAgent = ( const reload = useCallback(() => setGeneration((g) => g + 1), []) + // The build's own status. A rebuild while a run is in flight (a setting + // changed mid-run) must not flip the UI out of 'running': the run keeps the + // agent it started with, and the new one takes the next run. + type BuildStatus = { status: 'idle' | 'initializing' | 'ready' | 'error'; error?: string } + const buildStatusRef = useRef({ status: 'idle' }) + const setBuildStatus = useCallback((next: BuildStatus) => { + buildStatusRef.current = next + if (!abortRef.current) dispatch({ type: 'status', ...next }) + }, []) + // Build (and rebuild) the agent. Model resolution is async (dynamic provider // imports, vault key fetch, WebLLM weight download), so this lives in effect. const { autoStart, deps } = options @@ -176,7 +186,7 @@ export const useAgent = ( // undefined), in which case we sit idle rather than build a broken agent. const shouldBuild = (autoStart !== false || generation > 0) && Boolean(cfg.model) if (!shouldBuild) { - dispatch({ type: 'status', status: 'idle' }) + setBuildStatus({ status: 'idle' }) return } // Consent requests come to the hook unless the host handles them itself. @@ -188,17 +198,17 @@ export const useAgent = ( } : cfg let cancelled = false - dispatch({ type: 'status', status: 'initializing' }) + setBuildStatus({ status: 'initializing' }) createAgent(built) .then((agent) => { if (cancelled) return agent.setToolApprovalMode(modeRef.current) agentRef.current = agent - dispatch({ type: 'status', status: 'ready' }) + setBuildStatus({ status: 'ready' }) }) .catch((err) => { if (cancelled) return - dispatch({ type: 'status', status: 'error', error: errMessage(err) }) + setBuildStatus({ status: 'error', error: errMessage(err) }) }) return () => { cancelled = true @@ -259,7 +269,8 @@ export const useAgent = ( } finally { abortRef.current = undefined settleAll({ approved: false, reason: 'the run ended' }) - dispatch({ type: 'status', status: 'ready' }) + // Whatever the agent did meanwhile: rebuilt, still building, or gone. + dispatch({ type: 'status', ...buildStatusRef.current }) } }, [settleAll], diff --git a/src/hooks/use-webllm-model.ts b/src/hooks/use-webllm-model.ts index 189a5db..feae798 100644 --- a/src/hooks/use-webllm-model.ts +++ b/src/hooks/use-webllm-model.ts @@ -3,6 +3,7 @@ import { createWebLLMModel, isWebGPUAvailable, preloadWebLLMModel, + unloadWebLLMModel, type WebLLMModelOptions, } from '@dudko.dev/agent-web' import { errMessage } from '../util.js' @@ -39,24 +40,41 @@ export interface UseWebLLMModelOptions extends WebLLMModelOptions { } export interface UseWebLLMModelReturn { - /** The built model once loaded — pass to `createAgent({ model })`. */ + /** + * The built model, once **this** `modelId` is loaded — pass to + * `createAgent({ model })`. Undefined right after the id changes, until the + * new one is loaded. + */ model: WebLLMModel | undefined - /** Start the download / engine init. Idempotent-ish: safe to call again to retry. */ + /** + * Download and initialize the current `modelId`. Resolves with the model + * already loaded for it, if any. Loading another id first frees the previous + * model's GPU memory (one engine per hook); call again after an error to retry. + */ load: () => Promise - /** True while weights are downloading / the engine is initializing. */ + /** Free the loaded model's GPU memory (and drop a load in flight). */ + unload: () => Promise + /** The id of the model held in memory, whichever id is current. */ + loadedModelId: string | undefined + /** True while the current `modelId` is downloading / initializing. */ loading: boolean - /** Load progress, 0..1. */ + /** Load progress of the current `modelId`, 0..1. */ progress: number /** Human-readable progress text from WebLLM. */ text: string - /** Error message if the load failed. */ + /** Why the current `modelId` failed to load. */ error: string | undefined /** Whether WebGPU is available (required for local models). */ supported: boolean - /** True once the model is ready. */ + /** True once the current `modelId` is ready. */ ready: boolean } +interface Loaded { + id: string + model: WebLLMModel +} + /** * Load a local WebGPU model with WebLLM and track its download progress. * Nothing downloads until you call `load()` (models are large), so you can @@ -64,6 +82,12 @@ export interface UseWebLLMModelReturn { * the core's `preloadWebLLMModel`), so `ready` means "ready to chat" and * progress fills during the load rather than silently on the first message. * + * Everything it reports is about the **current** `modelId`: switch the id and + * `model` / `ready` / `progress` / `error` describe the new one (not loaded + * yet), while the previous model stays in memory — switching back is instant. + * Loading the new id frees the previous one first, so two models never hold + * GPU memory at once; `unload()` frees it on demand. + * * ```tsx * import { webLLM } from '@browser-ai/web-llm' // your app's optional peer * const create = (id: string, opts?: WebLLMModelOptions) => Promise.resolve(webLLM(id, opts)) @@ -77,59 +101,104 @@ export const useWebLLMModel = ( modelId: string, options?: UseWebLLMModelOptions, ): UseWebLLMModelReturn => { - const [model, setModel] = useState(undefined) - const [loading, setLoading] = useState(false) - const [progress, setProgress] = useState(0) - const [text, setText] = useState('') - const [error, setError] = useState(undefined) + const [loaded, setLoaded] = useState(undefined) + const [pending, setPending] = useState<{ id: string; progress: number; text: string }>() + const [failure, setFailure] = useState<{ id: string; message: string }>() const optionsRef = useRef(options) optionsRef.current = options + // The source of truth for async code (state lags a render behind). + const loadedRef = useRef(undefined) + const inflightRef = useRef<{ id: string; promise: Promise } | undefined>( + undefined, + ) + // Bumped by every load()/unload(): an older load that finishes late is stale. + const seqRef = useRef(0) + + const setHeld = (next: Loaded | undefined) => { + loadedRef.current = next + setLoaded(next) + } - const load = useCallback(async (): Promise => { + const load = useCallback((): Promise => { + const id = modelId + if (loadedRef.current?.id === id) return Promise.resolve(loadedRef.current.model) + if (inflightRef.current?.id === id) return inflightRef.current.promise if (!isWebGPUAvailable()) { - setError('WebGPU is not available in this browser.') - return undefined - } - setLoading(true) - setError(undefined) - setProgress(0) - setText('') - try { - const { create = createWebLLMModel, ...modelOptions } = optionsRef.current ?? {} - const built = await create(modelId, { - ...modelOptions, - // Drive preload here (below) so download progress is reported the same - // way whether `create` is the core's `createWebLLMModel` or an injected - // factory (e.g. a statically-imported `webLLM`, needed under bundlers). - preload: false, - initProgressCallback: (report) => { - setProgress(report.progress) - setText(report.text) - modelOptions.initProgressCallback?.(report) - }, - }) - // Download the weights + init the engine now via the core's helper (a - // 1-token warm-up). WebLLM builds are otherwise lazy — the ~GB download - // would only start on the first `run()`, long after we told the UI the - // model is "ready". Fast + idempotent once the weights are cached. - await preloadWebLLMModel(built) - setModel(built) - return built - } catch (err) { - setError(errMessage(err)) - return undefined - } finally { - setLoading(false) + setFailure({ id, message: 'WebGPU is not available in this browser.' }) + return Promise.resolve(undefined) } + const seq = ++seqRef.current + const current = () => seq === seqRef.current + const promise = (async (): Promise => { + setPending({ id, progress: 0, text: '' }) + setFailure(undefined) + let built: WebLLMModel | undefined + try { + // One engine per hook: free the previous model before the next download. + const previous = loadedRef.current + if (previous) { + setHeld(undefined) + await unloadWebLLMModel(previous.model) + } + const { create = createWebLLMModel, ...modelOptions } = optionsRef.current ?? {} + built = await create(id, { + ...modelOptions, + // Drive preload here (below) so download progress is reported the same + // way whether `create` is the core's `createWebLLMModel` or an injected + // factory (e.g. a statically-imported `webLLM`, needed under bundlers). + preload: false, + initProgressCallback: (report) => { + if (current()) setPending({ id, progress: report.progress, text: report.text }) + modelOptions.initProgressCallback?.(report) + }, + }) + // Download the weights + init the engine now via the core's helper (a + // 1-token warm-up). WebLLM builds are otherwise lazy — the ~GB download + // would only start on the first `run()`, long after we told the UI the + // model is "ready". Fast + idempotent once the weights are cached. + await preloadWebLLMModel(built) + if (!current()) { + // Superseded by another load() or an unload(): don't leak its engine. + await unloadWebLLMModel(built) + return undefined + } + setHeld({ id, model: built }) + return built + } catch (err) { + if (built) await unloadWebLLMModel(built) + if (current()) setFailure({ id, message: errMessage(err) }) + return undefined + } finally { + if (current()) { + inflightRef.current = undefined + setPending(undefined) + } + } + })() + inflightRef.current = { id, promise } + return promise }, [modelId]) + const unload = useCallback(async (): Promise => { + seqRef.current++ // a load in flight becomes stale and frees itself + inflightRef.current = undefined + setPending(undefined) + const previous = loadedRef.current + setHeld(undefined) + if (previous) await unloadWebLLMModel(previous.model) + }, []) + + const model = loaded?.id === modelId ? loaded.model : undefined + const progressing = pending?.id === modelId ? pending : undefined return { model, load, - loading, - progress, - text, - error, + unload, + loadedModelId: loaded?.id, + loading: progressing !== undefined, + progress: progressing?.progress ?? 0, + text: progressing?.text ?? '', + error: failure?.id === modelId ? failure.message : undefined, supported: isWebGPUAvailable(), ready: model !== undefined, } diff --git a/src/labels.tsx b/src/labels.tsx index bf2b075..2a00f30 100644 --- a/src/labels.tsx +++ b/src/labels.tsx @@ -76,7 +76,8 @@ export interface AgentLabels { readOnly: string think: (levels: string) => string } - thinkingLevels: { off: string; low: string; medium: string; high: string } + /** `off` = the provider's default; `none` = thinking switched off. */ + thinkingLevels: { off: string; none: string; low: string; medium: string; high: string } // ── files panel ──────────────────────────────────────────────────────────── filesEmpty: string @@ -195,7 +196,13 @@ export const defaultLabels: AgentLabels = { readOnly: 'Only read; refuse changes', think: (levels) => `Set thinking: ${levels}`, }, - thinkingLevels: { off: 'Default', low: 'Low', medium: 'Medium', high: 'High' }, + thinkingLevels: { + off: 'Default', + none: 'No thinking', + low: 'Low', + medium: 'Medium', + high: 'High', + }, filesEmpty: 'No files yet. Attachments and files the agent writes appear here.', upload: 'Upload',