From c91db60504e8aa45ffe800f4c56ddceb99fd6686 Mon Sep 17 00:00:00 2001 From: Jack <2075649045@qq.com> Date: Mon, 21 Sep 2026 12:42:48 +0800 Subject: [PATCH 1/8] feat(images): add configured batch generation and editing Let conversations create and refine image assets through a separate image model without replacing the chat default. Ship an imagegen skill and preserve generated files and per-item outcomes across reloads. Authorize billable image calls through host-core, propagate cancellation, and bound requests, downloads, and local inputs. Cover configuration, batch failures, edits, permissions, and persistence with isolated tests. --- apps/desktop/electron/main/builtin-skills.ts | 36 ++-- apps/desktop/electron/main/runtime/sidecar.ts | 2 + .../main/services/image-generation-service.ts | 114 ++++++++++ .../electron/main/services/image-inputs.ts | 88 ++++++++ .../resources/skills/image-generation.md | 59 +++++ .../settings/ImageGenerationModelRow.tsx | 125 +++++++++++ .../components/settings/ModelConfigPage.tsx | 14 +- .../settings/ModelSelectionPanes.tsx | 5 + .../settings/ProviderSetupDialog.tsx | 11 +- .../chat/transcript/GeneratedImages.tsx | 92 ++++++++ .../src/features/chat/transcript/ToolRow.tsx | 3 + apps/desktop/src/lib/settings-search.ts | 1 + apps/desktop/src/styles/generated-images.css | 23 ++ apps/desktop/test/image-generation.test.mjs | 126 +++++++++++ apps/desktop/test/plugin-skills.test.mjs | 2 +- crates/host-core/src/permissions.rs | 42 +++- crates/host-core/src/rpc/mod.rs | 42 +++- crates/host-core/src/tools/mod.rs | 3 + docs/adr/README.md | 1 + docs/adr/image-generation-capability.md | 41 ++++ .../03-runtime/03-tools-and-permissions.md | 7 + .../13-model-catalog-and-selection.md | 7 + docs/spec/03-runtime/21-image-generation.md | 77 +++++++ docs/spec/03-runtime/README.md | 2 + docs/spec/06-delivery/04-e2e-test-plan.md | 26 ++- package.json | 1 + .../src/image-generation/download.ts | 130 +++++++++++ .../image-generation/image-generation.test.ts | 171 +++++++++++++++ .../src/image-generation/index.ts | 169 +++++++++++++++ .../src/image-generation/tool.ts | 19 ++ packages/agent-runtime/src/index.ts | 1 + packages/agent-runtime/src/runtime.ts | 7 +- packages/host-runtime/src/agent-sidecar.ts | 44 +++- .../src/image-generation-bridge.test.ts | 83 ++++++++ packages/i18n/src/locales/de/index.ts | 13 ++ packages/i18n/src/locales/en/index.ts | 13 ++ packages/i18n/src/locales/es/index.ts | 13 ++ packages/i18n/src/locales/fr/index.ts | 13 ++ packages/i18n/src/locales/ko/index.ts | 13 ++ packages/i18n/src/locales/tr/index.ts | 13 ++ packages/i18n/src/locales/zh-CN/index.ts | 13 ++ packages/i18n/src/locales/zh-TW/index.ts | 13 ++ packages/shared/src/changelog-de.ts | 1 + packages/shared/src/changelog-es.ts | 1 + packages/shared/src/changelog-fr.ts | 1 + packages/shared/src/changelog-ko.ts | 1 + packages/shared/src/changelog-tr.ts | 1 + packages/shared/src/changelog.ts | 3 + packages/shared/src/image-generation.ts | 75 +++++++ packages/shared/src/index.ts | 1 + packages/shared/src/rpc-timeouts.ts | 3 + packages/shared/src/types/settings.ts | 1 + scripts/e2e-image-generation-ui.mjs | 90 ++++++++ scripts/e2e-image-generation.mjs | 139 ++++++++++++ scripts/e2e/image-generation-ui.tsx | 201 ++++++++++++++++++ scripts/test-image-generation-live.mjs | 167 +++++++++++++++ 56 files changed, 2322 insertions(+), 41 deletions(-) create mode 100644 apps/desktop/electron/main/services/image-generation-service.ts create mode 100644 apps/desktop/electron/main/services/image-inputs.ts create mode 100644 apps/desktop/resources/skills/image-generation.md create mode 100644 apps/desktop/src/components/settings/ImageGenerationModelRow.tsx create mode 100644 apps/desktop/src/features/chat/transcript/GeneratedImages.tsx create mode 100644 apps/desktop/src/styles/generated-images.css create mode 100644 apps/desktop/test/image-generation.test.mjs create mode 100644 docs/adr/image-generation-capability.md create mode 100644 docs/spec/03-runtime/21-image-generation.md create mode 100644 packages/agent-runtime/src/image-generation/download.ts create mode 100644 packages/agent-runtime/src/image-generation/image-generation.test.ts create mode 100644 packages/agent-runtime/src/image-generation/index.ts create mode 100644 packages/agent-runtime/src/image-generation/tool.ts create mode 100644 packages/host-runtime/src/image-generation-bridge.test.ts create mode 100644 packages/shared/src/image-generation.ts create mode 100644 scripts/e2e-image-generation-ui.mjs create mode 100644 scripts/e2e-image-generation.mjs create mode 100644 scripts/e2e/image-generation-ui.tsx create mode 100644 scripts/test-image-generation-live.mjs diff --git a/apps/desktop/electron/main/builtin-skills.ts b/apps/desktop/electron/main/builtin-skills.ts index 9e7d5d71f6..3dcb12005a 100644 --- a/apps/desktop/electron/main/builtin-skills.ts +++ b/apps/desktop/electron/main/builtin-skills.ts @@ -15,13 +15,16 @@ import type { PluginSkillDef } from "@pi-desktop/agent-runtime"; /** Bundled skill teaching the plugin-development loop. */ export const PLUGIN_DEV_SKILL_FILE = "plugin-development.md"; export const PLUGIN_DEV_SKILL_ID = "pi-desktop/plugin-development"; +export const IMAGE_GENERATION_SKILL_ID = "pi-desktop/imagegen"; +const IMAGE_GENERATION_SKILL_FILE = "image-generation.md"; /** electron-builder copies `resources/skills` to `/skills`. */ function resolveBuiltinSkillPath(fileName: string): string | null { + const moduleDir = typeof __dirname === "string" ? __dirname : import.meta.dirname; const candidates = [ join(process.resourcesPath || "", "skills", fileName), - join(__dirname, "../../resources/skills", fileName), - join(__dirname, "../../../resources/skills", fileName), + join(moduleDir, "../../resources/skills", fileName), + join(moduleDir, "../../../resources/skills", fileName), ]; for (const candidate of candidates) { if (candidate && existsSync(candidate)) return candidate; @@ -85,18 +88,15 @@ export type BuiltinSkillInput = { * fresh so a packaged update takes effect without a restart. */ export function builtinSkills(input: BuiltinSkillInput): PluginSkillDef[] { - if (!isPluginWorkspace(input.workspacePath, input.pluginPaths)) return []; - const raw = readBuiltinSkill(PLUGIN_DEV_SKILL_FILE); - if (!raw?.trim()) return []; - const parsed = parseSkillFrontmatter(raw); - if (!parsed.body) return []; - return [ - { - id: PLUGIN_DEV_SKILL_ID, - name: parsed.name ?? "PI-Desktop plugin development", - description: parsed.description, - }, - ]; + const ids = [IMAGE_GENERATION_SKILL_ID]; + if (isPluginWorkspace(input.workspacePath, input.pluginPaths)) ids.push(PLUGIN_DEV_SKILL_ID); + return ids.flatMap((id) => { + const file = id === IMAGE_GENERATION_SKILL_ID ? IMAGE_GENERATION_SKILL_FILE : PLUGIN_DEV_SKILL_FILE; + const raw = readBuiltinSkill(file); + if (!raw?.trim()) return []; + const parsed = parseSkillFrontmatter(raw); + return parsed.body ? [{ id, name: parsed.name ?? id, description: parsed.description }] : []; + }); } /** @@ -106,14 +106,14 @@ export function builtinSkills(input: BuiltinSkillInput): PluginSkillDef[] { export function loadBuiltinSkillBody( id: string, ): { id: string; name: string; body: string } | null { - if (id !== PLUGIN_DEV_SKILL_ID) return null; - const raw = readBuiltinSkill(PLUGIN_DEV_SKILL_FILE); + if (id !== PLUGIN_DEV_SKILL_ID && id !== IMAGE_GENERATION_SKILL_ID) return null; + const raw = readBuiltinSkill(id === IMAGE_GENERATION_SKILL_ID ? IMAGE_GENERATION_SKILL_FILE : PLUGIN_DEV_SKILL_FILE); if (!raw?.trim()) return null; const parsed = parseSkillFrontmatter(raw); if (!parsed.body) return null; return { - id: PLUGIN_DEV_SKILL_ID, - name: parsed.name ?? "PI-Desktop plugin development", + id, + name: parsed.name ?? id, body: parsed.body, }; } diff --git a/apps/desktop/electron/main/runtime/sidecar.ts b/apps/desktop/electron/main/runtime/sidecar.ts index 685734af07..fcd68d044a 100644 --- a/apps/desktop/electron/main/runtime/sidecar.ts +++ b/apps/desktop/electron/main/runtime/sidecar.ts @@ -7,6 +7,7 @@ import { subagentProviderLookupError, } from "@pi-desktop/agent-runtime"; import { loadBuiltinSkillBody } from "../builtin-skills"; +import { createImageGenerationTool } from "../services/image-generation-service"; import { registerPluginDevTools } from "../plugin-dev-tools"; import { resolveLocalFile } from "../browser-view"; import { modelConfigFromModelsDev } from "../models-dev-catalog"; @@ -469,6 +470,7 @@ export function createSidecarRuntime({ }); // Agent-driven work panel preview (D100): open a workspace HTML file in // the embedded browser; live reload keeps it current through later edits. + s.setLocalTool("GenerateImages", createImageGenerationTool({ dataDir, getHost: () => runtimeState.host })); s.setLocalTool("BrowserPreview", async ({ args, sessionId }) => { const raw = String((args as { path?: unknown })?.path ?? "").trim(); if (!raw) { diff --git a/apps/desktop/electron/main/services/image-generation-service.ts b/apps/desktop/electron/main/services/image-generation-service.ts new file mode 100644 index 0000000000..99da58cc40 --- /dev/null +++ b/apps/desktop/electron/main/services/image-generation-service.ts @@ -0,0 +1,114 @@ +import { randomUUID } from "node:crypto"; +import { mkdir, realpath, writeFile } from "node:fs/promises"; +import { isAbsolute, join, relative, resolve } from "node:path"; +import { generateImageBatch } from "@pi-desktop/agent-runtime"; +import { + imageGenerationPrompts, + parseImageGenerationBinding, + type AppSettings, + type ProviderPublic, +} from "@pi-desktop/shared"; +import type { HostProcess } from "../host-process"; +import type { LocalToolHandler } from "../agent-sidecar"; +import { imageInputLoader } from "./image-inputs"; + +function failure(errorCode: string, content: string) { + return { + ok: false, + isError: true, + errorCode, + content: { kind: "image-generation-error", errorCode, message: content }, + }; +} + +export function createImageGenerationTool(options: { + dataDir: string; + getHost: () => Pick | null; + fetchImpl?: typeof fetch; +}): LocalToolHandler { + return async ({ sessionId, args, signal }) => { + imageGenerationPrompts(args); + const host = options.getHost(); + if (!host) return failure("HOST_UNAVAILABLE", "Host unavailable."); + const settings = await host.call("settings.get"); + const binding = parseImageGenerationBinding(settings.imageGeneration); + if (!binding) + return failure( + "IMAGE_NOT_CONFIGURED", + "Configure an image generation model in Settings > AI > Image generation model before generating images. Do not substitute another model.", + ); + const { provider } = await host.call<{ provider?: ProviderPublic }>("providers.get", { + id: binding.providerId, + }); + if ( + !provider?.enabled || + !provider.baseUrl || + !provider.models.some((model) => model.id === binding.modelId) + ) { + return failure( + "IMAGE_MODEL_UNAVAILABLE", + "The configured image model is unavailable. Update Settings > AI > Image generation model.", + ); + } + if (provider.authKind === "oauth") + return failure( + "IMAGE_AUTH_UNSUPPORTED", + "Image generation requires an API-key or no-auth service.", + ); + const { value } = await host.call<{ value?: string }>("providers.getSecret", { + id: provider.id, + }); + if (provider.authKind !== "none" && !value) + return failure("IMAGE_AUTH_FAILED", "The image provider needs an API key."); + const { path } = await host.call<{ path: string }>("session.getScratchPath", { sessionId }); + const root = resolve(options.dataDir, "scratch"); + const within = (base: string, target: string) => { + const rel = relative(base, target); + return !!rel && !rel.startsWith("..") && !isAbsolute(rel); + }; + if (typeof path !== "string" || !within(root, resolve(path))) + return failure("INVALID_ARGUMENT", "Invalid image output directory."); + await mkdir(path, { recursive: true }); + const realRoot = await realpath(root); + const realDir = await realpath(path); + if (!within(realRoot, realDir)) + return failure("INVALID_ARGUMENT", "Invalid image output directory."); + signal.throwIfAborted(); + const { session } = await host.call<{ session?: { projectPath?: string } }>("session.get", { + id: sessionId, + }); + if (!session) return failure("SESSION_NOT_FOUND", "The image session no longer exists."); + const results = await generateImageBatch({ + input: args, + endpoint: { + baseUrl: provider.baseUrl, + modelId: binding.modelId, + apiKey: value, + headers: provider.headers, + }, + signal, + fetchImpl: options.fetchImpl, + loadImages: imageInputLoader({ + dataDir: options.dataDir, + scratchPath: realDir, + projectPath: session?.projectPath, + }), + save: async (image) => { + const target = join(realDir, `generated-${randomUUID()}.${image.extension}`); + await writeFile(target, image.bytes, { flag: "wx" }); + return target; + }, + }); + const ok = results.some((result) => result.status === "succeeded"); + return { + ok, + isError: !ok, + content: { + kind: "generated-images", + providerId: binding.providerId, + modelId: binding.modelId, + results, + }, + }; + }; +} diff --git a/apps/desktop/electron/main/services/image-inputs.ts b/apps/desktop/electron/main/services/image-inputs.ts new file mode 100644 index 0000000000..6a9fbfb6a9 --- /dev/null +++ b/apps/desktop/electron/main/services/image-inputs.ts @@ -0,0 +1,88 @@ +import { open, realpath } from "node:fs/promises"; +import { isAbsolute, relative, resolve } from "node:path"; +import { generatedImageType, MAX_IMAGE_BYTES } from "@pi-desktop/agent-runtime"; + +/** Session/project roots are captured by the host, never supplied by the model. */ +export function imageInputLoader(options: { + projectPath?: string; + scratchPath: string; + dataDir: string; +}) { + let loadedBytes = 0; + const cache = new Map< + string, + Promise<{ bytes: Uint8Array; mimeType: string; extension: string }> + >(); + const read = async (ref: string) => { + const candidate = /^attachments[\\/][a-f0-9]{64}$/.test(ref) + ? resolve(options.dataDir, ref) + : isAbsolute(ref) + ? ref + : options.projectPath + ? resolve(options.projectPath, ref) + : resolve(options.scratchPath, ref); + const path = await realpath(candidate); + const roots = await Promise.all( + [options.projectPath, options.scratchPath, resolve(options.dataDir, "attachments")] + .filter((root): root is string => !!root) + .map((root) => + realpath(root).catch((error: NodeJS.ErrnoException) => { + if (error.code === "ENOENT") return null; + throw error; + }), + ), + ); + if ( + !roots.some((root) => { + if (!root) return false; + const rel = relative(root, path); + return !!rel && !rel.startsWith("..") && !isAbsolute(rel); + }) + ) + throw Object.assign(new Error("Image input is outside the session and project roots"), { + errorCode: "IMAGE_INPUT_OUTSIDE_ROOT", + }); + const file = await open(path, "r"); + try { + const stat = await file.stat(); + if (!stat.isFile() || stat.size > MAX_IMAGE_BYTES) + throw Object.assign(new Error("Image input is too large"), { + errorCode: "IMAGE_INPUT_INVALID", + }); + // A bounded read still holds if another process grows the file after stat. + const bytes = Buffer.alloc(Math.min(stat.size + 1, MAX_IMAGE_BYTES + 1)); + let size = 0; + while (size < bytes.length) { + const read = await file.read(bytes, size, bytes.length - size, null); + if (!read.bytesRead) break; + size += read.bytesRead; + } + const data = bytes.subarray(0, size); + if (loadedBytes + size > 64 * 1024 * 1024) + throw Object.assign(new Error("Batch image inputs exceed 64 MB"), { + errorCode: "IMAGE_INPUT_TOO_LARGE", + }); + loadedBytes += size; + return { bytes: data, ...generatedImageType(data) }; + } finally { + await file.close(); + } + }; + return async (refs: string[]) => { + const images = await Promise.all( + refs.map((ref) => { + let image = cache.get(ref); + if (!image) { + image = read(ref); + cache.set(ref, image); + } + return image; + }), + ); + if (images.reduce((size, image) => size + image.bytes.length, 0) > 32 * 1024 * 1024) + throw Object.assign(new Error("Image inputs exceed 32 MB"), { + errorCode: "IMAGE_INPUT_TOO_LARGE", + }); + return images; + }; +} diff --git a/apps/desktop/resources/skills/image-generation.md b/apps/desktop/resources/skills/image-generation.md new file mode 100644 index 0000000000..31590959c7 --- /dev/null +++ b/apps/desktop/resources/skills/image-generation.md @@ -0,0 +1,59 @@ +--- +name: imagegen +description: Generate or edit raster images, illustrations, photos, banners, and project assets with the configured image model. Supports reference images, edits of earlier results, variants, and batches. Prefer existing code-native tools for SVG/CSS edits. +--- + +# Image generation + +Use the desktop `GenerateImages` tool. The user selects its provider and model +under Settings → AI → Image generation model; this is independent of the chat +model. If ToolSearch is available and GenerateImages is not loaded, discover it +there first. Do not install an SDK, run an API script, ask for a key in chat, or +substitute the conversation model. + +## Prepare the request + +- Preserve the requested subject, composition, style, exact text, and constraints. + Add useful detail to a vague prompt without inventing additional deliverables. +- For a project asset, include its intended use and required framing. Prefer the + existing SVG/CSS asset system for changes to code-native icons or diagrams. +- For an edit, pass the source file paths as `images` (one to four per item). + Specify what changes and what must stay unchanged. For a follow-up edit, use + the previous result path. Inputs must be under the current project, session + scratch directory, or attachment store. Do not use remote URLs. +- Editing uses OpenAI Images multipart requests and depends on the selected + model supporting that endpoint. Do not silently replace an edit with a fresh + generation if it fails. Mask painting is not a desktop UI feature. + +## Generate one image or a batch + +Call `GenerateImages` with `items`, where each item has `prompt` and optional +`count` (default 1), plus optional `images` for edits. Use one item with `count` for variants of the same prompt; +use separate items for different assets. A batch permits at most 10 images +total. For a larger explicitly requested set, split it into bounded batches. +Generate only the quantity the user asked for; do not add unrequested variants. + +Example: two cover variants and one distinct icon: + +```json +{"items":[{"prompt":"Editorial cover: a ceramic cup on a quiet desk, warm daylight, no text","count":2},{"prompt":"A small raster illustration of a green leaf on a white background","count":1}]} +``` + +Generation may incur cost. Do not retry failed or timed-out items automatically, +including after cancellation: the provider may already have processed them. +Report partial success and wait for a user request before retrying. If no image +model is configured, direct the user to Settings → AI; do not select one silently. + +## Deliver the result + +Results are ordered and contain a status and, for successes, a local image path. +Show the successful images with Markdown image links and report failed items. +The desktop also renders their previews directly from the tool result. + +Use available image inspection tools to check the result when possible. Do not +claim to have visually inspected an image if you only received its file path. +For project deliverables, copy the selected image into the requested project +location using existing file/shell tools and update the consuming reference. +Use a new filename unless replacement was requested. Preview-only images may +remain in session storage. Report the final project paths, the prompt used, +and any requested images that did not complete. diff --git a/apps/desktop/src/components/settings/ImageGenerationModelRow.tsx b/apps/desktop/src/components/settings/ImageGenerationModelRow.tsx new file mode 100644 index 0000000000..a245f1e685 --- /dev/null +++ b/apps/desktop/src/components/settings/ImageGenerationModelRow.tsx @@ -0,0 +1,125 @@ +import { useState } from "react"; +import { useTranslation } from "react-i18next"; +import type { AppSettings, ImageGenerationBinding, ProviderPublic } from "@pi-desktop/shared"; +import { api } from "../../lib/api"; +import { useAppStore } from "../../stores/app-store"; +import { Button, Input } from "../ui"; +import { AnchoredMenu } from "./AnchoredMenu"; + +export function ImageGenerationModelRow({ + settings, + providers, +}: { + settings: AppSettings; + providers: ProviderPublic[]; +}) { + const { t } = useTranslation(); + const [open, setOpen] = useState(false); + const [query, setQuery] = useState(""); + const [busy, setBusy] = useState(false); + const [error, setError] = useState(""); + const binding = settings.imageGeneration; + const provider = providers.find((entry) => entry.id === binding?.providerId); + const eligible = (entry: ProviderPublic) => + entry.enabled && + entry.authKind !== "oauth" && + !!entry.baseUrl && + (entry.hasSecret || entry.authKind === "none"); + const valid = + provider && + eligible(provider) && + provider.models.some((model) => model.id === binding?.modelId); + const options = providers + .filter(eligible) + .flatMap((entry) => entry.models.map((model) => ({ provider: entry, model }))) + .filter(({ provider: entry, model }) => + `${entry.name} ${model.id} ${model.alias ?? ""}` + .toLowerCase() + .includes(query.trim().toLowerCase()), + ); + const save = async (next: ImageGenerationBinding | null) => { + setBusy(true); + setError(""); + try { + await api.setSettings({ ...(await api.getSettings()), imageGeneration: next }); + useAppStore.setState({ settings: await api.getSettings() }); + setOpen(false); + } catch { + setError(t("settings.imageModelSaveFailed")); + } finally { + setBusy(false); + } + }; + return ( +
+
+
+
{t("settings.imageModel")}
+
+ {binding + ? `${provider?.name ?? binding.providerId} / ${binding.modelId}` + : t("settings.imageModelUnset")} +
+ {binding && !valid ? ( +
{t("settings.imageModelUnavailable")}
+ ) : null} + {error ?
{error}
: null} +
+ setOpen(false)} + label={t("settings.imageModel")} + align="end" + menuClassName="model-default-menu" + trigger={(ref) => ( + + )} + > + setQuery(event.target.value)} + placeholder={t("settings.defaultModelSearch")} + aria-label={t("settings.defaultModelSearch")} + autoFocus + /> +
    + {options.map(({ provider: entry, model }) => ( +
  • + +
  • + ))} +
+ {!options.length ? ( +
{t("settings.noModelMatches")}
+ ) : null} +
+ {binding ? ( + + ) : null} +
+
+ ); +} diff --git a/apps/desktop/src/components/settings/ModelConfigPage.tsx b/apps/desktop/src/components/settings/ModelConfigPage.tsx index dc9a342cae..8e98b76068 100644 --- a/apps/desktop/src/components/settings/ModelConfigPage.tsx +++ b/apps/desktop/src/components/settings/ModelConfigPage.tsx @@ -37,6 +37,7 @@ import { displayedDefaultModelId, } from "./default-model"; import { copyProviderConfiguration, type ProviderCopyDraft } from "./provider-copy"; +import { ImageGenerationModelRow } from "./ImageGenerationModelRow"; import { ProviderSetupDialog } from "./ProviderSetupDialog"; import { useProviderReorder } from "./useProviderReorder"; import { VendorAccountsSection } from "./VendorAccountsSection"; @@ -161,10 +162,14 @@ export function ModelConfigPage() { /** * Preserve the selected app default unless it was removed from the provider. */ - const afterSaved = async (saved: ProviderPublic, models: ModelBinding[]) => { + const afterSaved = async (saved: ProviderPublic, models: ModelBinding[], imageModelId?: string) => { const firstModelId = models[0]?.id; try { - if (copyDraft) { + if (imageModelId) { + await api.setSettings({ ...(await api.getSettings()), imageGeneration: { providerId: saved.id, modelId: imageModelId } }); + useAppStore.setState({ settings: await api.getSettings() }); + showToast(t("settings.providerSaved"), { variant: "success" }); + } else if (copyDraft) { showToast(t("settings.providerSaved"), { variant: "success" }); } else if (!editingProvider) { await api.setSettings({ @@ -421,6 +426,8 @@ export function ModelConfigPage() { + +
@@ -722,7 +729,8 @@ export function ModelConfigPage() { provider={editingProvider} initialDraft={copyDraft} onClose={() => { setSetupFor(null); setCopyDraft(null); }} - onSaved={(saved, models) => void afterSaved(saved, models)} + imageModelId={settings.imageGeneration?.providerId === editingProvider?.id ? settings.imageGeneration?.modelId : undefined} + onSaved={afterSaved} /> ) : null}
diff --git a/apps/desktop/src/components/settings/ModelSelectionPanes.tsx b/apps/desktop/src/components/settings/ModelSelectionPanes.tsx index b4a8872d80..33cf11ed48 100644 --- a/apps/desktop/src/components/settings/ModelSelectionPanes.tsx +++ b/apps/desktop/src/components/settings/ModelSelectionPanes.tsx @@ -179,6 +179,8 @@ export function applyVisibleModelSelection( } export type ModelSelectionPanesProps = { + imageModelId?: string; + onImageModelChange?: (id: string) => void; discovery: ProviderModelsState & { canReload?: boolean }; selection: ModelSelection; /** Heading of the discovered list: a service's models, or an account's. */ @@ -207,6 +209,8 @@ export function ModelSelectionPanes({ busy = false, onReload, apiStyle, + imageModelId, + onImageModelChange, }: ModelSelectionPanesProps) { const { t } = useTranslation(); const { rows, models, publishedLevelsById, setModels } = selection; @@ -598,6 +602,7 @@ export function ModelSelectionPanes({ id={advancedId} hidden={!expanded} > + {onImageModelChange ? : null}
diff --git a/apps/desktop/src/features/chat/transcript/GeneratedImages.tsx b/apps/desktop/src/features/chat/transcript/GeneratedImages.tsx new file mode 100644 index 0000000000..99a50e5e58 --- /dev/null +++ b/apps/desktop/src/features/chat/transcript/GeneratedImages.tsx @@ -0,0 +1,92 @@ +import { useTranslation } from "react-i18next"; +import type { UiMessage } from "@pi-desktop/shared"; +import { useReferencedImageDataUrl } from "../../../lib/use-referenced-image-data-url"; +import { toolResultPayload } from "../../../lib/tool-presentation"; +import { useOpenChatFileRef } from "../../../hooks/use-preview-target"; +import { useAppStore } from "../../../stores/app-store"; +import { Button } from "../../../components/ui"; + +function ImageResult({ + path, + index, + mimeType, +}: { + path: string; + index: number; + mimeType?: string; +}) { + const { t } = useTranslation(); + const dataUrl = useReferencedImageDataUrl(path, mimeType); + const open = useOpenChatFileRef(); + return ( +
+ +
+ ); +} + +export function GeneratedImages({ message }: { message: UiMessage }) { + const { t } = useTranslation(); + if (message.toolName !== "GenerateImages") return null; + const payload = toolResultPayload(message); + const record = + payload && typeof payload === "object" ? (payload as Record) : null; + const results = + record?.kind === "generated-images" && Array.isArray(record.results) ? record.results : []; + const needsSetup = + record?.kind === "image-generation-error" && + [ + "IMAGE_NOT_CONFIGURED", + "IMAGE_MODEL_UNAVAILABLE", + "IMAGE_AUTH_FAILED", + "IMAGE_AUTH_UNSUPPORTED", + ].includes(String(record.errorCode)); + return ( +
+ {needsSetup ? ( +
+

{t("settings.imageModelSetupHint")}

+ +
+ ) : null} + {results.map((raw, index) => { + if (!raw || typeof raw !== "object") return null; + const item = raw as Record; + return item.status === "succeeded" && typeof item.path === "string" ? ( + + ) : ( +

+ {t("settings.imageGenerationFailed", { index: index + 1 })}{" "} + {typeof item.errorCode === "string" ? item.errorCode : ""} +

+ ); + })} +
+ ); +} diff --git a/apps/desktop/src/features/chat/transcript/ToolRow.tsx b/apps/desktop/src/features/chat/transcript/ToolRow.tsx index de940394d9..e483d038d1 100644 --- a/apps/desktop/src/features/chat/transcript/ToolRow.tsx +++ b/apps/desktop/src/features/chat/transcript/ToolRow.tsx @@ -1,3 +1,5 @@ +import { GeneratedImages } from "./GeneratedImages"; +import "../../../styles/generated-images.css"; import { Fragment, memo, @@ -523,6 +525,7 @@ export const ToolRow = memo(function ToolRow({ ) : null} + {inlineOpen && delegate ? ( { + const dataDir = await mkdtemp(join(tmpdir(), "pi-images-test-")); + t.after(() => rm(dataDir, { recursive: true, force: true })); + const requests = []; + const server = createServer(async (request, response) => { + let body = ""; + for await (const chunk of request) body += chunk.toString("latin1"); + requests.push({ url: request.url, body, authorization: request.headers.authorization }); + response.setHeader("Content-Type", "application/json"); + response.end(JSON.stringify({ data: [{ b64_json: png.toString("base64") }] })); + }); + await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); + t.after(() => new Promise((resolve) => server.close(resolve))); + let settings = { + defaultModelId: "chat", + defaultProviderId: "chat-provider", + imageGeneration: { providerId: "image-provider", modelId: "image-one" }, + }; + const host = { + call: async (method) => { + if (method === "settings.get") return structuredClone(settings); + if (method === "providers.get") + return { + provider: { + id: "image-provider", + enabled: true, + authKind: "api_key_and_base_url", + models: [{ id: "image-one" }, { id: "image-two" }], + baseUrl: `http://127.0.0.1:${server.address().port}`, + }, + }; + if (method === "providers.getSecret") return { value: "fixture-key" }; + if (method === "session.getScratchPath") return { path: join(dataDir, "scratch", "session") }; + if (method === "session.get") return { session: {} }; + throw new Error(method); + }, + }; + const options = { dataDir, getHost: () => host }; + const call = (args) => + createImageGenerationTool(options)({ + sessionId: "session", + toolCallId: "call", + args, + signal: new AbortController().signal, + }); + const batch = await call({ items: [{ prompt: "cover", count: 2 }, { prompt: "icon" }] }); + assert.equal(batch.ok, true); + assert.equal(batch.content.results.length, 3); + for (const image of batch.content.results) assert.deepEqual(await readFile(image.path), png); + assert.ok( + requests.every( + (request) => + request.url === "/v1/images/generations" && request.authorization === "Bearer fixture-key", + ), + ); + const original = batch.content.results[0].path; + const edited = await call({ items: [{ prompt: "make it green", images: [original], count: 2 }] }); + assert.equal(edited.content.results.length, 2); + assert.ok( + requests + .slice(3) + .every( + (request) => request.url === "/v1/images/edits" && request.body.includes('name="image"'), + ), + ); + assert.notEqual(edited.content.results[0].path, original); + assert.deepEqual(await readFile(original), png); + settings.imageGeneration = { providerId: "image-provider", modelId: "image-two" }; + await call({ items: [{ prompt: "replacement" }] }); + assert.equal(JSON.parse(requests.at(-1).body).model, "image-two"); + settings.imageGeneration = null; + assert.equal((await call({ items: [{ prompt: "unset" }] })).errorCode, "IMAGE_NOT_CONFIGURED"); + assert.equal(requests.length, 6); + assert.equal(settings.defaultModelId, "chat"); +}); + +test("edit inputs reject outside paths, traversal and non-image files", async (t) => { + const dataDir = await mkdtemp(join(tmpdir(), "pi-image-input-")); + t.after(() => rm(dataDir, { recursive: true, force: true })); + const scratchPath = join(dataDir, "scratch", "s"); + await mkdir(scratchPath, { recursive: true }); + await writeFile(join(dataDir, "outside.png"), png); + await writeFile(join(scratchPath, "fake.png"), "not an image"); + const load = imageInputLoader({ dataDir, scratchPath }); + await assert.rejects(load([join(dataDir, "outside.png")]), /outside/); + await assert.rejects(load(["../../outside.png"]), /outside/); + await assert.rejects(load(["fake.png"]), /IMAGE_INVALID_CONTENT/); +}); + +test("the default imagegen skill is discoverable and loads in an ordinary session", async () => { + const previous = globalThis.__dirname; + globalThis.__dirname = fileURLToPath(new URL("../electron/main/", import.meta.url)); + try { + const { builtinSkills, loadBuiltinSkillBody } = await import( + "../electron/main/builtin-skills.ts" + ); + const skills = builtinSkills({}); + assert.equal(skills.find((skill) => skill.id === "pi-desktop/imagegen")?.name, "imagegen"); + assert.ok(!skills.some((skill) => skill.id === "pi-desktop/plugin-development")); + const body = loadBuiltinSkillBody("pi-desktop/imagegen").body; + assert.match(body, /GenerateImages/); + assert.match(body, /previous result path/); + assert.match(body, /Do not retry/); + assert.equal(loadBuiltinSkillBody("../../outside"), null); + } finally { + globalThis.__dirname = previous; + } +}); diff --git a/apps/desktop/test/plugin-skills.test.mjs b/apps/desktop/test/plugin-skills.test.mjs index 0b5ac01662..bbd50a64d6 100644 --- a/apps/desktop/test/plugin-skills.test.mjs +++ b/apps/desktop/test/plugin-skills.test.mjs @@ -105,7 +105,7 @@ test("the built-in plugin skill only activates for plugin workspaces", () => { assert.match(builtinSrc, /isPluginWorkspace/); assert.match(builtinSrc, /schemaVersion.*number/s); assert.match(builtinSrc, /pluginPaths\.some/); - assert.match(builtinSrc, /if \(!isPluginWorkspace\(input\.workspacePath, input\.pluginPaths\)\) return \[\]/); + assert.match(builtinSrc, /if \(isPluginWorkspace\(input\.workspacePath, input\.pluginPaths\)\) ids\.push\(PLUGIN_DEV_SKILL_ID\)/); assert.match(mainSrc, /builtinSkills\(\{/); }); diff --git a/crates/host-core/src/permissions.rs b/crates/host-core/src/permissions.rs index 87506e595a..475b9387f8 100644 --- a/crates/host-core/src/permissions.rs +++ b/crates/host-core/src/permissions.rs @@ -127,7 +127,7 @@ impl PermissionManager { pub fn tool_risk_with_declared(tool_name: &str, declared: Option<&str>) -> Risk { match tool_name { "Read" | "Glob" | "Grep" | "ScheduledTaskList" => Risk::Low, - "Write" | "Edit" | "Bash" => Risk::High, + "Write" | "Edit" | "Bash" | "GenerateImages" => Risk::High, name if name.starts_with("plugin_") => match declared { Some("low") => Risk::Low, Some("high") => Risk::High, @@ -736,3 +736,43 @@ mod tests { ); } } + +#[cfg(test)] +mod image_generation_tests { + use super::*; + + #[test] + fn image_generation_requires_approval_and_is_not_plan_safe() { + assert!(matches!( + PermissionManager::tool_risk_with_declared("GenerateImages", None), + Risk::High + )); + let manager = PermissionManager::default(); + let grants = HashMap::new(); + for mode in ["ask", "accept-edits"] { + assert!(manager + .evaluate_auto_with_permission_mode("s", "GenerateImages", "agent", mode, &grants) + .is_none()); + } + assert_eq!( + manager.evaluate_auto_with_permission_mode( + "s", + "GenerateImages", + "plan", + "auto", + &grants + ), + Some(PermissionDecision::Deny) + ); + assert_eq!( + manager.evaluate_auto_with_permission_mode( + "s", + "GenerateImages", + "goal", + "auto", + &grants + ), + Some(PermissionDecision::Deny) + ); + } +} diff --git a/crates/host-core/src/rpc/mod.rs b/crates/host-core/src/rpc/mod.rs index ef448b8118..837ef9ab93 100644 --- a/crates/host-core/src/rpc/mod.rs +++ b/crates/host-core/src/rpc/mod.rs @@ -685,6 +685,21 @@ fn validate_settings_value(value: &Value) -> Result<(), JsonRpcError> { let Some(object) = value.as_object() else { return Ok(()); }; + if let Some(binding) = object.get("imageGeneration").filter(|v| !v.is_null()) { + for (key, max) in [("providerId", 128), ("modelId", 256)] { + if !binding + .get(key) + .and_then(Value::as_str) + .is_some_and(|s| !s.trim().is_empty() && s.len() <= max) + { + return Err(rpc_err( + 1002, + "invalid image generation binding", + "INVALID_PARAMS", + )); + } + } + } if let Some(template_value) = object.get("promptEnhancementUserTemplate") { if let Some(message) = prompt_enhancement_template_error("promptEnhancementUserTemplate", template_value) @@ -1096,7 +1111,7 @@ fn bash_cancellation_requested(receiver: &Option>, p: &ToolsExecuteParams) { - if p.tool_name != "Bash" { + if !matches!(p.tool_name.as_str(), "Bash" | "GenerateImages") { return; } let mut st = state.lock().await; @@ -3204,7 +3219,8 @@ async fn handle_request( // Register before permission evaluation so tools.abort can cancel // an approval wait as well as an already-spawned process. - let cancellation_receiver = if p.tool_name == "Bash" { + let cancellation_receiver = if matches!(p.tool_name.as_str(), "Bash" | "GenerateImages") + { let mut st = state.lock().await; match st.register_bash_cancellation(&p.session_id, &p.tool_call_id) { Ok(receiver) => Some(receiver), @@ -8606,3 +8622,25 @@ mod tests { assert_eq!(st.plugins.locale(), "en-US"); } } + +#[cfg(test)] +mod image_generation_settings_tests { + use super::*; + #[test] + fn validates_optional_image_binding() { + for value in [ + json!({}), + json!({"imageGeneration": null}), + json!({"imageGeneration": {"providerId": "p", "modelId": "image"}}), + ] { + assert!(validate_settings_value(&value).is_ok()); + } + for value in [ + json!(false), + json!({}), + json!({"providerId": "p", "modelId": " "}), + ] { + assert!(validate_settings_value(&json!({"imageGeneration": value})).is_err()); + } + } +} diff --git a/crates/host-core/src/tools/mod.rs b/crates/host-core/src/tools/mod.rs index 58211d47ce..0479585e29 100644 --- a/crates/host-core/src/tools/mod.rs +++ b/crates/host-core/src/tools/mod.rs @@ -1068,6 +1068,9 @@ pub async fn execute_tool_with_path_access( } } let result: Result = match tool_name { + // Authorize the desktop-owned image request through the normal host gate. + // Only the trusted desktop runner performs the external call. + "GenerateImages" => Ok(serde_json::json!({ "authorized": true })), "Read" => tool_read( workspace, scratch, diff --git a/docs/adr/README.md b/docs/adr/README.md index 917bd6cb02..ac63987886 100644 --- a/docs/adr/README.md +++ b/docs/adr/README.md @@ -331,3 +331,4 @@ Each ADR includes: | turn-process-and-thinking-display | [Turn process and thinking presentation](turn-process-and-thinking-display.md) | Accepted | | provider-display-order | [Provider display order](provider-display-order.md) | Accepted | | provider-system-certificates | [Desktop sidecar uses OS-trusted certificates](provider-system-certificates.md) | Accepted | +| image-generation-capability | [Image generation as a configured Agent capability](image-generation-capability.md) | Accepted | diff --git a/docs/adr/image-generation-capability.md b/docs/adr/image-generation-capability.md new file mode 100644 index 0000000000..ba33fd25f4 --- /dev/null +++ b/docs/adr/image-generation-capability.md @@ -0,0 +1,41 @@ +# ADR: Image generation as a configured Agent capability + +- Status: Accepted +- Date: 2026-09-21 +- Related: ADR 0281, ADR 0101, ADR 0172; [image generation spec](../spec/03-runtime/21-image-generation.md) + +## Context + +The desktop understands image input and previews but has no image production +path. Users need one selectable image model, conversational generation/editing, +and batches without changing their default chat model. A Skill alone cannot +execute image requests. Pure image endpoints cannot be treated as chat adapters. + +## Decision + +Keep a separate optional settings binding referencing existing provider credentials. +Expose one Agent tool plus a bundled, lazily loaded imagegen skill. Use an independent +OpenAI Images adapter and batch scheduler in agent-runtime; the desktop service +coordinates host-owned settings/credentials, contained source files and saved output. +Renderer code handles configuration, previews and navigation only. + +The existing local-tool shortcut does not perform host permission checks. For this +billable tool, the bridge must obtain host-core authorization through `tools.execute` +before invoking the local service. Host-core never performs the HTTP call; its +result is authorization, while the sidecar result records actual generated files. +Local tools gain an abort signal so this path can stop requests on cancellation, +timeout and process loss. Existing tools retain their deadlines and behavior. + +## Alternatives and consequences + +- An independent image workspace would duplicate conversation/history ownership; + use existing chat and previews first. +- A shell/API-only skill would bypass model settings and require scripts to handle + credentials, files and retries. The skill instead calls the bounded host tool. +- A broad multi-protocol framework is unnecessary for the requested single Images + protocol. Generation and multipart editing share its adapter and scheduler. +- Optional settings preserve existing stored data and conversation defaults. No + schema migration, plugin API change or direct renderer credential access occurs. +- Partial results remain useful; no automatic retry avoids duplicate paid work. + Upstream model support and billing remain provider responsibilities. +- Headless hosts without the desktop image runner do not perform image requests. diff --git a/docs/spec/03-runtime/03-tools-and-permissions.md b/docs/spec/03-runtime/03-tools-and-permissions.md index 2af8ccf0e6..91c7c7b339 100644 --- a/docs/spec/03-runtime/03-tools-and-permissions.md +++ b/docs/spec/03-runtime/03-tools-and-permissions.md @@ -614,3 +614,10 @@ Naming: - command allowlist / denylist - dry-run mode - apply patches after preview + +## Image generation and editing + +`GenerateImages` is a high-risk Agent-only capability, authorized by host-core +before the trusted desktop executes the request. Plan/Goal remain denied even +under Auto. See [image generation](21-image-generation.md) for cancellation, +limits and result semantics. diff --git a/docs/spec/03-runtime/13-model-catalog-and-selection.md b/docs/spec/03-runtime/13-model-catalog-and-selection.md index 540176f357..f05863ab9f 100644 --- a/docs/spec/03-runtime/13-model-catalog-and-selection.md +++ b/docs/spec/03-runtime/13-model-catalog-and-selection.md @@ -538,3 +538,10 @@ same model to the check mark, the toggle and the duplicate guard. - [ ] compact limit text never reads above the published value, keeps the neighbouring 1M-line windows apart (`1M` / `1.05M` / `1.1M`), and never renders a `K` mantissa at or above 1000 + +## Image model binding + +The default conversation model has a separate **Image generation model** row below +it. Model Advanced can select that unique binding; provider form Save commits it, +Cancel discards it, and replacing it leaves the conversation default unchanged. +See [image generation and editing](21-image-generation.md) for the tool and batch contract. diff --git a/docs/spec/03-runtime/21-image-generation.md b/docs/spec/03-runtime/21-image-generation.md new file mode 100644 index 0000000000..00ae772159 --- /dev/null +++ b/docs/spec/03-runtime/21-image-generation.md @@ -0,0 +1,77 @@ +# Image generation and editing + +The desktop exposes one optional `AppSettings.imageGeneration` binding with +`providerId` and `modelId`. `null` clears it; absent means unconfigured. Host-core +validates and persists it through the existing settings store. No schema bump +is needed. The binding is independent of the conversation default and references +an existing enabled API-key or no-auth provider and one of its configured models. + +## Configuration + +Model Advanced offers **Set as image model**. A draft selection only takes effect +when the provider form saves; Cancel leaves settings unchanged. Saving a provider +as an image model does not replace the default conversation model. Below the +default model row, **Image generation model** displays the binding and offers +searchable replacement and Clear. OAuth accounts are not eligible. Missing, +disabled or removed bindings remain visible as unavailable; there is no fallback. + +## Agent contract + +`GenerateImages({items: [{prompt, count?, images?}]})` is an Agent-mode tool. +`count` defaults to 1. Both distinct prompts and same-prompt variants are supported, +with 1–10 output images total. An optional `images` array supplies 1–4 local source +files per item for editing; a previous generated path supports iterative edits. +Prompt length is bounded at 32,000 characters. Unsupported or excessive input is +rejected rather than truncated. Plan and Goal cannot execute the tool. + +The bundled `pi-desktop/imagegen` skill is discoverable in ordinary sessions and +loads through the existing Skill tool. It teaches prompting, batches, reference +edits, preserving originals, partial-failure handling and project asset delivery. +It does not grant permission or carry credentials. + +The trusted desktop bridge first calls host-core `tools.execute` with the same +identity, arguments and permission scope. Host-core applies the existing high-risk +tool policy and returns authorization only. The bridge executes the image service +only after approval. The host authorization audit is distinct from the actual +sidecar tool result. Runtime cancellation reaches both pending approval and the +network request; host loss or sidecar disposal aborts local work. + +## OpenAI Images adapter + +Only OpenAI-compatible Images is supported. A root base URL acquires `/v1`; an +explicit path prefix is retained. Generation uses `POST images/generations` with +`model`, `prompt`, `n: 1`. Editing uses `POST images/edits`, multipart image files, +the prompt, model and `n: 1`. No browser mask editor is included. A configured +service may implement generation without editing; its error is reported without +silently switching to generation or another model. + +Each batch snapshots the binding and provider before dispatch, uses two workers, +and returns results in input order. Each output has a 180-second request budget. +Requests are never automatically retried. Authentication failure stops queued +work. Cancellation stops queued work and aborts active HTTP; it cannot promise the +upstream provider stopped processing or billing. Completed output files survive. + +Responses accept exactly one Base64 image or HTTPS image URL per request. JSON and +download bodies are bounded; image files are capped at 16 MiB and restricted to +PNG, JPEG and WebP signatures. Downloads use checked, pinned public DNS addresses, +reject redirects and private destinations, and never receive provider headers. +Input edits accept the session project, that session's scratch directory and the +attachment store after realpath containment. Each edit input set is capped at +32 MiB, with a 64 MiB input cache budget for the batch. Credentials remain outside the renderer and tool results. + +## Results and recovery + +Each result records index, status (`succeeded`, `failed`, `cancelled`), successful +path/MIME type or a safe error code. New files get unique names in session scratch; +editing never overwrites its source. The tool result and transcript retain file +references, not Base64. Existing bounded image reads and file viewers serve previews. +Successful images and per-item failures render even when only part of a batch +completed. Missing configuration returns a structured error and a Settings → AI +navigation action. The same references render after session reload/restart. + +Validation: `node scripts/e2e-image-generation.mjs` covers host/stdio/HTTP/storage; +`node scripts/e2e-image-generation-ui.mjs` covers real React/Chromium interactions +with an API-boundary fixture. Unit/service tests cover limits, cancellation, +partial failures, authentication, timeout, unsafe paths, and bounded downloads. +Live verification is opt-in via `scripts/test-image-generation-live.mjs`, limited +to one generation plus one edit and never a default test command. diff --git a/docs/spec/03-runtime/README.md b/docs/spec/03-runtime/README.md index 3b2ded8dfd..e8b945e501 100644 --- a/docs/spec/03-runtime/README.md +++ b/docs/spec/03-runtime/README.md @@ -22,3 +22,5 @@ | [18-line-anchored-edit-contract.md](18-line-anchored-edit-contract.md) | Line-anchored Edit contract | | [19-remote-agent-control-protocol.md](19-remote-agent-control-protocol.md) | Remote Agent Control Protocol | | [20-speech.md](20-speech.md) | Host speech (ASR/TTS) | + +- [Image generation and editing](21-image-generation.md) diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index 010f2ee41d..4057bb062a 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -58,8 +58,8 @@ - Document every user-visible and protocol-visible behavior that MVP must verify. - Provide a scenario catalog that maps to acceptance criteria (A–H) and milestones (M1–M6). - Serve as the traceability backbone: scenario ID ↔ acceptance criterion ↔ spec. -- Define the required E2E gate for code-bearing changes on the integrated - `main` that precedes the pull request. +- Define the required E2E gate for code-bearing task candidates containing the + latest `origin/main`, before the pull request; do not merge into local `main`. - Keep validation evidence tied to the commit the gate ran on, plus any later commit that changes the landed executable content. @@ -14163,3 +14163,25 @@ the latest destination. These assertions measure work counts, not device FPS. - **Milestone**: Maintenance. - **Status**: Covered by the existing HTTP client integration fixture and a focused component-render validation; no live IDA process required. + +### E2E-IMAGE-generation-and-editing + +- **Preconditions:** Built task candidate containing latest origin/main; isolated + host data directory, local image HTTP fixture, no production credentials. +- **Steps:** Choose a model in Advanced, cancel and verify no change; save, replace + and clear it through Settings. Generate same-prompt variants and distinct images, + edit a generated image, inspect partial failures, then restart the host/session. + Attempt the tool in a durable Plan session and verify no HTTP request occurs. +- **Expected:** One image binding persists without changing the chat default; + generated files, edit sources and transcript references survive restart. + Images render in chat; unconfigured errors navigate to AI settings. +- **Specs:** 03-runtime/21-image-generation; 03-runtime/13-model-catalog-and-selection. +- **Acceptance:** Configured image generation/editing, safe cancellation and persistence. +- **Milestone:** Post-MVP. +- **Status:** Automated by `scripts/e2e-image-generation.mjs` and + `scripts/e2e-image-generation-ui.mjs`; backend test uses real Rust/stdio/HTTP/files, + UI test uses real Chromium and production components with API-boundary fixtures. + +| Scenario | Acceptance | Specification | Automation | +| --- | --- | --- | --- | +| E2E-IMAGE-generation-and-editing | Image capability and recovery | 03-runtime/21-image-generation | Host and UI suites above | diff --git a/package.json b/package.json index cea4a8a9e9..3b2d1ab2ee 100644 --- a/package.json +++ b/package.json @@ -31,6 +31,7 @@ "test:e2e:composer-paste": "node scripts/e2e-composer-paste.mjs", "test:e2e:transcript": "node scripts/e2e-transcript-render.mjs", "test:e2e:transcript-disclosure": "node scripts/e2e-transcript-disclosure-anchor.mjs", + "test:e2e:images": "node scripts/e2e-image-generation.mjs && node scripts/e2e-image-generation-ui.mjs", "test:e2e:provider-order": "node scripts/e2e-provider-order.mjs", "test:e2e:provider-api-style": "node scripts/e2e-provider-api-style.mjs", "test:e2e:oauth-retry": "node --test apps/desktop/test/anthropic-oauth-retry.test.mjs", diff --git a/packages/agent-runtime/src/image-generation/download.ts b/packages/agent-runtime/src/image-generation/download.ts new file mode 100644 index 0000000000..523fce36eb --- /dev/null +++ b/packages/agent-runtime/src/image-generation/download.ts @@ -0,0 +1,130 @@ +import { lookup } from "node:dns/promises"; +import { isIP } from "node:net"; +import { Agent, fetch as fetchPinned } from "undici"; + +export const MAX_IMAGE_BYTES = 16 * 1024 * 1024; +export function imageError(code: string): Error & { errorCode: string } { + return Object.assign(new Error(code), { errorCode: code }); +} + +export async function boundedBytes( + response: Pick, + max: number, +): Promise { + if (Number(response.headers.get("content-length")) > max) { + await response.body?.cancel(); + throw imageError("IMAGE_TOO_LARGE"); + } + if (!response.body) throw imageError("IMAGE_EMPTY_RESPONSE"); + const reader = response.body.getReader(); + const chunks: Uint8Array[] = []; + let size = 0; + try { + while (true) { + const part = await reader.read(); + if (part.done) break; + size += part.value.length; + if (size > max) throw imageError("IMAGE_TOO_LARGE"); + chunks.push(part.value); + } + return Buffer.concat(chunks); + } finally { + await reader.cancel(); + reader.releaseLock(); + } +} + +/** Deny non-public destinations, including mapped IPv6 and cloud metadata. */ +export function publicImageAddress(address: string): boolean { + if (isIP(address) === 4) { + const [a, b, c] = address.split(".").map(Number); + return !( + a === 0 || + a === 10 || + a === 127 || + a >= 224 || + (a === 100 && b >= 64 && b <= 127) || + (a === 169 && b === 254) || + (a === 172 && b >= 16 && b <= 31) || + (a === 192 && (b === 168 || b === 0 || (b === 88 && c === 99))) || + (a === 198 && (b === 18 || b === 19 || (b === 51 && c === 100))) || + (a === 203 && b === 0 && c === 113) + ); + } + if (isIP(address) === 6) { + const lower = address.toLowerCase(); + return ( + /^[23][0-9a-f]{3}:/.test(lower) && !lower.startsWith("2001:") && !lower.startsWith("2002:") + ); + } + return false; +} + +/** Pin the checked DNS answer to the connection; never forward provider headers. */ +export async function downloadGeneratedImage( + raw: string, + signal: AbortSignal, +): Promise { + let url: URL; + try { + url = new URL(raw); + } catch { + throw imageError("IMAGE_INVALID_URL"); + } + if ( + url.protocol !== "https:" || + url.username || + url.password || + (url.port && url.port !== "443") + ) { + throw imageError("IMAGE_INVALID_URL"); + } + const hostname = url.hostname.replace(/^\[|\]$/g, ""); + const answers = await lookup(hostname, { all: true }); + signal.throwIfAborted(); + if (!answers.length || answers.some((answer) => !publicImageAddress(answer.address))) + throw imageError("IMAGE_UNSAFE_URL"); + const address = answers[0]; + const dispatcher = new Agent({ + connect: { + lookup: (_name, options, callback) => { + if (options.all) callback(null, [address]); + else callback(null, address.address, address.family); + }, + }, + }); + try { + const response = await fetchPinned(url, { signal, redirect: "error", dispatcher }); + if (!response.ok) throw imageError("IMAGE_DOWNLOAD_FAILED"); + return await boundedBytes(response as unknown as Response, MAX_IMAGE_BYTES); + } finally { + await dispatcher.destroy(); + } +} + +export function generatedImageType(bytes: Uint8Array): { mimeType: string; extension: string } { + const data = Buffer.from(bytes); + if (data.length > MAX_IMAGE_BYTES) throw imageError("IMAGE_TOO_LARGE"); + if ( + data.length >= 24 && + data.subarray(0, 8).equals(Buffer.from([137, 80, 78, 71, 13, 10, 26, 10])) && + data.toString("ascii", 12, 16) === "IHDR" + ) + return { mimeType: "image/png", extension: "png" }; + if ( + data.length > 4 && + data[0] === 255 && + data[1] === 216 && + data[2] === 255 && + data[data.length - 2] === 255 && + data[data.length - 1] === 217 + ) + return { mimeType: "image/jpeg", extension: "jpg" }; + if ( + data.length >= 16 && + data.toString("ascii", 0, 4) === "RIFF" && + data.toString("ascii", 8, 12) === "WEBP" + ) + return { mimeType: "image/webp", extension: "webp" }; + throw imageError("IMAGE_INVALID_CONTENT"); +} diff --git a/packages/agent-runtime/src/image-generation/image-generation.test.ts b/packages/agent-runtime/src/image-generation/image-generation.test.ts new file mode 100644 index 0000000000..127c5865d3 --- /dev/null +++ b/packages/agent-runtime/src/image-generation/image-generation.test.ts @@ -0,0 +1,171 @@ +import { describe, expect, it, vi } from "vitest"; +import { generateImageBatch, generateOneImage, imageGenerationUrl } from "./index.js"; +import { publicImageAddress, generatedImageType, boundedBytes } from "./download.js"; +import { imageGenerationPrompts, parseImageGenerationBinding } from "@pi-desktop/shared"; + +const png = + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+a9mQAAAAASUVORK5CYII="; +const endpoint = { baseUrl: "https://example.com", modelId: "image-model", apiKey: "test-only" }; +const response = () => Response.json({ data: [{ b64_json: png }] }); +const save = async (_image: unknown, index: number) => `/scratch/image-${index}.png`; + +describe("image generation", () => { + it("normalizes root/versioned base URLs without doubling the version", () => { + expect(imageGenerationUrl(endpoint.baseUrl)).toBe("https://example.com/v1/images/generations"); + expect(imageGenerationUrl("https://example.com/gateway/v1/")).toBe( + "https://example.com/gateway/v1/images/generations", + ); + expect(() => imageGenerationUrl("https://user:pass@example.com")).toThrow(); + }); + it("validates binding and both batch forms without truncation", () => { + expect(parseImageGenerationBinding(null)).toBeNull(); + expect( + imageGenerationPrompts({ + items: [ + { prompt: "a", images: [] }, + { prompt: "b", images: null }, + ], + }), + ).toEqual(["a", "b"]); + expect(() => parseImageGenerationBinding({ modelId: "m" })).toThrow(); + expect(imageGenerationPrompts({ items: [{ prompt: "a", count: 2 }, { prompt: "b" }] })).toEqual( + ["a", "a", "b"], + ); + for (const count of [0, 1.5, 11, "2"]) + expect(() => imageGenerationPrompts({ items: [{ prompt: "a", count }] })).toThrow(); + expect(() => + imageGenerationPrompts({ + items: [ + { prompt: "a", count: 6 }, + { prompt: "b", count: 5 }, + ], + }), + ).toThrow(); + }); + it("uses the configured model, key and headers without chat protocol fields", async () => { + const fetchImpl = vi.fn().mockResolvedValue(response()); + const image = await generateOneImage( + { ...endpoint, headers: { "X-Test": "custom" } }, + "draw", + new AbortController().signal, + fetchImpl, + ); + const [url, init] = fetchImpl.mock.calls[0]; + expect(url).toBe("https://example.com/v1/images/generations"); + expect(JSON.parse(String(init?.body))).toEqual({ model: "image-model", prompt: "draw", n: 1 }); + expect(new Headers(init?.headers).get("Authorization")).toBe("Bearer test-only"); + expect(new Headers(init?.headers).get("X-Test")).toBe("custom"); + expect(init?.redirect).toBe("error"); + expect(image.mimeType).toBe("image/png"); + }); + it("limits concurrency to two and preserves input order through partial failure", async () => { + const releases: Array<() => void> = []; + let inFlight = 0; + let peak = 0; + let calls = 0; + const fetchImpl: typeof fetch = async () => { + const call = calls++; + peak = Math.max(peak, ++inFlight); + await new Promise((resolve) => releases.push(resolve)); + inFlight--; + return call === 1 ? new Response("failure", { status: 500 }) : response(); + }; + const batch = generateImageBatch({ + input: { items: [{ prompt: "a", count: 3 }] }, + endpoint, + signal: new AbortController().signal, + save, + fetchImpl, + }); + await vi.waitFor(() => expect(releases.length).toBe(2)); + releases[1](); + await vi.waitFor(() => expect(releases.length).toBe(3)); + releases[2](); + releases[0](); + expect((await batch).map((result) => result.status)).toEqual([ + "succeeded", + "failed", + "succeeded", + ]); + expect(peak).toBe(2); + expect(calls).toBe(3); + }); + it("cancels in-flight requests and never starts queued images", async () => { + const controller = new AbortController(); + const fetchImpl = vi.fn().mockImplementation( + async (_url, init) => + new Promise((_resolve, reject) => { + init?.signal?.addEventListener("abort", () => reject(new Error("aborted")), { + once: true, + }); + }), + ); + const batch = generateImageBatch({ + input: { items: [{ prompt: "a", count: 5 }] }, + endpoint, + signal: controller.signal, + save, + fetchImpl, + }); + controller.abort(); + expect((await batch).every((item) => item.status === "cancelled")).toBe(true); + expect(fetchImpl).toHaveBeenCalledTimes(2); + }); + it("stops queued work on authentication failure and does not disclose provider bodies", async () => { + const fetchImpl = vi + .fn() + .mockResolvedValue(new Response("secret provider payload", { status: 401 })); + const result = await generateImageBatch({ + input: { items: [{ prompt: "a", count: 5 }] }, + endpoint, + signal: new AbortController().signal, + save, + fetchImpl, + }); + expect(fetchImpl).toHaveBeenCalledTimes(2); + expect(JSON.stringify(result)).not.toContain("secret"); + expect(result[4].status).toBe("cancelled"); + }); + it("times out each image without retry", async () => { + vi.useFakeTimers(); + try { + const fetchImpl = vi.fn().mockImplementation( + async (_url, init) => + new Promise((_resolve, reject) => { + init?.signal?.addEventListener("abort", () => reject(new Error("timeout")), { + once: true, + }); + }), + ); + const batch = generateImageBatch({ + input: { items: [{ prompt: "a" }] }, + endpoint, + signal: new AbortController().signal, + save, + fetchImpl, + }); + await vi.advanceTimersByTimeAsync(180_000); + expect((await batch)[0].errorCode).toBe("IMAGE_TIMEOUT"); + expect(fetchImpl).toHaveBeenCalledTimes(1); + } finally { + vi.useRealTimers(); + } + }); + it("rejects unsafe image addresses, forged images and oversized bodies", async () => { + for (const address of [ + "127.0.0.1", + "10.0.0.1", + "169.254.169.254", + "172.16.0.1", + "192.168.0.1", + "::1", + "::ffff:127.0.0.1", + "fc00::1", + "2002:7f00:1::", + ]) + expect(publicImageAddress(address)).toBe(false); + expect(publicImageAddress("8.8.8.8")).toBe(true); + expect(() => generatedImageType(Buffer.from(""))).toThrow("IMAGE_INVALID_CONTENT"); + await expect(boundedBytes(new Response("oversized"), 2)).rejects.toThrow("IMAGE_TOO_LARGE"); + }); +}); diff --git a/packages/agent-runtime/src/image-generation/index.ts b/packages/agent-runtime/src/image-generation/index.ts new file mode 100644 index 0000000000..bf1ffb6947 --- /dev/null +++ b/packages/agent-runtime/src/image-generation/index.ts @@ -0,0 +1,169 @@ +import { + IMAGE_GENERATION_TIMEOUT_MS, + imageGenerationItems, + type GeneratedImageResult, +} from "@pi-desktop/shared"; +import { + boundedBytes, + downloadGeneratedImage, + generatedImageType, + imageError, + MAX_IMAGE_BYTES, +} from "./download.js"; +export { generatedImageType, MAX_IMAGE_BYTES } from "./download.js"; + +export type ImageEndpoint = { + baseUrl: string; + modelId: string; + apiKey?: string; + headers?: Record; +}; + +export function imageGenerationUrl(baseUrl: string, edit = false): string { + const url = new URL(baseUrl); + if ( + !["http:", "https:"].includes(url.protocol) || + url.username || + url.password || + url.search || + url.hash + ) + throw imageError("IMAGE_INVALID_ENDPOINT"); + const path = url.pathname.replace(/\/+$/, ""); + url.pathname = `${path || "/v1"}/images/${edit ? "edits" : "generations"}`; + return url.href; +} + +export type ImageEditInput = { bytes: Uint8Array; mimeType: string; extension: string }; + +export async function generateOneImage( + endpoint: ImageEndpoint, + prompt: string, + signal: AbortSignal, + fetchImpl: typeof fetch = fetch, + images: ImageEditInput[] = [], +) { + const headers = new Headers(endpoint.headers); + headers.set("Content-Type", "application/json"); + if (endpoint.apiKey) headers.set("Authorization", `Bearer ${endpoint.apiKey}`); + let body: BodyInit = JSON.stringify({ model: endpoint.modelId, prompt, n: 1 }); + if (images.length) { + const form = new FormData(); + form.set("model", endpoint.modelId); + form.set("prompt", prompt); + form.set("n", "1"); + images.forEach((image, index) => + form.append( + images.length === 1 ? "image" : "image[]", + new Blob([new Uint8Array(image.bytes)], { type: image.mimeType }), + `input-${index}.${image.extension}`, + ), + ); + headers.delete("Content-Type"); + body = form; + } + const response = await fetchImpl(imageGenerationUrl(endpoint.baseUrl, images.length > 0), { + method: "POST", + headers, + redirect: "error", + signal, + body, + }); + if (!response.ok) { + await response.body?.cancel(); + throw imageError( + response.status === 401 || response.status === 403 + ? "IMAGE_AUTH_FAILED" + : `IMAGE_HTTP_${response.status}`, + ); + } + const raw = await boundedBytes(response, Math.ceil((MAX_IMAGE_BYTES * 4) / 3) + 65536); + let payload: unknown; + try { + payload = JSON.parse(Buffer.from(raw).toString("utf8")); + } catch { + throw imageError("IMAGE_INVALID_RESPONSE"); + } + const data = (payload as { data?: unknown })?.data; + if (!Array.isArray(data) || data.length !== 1 || !data[0] || typeof data[0] !== "object") + throw imageError("IMAGE_INVALID_RESPONSE"); + const item = data[0] as Record; + let bytes: Uint8Array; + if ( + typeof item.b64_json === "string" && + item.b64_json.length && + /^(?:[A-Za-z0-9+/]{4})*(?:[A-Za-z0-9+/]{2}==|[A-Za-z0-9+/]{3}=)?$/.test(item.b64_json) + ) { + bytes = Buffer.from(item.b64_json, "base64"); + } else if (typeof item.url === "string") { + bytes = await downloadGeneratedImage(item.url, signal); + } else throw imageError("IMAGE_INVALID_RESPONSE"); + return { bytes, ...generatedImageType(bytes) }; +} + +/** Two workers, stable result ordering, no automatic retry of billable requests. */ +export async function generateImageBatch(options: { + input: unknown; + endpoint: ImageEndpoint; + signal: AbortSignal; + save: (image: Awaited>, index: number) => Promise; + fetchImpl?: typeof fetch; + loadImages?: (paths: string[]) => Promise; +}): Promise { + const prompts = imageGenerationItems(options.input); + const results: GeneratedImageResult[] = new Array(prompts.length); + let next = 0; + let authFailed = false; + const worker = async () => { + while (next < prompts.length) { + const index = next++; + if (options.signal.aborted || authFailed) { + results[index] = { + index, + status: "cancelled", + errorCode: authFailed ? "IMAGE_AUTH_FAILED" : "IMAGE_CANCELLED", + }; + continue; + } + const controller = new AbortController(); + const signal = AbortSignal.any([options.signal, controller.signal]); + const timer = setTimeout(() => controller.abort(), IMAGE_GENERATION_TIMEOUT_MS); + try { + const item = prompts[index]; + if (item.images && !options.loadImages) throw imageError("IMAGE_EDIT_UNAVAILABLE"); + const images = item.images ? await options.loadImages!(item.images) : []; + signal.throwIfAborted(); + const image = await generateOneImage( + options.endpoint, + item.prompt, + signal, + options.fetchImpl, + images, + ); + const path = await options.save(image, index); + results[index] = { index, status: "succeeded", path, mimeType: image.mimeType }; + } catch (error) { + const code = options.signal.aborted + ? "IMAGE_CANCELLED" + : controller.signal.aborted + ? "IMAGE_TIMEOUT" + : error && + typeof error === "object" && + "errorCode" in error && + typeof error.errorCode === "string" + ? error.errorCode + : "IMAGE_REQUEST_FAILED"; + if (code === "IMAGE_AUTH_FAILED") authFailed = true; + results[index] = { + index, + status: options.signal.aborted ? "cancelled" : "failed", + errorCode: code, + }; + } finally { + clearTimeout(timer); + } + } + }; + await Promise.all([worker(), worker()]); + return results; +} diff --git a/packages/agent-runtime/src/image-generation/tool.ts b/packages/agent-runtime/src/image-generation/tool.ts new file mode 100644 index 0000000000..e7d8779262 --- /dev/null +++ b/packages/agent-runtime/src/image-generation/tool.ts @@ -0,0 +1,19 @@ +import { Type } from "@earendil-works/pi-ai"; + +export const imageGenerationDescription = + "Generate raster images using the image model configured in Settings > AI. items supports distinct prompts and count variants; at most 10 images total. Each image may incur a charge. Generate only the requested number, report partial failures, and do not retry without the user's request. Results contain local image paths: display successful images with Markdown image links. For edits, provide images as local paths from the session, attachments, or project. Use previous result paths to refine generated images; preserve originals."; +export const imageGenerationParameters = { + items: Type.Array( + Type.Object({ + prompt: Type.String({ minLength: 1, maxLength: 32000 }), + images: Type.Optional( + Type.Union([ + Type.Array(Type.String({ minLength: 1, maxLength: 4096 }), { maxItems: 4 }), + Type.Null(), + ]), + ), + count: Type.Optional(Type.Integer({ minimum: 1, maximum: 10 })), + }), + { minItems: 1, maxItems: 10 }, + ), +}; diff --git a/packages/agent-runtime/src/index.ts b/packages/agent-runtime/src/index.ts index 5546ac24b1..2834f0aa61 100644 --- a/packages/agent-runtime/src/index.ts +++ b/packages/agent-runtime/src/index.ts @@ -23,3 +23,4 @@ export { startAuthenticatedProxyRelay } from "./authenticated-proxy-relay.js"; export type { AuthenticatedProxyRelay } from "./authenticated-proxy-relay.js"; export * from "./speech/index.js"; +export * from "./image-generation/index.js"; diff --git a/packages/agent-runtime/src/runtime.ts b/packages/agent-runtime/src/runtime.ts index 60f6a8014f..948a1997d4 100644 --- a/packages/agent-runtime/src/runtime.ts +++ b/packages/agent-runtime/src/runtime.ts @@ -1,3 +1,4 @@ +import { imageGenerationDescription, imageGenerationParameters } from "./image-generation/tool.js"; import { scheduledToolParameters, scheduledToolDescriptions } from "./scheduled-tools.js"; import { randomUUID } from "node:crypto"; import { @@ -2749,6 +2750,7 @@ Delegation rules: const externalPathHint = " An explicit path outside the workspace and session scratch roots requires permission unless the effective mode is Auto."; const describe = (toolName: string): string => { + if (toolName === "GenerateImages") return imageGenerationDescription; if (scheduledToolDescriptions[toolName]) return scheduledToolDescriptions[toolName]; switch (toolName) { case "BrowserPreview": @@ -2801,6 +2803,7 @@ Delegation rules: // One entry per tool: the shapes diverge enough that a chain of ternaries // stopped being readable. const parameters: Record[0]> = { + GenerateImages: imageGenerationParameters, Read: { path: pathParam( "Existing regular file only, never a directory; workspace-relative or explicitly approved.", @@ -2947,7 +2950,7 @@ Delegation rules: }) : undefined; const abort = () => { - if (!isBash || abortRequested || settled) return; + if ((!isBash && toolName !== "GenerateImages") || abortRequested || settled) return; abortRequested = true; abortPromise = this.host .call("tools.abort", { @@ -3260,7 +3263,7 @@ Delegation rules: ] : ["Read", "Glob", "Grep", "BrowserPreview", "Bash"]; if (this.mode === "agent") { - tools.push("PluginScaffold", "PluginPack", ...Object.keys(scheduledToolParameters)); + tools.push("PluginScaffold", "PluginPack", "GenerateImages", ...Object.keys(scheduledToolParameters)); } const builtins = tools.map(exec); diff --git a/packages/host-runtime/src/agent-sidecar.ts b/packages/host-runtime/src/agent-sidecar.ts index b89741ae86..3b35323ef3 100644 --- a/packages/host-runtime/src/agent-sidecar.ts +++ b/packages/host-runtime/src/agent-sidecar.ts @@ -1,6 +1,6 @@ import { spawn, type ChildProcessWithoutNullStreams } from "node:child_process"; import { randomUUID } from "node:crypto"; -import { DEFAULT_RPC_TIMEOUT_MS, readNdjsonLines, rpcTimeoutMs } from "@pi-desktop/shared"; +import { DEFAULT_RPC_TIMEOUT_MS, IMAGE_BATCH_TIMEOUT_MS, imageGenerationPrompts, readNdjsonLines, rpcTimeoutMs } from "@pi-desktop/shared"; import type { ProcessExitHandler, StderrHandler } from "./host-process.js"; // stderr lines kept per sidecar so an unexpected exit can be reported with the @@ -21,6 +21,7 @@ export type LocalToolHandler = (input: { sessionId: string; toolCallId: string; args: unknown; + signal: AbortSignal; }) => Promise; export type ProjectInstructionResolver = (input: { @@ -142,6 +143,7 @@ export class AgentSidecar { private stdoutReader?: ReturnType; // Tools served by the embedding host itself (e.g. BrowserPreview drives the // work panel's WebContentsView) — host-core never sees these. + private localToolControllers = new Map(); private localTools = new Map(); private localToolTimers = new Set>(); private projectInstructionResolver: ProjectInstructionResolver | null = null; @@ -210,6 +212,8 @@ export class AgentSidecar { this.pending.clear(); for (const timer of this.localToolTimers) clearTimeout(timer); this.localToolTimers.clear(); + for (const controller of this.localToolControllers.values()) controller.abort(); + this.localToolControllers.clear(); this.handlers.clear(); this.stdoutReader?.close(); this.stdoutReader = undefined; @@ -258,24 +262,39 @@ export class AgentSidecar { private async runLocalTool( handler: LocalToolHandler, - input: Parameters[0], + input: Omit[0], "signal">, + params: Record, ): Promise { + const key = `${input.sessionId}:${input.toolCallId}`; + if (this.localToolControllers.has(key)) throw new Error("duplicate local tool call"); + const controller = new AbortController(); + this.localToolControllers.set(key, controller); + const imageGeneration = params.toolName === "GenerateImages"; let timer: ReturnType | undefined; try { return await Promise.race([ - handler(input), + (async () => { + if (imageGeneration) { + imageGenerationPrompts(input.args); + if (!this.host) throw new Error("host unavailable"); + const gate = await this.host.call("tools.execute", params); + if (!gate.ok) return gate; + controller.signal.throwIfAborted(); + } + return handler({ ...input, signal: controller.signal }); + })(), new Promise((_, reject) => { timer = setTimeout(() => { + controller.abort(); reject(new Error("host-local tool timeout")); - }, DEFAULT_RPC_TIMEOUT_MS); + }, imageGeneration ? IMAGE_BATCH_TIMEOUT_MS + 130_000 : DEFAULT_RPC_TIMEOUT_MS); this.localToolTimers.add(timer); }), ]); } finally { - if (timer) { - clearTimeout(timer); - this.localToolTimers.delete(timer); - } + controller.abort(); + this.localToolControllers.delete(key); + if (timer) { clearTimeout(timer); this.localToolTimers.delete(timer); } } } @@ -340,6 +359,9 @@ export class AgentSidecar { setHost(host: SidecarHostLink) { if (this.closed) return; + if (this.host && this.host !== host) { + for (const controller of this.localToolControllers.values()) controller.abort(); + } this.host = host; this.unsubscribeHost?.(); this.unsubscribeHostExit?.(); @@ -358,6 +380,7 @@ export class AgentSidecar { this.writeToChild(payload); }); this.unsubscribeHostExit = host.onExit(() => { + for (const controller of this.localToolControllers.values()) controller.abort(); this.unsubscribeHost?.(); this.unsubscribeHost = null; this.unsubscribeHostExit = null; @@ -462,6 +485,9 @@ export class AgentSidecar { { code: -32000, data: { errorCode: "TOOL_DISABLED_IN_PLAN" } }, ); } + if (method === "tools.abort") { + this.localToolControllers.get(`${params.sessionId}:${params.toolCallId}`)?.abort(); + } if (method === "project.instructions.resolve") { if (!this.projectInstructionResolver) { throw new Error("project instruction resolver unavailable"); @@ -530,7 +556,7 @@ export class AgentSidecar { sessionId: String(params.sessionId ?? ""), toolCallId: String(params.toolCallId ?? ""), args: params.args, - }); + }, params); this.writeToChild( JSON.stringify({ jsonrpc: "2.0", id: msg.id, result }) + "\n", ); diff --git a/packages/host-runtime/src/image-generation-bridge.test.ts b/packages/host-runtime/src/image-generation-bridge.test.ts new file mode 100644 index 0000000000..d35ac2b58f --- /dev/null +++ b/packages/host-runtime/src/image-generation-bridge.test.ts @@ -0,0 +1,83 @@ +import { afterEach, expect, it, vi } from "vitest"; +import { AgentSidecar, type SidecarHostLink } from "./agent-sidecar.js"; + +// A real stdio child forwards requested calls over the production reverse RPC. +const child = `const rl=require('node:readline').createInterface({input:process.stdin}); +const pending=new Map(); rl.on('line',line=>{const m=JSON.parse(line); +if(m.method==='probe'){pending.set('r'+m.id,m.id);console.log(JSON.stringify({id:'r'+m.id,method:'host.proxy',params:m.params}));} +else if(pending.has(m.id)){console.log(JSON.stringify({...m,id:pending.get(m.id)}));pending.delete(m.id);}});`; +const sidecars: AgentSidecar[] = []; +afterEach(async () => { + await Promise.all(sidecars.splice(0).map((sidecar) => sidecar.dispose())); +}); +function harness(allowed = true) { + const sidecar = new AgentSidecar({ + launch: { command: process.execPath, args: ["-e", child] }, + onStderr: () => {}, + }); + sidecars.push(sidecar); + const calls: string[] = []; + const host: SidecarHostLink = { + async call(method: string): Promise { + calls.push(method); + return { ok: allowed, content: allowed ? { authorized: true } : "denied" } as T; + }, + onNotification: () => () => {}, + onExit: () => () => {}, + }; + sidecar.setHost(host); + const execute = (mode = "agent", toolCallId = "i") => + sidecar.call<{ ok: boolean }>("probe", { + method: "tools.execute", + params: { + sessionId: "s", + toolCallId, + mode, + toolName: "GenerateImages", + args: { items: [{ prompt: "image" }] }, + }, + }); + return { sidecar, calls, execute }; +} + +it("authorizes image calls through the host before executing the local handler", async () => { + const { sidecar, calls, execute } = harness(); + sidecar.setLocalTool("GenerateImages", async () => { + calls.push("generated"); + return { ok: true, content: "image" }; + }); + expect((await execute()).ok).toBe(true); + expect(calls).toEqual(["tools.execute", "generated"]); +}); + +it("denied and Plan calls never reach the image service", async () => { + const { sidecar, execute } = harness(false); + const generate = vi.fn().mockResolvedValue({ ok: true, content: "image" }); + sidecar.setLocalTool("GenerateImages", generate); + expect((await execute()).ok).toBe(false); + expect((await execute("plan")).ok).toBe(false); + expect(generate).not.toHaveBeenCalled(); +}); + +it("tools.abort reaches the in-flight request and still forwards host cancellation", async () => { + const { sidecar, calls, execute } = harness(); + let started!: () => void; + const ready = new Promise((resolve) => { + started = resolve; + }); + sidecar.setLocalTool("GenerateImages", async ({ signal }) => { + started(); + await new Promise((resolve) => + signal.addEventListener("abort", () => resolve(), { once: true }), + ); + return { ok: false, content: "cancelled" }; + }); + const pending = execute(); + await ready; + await sidecar.call("probe", { + method: "tools.abort", + params: { sessionId: "s", toolCallId: "i" }, + }); + expect((await pending).ok).toBe(false); + expect(calls).toEqual(["tools.execute", "tools.abort"]); +}); diff --git a/packages/i18n/src/locales/de/index.ts b/packages/i18n/src/locales/de/index.ts index 4c4fba2d24..7d1d8aeaff 100644 --- a/packages/i18n/src/locales/de/index.ts +++ b/packages/i18n/src/locales/de/index.ts @@ -550,6 +550,19 @@ export const de = { "dismiss": "Verwerfen" }, "settings": { + "imageModel": "Bildgenerierungsmodell", + "imageModelUnset": "Nicht konfiguriert", + "imageModelUnavailable": "Dieses Bildmodell ist nicht verfügbar. Wählen Sie ein anderes.", + "imageModelSaveFailed": "Bildmodell konnte nicht gespeichert werden.", + "clearImageModel": "Löschen", + "setImageModel": "Als Bildmodell festlegen", + "imageModelSelected": "Ausgewähltes Bildmodell", + "generatedImage": "Generiertes Bild {{index}}", + "imagePreviewUnavailable": "Vorschau nicht verfügbar; Datei öffnen", + "imageGenerationFailed": "Bild {{index}} wurde nicht fertiggestellt.", + "imageModelSetupHint": "Konfigurieren Sie ein Bildmodell unter Einstellungen → AI.", + "configureImageModel": "Bildmodell konfigurieren", + sklm: { browse: "Markt", diff --git a/packages/i18n/src/locales/en/index.ts b/packages/i18n/src/locales/en/index.ts index e3b69ebcc5..546d27a1c6 100644 --- a/packages/i18n/src/locales/en/index.ts +++ b/packages/i18n/src/locales/en/index.ts @@ -557,6 +557,19 @@ export const en = { dismiss: "Dismiss", }, settings: { + imageModel: "Image generation model", + imageModelUnset: "Not configured", + imageModelUnavailable: "This image model is unavailable. Choose another model.", + imageModelSaveFailed: "Could not save the image model.", + clearImageModel: "Clear", + setImageModel: "Set as image model", + imageModelSelected: "Selected image model", + generatedImage: "Generated image {{index}}", + imagePreviewUnavailable: "Preview unavailable; open file", + imageGenerationFailed: "Image {{index}} did not complete.", + imageModelSetupHint: "Configure an image model in Settings → AI before generating images.", + configureImageModel: "Configure image model", + sklm: { browse: "Market", diff --git a/packages/i18n/src/locales/es/index.ts b/packages/i18n/src/locales/es/index.ts index 667a54c07c..04311bd850 100644 --- a/packages/i18n/src/locales/es/index.ts +++ b/packages/i18n/src/locales/es/index.ts @@ -550,6 +550,19 @@ export const es = { "dismiss": "Descartar" }, "settings": { + "imageModel": "Modelo de imágenes", + "imageModelUnset": "Sin configurar", + "imageModelUnavailable": "Este modelo no está disponible. Elige otro.", + "imageModelSaveFailed": "No se pudo guardar el modelo.", + "clearImageModel": "Borrar", + "setImageModel": "Usar para generar imágenes", + "imageModelSelected": "Modelo de imágenes seleccionado", + "generatedImage": "Imagen generada {{index}}", + "imagePreviewUnavailable": "Vista previa no disponible; abrir archivo", + "imageGenerationFailed": "La imagen {{index}} no se completó.", + "imageModelSetupHint": "Configura un modelo de imágenes en Ajustes → AI.", + "configureImageModel": "Configurar modelo de imágenes", + sklm: { browse: "Mercado", diff --git a/packages/i18n/src/locales/fr/index.ts b/packages/i18n/src/locales/fr/index.ts index d69c1f39a1..b3cae75a30 100644 --- a/packages/i18n/src/locales/fr/index.ts +++ b/packages/i18n/src/locales/fr/index.ts @@ -550,6 +550,19 @@ export const fr = { "dismiss": "Ignorer" }, "settings": { + "imageModel": "Modèle de génération d’images", + "imageModelUnset": "Non configuré", + "imageModelUnavailable": "Ce modèle est indisponible. Choisissez-en un autre.", + "imageModelSaveFailed": "Impossible d’enregistrer le modèle.", + "clearImageModel": "Effacer", + "setImageModel": "Définir comme modèle d’images", + "imageModelSelected": "Modèle d’images sélectionné", + "generatedImage": "Image générée {{index}}", + "imagePreviewUnavailable": "Aperçu indisponible ; ouvrir le fichier", + "imageGenerationFailed": "L’image {{index}} n’a pas été terminée.", + "imageModelSetupHint": "Configurez un modèle d’images dans Paramètres → AI.", + "configureImageModel": "Configurer le modèle d’images", + sklm: { browse: "Marché", diff --git a/packages/i18n/src/locales/ko/index.ts b/packages/i18n/src/locales/ko/index.ts index 9c9cb78c7f..1396300aa5 100644 --- a/packages/i18n/src/locales/ko/index.ts +++ b/packages/i18n/src/locales/ko/index.ts @@ -559,6 +559,19 @@ export const ko = { dismiss: "닫기", }, settings: { + "imageModel": "이미지 생성 모델", + "imageModelUnset": "설정되지 않음", + "imageModelUnavailable": "이 이미지 모델을 사용할 수 없습니다. 다른 모델을 선택하세요.", + "imageModelSaveFailed": "이미지 모델을 저장할 수 없습니다.", + "clearImageModel": "지우기", + "setImageModel": "이미지 모델로 설정", + "imageModelSelected": "선택된 이미지 모델", + "generatedImage": "생성된 이미지 {{index}}", + "imagePreviewUnavailable": "미리보기 불가; 파일 열기", + "imageGenerationFailed": "이미지 {{index}} 생성이 완료되지 않았습니다.", + "imageModelSetupHint": "설정 → AI에서 이미지 모델을 설정하세요.", + "configureImageModel": "이미지 모델 설정", + sklm: { browse: "마켓", diff --git a/packages/i18n/src/locales/tr/index.ts b/packages/i18n/src/locales/tr/index.ts index a3e57db14f..e7d3a1a60d 100644 --- a/packages/i18n/src/locales/tr/index.ts +++ b/packages/i18n/src/locales/tr/index.ts @@ -559,6 +559,19 @@ export const tr = { dismiss: "Kapat", }, settings: { + "imageModel": "Görsel oluşturma modeli", + "imageModelUnset": "Yapılandırılmadı", + "imageModelUnavailable": "Bu görsel modeli kullanılamıyor. Başka bir model seçin.", + "imageModelSaveFailed": "Görsel modeli kaydedilemedi.", + "clearImageModel": "Temizle", + "setImageModel": "Görsel modeli olarak ayarla", + "imageModelSelected": "Seçili görsel modeli", + "generatedImage": "Oluşturulan görsel {{index}}", + "imagePreviewUnavailable": "Önizleme yok; dosyayı aç", + "imageGenerationFailed": "Görsel {{index}} tamamlanmadı.", + "imageModelSetupHint": "Ayarlar → AI bölümünde bir görsel modeli yapılandırın.", + "configureImageModel": "Görsel modelini yapılandır", + sklm: { browse: "Market", diff --git a/packages/i18n/src/locales/zh-CN/index.ts b/packages/i18n/src/locales/zh-CN/index.ts index 6e6020c2d8..172be5da83 100644 --- a/packages/i18n/src/locales/zh-CN/index.ts +++ b/packages/i18n/src/locales/zh-CN/index.ts @@ -554,6 +554,19 @@ export const zhCN = { dismiss: "关闭", }, settings: { + imageModel: "生图模型", + imageModelUnset: "未配置", + imageModelUnavailable: "当前生图模型不可用,请选择其他模型。", + imageModelSaveFailed: "无法保存生图模型。", + clearImageModel: "清除", + setImageModel: "设为生图模型", + imageModelSelected: "已设为生图模型", + generatedImage: "生成图片 {{index}}", + imagePreviewUnavailable: "预览不可用,打开文件", + imageGenerationFailed: "图片 {{index}} 未完成。", + imageModelSetupHint: "请前往 设置 → AI → 生图模型 完成配置。", + configureImageModel: "配置生图模型", + sklm: { browse: "市场", diff --git a/packages/i18n/src/locales/zh-TW/index.ts b/packages/i18n/src/locales/zh-TW/index.ts index 6bd8270358..0184c917c4 100644 --- a/packages/i18n/src/locales/zh-TW/index.ts +++ b/packages/i18n/src/locales/zh-TW/index.ts @@ -554,6 +554,19 @@ export const zhTW = { dismiss: "關閉", }, settings: { + imageModel: "生圖模型", + imageModelUnset: "未設定", + imageModelUnavailable: "目前生圖模型無法使用,請選擇其他模型。", + imageModelSaveFailed: "無法儲存生圖模型。", + clearImageModel: "清除", + setImageModel: "設為生圖模型", + imageModelSelected: "已設為生圖模型", + generatedImage: "生成圖片 {{index}}", + imagePreviewUnavailable: "無法預覽,開啟檔案", + imageGenerationFailed: "圖片 {{index}} 未完成。", + imageModelSetupHint: "請前往 設定 → AI → 生圖模型 完成設定。", + configureImageModel: "設定生圖模型", + sklm: { browse: "市場", diff --git a/packages/shared/src/changelog-de.ts b/packages/shared/src/changelog-de.ts index 7931ce40bf..e6425a2ff4 100644 --- a/packages/shared/src/changelog-de.ts +++ b/packages/shared/src/changelog-de.ts @@ -5,6 +5,7 @@ export const deEntries: ChangelogEntry[] = [ "version": "0.15.2", "date": "2026-09-21", "highlights": [ + "Bilder im Chat generieren und bearbeiten, ein Bildmodell wählen und mit dem integrierten imagegen-Skill Stapel erstellen.", "Werkzeugaktivitäten folgen der Gesprächsbreite; lange Aktivitätsnamen werden sauber begrenzt.", "Dateianhänge aus dem Einfügen bleiben erhalten, auch wenn der Vorgang nach einem Sitzungswechsel endet.", "Das ausgewählte Standardmodell bleibt beim Bearbeiten von Anbietern erhalten und fällt sicher zurück, wenn es entfernt wird.", diff --git a/packages/shared/src/changelog-es.ts b/packages/shared/src/changelog-es.ts index c5fc554905..03dc259050 100644 --- a/packages/shared/src/changelog-es.ts +++ b/packages/shared/src/changelog-es.ts @@ -5,6 +5,7 @@ export const esEntries: ChangelogEntry[] = [ "version": "0.15.2", "date": "2026-09-21", "highlights": [ + "Genera y edita imágenes en el chat, elige un modelo y crea lotes con la habilidad integrada imagegen.", "La actividad de herramientas sigue el ancho de la conversación y contiene correctamente las etiquetas largas.", "Los archivos adjuntos pegados se conservan aunque el pegado termine después de cambiar de sesión.", "El modelo predeterminado seleccionado se conserva al editar proveedores y se aplica un respaldo seguro si se elimina.", diff --git a/packages/shared/src/changelog-fr.ts b/packages/shared/src/changelog-fr.ts index aa9aaf9fa3..d2b6c77c9f 100644 --- a/packages/shared/src/changelog-fr.ts +++ b/packages/shared/src/changelog-fr.ts @@ -5,6 +5,7 @@ export const frEntries: ChangelogEntry[] = [ "version": "0.15.2", "date": "2026-09-21", "highlights": [ + "Générez et modifiez des images dans le chat, choisissez un modèle et créez des lots avec la compétence intégrée imagegen.", "L'activité des outils suit la largeur de la conversation et contient proprement les libellés longs.", "Les pièces jointes collées sont conservées même si le collage se termine après un changement de session.", "Le modèle par défaut sélectionné est conservé lors de la modification des fournisseurs, avec un repli sûr s'il est supprimé.", diff --git a/packages/shared/src/changelog-ko.ts b/packages/shared/src/changelog-ko.ts index 8d53800e38..ec5ecf31e0 100644 --- a/packages/shared/src/changelog-ko.ts +++ b/packages/shared/src/changelog-ko.ts @@ -5,6 +5,7 @@ export const koEntries: ChangelogEntry[] = [ version: "0.15.2", date: "2026-09-21", highlights: [ + "채팅에서 이미지를 생성하고 편집하며, 이미지 모델과 내장 imagegen 스킬로 일괄 생성할 수 있습니다.", "도구 활동이 대화 너비를 따라 정렬되고 긴 활동 이름도 깔끔하게 표시됩니다.", "세션을 전환한 뒤 붙여넣기가 완료되어도 붙여넣은 파일 첨부가 유지됩니다.", "제공자를 편집할 때 선택한 기본 모델을 유지하며, 모델이 삭제되면 안전하게 대체 모델로 전환합니다.", diff --git a/packages/shared/src/changelog-tr.ts b/packages/shared/src/changelog-tr.ts index 18ee141929..12986cec89 100644 --- a/packages/shared/src/changelog-tr.ts +++ b/packages/shared/src/changelog-tr.ts @@ -5,6 +5,7 @@ export const trEntries: ChangelogEntry[] = [ "version": "0.15.2", "date": "2026-09-21", "highlights": [ + "Sohbette görsel oluşturun ve düzenleyin; tek bir görsel modeli ve yerleşik imagegen becerisiyle toplu üretim yapın.", "Araç etkinlikleri konuşma genişliğini izler ve uzun etkinlik etiketleri düzgünce sığdırılır.", "Oturum değiştirdikten sonra yapıştırma tamamlansa bile yapıştırılan dosya ekleri korunur.", "Sağlayıcıları düzenlerken seçili varsayılan model korunur; model kaldırılırsa güvenli bir yedek kullanılır.", diff --git a/packages/shared/src/changelog.ts b/packages/shared/src/changelog.ts index 828aa03869..299f1afd98 100644 --- a/packages/shared/src/changelog.ts +++ b/packages/shared/src/changelog.ts @@ -32,6 +32,7 @@ const enEntries: ChangelogEntry[] = [ version: "0.15.2", date: "2026-09-21", highlights: [ + "Generate and edit images in chat, choose one image model, and create batches with the built-in imagegen skill.", "Keep tool activity aligned with the conversation width and contain long activity labels cleanly.", "Preserve pasted file attachments when a paste finishes after switching sessions.", "Keep the selected default model when editing providers, and fall back safely when it is removed.", @@ -827,6 +828,7 @@ const zhCNEntries: ChangelogEntry[] = [ version: "0.15.2", date: "2026-09-21", highlights: [ + "支持聊天生图与图片编辑,可设置唯一生图模型,并通过内置 imagegen 技能批量生成。", "让工具活动跟随对话宽度排列,并妥善收纳过长的活动名称。", "切换会话后,如果粘贴操作稍后完成,文件附件也会保留。", "编辑提供商时保留已选的默认模型;模型被移除后安全回退。", @@ -1622,6 +1624,7 @@ const zhTWEntries: ChangelogEntry[] = [ version: "0.15.2", date: "2026-09-21", highlights: [ + "支援聊天生圖與圖片編輯,可設定唯一生圖模型,並透過內建 imagegen 技能批次生成。", "讓工具活動跟隨對話寬度排列,並妥善收納過長的活動名稱。", "切換工作階段後,即使貼上操作稍後完成,檔案附件也會保留。", "編輯提供商時保留已選的預設模型;模型被移除後安全回退。", diff --git a/packages/shared/src/image-generation.ts b/packages/shared/src/image-generation.ts new file mode 100644 index 0000000000..8a88980210 --- /dev/null +++ b/packages/shared/src/image-generation.ts @@ -0,0 +1,75 @@ +/** A single host-owned binding, independent of the default conversation model. */ +export type ImageGenerationBinding = { providerId: string; modelId: string }; +export const MAX_GENERATED_IMAGES = 10; +export const IMAGE_GENERATION_TIMEOUT_MS = 180_000; +export const IMAGE_BATCH_TIMEOUT_MS = 950_000; +export type ImageGenerationItem = { prompt: string; count?: number; images?: string[] }; +export type ImageGenerationInput = { items: ImageGenerationItem[] }; +export type GeneratedImageResult = { + index: number; + status: "succeeded" | "failed" | "cancelled"; + path?: string; + mimeType?: string; + errorCode?: string; +}; + +function invalid(message: string): never { + throw Object.assign(new Error(message), { errorCode: "INVALID_ARGUMENT" }); +} + +export function parseImageGenerationBinding(value: unknown): ImageGenerationBinding | null { + if (value == null) return null; + if (typeof value !== "object" || Array.isArray(value)) + invalid("imageGeneration must be an object"); + const record = value as Record; + const field = (name: string, max: number) => { + const raw = record[name]; + if (typeof raw !== "string" || !raw.trim() || raw.length > max) + invalid(`imageGeneration.${name} is invalid`); + return raw.trim(); + }; + return { providerId: field("providerId", 128), modelId: field("modelId", 256) }; +} + +/** Expand variants into individually accountable requests; never silently truncate. */ +export function imageGenerationItems(value: unknown): ImageGenerationItem[] { + if (!value || typeof value !== "object" || Array.isArray(value)) invalid("items are required"); + const items = (value as Record).items; + if (!Array.isArray(items) || !items.length || items.length > MAX_GENERATED_IMAGES) + invalid("items must contain 1–10 entries"); + const prompts: ImageGenerationItem[] = []; + for (const item of items) { + if (!item || typeof item !== "object" || Array.isArray(item)) invalid("invalid image item"); + const { prompt, count = 1, images } = item as Record; + if (typeof prompt !== "string" || !prompt.trim() || prompt.length > 32_000) + invalid("prompt must contain 1–32000 characters"); + if ( + typeof count !== "number" || + !Number.isInteger(count) || + count < 1 || + count > MAX_GENERATED_IMAGES + ) + invalid("count must be an integer from 1 to 10"); + if ( + images != null && + (!Array.isArray(images) || + images.length > 4 || + images.some((path) => typeof path !== "string" || !path.trim() || path.length > 4096)) + ) + invalid("images must contain at most 4 local image paths"); + for (let i = 0; i < count; i++) + prompts.push({ + prompt: prompt.trim(), + ...(Array.isArray(images) && images.length + ? { images: (images as string[]).map((path) => path.trim()) } + : {}), + }); + if (prompts.length > MAX_GENERATED_IMAGES) + invalid("at most 10 images may be generated per batch"); + } + return prompts; +} + +export function imageGenerationPrompts(value: unknown): string[] { + return imageGenerationItems(value).map((item) => item.prompt); +} diff --git a/packages/shared/src/index.ts b/packages/shared/src/index.ts index c310ecbac2..bf6fc44b9c 100644 --- a/packages/shared/src/index.ts +++ b/packages/shared/src/index.ts @@ -49,6 +49,7 @@ export * from "./github-feedback.js"; export * from "./network-proxy.js"; export * from "./attachment-limits.js"; export * from "./speech.js"; +export * from "./image-generation.js"; export * from "./font-size.js"; export * from "./chat-content-width.js"; export * from "./racp.js"; diff --git a/packages/shared/src/rpc-timeouts.ts b/packages/shared/src/rpc-timeouts.ts index 57a69f682e..e84b754f68 100644 --- a/packages/shared/src/rpc-timeouts.ts +++ b/packages/shared/src/rpc-timeouts.ts @@ -1,3 +1,5 @@ +import { IMAGE_BATCH_TIMEOUT_MS } from "./image-generation.js"; + export const DEFAULT_RPC_TIMEOUT_MS = 130_000; export const PERMISSION_TIMEOUT_MS = 120_000; export const DEFAULT_COMMAND_TIMEOUT_MS = 60_000; @@ -53,6 +55,7 @@ export function rpcTimeoutMs( if (method !== "tools.execute") return DEFAULT_RPC_TIMEOUT_MS; const input = isRecord(params) ? params : undefined; + if (input?.toolName === "GenerateImages") return IMAGE_BATCH_TIMEOUT_MS + PERMISSION_TIMEOUT_MS + TOOL_QUEUE_WAIT_MS + COMMAND_RPC_BUFFER_MS; if (isDesktopDispatchedTool(input?.toolName)) { return executionRpcTimeoutMs( input?.timeoutMs, diff --git a/packages/shared/src/types/settings.ts b/packages/shared/src/types/settings.ts index b351f1ef39..f65060c472 100644 --- a/packages/shared/src/types/settings.ts +++ b/packages/shared/src/types/settings.ts @@ -22,6 +22,7 @@ export type ThemePreference = "system" | "light" | "dark" | `plugin:${string}`; export type CloseBehavior = "ask" | "tray" | "quit"; export type AppSettings = { + imageGeneration?: import("../image-generation.js").ImageGenerationBinding | null; defaultProviderId?: string; defaultModelId?: string; /** Host speech bindings. Absent means voice actions stay disabled. */ diff --git a/scripts/e2e-image-generation-ui.mjs b/scripts/e2e-image-generation-ui.mjs new file mode 100644 index 0000000000..1c5609e456 --- /dev/null +++ b/scripts/e2e-image-generation-ui.mjs @@ -0,0 +1,90 @@ +#!/usr/bin/env node +/** Real React/Chromium coverage for the custom-provider API-format boundary. */ +import assert from "node:assert/strict"; +import { spawn } from "node:child_process"; +import { createRequire } from "node:module"; +import { mkdtemp, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { dirname, join } from "node:path"; +import { fileURLToPath } from "node:url"; +import { resolveElectronBinary } from "./e2e/boot.mjs"; + +const root = join(dirname(fileURLToPath(import.meta.url)), ".."); +const require = createRequire(join(root, "packages/agent-runtime/package.json")); +const { build } = require("esbuild"); +const { electronBinary } = resolveElectronBinary(root); +const temp = await mkdtemp(join(tmpdir(), "pi-image-generation-ui-")); +try { + await build({ + entryPoints: [join(root, "scripts/e2e/image-generation-ui.tsx")], + outfile: join(temp, "renderer.js"), + bundle: true, + platform: "browser", + format: "iife", + jsx: "automatic", + define: { "process.env.NODE_ENV": '"production"' }, + // Exercise API-format interactions with real components/hooks, not visual layout. + loader: { ".css": "empty" }, + alias: { + "@pi-desktop/i18n": join(root, "packages/i18n/src/index.ts"), + // The fixture lives outside the desktop package; use its React instance. + react: join(root, "apps/desktop/node_modules/react"), + "react-dom": join(root, "apps/desktop/node_modules/react-dom"), + }, + nodePaths: [join(root, "apps/desktop/node_modules")], + }); + await writeFile( + join(temp, "index.html"), + 'Image generation interactions', + ); + await writeFile( + join(temp, "main.cjs"), + ` +const { app, BrowserWindow } = require("electron"); +const path = require("node:path"); +app.setPath("userData", path.join(__dirname, "profile")); +app.whenReady().then(async () => { + const window = new BrowserWindow({ show: false, webPreferences: { sandbox: true, contextIsolation: true, nodeIntegration: false } }); + window.webContents.on("console-message", (event) => console.error(event.message)); + try { + await window.loadFile(path.join(__dirname, "index.html")); + const result = await window.webContents.executeJavaScript("globalThis.imageGenerationProbe()"); + console.log("IMAGE_GENERATION_UI_PROBE " + JSON.stringify(result)); + app.quit(); + } catch (error) { + console.error("IMAGE_GENERATION_UI_PROBE " + JSON.stringify({ ok: false, error: String(error) })); + app.exit(1); + } +}); +`, + ); + const env = { ...process.env }; + delete env.ELECTRON_RUN_AS_NODE; + const child = spawn(electronBinary, [join(temp, "main.cjs")], { + env, + stdio: ["ignore", "pipe", "pipe"], + }); + let output = ""; + for (const stream of [child.stdout, child.stderr]) + stream.on("data", (data) => { + output += data; + }); + const timeout = setTimeout(() => child.kill("SIGKILL"), 45_000); + let code; + try { + code = await new Promise((resolve, reject) => { + child.once("error", reject); + child.once("close", resolve); + }); + } finally { + clearTimeout(timeout); + } + const line = output.split(/\r?\n/).find((line) => line.startsWith("IMAGE_GENERATION_UI_PROBE ")); + assert(line, `renderer returned no probe result (exit=${code}): ${output.slice(-2000)}`); + const result = JSON.parse(line.slice("IMAGE_GENERATION_UI_PROBE ".length)); + console.log("IMAGE_GENERATION_UI_PROBE " + JSON.stringify(result)); + assert.equal(code, 0, output.slice(-6000)); + assert.equal(result.ok, true); +} finally { + await rm(temp, { recursive: true, force: true }); +} diff --git a/scripts/e2e-image-generation.mjs b/scripts/e2e-image-generation.mjs new file mode 100644 index 0000000000..efaf9d1bcb --- /dev/null +++ b/scripts/e2e-image-generation.mjs @@ -0,0 +1,139 @@ +/** Isolated host/stdio/HTTP/filesystem candidate gate; no paid providers. */ +import assert from "node:assert/strict"; +import { createServer } from "node:http"; +import { register } from "node:module"; +import { mkdtemp, mkdir, readFile, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { randomUUID } from "node:crypto"; +import { Host, resolveHostBinary } from "./e2e/host.mjs"; +import { AgentSidecar } from "../packages/host-runtime/dist/agent-sidecar.js"; +register(new URL("../apps/desktop/test/helpers/ts-import-hooks.mjs", import.meta.url)); +const { createImageGenerationTool } = await import( + "../apps/desktop/electron/main/services/image-generation-service.ts" +); +const dataDir = await mkdtemp(join(tmpdir(), "pi-image-e2e-")); +const host = new Host(resolveHostBinary(), dataDir); +const png = Buffer.from( + "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+a9mQAAAAASUVORK5CYII=", + "base64", +); +let requests = 0; +const server = createServer(async (request, response) => { + for await (const _chunk of request) { + /* Drain the bounded fixture request. */ + } + requests++; + response.setHeader("Content-Type", "application/json"); + response.end(JSON.stringify({ data: [{ b64_json: png.toString("base64") }] })); +}); +await new Promise((resolve) => server.listen(0, "127.0.0.1", resolve)); +// Only the LLM edge is simulated; reverse RPC, authorization and execution are real. +const child = `const rl=require('node:readline').createInterface({input:process.stdin});const p=new Map();rl.on('line',line=>{const m=JSON.parse(line);if(m.method==='probe'){p.set('r'+m.id,m.id);console.log(JSON.stringify({id:'r'+m.id,method:'host.proxy',params:m.params}));}else if(p.has(m.id)){console.log(JSON.stringify({...m,id:p.get(m.id)}));p.delete(m.id);}});`; +const sidecar = new AgentSidecar({ + launch: { command: process.execPath, args: ["-e", child] }, + onStderr: (text) => process.stderr.write(text), +}); +sidecar.setHost({ + call: (method, params) => host.call(method, params), + onNotification: () => () => {}, + onExit: () => () => {}, +}); +sidecar.setLocalTool("GenerateImages", createImageGenerationTool({ dataDir, getHost: () => host })); +try { + await host.start(); + const project = join(dataDir, "project"); + await mkdir(project); + await host.call("workspace.set", { path: project }); + const { provider } = await host.call("providers.create", { + name: "Image fixture", + type: "openai_compatible", + protocol: "openai_compatible", + authKind: "none", + baseUrl: `http://127.0.0.1:${server.address().port}/v1`, + defaultModelId: "image-fixture", + apiStyle: "chat_completions", + }); + const binding = { providerId: provider.id, modelId: "image-fixture" }; + await host.call("settings.set", { + defaultPermissionMode: "auto", + imageGeneration: binding, + defaultModelId: "chat-fixture", + }); + const { session } = await host.call("session.create", { + title: "Image test", + mode: "agent", + projectPath: project, + providerId: provider.id, + modelId: "chat-fixture", + }); + const execute = (id, items) => + sidecar.call("probe", { + method: "tools.execute", + params: { + sessionId: id, + toolCallId: randomUUID(), + toolName: "GenerateImages", + mode: "agent", + args: { items }, + }, + }); + const generated = await execute(session.id, [{ prompt: "cover", count: 2 }, { prompt: "icon" }]); + assert.equal(generated.ok, true, JSON.stringify(generated)); + assert.equal(generated.content.results.length, 3); + const source = generated.content.results[0].path; + const edited = await execute(session.id, [{ prompt: "green background", images: [source] }]); + assert.equal(edited.ok, true, JSON.stringify(edited)); + assert.notEqual(edited.content.results[0].path, source); + await host.call("session.appendMessage", { + sessionId: session.id, + message: { + id: randomUUID(), + role: "tool", + content: "", + toolName: "GenerateImages", + toolResult: { details: generated.content }, + createdAt: new Date().toISOString(), + status: "complete", + }, + }); + const { session: plan } = await host.call("session.create", { + title: "Plan", + mode: "plan", + projectPath: project, + }); + const denied = await execute(plan.id, [{ prompt: "must not run" }]); + assert.equal(denied.ok, false); + assert.equal(requests, 4); + await host.restart(); + const restored = await host.call("settings.get"); + assert.deepEqual(restored.imageGeneration, binding); + assert.equal(restored.defaultModelId, "chat-fixture"); + const recovered = await host.call("session.get", { id: session.id }); + assert.deepEqual(recovered.session.messages[0].toolResult.details, generated.content); + assert.deepEqual(await readFile(source), png); + await host.call("settings.set", { imageGeneration: null }); + assert.equal( + (await execute(session.id, [{ prompt: "unset" }])).errorCode, + "IMAGE_NOT_CONFIGURED", + ); + assert.equal(requests, 4); + console.log( + JSON.stringify({ + ok: true, + fixtureRequests: requests, + scenarios: [ + "batch-generation", + "edit-result", + "host-plan-denial", + "settings-and-transcript-restart", + "clear-binding", + ], + }), + ); +} finally { + await sidecar.dispose(); + await host.stop(); + await new Promise((resolve) => server.close(resolve)); + await rm(dataDir, { recursive: true, force: true }); +} diff --git a/scripts/e2e/image-generation-ui.tsx b/scripts/e2e/image-generation-ui.tsx new file mode 100644 index 0000000000..5bf1bb2c85 --- /dev/null +++ b/scripts/e2e/image-generation-ui.tsx @@ -0,0 +1,201 @@ +import { createRoot } from "react-dom/client"; +import { flushSync } from "react-dom"; +import { createInstance } from "i18next"; +import { I18nextProvider } from "react-i18next"; +import { catalogs } from "@pi-desktop/i18n"; +import type { AppSettings, ProviderPublic, UiMessage } from "@pi-desktop/shared"; +import { ModelConfigPage } from "../../apps/desktop/src/components/settings/ModelConfigPage"; +import { GeneratedImages } from "../../apps/desktop/src/features/chat/transcript/GeneratedImages"; +import { api } from "../../apps/desktop/src/lib/api"; +import { useAppStore } from "../../apps/desktop/src/stores/app-store"; + +declare global { + var imageGenerationProbe: () => Promise; +} +const assert = (condition: unknown, message: string) => { + if (!condition) throw new Error(message); +}; +const painted = () => new Promise((resolve) => requestAnimationFrame(() => resolve())); +async function until(condition: () => boolean, message: string) { + const deadline = performance.now() + 6000; + while (!condition() && performance.now() < deadline) await painted(); + assert(condition(), message); +} + +globalThis.imageGenerationProbe = async () => { + const i18n = createInstance(); + await i18n.init({ + lng: "en", + resources: { en: { translation: catalogs.en }, "zh-CN": { translation: catalogs["zh-CN"] } }, + }); + let settings = { + defaultMode: "agent", + defaultProviderId: "chat", + defaultModelId: "chat-model", + language: "en", + } as AppSettings; + const provider: ProviderPublic = { + id: "p", + name: "Images", + baseUrl: "https://example.com/v1", + vendorKey: "custom", + type: "openai_compatible", + protocol: "openai_compatible", + apiStyle: "chat_completions", + authKind: "api_key_and_base_url", + enabled: true, + hasSecret: true, + supportsReasoning: false, + supportedThinkingLevels: [], + createdAt: "", + updatedAt: "", + models: ["image-one", "image-two"].map((id) => ({ + id, + contextWindow: 32000, + maxTokens: 4096, + thinkingLevels: [], + defaultThinkingLevel: null, + })), + }; + api.getSettings = async () => structuredClone(settings); + api.setSettings = async (next) => { + settings = structuredClone(next); + return { ok: true }; + }; + api.listProviders = async () => ({ providers: [provider] }); + api.listSessions = async () => ({ sessions: [] }); + api.getOnboarding = async () => ({ dismissed: true }); + api.listProviderModels = async () => ({ models: [], source: "remote" }); + api.modelCatalogStatus = async () => ({ + loaded: true, + source: "bundled", + catalogPath: "", + providerCount: 1, + modelCount: 2, + }); + api.updateProvider = async () => ({ provider }); + api.fsReadImageDataUrl = async () => ({ + kind: "image", + dataUrl: + "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+a9mQAAAAASUVORK5CYII=", + }); + useAppStore.setState({ settings, providers: [provider] }); + const container = document.createElement("div"); + document.body.append(container); + const root = createRoot(container); + const render = (message?: UiMessage) => + flushSync(() => + root.render( + + {message ? : } + , + ), + ); + const click = (element: HTMLElement | undefined | null) => { + assert(element, "missing click target"); + flushSync(() => element!.click()); + }; + const button = (text: string) => + [...document.querySelectorAll("button")].find( + (element) => element.textContent?.trim() === text, + ); + try { + for (const locale of ["en", "zh-CN"]) { + await i18n.changeLanguage(locale); + render(); + const edit = + container.querySelector( + 'button[aria-label="' + i18n.t("settings.editProvider") + '"]', + ) ?? + container.querySelector( + ".model-provider-row button[aria-label*='Edit']", + ); + // Enter from the real model settings page, not the advanced pane alone. + click(edit); + await until( + () => !!button(i18n.t("settings.setImageModel")), + "advanced image-model action missing", + ); + click(button(i18n.t("settings.setImageModel"))); + assert(!settings.imageGeneration, "draft selection persisted before Save"); + click(button(i18n.t("settings.cancel"))); + assert(!settings.imageGeneration, "cancel changed binding"); + click( + container.querySelector( + 'button[aria-label="' + i18n.t("settings.editProvider") + '"]', + ), + ); + await until(() => !!button(i18n.t("settings.setImageModel")), "second edit did not mount"); + click(button(i18n.t("settings.setImageModel"))); + click(button(i18n.t("settings.saveProvider"))); + await until( + () => + settings.imageGeneration?.modelId === "image-one" && + !document.querySelector(".provider-setup"), + "saved image model missing", + ); + assert(settings.defaultModelId === "chat-model", "image model changed default chat model"); + const row = [...container.querySelectorAll(".settings-row")].find((element) => + element.textContent?.includes(i18n.t("settings.imageModel")), + )!; + click( + [...row.querySelectorAll("button")].find( + (element) => element.textContent === i18n.t("settings.changeDefaultModel"), + ), + ); + await until(() => !!button("Images / image-two"), "replacement picker missing"); + click(button("Images / image-two")); + await until( + () => settings.imageGeneration?.modelId === "image-two", + "replacement did not persist", + ); + click(button(i18n.t("settings.clearImageModel"))); + await until(() => settings.imageGeneration === null, "clear did not persist"); + } + const message = { + id: "image", + role: "tool", + content: "", + createdAt: "", + toolName: "GenerateImages", + toolResult: { + details: { + kind: "generated-images", + results: [ + { index: 0, status: "succeeded", path: "/scratch/result.png", mimeType: "image/png" }, + { index: 1, status: "failed", errorCode: "IMAGE_TIMEOUT" }, + ], + }, + }, + } as UiMessage; + render(message); + await until(() => !!container.querySelector("img"), "generated preview missing"); + assert(container.textContent?.includes("IMAGE_TIMEOUT"), "partial failure missing"); + render({ + ...message, + toolResult: { + details: { kind: "image-generation-error", errorCode: "IMAGE_NOT_CONFIGURED" }, + }, + }); + click(button(i18n.t("settings.configureImageModel"))); + assert( + useAppStore.getState().settingsTab === "ai" && useAppStore.getState().page === "settings", + "setup action did not navigate", + ); + return { + ok: true, + locales: ["en", "zh-CN"], + scenarios: [ + "advanced-save-cancel", + "replace-clear", + "chat-default-preserved", + "image-preview-partial-failure", + "setup-navigation", + ], + apiBoundary: "fixture", + }; + } finally { + flushSync(() => root.unmount()); + container.remove(); + } +}; diff --git a/scripts/test-image-generation-live.mjs b/scripts/test-image-generation-live.mjs new file mode 100644 index 0000000000..082adfc186 --- /dev/null +++ b/scripts/test-image-generation-live.mjs @@ -0,0 +1,167 @@ +/** Explicitly authorized opt-in smoke: at most one generation and one edit. */ +import assert from "node:assert/strict"; +import { register } from "node:module"; +import { mkdtemp, rm } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +register(new URL("../apps/desktop/test/helpers/ts-import-hooks.mjs", import.meta.url)); +const { createImageGenerationTool } = await import( + "../apps/desktop/electron/main/services/image-generation-service.ts" +); +const { imageGenerationParameters, imageGenerationDescription } = await import( + "../packages/agent-runtime/src/image-generation/tool.ts" +); +const { imageGenerationItems } = await import("../packages/shared/dist/image-generation.js"); + +let inputConfig = {}; +if (process.argv.includes("--stdin")) { + const { createInterface } = await import("node:readline"); + if (process.stdin.isTTY) process.stdin.setRawMode(true); + const reader = createInterface({ input: process.stdin, terminal: false }); + const line = await new Promise((resolve) => reader.once("line", resolve)); + reader.close(); + inputConfig = JSON.parse(line); +} +const base = inputConfig.baseUrl || process.env.PI_IMAGE_TEST_BASE_URL; +const chatKey = inputConfig.chatKey || process.env.PI_IMAGE_TEST_CHAT_KEY; +const imageKey = inputConfig.imageKey || process.env.PI_IMAGE_TEST_IMAGE_KEY; +const imageModel = inputConfig.imageModel || process.env.PI_IMAGE_TEST_MODEL; +assert(base && chatKey && imageKey && imageModel, "Set the opt-in image test environment first."); +const apiBase = new URL(base); +if (!apiBase.pathname.replace(/\/+$/, "")) apiBase.pathname = "/v1"; +const url = (path) => `${apiBase.href.replace(/\/+$/, "")}/${path}`; +const chatHeaders = { Authorization: `Bearer ${chatKey}`, "Content-Type": "application/json" }; +const directory = await mkdtemp(join(tmpdir(), "pi-image-live-")); +let attempts = 0; +try { + const list = await fetch(url("models"), { + headers: chatHeaders, + redirect: "error", + signal: AbortSignal.timeout(30_000), + }); + assert(list.ok, `Model discovery HTTP ${list.status}`); + const ids = ((await list.json()).data ?? []) + .map((row) => row.id) + .filter((id) => typeof id === "string"); + const chatModel = + process.env.PI_IMAGE_TEST_CHAT_MODEL || + ["deepseek", "deepseek-chat", "deepseek-v3"].find((id) => ids.includes(id)) || + ids.find((id) => /deepseek/i.test(id)); + assert(chatModel, "No DeepSeek model was advertised by the supplied endpoint."); + const chat = await fetch(url("chat/completions"), { + method: "POST", + headers: chatHeaders, + redirect: "error", + signal: AbortSignal.timeout(90_000), + body: JSON.stringify({ + model: chatModel, + messages: [ + { + role: "user", + content: + "Use GenerateImages exactly once to generate ONE simple raster illustration of a blue ceramic cup on a plain white background. No text, no variants, no input images. Return a tool call.", + }, + ], + tools: [ + { + type: "function", + function: { + name: "GenerateImages", + description: imageGenerationDescription, + parameters: { + type: "object", + properties: imageGenerationParameters, + required: ["items"], + }, + }, + }, + ], + tool_choice: "auto", + max_tokens: 1200, + }), + }); + assert(chat.ok, `DeepSeek HTTP ${chat.status}`); + const call = (await chat.json()).choices?.[0]?.message?.tool_calls?.[0]; + assert.equal( + call?.function?.name, + "GenerateImages", + "DeepSeek did not return the expected tool call.", + ); + const input = JSON.parse(call.function.arguments); + const jobs = imageGenerationItems(input); + assert(jobs.length === 1 && !jobs[0].images, "Live test refuses more than one initial image."); + const host = { + call: async (method) => { + if (method === "settings.get") + return { imageGeneration: { providerId: "live-image", modelId: imageModel } }; + if (method === "providers.get") + return { + provider: { + id: "live-image", + enabled: true, + authKind: "api_key_and_base_url", + baseUrl: apiBase.href, + models: [{ id: imageModel }], + }, + }; + if (method === "providers.getSecret") return { value: imageKey }; + if (method === "session.getScratchPath") return { path: join(directory, "scratch", "live") }; + if (method === "session.get") return { session: {} }; + throw new Error("Unexpected host method"); + }, + }; + const tool = createImageGenerationTool({ + dataDir: directory, + getHost: () => host, + fetchImpl: async (...args) => { + assert(++attempts <= 2, "Live image request budget exhausted"); + return fetch(...args); + }, + }); + const run = (args) => + tool({ + sessionId: "live", + toolCallId: String(attempts), + args, + signal: new AbortController().signal, + }); + const generated = await run(input); + console.log( + JSON.stringify({ + stage: "generation", + chatModel, + imageModel, + toolCalled: true, + results: generated.content.results?.map(({ status, errorCode, mimeType }) => ({ + status, + errorCode, + mimeType, + })), + }), + ); + assert(generated.ok, "Live generation failed; no retry was attempted."); + const edited = await run({ + items: [ + { + prompt: + "Change only the blue ceramic cup to green. Preserve the plain white background and the composition.", + images: [generated.content.results[0].path], + }, + ], + }); + console.log( + JSON.stringify({ + stage: "edit", + imageModel, + results: edited.content.results?.map(({ status, errorCode, mimeType }) => ({ + status, + errorCode, + mimeType, + })), + attempts, + }), + ); + assert(edited.ok, "Live edit failed; no retry was attempted."); +} finally { + await rm(directory, { recursive: true, force: true }); +} From e7ce98840234f59190449684bef060b73b34077c Mon Sep 17 00:00:00 2001 From: Jack <2075649045@qq.com> Date: Mon, 21 Sep 2026 12:44:40 +0800 Subject: [PATCH 2/8] test(images): verify generated previews decode in Chromium Allow data images in the isolated UI fixture and wait for a decoded image rather than accepting an empty img element as preview coverage. --- scripts/e2e-image-generation-ui.mjs | 2 +- scripts/e2e/image-generation-ui.tsx | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/scripts/e2e-image-generation-ui.mjs b/scripts/e2e-image-generation-ui.mjs index 1c5609e456..6ac9a6dedc 100644 --- a/scripts/e2e-image-generation-ui.mjs +++ b/scripts/e2e-image-generation-ui.mjs @@ -35,7 +35,7 @@ try { }); await writeFile( join(temp, "index.html"), - 'Image generation interactions', + 'Image generation interactions', ); await writeFile( join(temp, "main.cjs"), diff --git a/scripts/e2e/image-generation-ui.tsx b/scripts/e2e/image-generation-ui.tsx index 5bf1bb2c85..dae3435b68 100644 --- a/scripts/e2e/image-generation-ui.tsx +++ b/scripts/e2e/image-generation-ui.tsx @@ -169,7 +169,7 @@ globalThis.imageGenerationProbe = async () => { }, } as UiMessage; render(message); - await until(() => !!container.querySelector("img"), "generated preview missing"); + await until(() => (container.querySelector("img")?.naturalWidth ?? 0) > 0, "generated preview did not decode"); assert(container.textContent?.includes("IMAGE_TIMEOUT"), "partial failure missing"); render({ ...message, From f5c19739ec2330788914af30c6176335b0d0e8d6 Mon Sep 17 00:00:00 2001 From: Jack <2075649045@qq.com> Date: Mon, 21 Sep 2026 17:19:32 +0800 Subject: [PATCH 3/8] fix(images): refine settings and keep generated previews visible Align the image model summary with conversation defaults and keep image results available when process details collapse. Resolve absolute image references through the existing contained reader, including Windows paths. Cover advanced selection, unavailable states and generation-to-edit conversations with isolated desktop and renderer regression tests. --- .../main/services/image-generation-service.ts | 4 +- .../resources/skills/image-generation.md | 4 +- apps/desktop/src/components/Markdown.tsx | 5 +- .../settings/ImageGenerationModelRow.tsx | 128 ++-------- .../components/settings/ModelConfigPage.tsx | 3 +- .../chat/transcript/ActivityGroup.tsx | 1 + .../chat/transcript/AssistantTurn.tsx | 4 + .../chat/transcript/GeneratedImages.tsx | 2 +- .../src/features/chat/transcript/ToolRow.tsx | 6 +- apps/desktop/src/lib/markdown-image-paths.ts | 31 +++ apps/desktop/src/styles/model-config.css | 4 + .../test/markdown-image-paths.test.mjs | 30 +++ docs/spec/03-runtime/21-image-generation.md | 18 +- docs/spec/06-delivery/04-e2e-test-plan.md | 20 ++ package.json | 2 +- .../src/image-generation/tool.ts | 2 +- packages/i18n/src/locales/de/index.ts | 4 +- packages/i18n/src/locales/en/index.ts | 4 +- packages/i18n/src/locales/es/index.ts | 4 +- packages/i18n/src/locales/fr/index.ts | 4 +- packages/i18n/src/locales/ko/index.ts | 4 +- packages/i18n/src/locales/tr/index.ts | 4 +- packages/i18n/src/locales/zh-CN/index.ts | 4 +- packages/i18n/src/locales/zh-TW/index.ts | 4 +- scripts/e2e-image-chat.mjs | 222 ++++++++++++++++++ scripts/e2e-image-generation-ui.mjs | 5 +- scripts/e2e/image-chat-model.mjs | 98 ++++++++ scripts/e2e/image-generation-ui.tsx | 81 +++++-- 28 files changed, 539 insertions(+), 163 deletions(-) create mode 100644 apps/desktop/src/lib/markdown-image-paths.ts create mode 100644 apps/desktop/test/markdown-image-paths.test.mjs create mode 100644 scripts/e2e-image-chat.mjs create mode 100644 scripts/e2e/image-chat-model.mjs diff --git a/apps/desktop/electron/main/services/image-generation-service.ts b/apps/desktop/electron/main/services/image-generation-service.ts index 99da58cc40..b85bf7ac5e 100644 --- a/apps/desktop/electron/main/services/image-generation-service.ts +++ b/apps/desktop/electron/main/services/image-generation-service.ts @@ -35,7 +35,7 @@ export function createImageGenerationTool(options: { if (!binding) return failure( "IMAGE_NOT_CONFIGURED", - "Configure an image generation model in Settings > AI > Image generation model before generating images. Do not substitute another model.", + "Configure an image generation model in Settings > Models > Image generation model before generating images. Do not substitute another model.", ); const { provider } = await host.call<{ provider?: ProviderPublic }>("providers.get", { id: binding.providerId, @@ -47,7 +47,7 @@ export function createImageGenerationTool(options: { ) { return failure( "IMAGE_MODEL_UNAVAILABLE", - "The configured image model is unavailable. Update Settings > AI > Image generation model.", + "The configured image model is unavailable. Update Settings > Models > Image generation model.", ); } if (provider.authKind === "oauth") diff --git a/apps/desktop/resources/skills/image-generation.md b/apps/desktop/resources/skills/image-generation.md index 31590959c7..7e4178b48d 100644 --- a/apps/desktop/resources/skills/image-generation.md +++ b/apps/desktop/resources/skills/image-generation.md @@ -6,7 +6,7 @@ description: Generate or edit raster images, illustrations, photos, banners, and # Image generation Use the desktop `GenerateImages` tool. The user selects its provider and model -under Settings → AI → Image generation model; this is independent of the chat +under Settings → Models → Image generation model; this is independent of the chat model. If ToolSearch is available and GenerateImages is not loaded, discover it there first. Do not install an SDK, run an API script, ask for a key in chat, or substitute the conversation model. @@ -42,7 +42,7 @@ Example: two cover variants and one distinct icon: Generation may incur cost. Do not retry failed or timed-out items automatically, including after cancellation: the provider may already have processed them. Report partial success and wait for a user request before retrying. If no image -model is configured, direct the user to Settings → AI; do not select one silently. +model is configured, direct the user to Settings → Models; do not select one silently. ## Deliver the result diff --git a/apps/desktop/src/components/Markdown.tsx b/apps/desktop/src/components/Markdown.tsx index 9a03b77997..253829c8c8 100644 --- a/apps/desktop/src/components/Markdown.tsx +++ b/apps/desktop/src/components/Markdown.tsx @@ -51,6 +51,7 @@ import { } from "../lib/latex-math"; import { useAppStore } from "../stores/app-store"; import { useReferencedImageDataUrl } from "../lib/use-referenced-image-data-url"; +import { absoluteImagePath, remarkLocalImagePaths } from "../lib/markdown-image-paths"; import { useOpenChatFileRef } from "../hooks/use-preview-target"; import { remarkChatFileLinks, @@ -621,7 +622,7 @@ function MarkdownImage({ !isRemote && /^attachments\/[0-9a-f]{64}$/i.test(decoded.replace(/\\/g, "/")) ? decoded.replace(/\\/g, "/") : null; - const localRef = rel ?? attachmentRef; + const localRef = (isRemote ? null : absoluteImagePath(source)) ?? rel ?? attachmentRef; // Always run the hook before any branch so hook order stays stable when a // streaming src flips between remote and local. Remote images pass null. const dataUrl = useReferencedImageDataUrl(isRemote ? null : localRef); @@ -716,7 +717,7 @@ const markdownComponents: Components = { table: Table, }; -const staticRemarkPlugins = [remarkGfm, remarkMath]; +const staticRemarkPlugins = [remarkGfm, remarkMath, remarkLocalImagePaths]; // Extend the default schema only for the media elements rendered above, plus // `remark-math`'s math classes on ``: the default `language-*` allow list diff --git a/apps/desktop/src/components/settings/ImageGenerationModelRow.tsx b/apps/desktop/src/components/settings/ImageGenerationModelRow.tsx index a245f1e685..0f829985f6 100644 --- a/apps/desktop/src/components/settings/ImageGenerationModelRow.tsx +++ b/apps/desktop/src/components/settings/ImageGenerationModelRow.tsx @@ -1,124 +1,32 @@ -import { useState } from "react"; import { useTranslation } from "react-i18next"; -import type { AppSettings, ImageGenerationBinding, ProviderPublic } from "@pi-desktop/shared"; -import { api } from "../../lib/api"; -import { useAppStore } from "../../stores/app-store"; -import { Button, Input } from "../ui"; -import { AnchoredMenu } from "./AnchoredMenu"; +import type { AppSettings, ProviderPublic } from "@pi-desktop/shared"; -export function ImageGenerationModelRow({ - settings, - providers, -}: { +export function ImageGenerationModelRow({ settings, providers }: { settings: AppSettings; providers: ProviderPublic[]; }) { const { t } = useTranslation(); - const [open, setOpen] = useState(false); - const [query, setQuery] = useState(""); - const [busy, setBusy] = useState(false); - const [error, setError] = useState(""); const binding = settings.imageGeneration; const provider = providers.find((entry) => entry.id === binding?.providerId); - const eligible = (entry: ProviderPublic) => - entry.enabled && - entry.authKind !== "oauth" && - !!entry.baseUrl && - (entry.hasSecret || entry.authKind === "none"); - const valid = - provider && - eligible(provider) && + const valid = provider?.enabled && provider.authKind !== "oauth" && + !!provider.baseUrl && (provider.hasSecret || provider.authKind === "none") && provider.models.some((model) => model.id === binding?.modelId); - const options = providers - .filter(eligible) - .flatMap((entry) => entry.models.map((model) => ({ provider: entry, model }))) - .filter(({ provider: entry, model }) => - `${entry.name} ${model.id} ${model.alias ?? ""}` - .toLowerCase() - .includes(query.trim().toLowerCase()), - ); - const save = async (next: ImageGenerationBinding | null) => { - setBusy(true); - setError(""); - try { - await api.setSettings({ ...(await api.getSettings()), imageGeneration: next }); - useAppStore.setState({ settings: await api.getSettings() }); - setOpen(false); - } catch { - setError(t("settings.imageModelSaveFailed")); - } finally { - setBusy(false); - } - }; + return ( -
-
-
-
{t("settings.imageModel")}
-
- {binding - ? `${provider?.name ?? binding.providerId} / ${binding.modelId}` - : t("settings.imageModelUnset")} -
- {binding && !valid ? ( -
{t("settings.imageModelUnavailable")}
- ) : null} - {error ?
{error}
: null} -
- setOpen(false)} - label={t("settings.imageModel")} - align="end" - menuClassName="model-default-menu" - trigger={(ref) => ( - +
+
+
{t("settings.imageModel")}
+
+ {valid && binding ? ( + <> + {provider.name} + · + {binding.modelId} + + ) : ( + {t("settings.imageModelUnavailable")} )} - > - setQuery(event.target.value)} - placeholder={t("settings.defaultModelSearch")} - aria-label={t("settings.defaultModelSearch")} - autoFocus - /> -
    - {options.map(({ provider: entry, model }) => ( -
  • - -
  • - ))} -
- {!options.length ? ( -
{t("settings.noModelMatches")}
- ) : null} - - {binding ? ( - - ) : null} +
); diff --git a/apps/desktop/src/components/settings/ModelConfigPage.tsx b/apps/desktop/src/components/settings/ModelConfigPage.tsx index 8e98b76068..8b0d3f8d1c 100644 --- a/apps/desktop/src/components/settings/ModelConfigPage.tsx +++ b/apps/desktop/src/components/settings/ModelConfigPage.tsx @@ -423,11 +423,10 @@ export function ModelConfigPage() {
+
- -
diff --git a/apps/desktop/src/features/chat/transcript/ActivityGroup.tsx b/apps/desktop/src/features/chat/transcript/ActivityGroup.tsx index 9b95b54429..18c1718c19 100644 --- a/apps/desktop/src/features/chat/transcript/ActivityGroup.tsx +++ b/apps/desktop/src/features/chat/transcript/ActivityGroup.tsx @@ -359,6 +359,7 @@ export const ActivityGroup = memo(function ActivityGroup({ return ( item.kind === "tool" && item.message.toolName === "GenerateImages").map((item) => ( + + ))} {!isActive && metaMessage ? ( { const state = useAppStore.getState(); - state.setSettingsTab("ai"); + state.setSettingsTab("agent"); state.setSettingsAnchor("settings.imageModel"); state.setPage("settings"); }} diff --git a/apps/desktop/src/features/chat/transcript/ToolRow.tsx b/apps/desktop/src/features/chat/transcript/ToolRow.tsx index e483d038d1..37640f490b 100644 --- a/apps/desktop/src/features/chat/transcript/ToolRow.tsx +++ b/apps/desktop/src/features/chat/transcript/ToolRow.tsx @@ -87,6 +87,8 @@ type ToolRowProps = { variant?: "default" | "topology"; /** Open the latest detailed-mode tool unless the user took over. */ autoOpen?: boolean; + /** The containing turn renders image results outside its process disclosure. */ + imagesInTurn?: boolean; /** Claims the containing activity group when this row is manually used. */ onUserInteraction?: () => void; /** Live delegation statuses read from the turn's lifecycle-tool rows. */ @@ -117,6 +119,7 @@ function toolRowPropsEqual( previous.message !== next.message || previous.variant !== next.variant || previous.autoOpen !== next.autoOpen || + previous.imagesInTurn !== next.imagesInTurn || previous.onUserInteraction !== next.onUserInteraction || !subagentRunsEqual(previous.delegate, next.delegate) ) { @@ -142,6 +145,7 @@ export const ToolRow = memo(function ToolRow({ delegate, variant = "default", autoOpen = false, + imagesInTurn = false, onUserInteraction, delegationStatuses, delegationTimings, @@ -525,7 +529,7 @@ export const ToolRow = memo(function ToolRow({
) : null} - + {!imagesInTurn && } {inlineOpen && delegate ? ( { + const visit = (node: ImagePathNode) => { + if (node.type === "image" && typeof node.url === "string") { + const path = absoluteImagePath(node.url); + if (path) node.url = encodeURIComponent(path); + } + node.children?.forEach(visit); + }; + visit(tree); + }; +} diff --git a/apps/desktop/src/styles/model-config.css b/apps/desktop/src/styles/model-config.css index 0f16dd36c8..c98e4b6f75 100644 --- a/apps/desktop/src/styles/model-config.css +++ b/apps/desktop/src/styles/model-config.css @@ -10,6 +10,10 @@ /* ------------------------------------------------- catalog status footer */ +.model-config-page .model-default-panel { + gap: 12px; +} + /* models.dev is the enrichment source, so the page only reports its snapshot. */ .model-catalog-status { display: flex; diff --git a/apps/desktop/test/markdown-image-paths.test.mjs b/apps/desktop/test/markdown-image-paths.test.mjs new file mode 100644 index 0000000000..f867f92ef4 --- /dev/null +++ b/apps/desktop/test/markdown-image-paths.test.mjs @@ -0,0 +1,30 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { absoluteImagePath, remarkLocalImagePaths } from "../src/lib/markdown-image-paths.ts"; + +test("absolute image paths survive URL encoding without enabling URL protocols", () => { + for (const path of [String.raw`C:\scratch\cup.png`, "D:/work/cup image.png", "/tmp/scratch/杯子.png"]) { + assert.equal(absoluteImagePath(path), path); + assert.equal(absoluteImagePath(encodeURIComponent(path)), path); + const tree = { type: "root", children: [{ type: "image", url: path }] }; + remarkLocalImagePaths()(tree); + assert.equal(tree.children[0].url, encodeURIComponent(path)); + } +}); + +test("remote, executable, network and malformed paths are not local image refs", () => { + for (const source of ["https://example.com/image.png", "javascript:alert(1)", "data:image/png;base64,AA", "//server/share.png", String.raw`\\server\share.png`, "file://server/share.png", "C:relative.png", "../image.png", "%ZZ", "C:/image%00.png", "C:/image\n.png"]) { + assert.equal(absoluteImagePath(source), null, source); + } +}); + +test("the image transformer preserves ordinary links, relative and remote images", () => { + const tree = { type: "root", children: [ + { type: "link", url: "C:/file.md" }, + { type: "image", url: "relative.png" }, + { type: "image", url: "https://example.com/image.png" }, + ] }; + const original = structuredClone(tree); + remarkLocalImagePaths()(tree); + assert.deepEqual(tree, original); +}); diff --git a/docs/spec/03-runtime/21-image-generation.md b/docs/spec/03-runtime/21-image-generation.md index 00ae772159..89bbe1be60 100644 --- a/docs/spec/03-runtime/21-image-generation.md +++ b/docs/spec/03-runtime/21-image-generation.md @@ -11,9 +11,11 @@ an existing enabled API-key or no-auth provider and one of its configured models Model Advanced offers **Set as image model**. A draft selection only takes effect when the provider form saves; Cancel leaves settings unchanged. Saving a provider as an image model does not replace the default conversation model. Below the -default model row, **Image generation model** displays the binding and offers -searchable replacement and Clear. OAuth accounts are not eligible. Missing, -disabled or removed bindings remain visible as unavailable; there is no fallback. +default model row in the same defaults panel, **Image generation model** is a +read-only summary with the same provider/model typography and a 12px row gap. +It has no Change or Clear actions; replacement uses the provider's Advanced +settings. Missing, disabled, credential-less or removed bindings display only +**Currently unavailable**. OAuth accounts are not eligible; there is no fallback. ## Agent contract @@ -66,12 +68,20 @@ path/MIME type or a safe error code. New files get unique names in session scrat editing never overwrites its source. The tool result and transcript retain file references, not Base64. Existing bounded image reads and file viewers serve previews. Successful images and per-item failures render even when only part of a batch -completed. Missing configuration returns a structured error and a Settings → AI +completed. Missing configuration returns a structured error and a Settings → Models navigation action. The same references render after session reload/restart. +Image results and setup actions remain visible outside the turn's collapsible +process details; opening tool details does not duplicate the image gallery. +Markdown image references to generated absolute paths, including Windows drive +paths, use the existing bounded host image reader. URL sanitization stays enabled; +the host still rejects files outside its permitted workspace/scratch/attachment roots. Validation: `node scripts/e2e-image-generation.mjs` covers host/stdio/HTTP/storage; `node scripts/e2e-image-generation-ui.mjs` covers real React/Chromium interactions with an API-boundary fixture. Unit/service tests cover limits, cancellation, partial failures, authentication, timeout, unsafe paths, and bounded downloads. +`node scripts/e2e-image-chat.mjs` exercises the full isolated desktop with local +model/image HTTP fixtures: adjacent default settings, composer submission, batch +previews, editing a generated file, collapsed results, and setup navigation. Live verification is opt-in via `scripts/test-image-generation-live.mjs`, limited to one generation plus one edit and never a default test command. diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index 4057bb062a..98f3878928 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -8,6 +8,26 @@ ## 1. Goals +### E2E-IMAGES-desktop-conversation + +- **Preconditions:** Isolated desktop profile and workspace, built image feature, + local chat and OpenAI Images HTTP fixtures; no live provider credentials. +- **Steps:** Open Models settings; verify the conversation and image defaults + share a compact panel. Submit a two-image request through the composer, then + edit the first output through a follow-up message. Collapse tool details. + Clear the binding and follow the visible configuration action back to Models. +- **Expected:** A 12px default-row gap, decoded image previews outside collapsed + process details, multipart source upload for editing, preserved originals, + and no image HTTP request while unconfigured. +- **Settings interactions:** The image summary has no Change/Clear buttons and + matches the default model's provider/model text styles. Select another + provider's image model in Advanced and save; the summary changes while the + chat default stays unchanged. Missing/disabled bindings show only Currently + unavailable. Covered in `scripts/e2e-image-generation-ui.mjs`. +- **Status:** Automated in `node scripts/e2e-image-chat.mjs`; optional screenshots + use `PI_IMAGE_CHAT_EVIDENCE_DIR`. The images are deterministic raster fixtures, + not evidence of real-model quality or provider compatibility. + ### E2E-SCHEDULED-desktop-automation-lifecycle - **Preconditions:** Isolated desktop profile, built task candidate, local SSE diff --git a/package.json b/package.json index 3b2d1ab2ee..3a211d5fab 100644 --- a/package.json +++ b/package.json @@ -31,7 +31,7 @@ "test:e2e:composer-paste": "node scripts/e2e-composer-paste.mjs", "test:e2e:transcript": "node scripts/e2e-transcript-render.mjs", "test:e2e:transcript-disclosure": "node scripts/e2e-transcript-disclosure-anchor.mjs", - "test:e2e:images": "node scripts/e2e-image-generation.mjs && node scripts/e2e-image-generation-ui.mjs", + "test:e2e:images": "node scripts/e2e-image-generation.mjs && node scripts/e2e-image-generation-ui.mjs && node scripts/e2e-image-chat.mjs", "test:e2e:provider-order": "node scripts/e2e-provider-order.mjs", "test:e2e:provider-api-style": "node scripts/e2e-provider-api-style.mjs", "test:e2e:oauth-retry": "node --test apps/desktop/test/anthropic-oauth-retry.test.mjs", diff --git a/packages/agent-runtime/src/image-generation/tool.ts b/packages/agent-runtime/src/image-generation/tool.ts index e7d8779262..07b36d6bd2 100644 --- a/packages/agent-runtime/src/image-generation/tool.ts +++ b/packages/agent-runtime/src/image-generation/tool.ts @@ -1,7 +1,7 @@ import { Type } from "@earendil-works/pi-ai"; export const imageGenerationDescription = - "Generate raster images using the image model configured in Settings > AI. items supports distinct prompts and count variants; at most 10 images total. Each image may incur a charge. Generate only the requested number, report partial failures, and do not retry without the user's request. Results contain local image paths: display successful images with Markdown image links. For edits, provide images as local paths from the session, attachments, or project. Use previous result paths to refine generated images; preserve originals."; + "Generate raster images using the image model configured in Settings > Models. items supports distinct prompts and count variants; at most 10 images total. Each image may incur a charge. Generate only the requested number, report partial failures, and do not retry without the user's request. Results contain local image paths: display successful images with Markdown image links. For edits, provide images as local paths from the session, attachments, or project. Use previous result paths to refine generated images; preserve originals."; export const imageGenerationParameters = { items: Type.Array( Type.Object({ diff --git a/packages/i18n/src/locales/de/index.ts b/packages/i18n/src/locales/de/index.ts index 7d1d8aeaff..dcfb173e67 100644 --- a/packages/i18n/src/locales/de/index.ts +++ b/packages/i18n/src/locales/de/index.ts @@ -552,7 +552,7 @@ export const de = { "settings": { "imageModel": "Bildgenerierungsmodell", "imageModelUnset": "Nicht konfiguriert", - "imageModelUnavailable": "Dieses Bildmodell ist nicht verfügbar. Wählen Sie ein anderes.", + "imageModelUnavailable": "Derzeit nicht verfügbar", "imageModelSaveFailed": "Bildmodell konnte nicht gespeichert werden.", "clearImageModel": "Löschen", "setImageModel": "Als Bildmodell festlegen", @@ -560,7 +560,7 @@ export const de = { "generatedImage": "Generiertes Bild {{index}}", "imagePreviewUnavailable": "Vorschau nicht verfügbar; Datei öffnen", "imageGenerationFailed": "Bild {{index}} wurde nicht fertiggestellt.", - "imageModelSetupHint": "Konfigurieren Sie ein Bildmodell unter Einstellungen → AI.", + "imageModelSetupHint": "Konfigurieren Sie ein Bildmodell unter Einstellungen → Modelle.", "configureImageModel": "Bildmodell konfigurieren", diff --git a/packages/i18n/src/locales/en/index.ts b/packages/i18n/src/locales/en/index.ts index 546d27a1c6..05e14dfa0b 100644 --- a/packages/i18n/src/locales/en/index.ts +++ b/packages/i18n/src/locales/en/index.ts @@ -559,7 +559,7 @@ export const en = { settings: { imageModel: "Image generation model", imageModelUnset: "Not configured", - imageModelUnavailable: "This image model is unavailable. Choose another model.", + imageModelUnavailable: "Currently unavailable", imageModelSaveFailed: "Could not save the image model.", clearImageModel: "Clear", setImageModel: "Set as image model", @@ -567,7 +567,7 @@ export const en = { generatedImage: "Generated image {{index}}", imagePreviewUnavailable: "Preview unavailable; open file", imageGenerationFailed: "Image {{index}} did not complete.", - imageModelSetupHint: "Configure an image model in Settings → AI before generating images.", + imageModelSetupHint: "Configure an image model in Settings → Models before generating images.", configureImageModel: "Configure image model", diff --git a/packages/i18n/src/locales/es/index.ts b/packages/i18n/src/locales/es/index.ts index 04311bd850..ad5afaf0ce 100644 --- a/packages/i18n/src/locales/es/index.ts +++ b/packages/i18n/src/locales/es/index.ts @@ -552,7 +552,7 @@ export const es = { "settings": { "imageModel": "Modelo de imágenes", "imageModelUnset": "Sin configurar", - "imageModelUnavailable": "Este modelo no está disponible. Elige otro.", + "imageModelUnavailable": "No disponible por ahora", "imageModelSaveFailed": "No se pudo guardar el modelo.", "clearImageModel": "Borrar", "setImageModel": "Usar para generar imágenes", @@ -560,7 +560,7 @@ export const es = { "generatedImage": "Imagen generada {{index}}", "imagePreviewUnavailable": "Vista previa no disponible; abrir archivo", "imageGenerationFailed": "La imagen {{index}} no se completó.", - "imageModelSetupHint": "Configura un modelo de imágenes en Ajustes → AI.", + "imageModelSetupHint": "Configura un modelo de imágenes en Ajustes → Modelos.", "configureImageModel": "Configurar modelo de imágenes", diff --git a/packages/i18n/src/locales/fr/index.ts b/packages/i18n/src/locales/fr/index.ts index b3cae75a30..9dfb105754 100644 --- a/packages/i18n/src/locales/fr/index.ts +++ b/packages/i18n/src/locales/fr/index.ts @@ -552,7 +552,7 @@ export const fr = { "settings": { "imageModel": "Modèle de génération d’images", "imageModelUnset": "Non configuré", - "imageModelUnavailable": "Ce modèle est indisponible. Choisissez-en un autre.", + "imageModelUnavailable": "Indisponible pour le moment", "imageModelSaveFailed": "Impossible d’enregistrer le modèle.", "clearImageModel": "Effacer", "setImageModel": "Définir comme modèle d’images", @@ -560,7 +560,7 @@ export const fr = { "generatedImage": "Image générée {{index}}", "imagePreviewUnavailable": "Aperçu indisponible ; ouvrir le fichier", "imageGenerationFailed": "L’image {{index}} n’a pas été terminée.", - "imageModelSetupHint": "Configurez un modèle d’images dans Paramètres → AI.", + "imageModelSetupHint": "Configurez un modèle d’images dans Paramètres → Modèles.", "configureImageModel": "Configurer le modèle d’images", diff --git a/packages/i18n/src/locales/ko/index.ts b/packages/i18n/src/locales/ko/index.ts index 1396300aa5..586504577e 100644 --- a/packages/i18n/src/locales/ko/index.ts +++ b/packages/i18n/src/locales/ko/index.ts @@ -561,7 +561,7 @@ export const ko = { settings: { "imageModel": "이미지 생성 모델", "imageModelUnset": "설정되지 않음", - "imageModelUnavailable": "이 이미지 모델을 사용할 수 없습니다. 다른 모델을 선택하세요.", + "imageModelUnavailable": "현재 사용 불가", "imageModelSaveFailed": "이미지 모델을 저장할 수 없습니다.", "clearImageModel": "지우기", "setImageModel": "이미지 모델로 설정", @@ -569,7 +569,7 @@ export const ko = { "generatedImage": "생성된 이미지 {{index}}", "imagePreviewUnavailable": "미리보기 불가; 파일 열기", "imageGenerationFailed": "이미지 {{index}} 생성이 완료되지 않았습니다.", - "imageModelSetupHint": "설정 → AI에서 이미지 모델을 설정하세요.", + "imageModelSetupHint": "설정 → 모델에서 이미지 모델을 설정하세요.", "configureImageModel": "이미지 모델 설정", diff --git a/packages/i18n/src/locales/tr/index.ts b/packages/i18n/src/locales/tr/index.ts index e7d3a1a60d..d2036d3275 100644 --- a/packages/i18n/src/locales/tr/index.ts +++ b/packages/i18n/src/locales/tr/index.ts @@ -561,7 +561,7 @@ export const tr = { settings: { "imageModel": "Görsel oluşturma modeli", "imageModelUnset": "Yapılandırılmadı", - "imageModelUnavailable": "Bu görsel modeli kullanılamıyor. Başka bir model seçin.", + "imageModelUnavailable": "Şu anda kullanılamıyor", "imageModelSaveFailed": "Görsel modeli kaydedilemedi.", "clearImageModel": "Temizle", "setImageModel": "Görsel modeli olarak ayarla", @@ -569,7 +569,7 @@ export const tr = { "generatedImage": "Oluşturulan görsel {{index}}", "imagePreviewUnavailable": "Önizleme yok; dosyayı aç", "imageGenerationFailed": "Görsel {{index}} tamamlanmadı.", - "imageModelSetupHint": "Ayarlar → AI bölümünde bir görsel modeli yapılandırın.", + "imageModelSetupHint": "Ayarlar → Modeller bölümünde bir görsel modeli yapılandırın.", "configureImageModel": "Görsel modelini yapılandır", diff --git a/packages/i18n/src/locales/zh-CN/index.ts b/packages/i18n/src/locales/zh-CN/index.ts index 172be5da83..4b321c26bb 100644 --- a/packages/i18n/src/locales/zh-CN/index.ts +++ b/packages/i18n/src/locales/zh-CN/index.ts @@ -556,7 +556,7 @@ export const zhCN = { settings: { imageModel: "生图模型", imageModelUnset: "未配置", - imageModelUnavailable: "当前生图模型不可用,请选择其他模型。", + imageModelUnavailable: "暂不可用", imageModelSaveFailed: "无法保存生图模型。", clearImageModel: "清除", setImageModel: "设为生图模型", @@ -564,7 +564,7 @@ export const zhCN = { generatedImage: "生成图片 {{index}}", imagePreviewUnavailable: "预览不可用,打开文件", imageGenerationFailed: "图片 {{index}} 未完成。", - imageModelSetupHint: "请前往 设置 → AI → 生图模型 完成配置。", + imageModelSetupHint: "请前往 设置 → 模型 → 生图模型 完成配置。", configureImageModel: "配置生图模型", diff --git a/packages/i18n/src/locales/zh-TW/index.ts b/packages/i18n/src/locales/zh-TW/index.ts index 0184c917c4..a4de10a8ee 100644 --- a/packages/i18n/src/locales/zh-TW/index.ts +++ b/packages/i18n/src/locales/zh-TW/index.ts @@ -556,7 +556,7 @@ export const zhTW = { settings: { imageModel: "生圖模型", imageModelUnset: "未設定", - imageModelUnavailable: "目前生圖模型無法使用,請選擇其他模型。", + imageModelUnavailable: "暫不可用", imageModelSaveFailed: "無法儲存生圖模型。", clearImageModel: "清除", setImageModel: "設為生圖模型", @@ -564,7 +564,7 @@ export const zhTW = { generatedImage: "生成圖片 {{index}}", imagePreviewUnavailable: "無法預覽,開啟檔案", imageGenerationFailed: "圖片 {{index}} 未完成。", - imageModelSetupHint: "請前往 設定 → AI → 生圖模型 完成設定。", + imageModelSetupHint: "請前往 設定 → 模型 → 生圖模型 完成設定。", configureImageModel: "設定生圖模型", diff --git a/scripts/e2e-image-chat.mjs b/scripts/e2e-image-chat.mjs new file mode 100644 index 0000000000..fdb6134426 --- /dev/null +++ b/scripts/e2e-image-chat.mjs @@ -0,0 +1,222 @@ +#!/usr/bin/env node +// Real desktop, preload, Host SQLite and agent sidecar; only the model is local SSE. +import assert from "node:assert/strict"; +import { spawn } from "node:child_process"; +import { createServer } from "node:http"; +import { mkdirSync, mkdtempSync, readFileSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; +import { once } from "node:events"; +import { Host, resolveHostBinary } from "./e2e/host.mjs"; +import { resolveElectronBinary } from "./e2e/boot.mjs"; +import { waitFor } from "./e2e/wait.mjs"; +import { imageChatModel } from "./e2e/image-chat-model.mjs"; + +const root = mkdtempSync(join(tmpdir(), "pi-image-chat-e2e-")); +const dataDir = join(root, "data"), + project = join(root, "project"); +mkdirSync(project); +const evidence = process.env.PI_IMAGE_CHAT_EVIDENCE_DIR; +if (evidence) mkdirSync(evidence, { recursive: true }); +const model = imageChatModel(); +const server = createServer(model.handler); +await new Promise((done) => server.listen(0, "127.0.0.1", done)); +const host = new Host(resolveHostBinary(), dataDir); +await host.start(); +await host.call("workspace.set", { path: project }); +const { provider } = await host.call("providers.create", { + name: "本地测试服务", + vendorKey: "custom", + type: "openai_compatible", + protocol: "openai_compatible", + baseUrl: `http://127.0.0.1:${server.address().port}/v1`, + authKind: "none", + defaultModelId: "fixture", + apiStyle: "chat_completions", +}); +await host.call("settings.set", { + language: "zh-CN", + defaultProviderId: provider.id, + defaultModelId: "fixture", + defaultMode: "agent", + defaultPermissionMode: "auto", +}); +const { provider: imageProvider } = await host.call("providers.create", { + name: "Image-only fixture", vendorKey: "custom", type: "openai_compatible", + protocol: "openai_compatible", baseUrl: `http://127.0.0.1:${server.address().port}/v1`, + authKind: "none", defaultModelId: "fixture", apiStyle: "chat_completions", +}); +await host.call("settings.set", { imageGeneration: { providerId: imageProvider.id, modelId: "fixture" } }); +await host.stop(); + +const { appDir, electronBinary } = resolveElectronBinary(); +const port = Number(process.env.PI_IMAGE_CHAT_CDP_PORT || 9386); +const env = { + ...process.env, + PI_DESKTOP_DATA_DIR: dataDir, + PI_DESKTOP_START_MAXIMIZED: "0", + ELECTRON_RENDERER_URL: "", +}; +delete env.ELECTRON_RUN_AS_NODE; +const child = spawn( + electronBinary, + [`--remote-debugging-port=${port}`, `--user-data-dir=${join(root, "profile")}`, "."], + { cwd: appDir, env, stdio: ["ignore", "pipe", "pipe"] }, +); +let output = ""; +child.stdout.on("data", (data) => { + output += data; +}); +child.stderr.on("data", (data) => { + output += data; +}); +let ws; +try { + let target; + await waitFor( + async () => { + try { + target = (await (await fetch(`http://127.0.0.1:${port}/json/list`)).json()).find( + (entry) => + entry.type === "page" && + entry.url.includes("index.html") && + !entry.url.includes("plugin-launcher"), + ); + return !!target; + } catch { + return false; + } + }, + 30_000, + "desktop CDP target", + ); + ws = new WebSocket(target.webSocketDebuggerUrl); + console.log("Desktop target", target.url); + await once(ws, "open"); + let sequence = 0; + const pending = new Map(); + ws.onmessage = ({ data }) => { + const message = JSON.parse(data), + entry = pending.get(message.id); + if (!entry) return; + pending.delete(message.id); + if (message.error) entry.reject(new Error(JSON.stringify(message.error))); + else entry.resolve(message.result); + }; + const send = (method, params = {}) => + new Promise((resolveCall, reject) => { + const id = ++sequence; + pending.set(id, { resolve: resolveCall, reject }); + ws.send(JSON.stringify({ id, method, params })); + }); + const evaluate = async (expression) => { + const result = await send("Runtime.evaluate", { + expression, + awaitPromise: true, + returnByValue: true, + }); + if (result.exceptionDetails) + throw new Error( + result.exceptionDetails.exception?.description ?? JSON.stringify(result.exceptionDetails), + ); + return result.result.value; + }; + const invoke = (name, ...args) => + evaluate( + `(async () => { const r = await window.piDesktop.invoke(window.piDesktop.channels.invoke[${JSON.stringify(name)}], ...${JSON.stringify(args)}); if (!r.ok) throw new Error(JSON.stringify(r.error)); return r.data; })()`, + ); + await send("Page.enable"); + const click = async (text, selector = "button", byLabel = false) => { + await waitFor( + () => + evaluate( + `(() => { const e = [...document.querySelectorAll(${JSON.stringify(selector)})].find(e => (${byLabel} ? e.getAttribute('aria-label') : e.textContent.trim()) === ${JSON.stringify(text)}); return !!e && !e.disabled; })()`, + ), + 5000, + `enabled ${text}`, + ); + const point = await evaluate( + `(() => { const button = [...document.querySelectorAll(${JSON.stringify(selector)})].find(e => (${byLabel} ? e.getAttribute('aria-label') : e.textContent.trim()) === ${JSON.stringify(text)}); if (!button || button.disabled) return null; button.scrollIntoView({block:'nearest'}); const r=button.getBoundingClientRect(), x=r.x+r.width/2, y=r.y+r.height/2; return button.contains(document.elementFromPoint(x,y)) ? {x,y} : null; })()`, + ); + assert.ok(point, `button hit target: ${text}`); + await send("Input.dispatchMouseEvent", { + type: "mousePressed", + ...point, + button: "left", + clickCount: 1, + }); + await send("Input.dispatchMouseEvent", { + type: "mouseReleased", + ...point, + button: "left", + clickCount: 1, + }); + }; + const key = async (name, code) => { + await send("Input.dispatchKeyEvent", { type: "keyDown", key: name, code: name, windowsVirtualKeyCode: code, ...(name === "Enter" ? { text: "\r" } : {}) }); + await send("Input.dispatchKeyEvent", { type: "keyUp", key: name, code: name, windowsVirtualKeyCode: code }); + }; + const screenshot = async (name) => { + if (evidence) { + const result = await send("Page.captureScreenshot", { format: "png" }); + writeFileSync(join(evidence, name), Buffer.from(result.data, "base64")); + } + }; + await send("Emulation.setDeviceMetricsOverride", {width:1280,height:900,deviceScaleFactor:1,mobile:false}); + await waitFor(() => evaluate(`!!document.querySelector('[data-nav="settings"]') && !document.querySelector('.startup-splash')`),30000,"desktop ready"); + await evaluate(`document.querySelector('[data-nav="settings"]').click()`); + await waitFor(() => evaluate(`!![...document.querySelectorAll('button')].find(e=>e.textContent.trim()==='AI')`),10000,"AI settings"); + await click("AI"); + await click("模型"); + await waitFor(() => evaluate(`!!document.querySelector('.model-default-row')`),10000,"model defaults"); + const gap = await evaluate(`(() => {const rows=[...document.querySelectorAll('.settings-row')];const image=rows.find(e=>e.innerText.includes('生图模型'));return image.getBoundingClientRect().top-document.querySelector('.model-default-row').getBoundingClientRect().bottom})()`); + await screenshot("settings.png"); + assert.ok(gap>=11 && gap<=13, `defaults gap ${gap}px`); + await click("返回应用"); + const prompt = async (content) => { + await waitFor(() => evaluate(`!!document.querySelector('.composer-input[contenteditable="true"]')`),10000,"composer ready"); + await evaluate(`document.querySelector('.composer-input').focus()`); + await send("Input.insertText",{text:content}); + await key("Enter",13); + }; + await prompt("请生成两张橙色球体、蓝色背景的图片。"); + await waitFor(() => evaluate(`document.querySelectorAll('.generated-image-preview img').length===2 && [...document.querySelectorAll('.generated-image-preview img')].every(e=>e.naturalWidth===480) && document.body.innerText.includes('已生成两张图片')`),60000,"batch images decoded in chat"); + assert.ok(await evaluate(`[...document.querySelectorAll('.generated-image-preview img')].every(e=>e.checkVisibility() && !e.closest('[inert]'))`), "generated images remain visible outside collapsed tool details"); + await screenshot("chat-batch.png"); + const source = model.results[0].value.results[0].path; + const sourceBytes = readFileSync(source); + assert.equal(model.imageRequests.length,2); + model.setScenario("edit"); + await prompt("把第一张图片背景改成绿色,保留橙色球体和原图。"); + await waitFor(() => evaluate(`document.querySelectorAll('.generated-image-preview img').length===3 && [...document.querySelectorAll('.generated-image-preview img')].every(e=>e.naturalWidth===480) && document.body.innerText.includes('已将第一张图的背景改为绿色')`),60000,"edited image decoded in chat"); + await waitFor(() => evaluate(`!document.querySelector('.assistant-turn.streaming')`),10000,"turn completed"); + await evaluate(`document.querySelectorAll('.turn-process > button[aria-expanded="true"]').forEach(e=>e.click())`); + await evaluate(`Promise.all(document.getAnimations().filter(a=>a.effect?.getTiming().iterations!==Infinity).map(a=>a.finished.catch(()=>{})))`); + assert.ok(await evaluate(`[...document.querySelectorAll('.generated-image-preview img')].every(e=>e.checkVisibility() && !e.closest('[inert]'))`), "collapse preserves generated and edited previews"); + await evaluate(`[...document.querySelectorAll('.message-row.user')].at(-1)?.scrollIntoView({block:'start'})`); + await screenshot("chat-edit.png"); + assert.equal(model.imageRequests.length,3); + assert.ok(model.imageRequests[2].edited); + assert.deepEqual(readFileSync(source),sourceBytes,"editing preserves the original file"); + assert.notEqual(model.results[1].value.results[0].path,source,"editing creates a new file"); + await invoke("settingsSet", {...await invoke("settingsGet"),imageGeneration:null}); + model.setScenario("unset"); + await prompt("再生成一张图片。"); + await waitFor(() => evaluate(`!![...document.querySelectorAll('button')].find(e=>e.textContent.trim()==='配置生图模型' && e.checkVisibility())`),60000,"visible setup action"); + await waitFor(() => evaluate(`!document.querySelector('.stop-btn') && document.body.innerText.includes('请先到模型设置选择生图模型,再重试。')`),10000,"configuration error turn completed"); + await evaluate(`new Promise(resolve=>requestAnimationFrame(()=>requestAnimationFrame(resolve)))`); + await screenshot("chat-unconfigured.png"); + await click("配置生图模型"); + await waitFor(() => evaluate(`!!document.querySelector('.model-default-row')`),10000,"setup action opens model settings"); + assert.equal(model.imageRequests.length,3,"unconfigured generation makes no image request"); + assert.deepEqual(model.failures,[]); + console.log(JSON.stringify({ok:true,gap,scenarios:["settings-spacing","composer-batch-generation","composer-edit-generated-image","collapsed-previews","unconfigured-setup-navigation"],imageRequests:model.imageRequests,evidence})); +} catch (error) { + console.error("MODEL_FIXTURE_ERRORS",JSON.stringify(model.failures)); + console.error(output.slice(-4000)); + throw error; +} finally { + ws?.close(); + child.kill(); + server.close(); +} diff --git a/scripts/e2e-image-generation-ui.mjs b/scripts/e2e-image-generation-ui.mjs index 6ac9a6dedc..eb2a3f1495 100644 --- a/scripts/e2e-image-generation-ui.mjs +++ b/scripts/e2e-image-generation-ui.mjs @@ -22,9 +22,8 @@ try { platform: "browser", format: "iife", jsx: "automatic", + loader: { ".woff": "file", ".woff2": "file", ".ttf": "file" }, define: { "process.env.NODE_ENV": '"production"' }, - // Exercise API-format interactions with real components/hooks, not visual layout. - loader: { ".css": "empty" }, alias: { "@pi-desktop/i18n": join(root, "packages/i18n/src/index.ts"), // The fixture lives outside the desktop package; use its React instance. @@ -35,7 +34,7 @@ try { }); await writeFile( join(temp, "index.html"), - 'Image generation interactions', + 'Image generation interactions', ); await writeFile( join(temp, "main.cjs"), diff --git a/scripts/e2e/image-chat-model.mjs b/scripts/e2e/image-chat-model.mjs new file mode 100644 index 0000000000..836acf00a2 --- /dev/null +++ b/scripts/e2e/image-chat-model.mjs @@ -0,0 +1,98 @@ +import assert from "node:assert/strict"; +import { deflateSync } from "node:zlib"; + +// Deterministic raster fixtures, not AI artwork. Only the HTTP provider is mocked. +function png(edited, variant) { + const width = 480, height = 320; + const pixels = Buffer.alloc((width * 4 + 1) * height); + for (let y = 0; y < height; y++) { + for (let x = 0; x < width; x++) { + const offset = y * (width * 4 + 1) + 1 + x * 4; + const circle = (x - 240) ** 2 + (y - 145) ** 2 < 82 ** 2; + const floor = y > 250; + const color = circle ? [245, 169 + variant * 15, 85] : floor ? [40, 56, 66] : edited ? [49, 105 + Math.floor(y / 8), 92] : [42 + Math.floor(y / 12), 70, 119 + variant * 15]; + pixels.set([...color, 255], offset); + } + } + const chunk = (type, data) => { + const payload = Buffer.concat([Buffer.from(type), data]); + let crc = 0xffffffff; + for (const byte of payload) { + crc ^= byte; + for (let i = 0; i < 8; i++) crc = (crc >>> 1) ^ ((crc & 1) ? 0xedb88320 : 0); + } + const result = Buffer.alloc(data.length + 12); + result.writeUInt32BE(data.length); payload.copy(result, 4); + result.writeUInt32BE((crc ^ 0xffffffff) >>> 0, result.length - 4); + return result; + }; + const header = Buffer.alloc(13); + header.writeUInt32BE(width); header.writeUInt32BE(height, 4); header[8] = 8; header[9] = 6; + return Buffer.concat([Buffer.from("89504e470d0a1a0a", "hex"), chunk("IHDR", header), chunk("IDAT", deflateSync(pixels)), chunk("IEND", Buffer.alloc(0))]); +} + +export function imageChatModel() { + let scenario = "batch", pending, done = false, calls = 0; + const results = [], imageRequests = [], failures = []; + const handler = async (req, res) => { + try { + if (req.method === "GET" && req.url.endsWith("/models")) { + res.writeHead(200, { "content-type": "application/json" }); + res.end(JSON.stringify({ data: [{ id: "fixture", object: "model" }] })); + return; + } + const chunks = []; + for await (const chunk of req) chunks.push(chunk); + const raw = Buffer.concat(chunks); + if (req.url.includes("/images/")) { + const edited = req.url.endsWith("/edits"); + if (edited) { + assert.match(req.headers["content-type"], /multipart\/form-data/); + assert.ok(raw.includes(Buffer.from("image/png"))); + assert.ok(raw.includes(Buffer.from("89504e470d0a1a0a", "hex")), "edit uploads source PNG"); + } + imageRequests.push({ edited, bytes: raw.length }); + res.writeHead(200, { "content-type": "application/json" }); + res.end(JSON.stringify({ data: [{ b64_json: png(edited, imageRequests.length % 2).toString("base64") }] })); + return; + } + const request = JSON.parse(raw.toString()); + if (pending) { + const result = request.messages.find((item) => item.role === "tool" && item.tool_call_id === pending.id); + assert.ok(result, "tool result returns to model"); + if (pending.name === "GenerateImages") { + const text = typeof result.content === "string" ? result.content : result.content.map((item) => item.text ?? "").join(""); + const value = JSON.parse(text); + results.push({ scenario, value }); + done = true; + } + pending = undefined; + } + res.writeHead(200, { "content-type": "text/event-stream" }); + const base = { id: `image-chat-${++calls}`, object: "chat.completion.chunk", created: 1, model: request.model }; + const emit = (delta, finish_reason = null) => res.write(`data: ${JSON.stringify({ ...base, choices: [{ index: 0, delta, finish_reason }] })}\n\n`); + if (!done) { + const available = request.tools?.some((tool) => tool.function?.name === "GenerateImages"); + const name = available ? "GenerateImages" : "ToolSearch"; + let args = { query: "GenerateImages" }; + if (available) { + const source = results.find((result) => result.scenario === "batch")?.value.results?.[0]?.path; + if (scenario === "edit") assert.ok(source, "generation returns an editable path"); + args = { items: [{ prompt: scenario === "edit" ? "Keep the orange sphere and change the background to green." : "An orange sphere on a blue background.", count: scenario === "batch" ? 2 : 1, ...(scenario === "edit" ? { images: [source] } : {}) }] }; + } + pending = { id: `image-call-${calls}`, name }; + emit({ role: "assistant", tool_calls: [{ index: 0, id: pending.id, type: "function", function: { name, arguments: JSON.stringify(args) } }] }); + emit({}, "tool_calls"); + } else { + const text = scenario === "batch" ? "已生成两张图片,可在对话中查看。" : scenario === "edit" ? "已将第一张图的背景改为绿色,原图保留。" : "请先到模型设置选择生图模型,再重试。"; + emit({ role: "assistant", content: text }); emit({}, "stop"); + } + res.end("data: [DONE]\n\n"); + } catch (error) { + failures.push(String(error)); + if (!res.headersSent) res.writeHead(500); + res.end(); + } + }; + return { handler, results, imageRequests, failures, setScenario(value) { scenario = value; pending = undefined; done = false; } }; +} diff --git a/scripts/e2e/image-generation-ui.tsx b/scripts/e2e/image-generation-ui.tsx index dae3435b68..a62aa76c9d 100644 --- a/scripts/e2e/image-generation-ui.tsx +++ b/scripts/e2e/image-generation-ui.tsx @@ -6,8 +6,12 @@ import { catalogs } from "@pi-desktop/i18n"; import type { AppSettings, ProviderPublic, UiMessage } from "@pi-desktop/shared"; import { ModelConfigPage } from "../../apps/desktop/src/components/settings/ModelConfigPage"; import { GeneratedImages } from "../../apps/desktop/src/features/chat/transcript/GeneratedImages"; +import { Markdown } from "../../apps/desktop/src/components/Markdown"; import { api } from "../../apps/desktop/src/lib/api"; import { useAppStore } from "../../apps/desktop/src/stores/app-store"; +import "../../apps/desktop/src/styles/tokens.css"; +import "../../apps/desktop/src/styles/settings.css"; +import "../../apps/desktop/src/styles/model-config.css"; declare global { var imageGenerationProbe: () => Promise; @@ -57,12 +61,15 @@ globalThis.imageGenerationProbe = async () => { defaultThinkingLevel: null, })), }; + const alternate = { ...provider, id: "q", name: "Images B" }; + const chatProvider = { ...provider, id: "chat", name: "Chat", models: [{ ...provider.models[0], id: "chat-model" }] }; + const providers = [provider, alternate, chatProvider]; api.getSettings = async () => structuredClone(settings); api.setSettings = async (next) => { settings = structuredClone(next); return { ok: true }; }; - api.listProviders = async () => ({ providers: [provider] }); + api.listProviders = async () => ({ providers }); api.listSessions = async () => ({ sessions: [] }); api.getOnboarding = async () => ({ dismissed: true }); api.listProviderModels = async () => ({ models: [], source: "remote" }); @@ -73,13 +80,13 @@ globalThis.imageGenerationProbe = async () => { providerCount: 1, modelCount: 2, }); - api.updateProvider = async () => ({ provider }); + api.updateProvider = async (input) => ({ provider: providers.find((entry) => entry.id === input.id)! }); api.fsReadImageDataUrl = async () => ({ kind: "image", dataUrl: "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+a9mQAAAAASUVORK5CYII=", }); - useAppStore.setState({ settings, providers: [provider] }); + useAppStore.setState({ settings, providers }); const container = document.createElement("div"); document.body.append(container); const root = createRoot(container); @@ -103,6 +110,12 @@ globalThis.imageGenerationProbe = async () => { for (const locale of ["en", "zh-CN"]) { await i18n.changeLanguage(locale); render(); + const defaultRow = container.querySelector(".model-default-row")!; + const imageRow = [...container.querySelectorAll(".settings-row")].find( + (element) => element.textContent?.includes(i18n.t("settings.imageModel")), + )!; + const gap = imageRow.getBoundingClientRect().top - defaultRow.getBoundingClientRect().bottom; + assert(gap >= 11 && gap <= 13, `model defaults should be adjacent rows, got ${gap}px`); const edit = container.querySelector( 'button[aria-label="' + i18n.t("settings.editProvider") + '"]', @@ -138,19 +151,35 @@ globalThis.imageGenerationProbe = async () => { const row = [...container.querySelectorAll(".settings-row")].find((element) => element.textContent?.includes(i18n.t("settings.imageModel")), )!; - click( - [...row.querySelectorAll("button")].find( - (element) => element.textContent === i18n.t("settings.changeDefaultModel"), - ), - ); - await until(() => !!button("Images / image-two"), "replacement picker missing"); - click(button("Images / image-two")); - await until( - () => settings.imageGeneration?.modelId === "image-two", - "replacement did not persist", - ); - click(button(i18n.t("settings.clearImageModel"))); - await until(() => settings.imageGeneration === null, "clear did not persist"); + assert(row.querySelectorAll("button").length === 0, "image summary still offers change or clear"); + const textStyle = (element: Element | null) => { + assert(element, "missing model text"); + const style = getComputedStyle(element!); + return [style.fontSize, style.fontWeight, style.fontFamily].join("|"); + }; + assert(textStyle(row.querySelector(".model-default-provider")) === textStyle(defaultRow.querySelector(".model-default-provider")), "provider typography differs from default model"); + assert(textStyle(row.querySelector(".model-default-model")) === textStyle(defaultRow.querySelector(".model-default-model")), "model typography differs from default model"); + const alternateRow = [...container.querySelectorAll(".model-provider-row")].find((element) => element.textContent?.includes("Images B")); + click(alternateRow?.querySelector('button[aria-label="' + i18n.t("settings.editProvider") + '"]')); + await until(() => !!button(i18n.t("settings.setImageModel")), "alternate provider edit missing"); + click(button(i18n.t("settings.setImageModel"))); + click(button(i18n.t("settings.saveProvider"))); + await until(() => settings.imageGeneration?.providerId === "q" && !document.querySelector(".provider-setup"), "advanced provider switch did not persist"); + assert(settings.defaultProviderId === "chat" && settings.defaultModelId === "chat-model", "image switch changed chat default"); + assert(row.textContent?.includes("Images B"), "new provider name missing"); + for (const invalid of [ + { ...alternate, enabled: false }, + { ...alternate, hasSecret: false }, + { ...alternate, models: [] }, + ]) { + flushSync(() => useAppStore.setState({ providers: [provider, invalid, chatProvider] })); + assert(row.querySelector('[role="status"]')?.textContent === i18n.t("settings.imageModelUnavailable"), "unavailable state missing"); + assert(!row.querySelector(".model-default-provider"), "unavailable binding still displays provider"); + } + settings = { ...settings, imageGeneration: null }; + flushSync(() => useAppStore.setState({ settings, providers })); + assert(row.querySelector('[role="status"]')?.textContent === i18n.t("settings.imageModelUnavailable"), "unset state missing"); + } const message = { id: "image", @@ -179,18 +208,34 @@ globalThis.imageGenerationProbe = async () => { }); click(button(i18n.t("settings.configureImageModel"))); assert( - useAppStore.getState().settingsTab === "ai" && useAppStore.getState().page === "settings", + useAppStore.getState().settingsTab === "agent" && useAppStore.getState().page === "settings", "setup action did not navigate", ); + const imageReads: string[] = []; + const readImage = api.fsReadImageDataUrl; + api.fsReadImageDataUrl = async (ref, mimeType) => { + imageReads.push(ref); + return readImage(ref, mimeType); + }; + for (const ref of [String.raw`C:\scratch\cup.png`, "/tmp/scratch/cup.png"]) { + flushSync(() => root.render( + , + )); + await until(() => (container.querySelector("img")?.naturalWidth ?? 0) > 0, "absolute generated image Markdown did not decode"); + await until(() => imageReads.includes(ref), `Markdown image did not use the contained host reader: ${JSON.stringify(imageReads)}`); + } return { ok: true, locales: ["en", "zh-CN"], scenarios: [ "advanced-save-cancel", - "replace-clear", + "advanced-provider-switch", + "read-only-summary-typography", + "unavailable-summary", "chat-default-preserved", "image-preview-partial-failure", "setup-navigation", + "absolute-image-markdown", ], apiBoundary: "fixture", }; From 4b730486af50cbe680e656d844619add5f0fc639 Mon Sep 17 00:00:00 2001 From: Jack <2075649045@qq.com> Date: Mon, 21 Sep 2026 17:19:56 +0800 Subject: [PATCH 4/8] fix(models): exclude image bindings from conversation selection Keep the configured image provider/model pair out of conversation defaults and Composer candidates. Reject retained image bindings before inference without rewriting existing conversation history. Verify provider-scoped filtering, binding replacement and the runtime guard through selection tests and complete desktop interactions. --- .../electron/main/runtime/session-launch.ts | 6 +++ apps/desktop/src/components/Composer.tsx | 2 + .../components/settings/ModelConfigPage.tsx | 13 +++-- .../src/components/settings/default-model.ts | 7 ++- .../composer/hooks/useComposerModelMenu.ts | 6 ++- apps/desktop/src/lib/composer-models.ts | 7 ++- .../image-conversation-selection.test.mjs | 51 +++++++++++++++++++ docs/spec/03-runtime/21-image-generation.md | 5 ++ docs/spec/06-delivery/04-e2e-test-plan.md | 3 ++ packages/shared/src/image-generation.ts | 10 ++++ scripts/e2e-image-chat.mjs | 10 +++- scripts/e2e/image-generation-ui.tsx | 7 +++ 12 files changed, 117 insertions(+), 10 deletions(-) create mode 100644 apps/desktop/test/image-conversation-selection.test.mjs diff --git a/apps/desktop/electron/main/runtime/session-launch.ts b/apps/desktop/electron/main/runtime/session-launch.ts index d901c2e828..0e4a326098 100644 --- a/apps/desktop/electron/main/runtime/session-launch.ts +++ b/apps/desktop/electron/main/runtime/session-launch.ts @@ -3,6 +3,7 @@ import { ErrorCodes as SharedErrorCodes, isActiveInProject, isCommandShellCatalog, + isImageGenerationModel, normalizeMode, resolveBindingContextWindow, trustedExtensionAgentKeyFromProviderId, @@ -326,6 +327,11 @@ export function createSessionLaunchRuntime({ errorCode: ErrorCodes.MODEL_NOT_CONFIGURED, }); } + if (isImageGenerationModel(settings.imageGeneration, provider.id, modelId)) { + throw Object.assign(new Error("The image model cannot be used for conversation; select a chat model"), { + errorCode: ErrorCodes.MODEL_NOT_CONFIGURED, + }); + } // The authenticated collection owns a vendor account's available model IDs // and wire endpoint. models.dev owns metadata; one account can span multiple // wire APIs and gateway catalogs. diff --git a/apps/desktop/src/components/Composer.tsx b/apps/desktop/src/components/Composer.tsx index a60883b20b..58631f85c2 100644 --- a/apps/desktop/src/components/Composer.tsx +++ b/apps/desktop/src/components/Composer.tsx @@ -12,6 +12,7 @@ import type { } from "@pi-desktop/shared"; import { initialThinkingLevelForBinding, + isImageGenerationModel, modelIdsMatch, normalizeLargePasteThreshold, stripInlineComposerFileReferenceTokens, @@ -391,6 +392,7 @@ export function Composer({ : !!provider && provider.enabled && !!modelId && + !isImageGenerationModel(settings?.imageGeneration, provider.id, modelId) && (provider.hasSecret || provider.authKind === "none"); const enterToSend = settings?.enterToSend ?? true; const hasDraftContent = Boolean(value.trim() || activeFileReferences.length); diff --git a/apps/desktop/src/components/settings/ModelConfigPage.tsx b/apps/desktop/src/components/settings/ModelConfigPage.tsx index 8b0d3f8d1c..074af7699f 100644 --- a/apps/desktop/src/components/settings/ModelConfigPage.tsx +++ b/apps/desktop/src/components/settings/ModelConfigPage.tsx @@ -10,6 +10,7 @@ import { useEffect, useMemo, useRef, useState } from "react"; import { useTranslation } from "react-i18next"; import { OAUTH_AUTH_KIND, + isImageGenerationModel, modelIdsMatch, type ModelBinding, type ProviderPublic, @@ -32,7 +33,6 @@ import { } from "../icons"; import { AnchoredMenu } from "./AnchoredMenu"; import { - defaultModelIdOf, defaultModelOptions, displayedDefaultModelId, } from "./default-model"; @@ -111,7 +111,7 @@ export function ModelConfigPage() { const providerReady = (provider: ProviderPublic) => provider.enabled && - !!defaultModelIdOf(provider) && + defaultModelOptions([provider], settings?.imageGeneration).length > 0 && (provider.hasSecret || provider.hasOauth || provider.authKind === "none"); const aiProviders = useMemo( @@ -120,7 +120,7 @@ export function ModelConfigPage() { ); const reorder = useProviderReorder(aiProviders, busyId !== null || testingId !== null || setupFor !== null); const readyProviders = providers.filter(providerReady); - const defaultModelOptionsList = defaultModelOptions(readyProviders); + const defaultModelOptionsList = defaultModelOptions(readyProviders, settings?.imageGeneration); const visibleDefaultModelOptions = useMemo(() => { const query = defaultModelQuery.trim().toLowerCase(); if (!query) return defaultModelOptionsList; @@ -136,10 +136,13 @@ export function ModelConfigPage() { providers.find((provider) => provider.id === settings.defaultProviderId) ?? null; const editingProvider = setupFor ? providers.find((provider) => provider.id === setupFor) ?? null : null; - const defaultProviderReady = defaultProvider !== null && providerReady(defaultProvider); + const defaultProviderReady = defaultProvider !== null && providerReady(defaultProvider) && + !isImageGenerationModel(settings.imageGeneration, defaultProvider.id, + displayedDefaultModelId(defaultProvider, settings.defaultModelId)); const setDefaultModel = async (provider: ProviderPublic, modelId: string) => { + if (isImageGenerationModel(useAppStore.getState().settings?.imageGeneration, provider.id, modelId)) return; setBusyId(provider.id); try { await api.setSettings({ @@ -525,7 +528,7 @@ export function ModelConfigPage() { variant="ghost" disabled={rowBusy || !providerReady(provider)} onClick={() => - void setDefaultModel(provider, defaultModelIdOf(provider) ?? "") + void setDefaultModel(provider, defaultModelOptions([provider], settings.imageGeneration)[0]?.modelId ?? "") } > {t("settings.makeDefault")} diff --git a/apps/desktop/src/components/settings/default-model.ts b/apps/desktop/src/components/settings/default-model.ts index a95b018ddd..dc69df8f78 100644 --- a/apps/desktop/src/components/settings/default-model.ts +++ b/apps/desktop/src/components/settings/default-model.ts @@ -7,7 +7,7 @@ * is not configured, so these helpers keep the two notions apart: what a * provider itself offers, and what is safe to display for it. */ -import { modelIdsMatch, type ProviderPublic } from "@pi-desktop/shared"; +import { isImageGenerationModel, modelIdsMatch, type ImageGenerationBinding, type ProviderPublic } from "@pi-desktop/shared"; export type DefaultModelOption = { provider: ProviderPublic; @@ -17,13 +17,16 @@ export type DefaultModelOption = { /** Expand runnable providers into the model choices they actually configure. */ export function defaultModelOptions( providers: readonly ProviderPublic[], + imageGeneration?: ImageGenerationBinding | null, ): DefaultModelOption[] { return providers.flatMap((provider) => { const modelIds = (provider.models ?? []) .map((binding) => binding.id.trim()) .filter(Boolean); const ids = modelIds.length > 0 ? modelIds : [defaultModelIdOf(provider)?.trim() ?? ""]; - return [...new Set(ids)].filter(Boolean).map((modelId) => ({ provider, modelId })); + return [...new Set(ids)].filter((modelId) => !!modelId && + !isImageGenerationModel(imageGeneration, provider.id, modelId), + ).map((modelId) => ({ provider, modelId })); }); } diff --git a/apps/desktop/src/features/chat/composer/hooks/useComposerModelMenu.ts b/apps/desktop/src/features/chat/composer/hooks/useComposerModelMenu.ts index a457d5adfa..61d62fc2ab 100644 --- a/apps/desktop/src/features/chat/composer/hooks/useComposerModelMenu.ts +++ b/apps/desktop/src/features/chat/composer/hooks/useComposerModelMenu.ts @@ -6,6 +6,7 @@ import type { } from "@pi-desktop/shared"; import { initialThinkingLevelForBinding, + isImageGenerationModel, modelIdsMatch, } from "@pi-desktop/shared"; import { useAppStore } from "../../../../stores/app-store"; @@ -44,6 +45,7 @@ export function useComposerModelMenu({ controlsBlocked, }: UseComposerModelMenuOptions) { const providers = useAppStore((s) => s.providers); + const imageGeneration = useAppStore((s) => s.settings?.imageGeneration); const providerModels = useAppStore((s) => s.providerModels); const loadProviderModels = useAppStore((s) => s.loadProviderModels); const configureActiveSession = useAppStore((s) => s.configureActiveSession); @@ -115,6 +117,7 @@ export function useComposerModelMenu({ const models = composerModelsForProvider( candidate, providerModels[candidate.id], + imageGeneration, ); return { provider: candidate, @@ -124,7 +127,7 @@ export function useComposerModelMenu({ }; }) .filter((group) => group.models.length > 0), - [providers, providerModels], + [providers, providerModels, imageGeneration], ); const queryNeedle = query.trim().toLowerCase(); const filteredModelGroups = useMemo( @@ -247,6 +250,7 @@ export function useComposerModelMenu({ const selectModel = async (candidate: ProviderPublic, nextModelId: string) => { thinkingQueueRef.current?.invalidate(); await thinkingQueueRef.current?.idle(); + if (isImageGenerationModel(useAppStore.getState().settings?.imageGeneration, candidate.id, nextModelId)) return; try { const nextModelProvider = thinkingProviderForModel( candidate, diff --git a/apps/desktop/src/lib/composer-models.ts b/apps/desktop/src/lib/composer-models.ts index 9dd3f3866e..2707859f00 100644 --- a/apps/desktop/src/lib/composer-models.ts +++ b/apps/desktop/src/lib/composer-models.ts @@ -1,5 +1,7 @@ import { bindingSupportsImages, + isImageGenerationModel, + type ImageGenerationBinding, modelIdsMatch, modelMatchesFilter, type ModelBinding, @@ -51,8 +53,11 @@ function configuredModelIds(provider: ConfiguredProvider): string[] { export function composerModelsForProvider( provider: ConfiguredProvider, discovered: readonly ModelInfo[] | undefined, + imageGeneration?: ImageGenerationBinding | null, ): ModelInfo[] { - return configuredModelIds(provider).map((modelId) => { + return configuredModelIds(provider).filter((modelId) => + !isImageGenerationModel(imageGeneration, provider.id, modelId), + ).map((modelId) => { const metadata = (discovered ?? []).find((model) => modelIdsMatch(model.modelId, modelId), ); diff --git a/apps/desktop/test/image-conversation-selection.test.mjs b/apps/desktop/test/image-conversation-selection.test.mjs new file mode 100644 index 0000000000..88aa2f5b72 --- /dev/null +++ b/apps/desktop/test/image-conversation-selection.test.mjs @@ -0,0 +1,51 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import { register } from "node:module"; +import { isImageGenerationModel } from "@pi-desktop/shared"; +import { composerModelsForProvider } from "../src/lib/composer-models.ts"; +import { defaultModelOptions } from "../src/components/settings/default-model.ts"; + +const image = { providerId: "images", modelId: "gpt-image-2.5" }; +const providers = [ + { id: "images", models: [{ id: "gpt-image-2.5" }, { id: "chat" }] }, + { id: "other", models: [{ id: "gpt-image-2.5" }] }, +]; + +test("conversation candidates exclude only the configured image binding", () => { + assert.deepEqual(defaultModelOptions(providers, image).map(x => [x.provider.id, x.modelId]), [ + ["images", "chat"], ["other", "gpt-image-2.5"], + ]); + assert.deepEqual(composerModelsForProvider(providers[0], undefined, image).map(x => x.modelId), ["chat"]); + assert.equal(isImageGenerationModel(image, "other", image.modelId), false); + assert.equal(isImageGenerationModel(image, "images", image.modelId), true); +}); + +test("replacement and clearing recompute candidates without removing provider models", () => { + const replacement = { providerId: "other", modelId: "gpt-image-2.5" }; + assert.equal(defaultModelOptions(providers, replacement).length, 2); + assert.equal(composerModelsForProvider(providers[0], undefined, replacement).length, 2); + assert.equal(defaultModelOptions(providers, null).length, 3); + assert.equal(providers[0].models.length, 2); + const legacy = { id: "images", models: [], defaultModelId: image.modelId }; + assert.deepEqual(defaultModelOptions([legacy], image), []); + assert.deepEqual(composerModelsForProvider(legacy, undefined, image), []); +}); + +test("runtime rejects an image binding retained by an existing conversation", async () => { + register(new URL("./helpers/ts-import-hooks.mjs", import.meta.url)); + const { createSessionLaunchRuntime } = await import("../electron/main/runtime/session-launch.ts"); + const shell = { id: "cmd", label: "Command Prompt", dialect: "cmd", available: true, isDefault: true }; + const runtime = createSessionLaunchRuntime({ + runtimeState: { host: { call: async (method) => { + if (method === "commandShells.list") return { configuredId: null, effective: shell, fallback: false, choices: [shell] }; + if (method === "providers.list") return { providers: [{ ...providers[0], authKind: "none" }] }; + if (method === "providers.getSecret") return {}; + throw new Error(`Unexpected host call: ${method}`); + } } }, + modelsDevCatalog: { ensureLoaded: async () => {} }, + }); + await assert.rejects( + runtime.resolveAgentRuntimeLaunch("session", { providerId: "images", modelId: image.modelId }, { imageGeneration: image }), + error => error.errorCode === "MODEL_NOT_CONFIGURED" && /image model/.test(error.message), + ); +}); diff --git a/docs/spec/03-runtime/21-image-generation.md b/docs/spec/03-runtime/21-image-generation.md index 89bbe1be60..831a9fa8b9 100644 --- a/docs/spec/03-runtime/21-image-generation.md +++ b/docs/spec/03-runtime/21-image-generation.md @@ -16,6 +16,11 @@ read-only summary with the same provider/model typography and a 12px row gap. It has no Change or Clear actions; replacement uses the provider's Advanced settings. Missing, disabled, credential-less or removed bindings display only **Currently unavailable**. OAuth accounts are not eligible; there is no fallback. +The selected provider/model pair is excluded from the default conversation picker, +provider quick-default action, and Composer model menu. Other providers with the +same model ID remain independent. Existing conversation bindings and history are +preserved; a conversation still pinned to the image binding must select a chat +model before sending. Runtime launch also rejects that binding before inference. ## Agent contract diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index 98f3878928..964130016e 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -24,6 +24,9 @@ provider's image model in Advanced and save; the summary changes while the chat default stays unchanged. Missing/disabled bindings show only Currently unavailable. Covered in `scripts/e2e-image-generation-ui.mjs`. +- **Conversation selection:** The selected image provider/model is absent from + default and Composer candidates. Other providers retain same-ID models. An + existing session pinned to the image binding is rejected before inference. - **Status:** Automated in `node scripts/e2e-image-chat.mjs`; optional screenshots use `PI_IMAGE_CHAT_EVIDENCE_DIR`. The images are deterministic raster fixtures, not evidence of real-model quality or provider compatibility. diff --git a/packages/shared/src/image-generation.ts b/packages/shared/src/image-generation.ts index 8a88980210..8d184bc1fb 100644 --- a/packages/shared/src/image-generation.ts +++ b/packages/shared/src/image-generation.ts @@ -1,5 +1,15 @@ +import { modelIdsMatch } from "./types/models.js"; + /** A single host-owned binding, independent of the default conversation model. */ export type ImageGenerationBinding = { providerId: string; modelId: string }; +export function isImageGenerationModel( + binding: ImageGenerationBinding | null | undefined, + providerId: string | undefined, + modelId: string | undefined, +): boolean { + return !!binding && binding.providerId === providerId && !!modelId && + modelIdsMatch(binding.modelId, modelId); +} export const MAX_GENERATED_IMAGES = 10; export const IMAGE_GENERATION_TIMEOUT_MS = 180_000; export const IMAGE_BATCH_TIMEOUT_MS = 950_000; diff --git a/scripts/e2e-image-chat.mjs b/scripts/e2e-image-chat.mjs index fdb6134426..d58ce5e91a 100644 --- a/scripts/e2e-image-chat.mjs +++ b/scripts/e2e-image-chat.mjs @@ -173,6 +173,14 @@ try { await screenshot("settings.png"); assert.ok(gap>=11 && gap<=13, `defaults gap ${gap}px`); await click("返回应用"); + await evaluate(`document.querySelector('.composer-model-thinking-chip').click()`); + await waitFor(() => evaluate(`!!document.querySelector('.composer-menu-entry')`),5000,"composer model root"); + await evaluate(`document.querySelector('.composer-menu-entry').click()`); + await waitFor(() => evaluate(`document.querySelectorAll('.composer-model-option').length>0`),5000,"composer model options"); + assert.equal(await evaluate(`document.querySelectorAll('.composer-model-option').length`),1,"image-only provider must not be a chat candidate"); + assert.ok(await evaluate(`![...document.querySelectorAll('.composer-model-group-label')].some(e=>e.textContent.includes('Image-only fixture'))`)); + await key("Escape",27); + await key("Escape",27); const prompt = async (content) => { await waitFor(() => evaluate(`!!document.querySelector('.composer-input[contenteditable="true"]')`),10000,"composer ready"); await evaluate(`document.querySelector('.composer-input').focus()`); @@ -210,7 +218,7 @@ try { await waitFor(() => evaluate(`!!document.querySelector('.model-default-row')`),10000,"setup action opens model settings"); assert.equal(model.imageRequests.length,3,"unconfigured generation makes no image request"); assert.deepEqual(model.failures,[]); - console.log(JSON.stringify({ok:true,gap,scenarios:["settings-spacing","composer-batch-generation","composer-edit-generated-image","collapsed-previews","unconfigured-setup-navigation"],imageRequests:model.imageRequests,evidence})); + console.log(JSON.stringify({ok:true,gap,scenarios:["image-excluded-from-composer","settings-spacing","composer-batch-generation","composer-edit-generated-image","collapsed-previews","unconfigured-setup-navigation"],imageRequests:model.imageRequests,evidence})); } catch (error) { console.error("MODEL_FIXTURE_ERRORS",JSON.stringify(model.failures)); console.error(output.slice(-4000)); diff --git a/scripts/e2e/image-generation-ui.tsx b/scripts/e2e/image-generation-ui.tsx index a62aa76c9d..419aad4e4d 100644 --- a/scripts/e2e/image-generation-ui.tsx +++ b/scripts/e2e/image-generation-ui.tsx @@ -148,6 +148,12 @@ globalThis.imageGenerationProbe = async () => { "saved image model missing", ); assert(settings.defaultModelId === "chat-model", "image model changed default chat model"); + click(container.querySelector(".model-default-trigger")); + await until(() => !!document.querySelector(".model-default-list"), "default picker missing"); + assert(!document.querySelector('[aria-label="Images · image-one"]'), "image binding leaked into chat defaults"); + assert(document.querySelector('[aria-label="Images · image-two"]'), "other configured chat model disappeared"); + assert(document.querySelector('[aria-label="Images B · image-one"]'), "same model on another provider disappeared"); + click(container.querySelector(".model-default-trigger")); const row = [...container.querySelectorAll(".settings-row")].find((element) => element.textContent?.includes(i18n.t("settings.imageModel")), )!; @@ -233,6 +239,7 @@ globalThis.imageGenerationProbe = async () => { "read-only-summary-typography", "unavailable-summary", "chat-default-preserved", + "image-excluded-from-chat-defaults", "image-preview-partial-failure", "setup-navigation", "absolute-image-markdown", From 9df3891e717f9127228a64ac25af437ebb6c407d Mon Sep 17 00:00:00 2001 From: Jack <2075649045@qq.com> Date: Mon, 21 Sep 2026 17:20:13 +0800 Subject: [PATCH 5/8] fix(settings): preserve retry boolean across settings round trips Normalize disabled or absent infinite retry to false so saving unrelated settings does not fail boolean validation. Keep invalid writes rejected and cover both renderer and main-process read-modify-write boundaries. --- .../electron/main/runtime/provider-catalog.ts | 4 +-- apps/desktop/src/lib/api.ts | 4 +-- apps/desktop/test/settings-roundtrip.test.mjs | 36 +++++++++++++++++++ docs/spec/03-runtime/02-agent-runtime.md | 3 ++ scripts/e2e-image-chat.mjs | 6 ++++ 5 files changed, 47 insertions(+), 6 deletions(-) create mode 100644 apps/desktop/test/settings-roundtrip.test.mjs diff --git a/apps/desktop/electron/main/runtime/provider-catalog.ts b/apps/desktop/electron/main/runtime/provider-catalog.ts index 8da390e3d7..e3d0eccd46 100644 --- a/apps/desktop/electron/main/runtime/provider-catalog.ts +++ b/apps/desktop/electron/main/runtime/provider-catalog.ts @@ -181,9 +181,7 @@ export function createProviderCatalogRuntime({ return { ...(value as T), infiniteProviderRetry: (value as T & { infiniteProviderRetry?: unknown }) - .infiniteProviderRetry === true - ? true - : undefined, + .infiniteProviderRetry === true, defaultCommandShell: isCommandShellId(value.defaultCommandShell) ? value.defaultCommandShell : defaultCommandShellForPlatform(process.platform), diff --git a/apps/desktop/src/lib/api.ts b/apps/desktop/src/lib/api.ts index 3ef5e7f7ba..845162960c 100644 --- a/apps/desktop/src/lib/api.ts +++ b/apps/desktop/src/lib/api.ts @@ -344,9 +344,7 @@ export function normalizeSettings(settings: AppSettings): AppSettings { ...settings, defaultMode: normalizeMode((settings as { defaultMode?: unknown }).defaultMode), infiniteProviderRetry: - (settings as { infiniteProviderRetry?: unknown }).infiniteProviderRetry === true - ? true - : undefined, + (settings as { infiniteProviderRetry?: unknown }).infiniteProviderRetry === true, defaultCommandShell: isCommandShellId( (settings as { defaultCommandShell?: unknown }).defaultCommandShell, ) diff --git a/apps/desktop/test/settings-roundtrip.test.mjs b/apps/desktop/test/settings-roundtrip.test.mjs new file mode 100644 index 0000000000..3c391a2fb5 --- /dev/null +++ b/apps/desktop/test/settings-roundtrip.test.mjs @@ -0,0 +1,36 @@ +import assert from "node:assert/strict"; +import { register } from "node:module"; +import test from "node:test"; +register(new URL("./helpers/ts-import-hooks.mjs", import.meta.url)); +const renderer = await import("../src/lib/api.ts"); +const { createProviderCatalogRuntime } = await import("../electron/main/runtime/provider-catalog.ts"); +const main = createProviderCatalogRuntime({ getHost: () => null, modelsDevCatalog: {} }); + +test("settings read-modify-write preserves disabled retry and unrelated preferences", () => { + const previousWindow = globalThis.window; + globalThis.window = { piDesktop: { platform: "win32" } }; + try { + for (const stored of [{}, { infiniteProviderRetry: false }, { infiniteProviderRetry: true }]) { + const original = { defaultMode: "agent", theme: "light", ...stored }; + const received = structuredClone(main.normalizeSettings(original)); + const displayed = renderer.normalizeSettings(received); + const edited = { ...displayed, imageGeneration: { providerId: "images", modelId: "image" } }; + const outgoing = renderer.validateSettingsWrite(edited); + const persisted = main.validateSettingsWrite(structuredClone(outgoing)); + assert.equal(persisted.infiniteProviderRetry, stored.infiniteProviderRetry === true); + assert.equal(persisted.theme, "light"); + assert.deepEqual(persisted.imageGeneration, edited.imageGeneration); + } + } finally { + if (previousWindow === undefined) delete globalThis.window; + else globalThis.window = previousWindow; + } +}); + +test("invalid retry writes stay rejected at both boundaries", () => { + for (const value of [undefined, null, "yes", 0, {}]) { + const settings = { infiniteProviderRetry: value }; + assert.throws(() => renderer.validateSettingsWrite(settings), /infiniteProviderRetry is invalid/); + assert.throws(() => main.validateSettingsWrite(settings), /infiniteProviderRetry is invalid/); + } +}); diff --git a/docs/spec/03-runtime/02-agent-runtime.md b/docs/spec/03-runtime/02-agent-runtime.md index eaa9d10f19..02c1c6b639 100644 --- a/docs/spec/03-runtime/02-agent-runtime.md +++ b/docs/spec/03-runtime/02-agent-runtime.md @@ -236,6 +236,9 @@ the main session and its builtin subagents skip only the ten-retry ceiling for force. Non-retryable errors, context recovery, compaction, tool execution, and one-shot completions are unchanged. The setting can keep billing requests alive indefinitely until the user stops the turn. +Settings reads expose an explicit boolean for this flag: absent or disabled +values normalize to `false`. Read-modify-write operations on unrelated settings +must remain valid without enabling retries; non-boolean writes stay invalid. Each retry is abortable and reports its current backoff through the normalized status event. The `retrying` activity carries the classified error code, the bounded/redacted provider message, and the HTTP status when known. The main diff --git a/scripts/e2e-image-chat.mjs b/scripts/e2e-image-chat.mjs index d58ce5e91a..6f06cade18 100644 --- a/scripts/e2e-image-chat.mjs +++ b/scripts/e2e-image-chat.mjs @@ -172,6 +172,12 @@ try { const gap = await evaluate(`(() => {const rows=[...document.querySelectorAll('.settings-row')];const image=rows.find(e=>e.innerText.includes('生图模型'));return image.getBoundingClientRect().top-document.querySelector('.model-default-row').getBoundingClientRect().bottom})()`); await screenshot("settings.png"); assert.ok(gap>=11 && gap<=13, `defaults gap ${gap}px`); + await evaluate(`document.querySelector('.model-default-trigger').click()`); + await waitFor(() => evaluate(`!!document.querySelector('.model-default-option')`),5000,"default option ready"); + await evaluate(`document.querySelector('.model-default-option').click()`); + await waitFor(() => evaluate(`document.body.innerText.includes('默认 AI 服务已更新')`),5000,"settings save success feedback"); + await waitFor(async () => (await invoke("settingsGet")).infiniteProviderRetry === false,5000,"normalized settings save round trip"); + assert.ok(await evaluate(`!document.body.innerText.includes('infiniteProviderRetry is invalid')`)); await click("返回应用"); await evaluate(`document.querySelector('.composer-model-thinking-chip').click()`); await waitFor(() => evaluate(`!!document.querySelector('.composer-menu-entry')`),5000,"composer model root"); From a0ab265fa4d6fac13f04dbe2d0614094410df4a2 Mon Sep 17 00:00:00 2001 From: Jack <2075649045@qq.com> Date: Mon, 21 Sep 2026 17:20:28 +0800 Subject: [PATCH 6/8] fix(images): preserve RPC failures and image format compatibility Carry stable local error codes through the sidecar RPC boundary while preserving existing data and excluding arbitrary Error properties. Request base64 responses for DALL-E without sending unsupported format parameters to GPT Image models. Cover the production proxy in a real child and multipart payloads through a local HTTP endpoint, including bounded responses and reference limits. --- docs/spec/03-runtime/21-image-generation.md | 9 ++ docs/spec/06-delivery/04-e2e-test-plan.md | 6 ++ .../src/image-generation/index.ts | 7 +- .../openai-images-contract.test.ts | 91 +++++++++++++++++++ .../src/parent-host-proxy.test.ts | 15 +++ .../agent-runtime/src/parent-host-proxy.ts | 6 +- packages/host-runtime/src/agent-sidecar.ts | 18 +--- .../src/image-generation-bridge.test.ts | 40 +++++++- packages/shared/src/index.ts | 1 + packages/shared/src/rpc-error.test.ts | 23 +++++ packages/shared/src/rpc-error.ts | 42 +++++++++ 11 files changed, 237 insertions(+), 21 deletions(-) create mode 100644 packages/agent-runtime/src/image-generation/openai-images-contract.test.ts create mode 100644 packages/shared/src/rpc-error.test.ts create mode 100644 packages/shared/src/rpc-error.ts diff --git a/docs/spec/03-runtime/21-image-generation.md b/docs/spec/03-runtime/21-image-generation.md index 831a9fa8b9..09896f474c 100644 --- a/docs/spec/03-runtime/21-image-generation.md +++ b/docs/spec/03-runtime/21-image-generation.md @@ -51,6 +51,11 @@ explicit path prefix is retained. Generation uses `POST images/generations` with the prompt, model and `n: 1`. No browser mask editor is included. A configured service may implement generation without editing; its error is reported without silently switching to generation or another model. +For exact DALL-E 2/3 model IDs, generation requests explicitly ask for +`response_format: b64_json`; the same form field is used for DALL-E edits where +supported. GPT Image and unknown compatible model IDs omit that parameter. +Editing retains binary multipart uploads (`image` for one reference, `image[]` +for multiple), with the transport generating the Content-Type boundary. Each batch snapshots the binding and provider before dispatch, uses two workers, and returns results in input order. Each output has a 180-second request budget. @@ -72,6 +77,10 @@ Each result records index, status (`succeeded`, `failed`, `cancelled`), successf path/MIME type or a safe error code. New files get unique names in session scratch; editing never overwrites its source. The tool result and transcript retain file references, not Base64. Existing bounded image reads and file viewers serve previews. +Thrown local-tool errors preserve stable `errorCode` values across the real +sidecar RPC boundary, independently of ordinary structured tool-result failures. +Existing RPC code, message and data are retained; arbitrary Error properties +are not serialized. The sidecar receiver exposes the code alongside its data. Successful images and per-item failures render even when only part of a batch completed. Missing configuration returns a structured error and a Settings → Models navigation action. The same references render after session reload/restart. diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index 964130016e..f06ef8899d 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -27,6 +27,12 @@ - **Conversation selection:** The selected image provider/model is absent from default and Composer candidates. Other providers retain same-ID models. An existing session pinned to the image binding is rejected before inference. +- **Transport contracts:** Real stdio reverse RPC retains a thrown local image + error's stable code in the production ParentHostProxy. Local HTTP tests check + single/multiple binary multipart fields and boundaries, DALL-E `b64_json` + requests, GPT Image parameter omission, bounded responses, and rejection of + more than four references before I/O. These are protocol tests, not official + provider account/live compatibility certification. - **Status:** Automated in `node scripts/e2e-image-chat.mjs`; optional screenshots use `PI_IMAGE_CHAT_EVIDENCE_DIR`. The images are deterministic raster fixtures, not evidence of real-model quality or provider compatibility. diff --git a/packages/agent-runtime/src/image-generation/index.ts b/packages/agent-runtime/src/image-generation/index.ts index bf1ffb6947..7d437411e4 100644 --- a/packages/agent-runtime/src/image-generation/index.ts +++ b/packages/agent-runtime/src/image-generation/index.ts @@ -46,12 +46,17 @@ export async function generateOneImage( const headers = new Headers(endpoint.headers); headers.set("Content-Type", "application/json"); if (endpoint.apiKey) headers.set("Authorization", `Bearer ${endpoint.apiKey}`); - let body: BodyInit = JSON.stringify({ model: endpoint.modelId, prompt, n: 1 }); + const responseFormat = /^dall-e-[23]$/i.test(endpoint.modelId) ? "b64_json" : undefined; + let body: BodyInit = JSON.stringify({ + model: endpoint.modelId, prompt, n: 1, + ...(responseFormat ? { response_format: responseFormat } : {}), + }); if (images.length) { const form = new FormData(); form.set("model", endpoint.modelId); form.set("prompt", prompt); form.set("n", "1"); + if (responseFormat) form.set("response_format", responseFormat); images.forEach((image, index) => form.append( images.length === 1 ? "image" : "image[]", diff --git a/packages/agent-runtime/src/image-generation/openai-images-contract.test.ts b/packages/agent-runtime/src/image-generation/openai-images-contract.test.ts new file mode 100644 index 0000000000..477f59f030 --- /dev/null +++ b/packages/agent-runtime/src/image-generation/openai-images-contract.test.ts @@ -0,0 +1,91 @@ +import { createServer } from "node:http"; +import type { AddressInfo } from "node:net"; +import { afterEach, expect, it } from "vitest"; +import { generateImageBatch, generateOneImage, type ImageEditInput } from "./index.js"; + +const png = Buffer.from("iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAQAAAC1HAwCAAAAC0lEQVR42mP8/x8AAwMCAO+a9mQAAAAASUVORK5CYII=", "base64"); +const input: ImageEditInput = { bytes: png, mimeType: "image/png", extension: "png" }; +const servers: ReturnType[] = []; +afterEach(async () => { await Promise.all(servers.splice(0).map(server => new Promise((resolve) => server.close(() => resolve())))); }); + +async function fixture() { + const requests: Array<{ path: string; body: Record | FormData; type: string }> = []; + const failures: unknown[] = []; + const server = createServer(async (req, res) => { + try { + const chunks: Buffer[] = []; + for await (const chunk of req) chunks.push(Buffer.from(chunk)); + const raw = Buffer.concat(chunks); + const type = String(req.headers["content-type"]); + const body = type.startsWith("multipart/form-data") + ? await new Request("http://localhost", { method: "POST", headers: { "content-type": type }, body: raw }).formData() + : JSON.parse(raw.toString()); + requests.push({ path: req.url ?? "", body, type }); + res.writeHead(200, { "content-type": "application/json" }); + res.end(JSON.stringify({ data: [{ b64_json: png.toString("base64") }] })); + } catch (error) { failures.push(error); res.writeHead(400); res.end(); } + }); + servers.push(server); + await new Promise(resolve => server.listen(0, "127.0.0.1", resolve)); + const baseUrl = `http://127.0.0.1:${(server.address() as AddressInfo).port}/v1`; + return { requests, failures, call: (modelId: string, images: ImageEditInput[] = []) => generateOneImage({ baseUrl, modelId }, "A cup", new AbortController().signal, fetch, images) }; +} + +it("requests b64_json for DALL-E without sending response_format to GPT Image", async () => { + const f = await fixture(); + for (const model of ["dall-e-2", "dall-e-3", "gpt-image-1.5", "gpt-image-2.5"]) { + expect((await f.call(model)).bytes).toEqual(png); + const request = f.requests.at(-1)!; + expect(request.path).toBe("/v1/images/generations"); + expect(request.body).toEqual({ model, prompt: "A cup", n: 1, ...(model.startsWith("dall-e-") ? { response_format: "b64_json" } : {}) }); + } + expect(f.failures).toEqual([]); +}); + +it("sends real multipart edits with boundary, binary files and model-specific response format", async () => { + const f = await fixture(); + for (const [model, images] of [["gpt-image-1.5", [input]], ["gpt-image-2.5", [input, input]], ["gpt-image-2.5", [input, input, input, input]], ["dall-e-2", [input]]] as const) { + expect((await f.call(model, [...images])).bytes).toEqual(png); + const request = f.requests.at(-1)!; + expect(request.path).toBe("/v1/images/edits"); + expect(request.type).toMatch(/^multipart\/form-data; boundary=/); + expect(request.body).toBeInstanceOf(FormData); + const form = request.body as FormData; + expect(form.get("model")).toBe(model); + expect(form.get("prompt")).toBe("A cup"); + expect(form.get("n")).toBe("1"); + expect(form.get("response_format")).toBe(model === "dall-e-2" ? "b64_json" : null); + const files = form.getAll(images.length === 1 ? "image" : "image[]"); + expect(files).toHaveLength(images.length); + for (const file of files) { + expect(file).toBeInstanceOf(Blob); + expect((file as Blob).type).toBe("image/png"); + expect(Buffer.from(await (file as Blob).arrayBuffer())).toEqual(png); + } + } + expect(f.failures).toEqual([]); +}); + +it("rejects more than four reference images before reading files or sending HTTP", async () => { + let reads = 0, requests = 0; + await expect(generateImageBatch({ + input: { items: [{ prompt: "cup", images: ["1.png", "2.png", "3.png", "4.png", "5.png"] }] }, + endpoint: { baseUrl: "https://example.invalid", modelId: "gpt-image-2.5" }, + signal: new AbortController().signal, + loadImages: async () => { reads++; return [input]; }, + fetchImpl: async () => { requests++; return new Response(); }, + save: async () => "unused", + })).rejects.toMatchObject({ errorCode: "INVALID_ARGUMENT" }); + expect({ reads, requests }).toEqual({ reads: 0, requests: 0 }); +}); + +it("does not leak provider error bodies and bounds JSON before decoding", async () => { + for (const status of [401, 403, 429, 500]) { + await expect(generateOneImage({ baseUrl: "https://example.invalid", modelId: "gpt-image-2.5" }, "cup", new AbortController().signal, + async () => new Response("private upstream token", { status }), + )).rejects.toMatchObject({ errorCode: [401, 403].includes(status) ? "IMAGE_AUTH_FAILED" : `IMAGE_HTTP_${status}` }); + } + await expect(generateOneImage({ baseUrl: "https://example.invalid", modelId: "gpt-image-2.5" }, "cup", new AbortController().signal, + async () => new Response("", { headers: { "content-length": "999999999" } }), + )).rejects.toMatchObject({ errorCode: "IMAGE_TOO_LARGE" }); +}); diff --git a/packages/agent-runtime/src/parent-host-proxy.test.ts b/packages/agent-runtime/src/parent-host-proxy.test.ts index 45e19f6457..839295dcd3 100644 --- a/packages/agent-runtime/src/parent-host-proxy.test.ts +++ b/packages/agent-runtime/src/parent-host-proxy.test.ts @@ -14,6 +14,21 @@ describe("ParentHostProxy RPC deadlines", () => { vi.restoreAllMocks(); }); + it("exposes the stable error code received from the host", async () => { + const stdout = stubStdout(); + const proxy = new ParentHostProxy(); + try { + const pending = proxy.call("tools.execute", { toolName: "GenerateImages" }); + const request = JSON.parse(String(stdout.mock.calls.at(-1)?.[0])); + proxy.handleParentMessage({ id: request.id, error: { + code: -32000, message: "Denied", data: { errorCode: "PERMISSION_DENIED" }, + } }); + await expect(pending).rejects.toMatchObject({ + code: -32000, errorCode: "PERMISSION_DENIED", data: { errorCode: "PERMISSION_DENIED" }, + }); + } finally { await proxy.dispose(); stdout.mockRestore(); } + }); + it("honors a per-call timeout", async () => { vi.useFakeTimers(); const stdout = stubStdout(); diff --git a/packages/agent-runtime/src/parent-host-proxy.ts b/packages/agent-runtime/src/parent-host-proxy.ts index 167a3575f9..f44991993c 100644 --- a/packages/agent-runtime/src/parent-host-proxy.ts +++ b/packages/agent-runtime/src/parent-host-proxy.ts @@ -1,5 +1,5 @@ import { randomUUID } from "node:crypto"; -import { rpcTimeoutMs } from "@pi-desktop/shared"; +import { rpcErrorFromWire, rpcTimeoutMs } from "@pi-desktop/shared"; export type ParentHostCloseHandler = (error: Error) => void; @@ -62,9 +62,7 @@ export class ParentHostProxy { this.pending.delete(String(msg.id)); if (pending.timer) clearTimeout(pending.timer); if (msg.error) { - const err = new Error(msg.error.message) as Error & { data?: unknown }; - err.data = msg.error.data; - pending.reject(err); + pending.reject(rpcErrorFromWire(msg.error)); } else { pending.resolve(msg.result); } diff --git a/packages/host-runtime/src/agent-sidecar.ts b/packages/host-runtime/src/agent-sidecar.ts index 3b35323ef3..f271680520 100644 --- a/packages/host-runtime/src/agent-sidecar.ts +++ b/packages/host-runtime/src/agent-sidecar.ts @@ -1,6 +1,6 @@ import { spawn, type ChildProcessWithoutNullStreams } from "node:child_process"; import { randomUUID } from "node:crypto"; -import { DEFAULT_RPC_TIMEOUT_MS, IMAGE_BATCH_TIMEOUT_MS, imageGenerationPrompts, readNdjsonLines, rpcTimeoutMs } from "@pi-desktop/shared"; +import { DEFAULT_RPC_TIMEOUT_MS, IMAGE_BATCH_TIMEOUT_MS, imageGenerationPrompts, readNdjsonLines, rpcTimeoutMs, rpcErrorFromWire, rpcErrorToWire } from "@pi-desktop/shared"; import type { ProcessExitHandler, StderrHandler } from "./host-process.js"; // stderr lines kept per sidecar so an unexpected exit can be reported with the @@ -567,16 +567,12 @@ export class AgentSidecar { this.writeToChild( JSON.stringify({ jsonrpc: "2.0", id: msg.id, result }) + "\n", ); - } catch (e: any) { + } catch (e: unknown) { this.writeToChild( JSON.stringify({ jsonrpc: "2.0", id: msg.id, - error: { - code: e?.code ?? -32000, - message: e instanceof Error ? e.message : String(e), - data: e?.data, - }, + error: rpcErrorToWire(e), }) + "\n", ); } @@ -589,13 +585,7 @@ export class AgentSidecar { this.pending.delete(String(msg.id)); if (pending.timer) clearTimeout(pending.timer); if (msg.error) { - const err = new Error(msg.error.message) as Error & { - code?: number; - data?: unknown; - }; - err.code = msg.error.code; - err.data = msg.error.data; - pending.reject(err); + pending.reject(rpcErrorFromWire(msg.error)); } else { pending.resolve(msg.result); } diff --git a/packages/host-runtime/src/image-generation-bridge.test.ts b/packages/host-runtime/src/image-generation-bridge.test.ts index d35ac2b58f..7eeb8e8cce 100644 --- a/packages/host-runtime/src/image-generation-bridge.test.ts +++ b/packages/host-runtime/src/image-generation-bridge.test.ts @@ -10,9 +10,9 @@ const sidecars: AgentSidecar[] = []; afterEach(async () => { await Promise.all(sidecars.splice(0).map((sidecar) => sidecar.dispose())); }); -function harness(allowed = true) { +function harness(allowed = true, childSource = child, esm = false) { const sidecar = new AgentSidecar({ - launch: { command: process.execPath, args: ["-e", child] }, + launch: { command: process.execPath, args: [...(esm ? ["--input-type=module"] : []), "-e", childSource] }, onStderr: () => {}, }); sidecars.push(sidecar); @@ -50,6 +50,42 @@ it("authorizes image calls through the host before executing the local handler", expect(calls).toEqual(["tools.execute", "generated"]); }); +it("preserves stable local error codes through real reverse RPC", async () => { + const { sidecar, execute } = harness(); + sidecar.setLocalTool("GenerateImages", async () => { + throw Object.assign(new Error("Image request failed"), { + errorCode: "IMAGE_TIMEOUT", data: { retryable: false }, secret: "must-not-cross", + }); + }); + await expect(execute()).rejects.toMatchObject({ + code: -32000, errorCode: "IMAGE_TIMEOUT", + data: { errorCode: "IMAGE_TIMEOUT", retryable: false }, + }); +}); + +it("delivers stable error codes to the production ParentHostProxy in a real child", async () => { + const receiverUrl = new URL("../../agent-runtime/src/parent-host-proxy.ts", import.meta.url).href; + const source = ` + import { createInterface } from 'node:readline'; + const { ParentHostProxy } = await import(${JSON.stringify(receiverUrl)}); + const proxy = new ParentHostProxy(); + createInterface({input:process.stdin}).on('line', async line => { + const message = JSON.parse(line); + if (proxy.handleParentMessage(message)) return; + try { + const result = await proxy.call(message.params.method, message.params.params); + console.log(JSON.stringify({id:message.id,result})); + } catch (error) { + console.log(JSON.stringify({id:message.id,result:{code:error.code,errorCode:error.errorCode,data:error.data}})); + } + });`; + const { sidecar, execute } = harness(true, source, true); + sidecar.setLocalTool("GenerateImages", async () => { + throw Object.assign(new Error("limited"), { errorCode: "IMAGE_HTTP_429" }); + }); + expect(await execute()).toEqual({ code: -32000, errorCode: "IMAGE_HTTP_429", data: { errorCode: "IMAGE_HTTP_429" } }); +}); + it("denied and Plan calls never reach the image service", async () => { const { sidecar, execute } = harness(false); const generate = vi.fn().mockResolvedValue({ ok: true, content: "image" }); diff --git a/packages/shared/src/index.ts b/packages/shared/src/index.ts index bf6fc44b9c..6d4a8b34b4 100644 --- a/packages/shared/src/index.ts +++ b/packages/shared/src/index.ts @@ -1,6 +1,7 @@ export * from "./activation.js"; export * from "./protocol.js"; export * from "./errors.js"; +export * from "./rpc-error.js"; export * from "./certificate-errors.js"; export * from "./types.js"; export * from "./transcript-truncation.js"; diff --git a/packages/shared/src/rpc-error.test.ts b/packages/shared/src/rpc-error.test.ts new file mode 100644 index 0000000000..005faea767 --- /dev/null +++ b/packages/shared/src/rpc-error.test.ts @@ -0,0 +1,23 @@ +import { expect, it } from "vitest"; +import { rpcErrorFromWire, rpcErrorToWire } from "./rpc-error.js"; + +it("retains stable codes and existing metadata without copying private properties", () => { + const wire = rpcErrorToWire(Object.assign(new Error("Denied"), { + code: -32602, errorCode: "PERMISSION_DENIED", data: { retryable: false }, + apiKey: "private-test-key", + })); + expect(wire).toEqual({ code: -32602, message: "Denied", data: { retryable: false, errorCode: "PERMISSION_DENIED" } }); + expect(JSON.stringify(wire)).not.toContain("private-test-key"); + expect(rpcErrorFromWire(wire)).toMatchObject({ code: -32602, errorCode: "PERMISSION_DENIED", data: wire.data }); +}); + +it("keeps an existing nested code authoritative and preserves legacy data", () => { + const existing = { errorCode: "HOST_OVERLOADED", retryAfterMs: 100 }; + expect(rpcErrorToWire(Object.assign(new Error("busy"), { errorCode: "GENERIC", data: existing })).data).toEqual(existing); + for (const data of [null, "legacy", [1, 2]]) { + const wire = rpcErrorToWire(Object.assign(new Error("failed"), { data, errorCode: "IMAGE_FAILED" })); + expect(wire.data).toEqual(data); + expect(rpcErrorFromWire(wire).errorCode).toBe("IMAGE_FAILED"); + } + expect(rpcErrorToWire(new Error("ordinary"))).toEqual({ code: -32000, message: "ordinary", data: undefined }); +}); diff --git a/packages/shared/src/rpc-error.ts b/packages/shared/src/rpc-error.ts new file mode 100644 index 0000000000..a14158f8eb --- /dev/null +++ b/packages/shared/src/rpc-error.ts @@ -0,0 +1,42 @@ +export type RpcErrorValue = { + code: number; + message: string; + data?: unknown; + errorCode?: string; +}; + +function record(value: unknown): Record | undefined { + return value !== null && typeof value === "object" && !Array.isArray(value) + ? value as Record : undefined; +} + +function stableCode(error: Record | undefined): string | undefined { + const nested = record(error?.data)?.errorCode; + return typeof nested === "string" ? nested + : typeof error?.errorCode === "string" ? error.errorCode : undefined; +} + +/** Preserve wire data and copy only established fields, not arbitrary Error properties. */ +export function rpcErrorToWire(error: unknown): RpcErrorValue { + const value = record(error); + const errorCode = stableCode(value); + const data = value?.data; + return { + code: typeof value?.code === "number" ? value.code : -32000, + message: error instanceof Error ? error.message : String(error), + data: errorCode && (data === undefined || record(data)) + ? { ...record(data), errorCode } : data, + // Legacy scalar/array data keeps its shape, with an additive error code. + ...(errorCode && data !== undefined && !record(data) ? { errorCode } : {}), + }; +} + +export function rpcErrorFromWire(value: RpcErrorValue): Error & { + code: number; data?: unknown; errorCode?: string; +} { + const errorCode = stableCode(value); + return Object.assign(new Error(value.message), { + code: value.code, data: value.data, + ...(errorCode ? { errorCode } : {}), + }); +} From 9f034cc29f13d95ec80d48a508c0998a08bb102a Mon Sep 17 00:00:00 2001 From: Jack <2075649045@qq.com> Date: Mon, 21 Sep 2026 17:37:08 +0800 Subject: [PATCH 7/8] docs(images): complete localized specification coverage Keep the new image capability discoverable in both documentation trees and satisfy locale, navigation and E2E contract checks. Clarify that configuration recovery opens Models settings. --- docs/spec/06-delivery/04-e2e-test-plan.md | 2 +- docs/spec/NAV.md | 1 + .../zh-CN/spec/03-runtime/02-agent-runtime.md | 1 + .../03-runtime/03-tools-and-permissions.md | 4 ++ .../13-model-catalog-and-selection.md | 4 ++ .../spec/03-runtime/21-image-generation.md | 45 ++++++++++++++++ docs/zh-CN/spec/03-runtime/README.md | 2 + .../spec/06-delivery/04-e2e-test-plan.md | 53 +++++++++++++++++++ docs/zh-CN/spec/NAV.md | 1 + 9 files changed, 112 insertions(+), 1 deletion(-) create mode 100644 docs/zh-CN/spec/03-runtime/21-image-generation.md diff --git a/docs/spec/06-delivery/04-e2e-test-plan.md b/docs/spec/06-delivery/04-e2e-test-plan.md index 89335766b9..b2e86fa85b 100644 --- a/docs/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/spec/06-delivery/04-e2e-test-plan.md @@ -14265,7 +14265,7 @@ the latest destination. These assertions measure work counts, not device FPS. Attempt the tool in a durable Plan session and verify no HTTP request occurs. - **Expected:** One image binding persists without changing the chat default; generated files, edit sources and transcript references survive restart. - Images render in chat; unconfigured errors navigate to AI settings. + Images render in chat; unconfigured errors navigate to Models settings. - **Specs:** 03-runtime/21-image-generation; 03-runtime/13-model-catalog-and-selection. - **Acceptance:** Configured image generation/editing, safe cancellation and persistence. - **Milestone:** Post-MVP. diff --git a/docs/spec/NAV.md b/docs/spec/NAV.md index 6d92dc6089..e085bacfa6 100644 --- a/docs/spec/NAV.md +++ b/docs/spec/NAV.md @@ -42,6 +42,7 @@ - [18-line-anchored-edit-contract.md](03-runtime/18-line-anchored-edit-contract.md) - [19-remote-agent-control-protocol.md](03-runtime/19-remote-agent-control-protocol.md) - [20-speech.md](03-runtime/20-speech.md) +- [21-image-generation.md](03-runtime/21-image-generation.md) ## 4. UX - [README.md](04-ux/README.md) diff --git a/docs/zh-CN/spec/03-runtime/02-agent-runtime.md b/docs/zh-CN/spec/03-runtime/02-agent-runtime.md index 9e03f3076a..026d712c13 100644 --- a/docs/zh-CN/spec/03-runtime/02-agent-runtime.md +++ b/docs/zh-CN/spec/03-runtime/02-agent-runtime.md @@ -194,6 +194,7 @@ not temporary retry activity. See the English source section 5d and ADR 0206. 网络/瞬时故障(含 `PROVIDER_RATE_LIMITED`)的次数上限。退避、`Retry-After`、可见重试状态和 停止路径不变。不可重试错误、上下文恢复、压缩、工具执行和一次性补全仍走原有有界预算。 开启后可能在用户停止回合前持续消耗 API 用量。 +设置读取对此开关返回明确布尔值:缺省或关闭均规范化为 `false`。修改并保存其他设置不得因此校验失败或启用重试;非布尔值写入仍被拒绝。 当 429 预算耗尽时,最终的助手错误和生命周期 `error` 只发出一次。 提供程序故障在可用时于 `AppError.details` 中携带有界诊断: diff --git a/docs/zh-CN/spec/03-runtime/03-tools-and-permissions.md b/docs/zh-CN/spec/03-runtime/03-tools-and-permissions.md index 9e23a01bd4..b141878b32 100644 --- a/docs/zh-CN/spec/03-runtime/03-tools-and-permissions.md +++ b/docs/zh-CN/spec/03-runtime/03-tools-and-permissions.md @@ -566,3 +566,7 @@ sidecar 不附加覆盖,委托使用会话的有效权限模式;因此父会 - 命令允许列表/拒绝列表 - 空运行模式 - 预览后应用补丁 + +## 图片生成与编辑 + +`GenerateImages` 是仅供 Agent 使用的高风险能力,可信桌面执行请求前必须由 host-core 授权。即使处于 Auto,Plan/Goal 仍被拒绝。取消、限制和结果语义见[图片生成规格](/zh-CN/spec/03-runtime/21-image-generation)。 diff --git a/docs/zh-CN/spec/03-runtime/13-model-catalog-and-selection.md b/docs/zh-CN/spec/03-runtime/13-model-catalog-and-selection.md index 28d0cf990d..ad0ee8dba9 100644 --- a/docs/zh-CN/spec/03-runtime/13-model-catalog-and-selection.md +++ b/docs/zh-CN/spec/03-runtime/13-model-catalog-and-selection.md @@ -329,3 +329,7 @@ Electron 使用本地 `models.dev` 记录装饰缓存和新发现的模型行。 - [ ] 来源标记能在提供商保存/读取往返后保留,未标记记录仍可正常使用 - [ ] 紧凑上限文本不会高于已发布值,1M 附近的相邻窗口保持可区分 (`1M` / `1.05M` / `1.1M`),且永远不会渲染出大于等于 1000 的 `K` 尾数 + +## 生图模型绑定 + +默认对话模型下方有独立的生图模型行。模型高级设置可指定唯一绑定;保存服务商表单才生效,取消丢弃选择,替换不会改变对话默认值。工具和批量合约见[图片生成与编辑](/zh-CN/spec/03-runtime/21-image-generation)。 diff --git a/docs/zh-CN/spec/03-runtime/21-image-generation.md b/docs/zh-CN/spec/03-runtime/21-image-generation.md new file mode 100644 index 0000000000..791c021c30 --- /dev/null +++ b/docs/zh-CN/spec/03-runtime/21-image-generation.md @@ -0,0 +1,45 @@ +# 图片生成与编辑 + +> **翻译说明:** 本页对应 [英文源规格](/spec/03-runtime/21-image-generation)。代码、协议字段和标识符保持原文;如有歧义,以英文版本为准。 + +桌面通过可选的 `AppSettings.imageGeneration` 绑定唯一生图模型,包含 `providerId` 和 `modelId`。`null` 清除绑定,缺省表示未配置。host-core 使用现有设置存储校验并持久化,无需数据库版本升级。绑定独立于默认对话模型,引用已启用的 API-key 或免认证服务商及其已配置模型。 + +## 配置 + +模型高级设置提供“设为生图模型”。只有保存服务商表单后选择才生效,取消不改变设置;保存不会替换默认对话模型。默认模型下方同一面板中的生图模型行只展示摘要,服务商和模型字号与默认模型一致,两行间距为 12px。摘要不提供更改或清除操作;替换通过服务商高级设置完成。绑定缺失、服务商停用、缺少凭据或模型移除时仅显示“暂不可用”。OAuth 账户不适用,也不会自动回退。 + +所选服务商与模型组合从默认对话模型选择器、服务商快速设为默认操作和 Composer 模型菜单中排除。其他服务商的同名模型独立保留。已有会话绑定和历史不改写;仍绑定生图模型的会话必须选择对话模型才能发送,运行时也会在推理前拒绝该绑定。 + +## Agent 合约 + +`GenerateImages({items: [{prompt, count?, images?}]})` 仅供 Agent 模式使用。`count` 默认为 1,支持不同提示词和同一提示词的多个变体,总输出为 1–10 张。每项可通过 `images` 提供 1–4 张本地参考图,之前生成的路径可用于迭代编辑。提示词最多 32,000 个字符;不支持或超限输入直接拒绝,不截断。Plan 和 Goal 不得执行此工具。 + +内置 `pi-desktop/imagegen` skill 可在普通会话中发现,并通过现有 Skill 工具加载。它说明提示词、批量、参考图编辑、保留原图、部分失败处理和项目素材交付,不授予权限,也不携带凭据。 + +可信桌面桥先使用相同身份、参数和权限范围调用 host-core `tools.execute`。host-core 应用现有高风险工具策略,仅返回授权结果;获得批准后桥才执行图片服务。宿主授权审计与 sidecar 的实际工具结果分开记录。取消会传递给待批准操作和网络请求;宿主断开或 sidecar 销毁时中止本地工作。 + +## OpenAI Images 适配器 + +仅支持 OpenAI-compatible Images。根地址自动补充 `/v1`,显式路径前缀保留。生成通过 `POST images/generations` 发送 `model`、`prompt` 和 `n: 1`;编辑通过 `POST images/edits` 发送 multipart 图片、提示词、模型和 `n: 1`。不包含浏览器蒙版编辑器。服务商可能只支持生成;编辑错误直接报告,不悄悄改为生成或切换模型。 + +对于精确匹配的 DALL-E 2/3 模型 ID,生成请求显式发送 `response_format: b64_json`,支持的 DALL-E 编辑请求使用同名表单字段。GPT Image 和未知兼容模型省略该参数。编辑使用二进制 multipart 上传,单图字段为 `image`,多图为 `image[]`,Content-Type boundary 由传输层生成。 + +每批请求在分发前快照模型绑定和服务商,使用两个工作线程,并按输入顺序返回结果。每张输出的请求预算为 180 秒。请求不会自动重试;认证失败停止队列工作。取消停止队列并中止活动 HTTP 请求,但不能保证上游停止处理或收费。已完成的输出文件保留。 + +每次响应接受一张 Base64 图片或 HTTPS 图片地址。JSON 和下载响应体有大小限制;图片最多 16 MiB,只接受 PNG、JPEG、WebP 文件签名。下载使用已校验并固定的公网 DNS 地址,拒绝重定向和私网目标,也不携带服务商请求头。 + +编辑输入必须在 realpath 校验后属于会话项目、该会话 scratch 目录或附件存储。每组参考图最多 32 MiB,每批输入缓存预算为 64 MiB。凭据不会进入渲染进程或工具结果。 + +## 结果与恢复 + +每项结果记录序号、状态(`succeeded`、`failed`、`cancelled`)、成功路径/MIME 类型或安全错误码。输出以唯一文件名保存至会话 scratch,编辑不覆盖原图。工具结果和对话只保留文件引用,不保存 Base64;预览复用现有受限图片读取和文件查看器。 + +本地工具抛出的稳定 `errorCode` 跨真实 sidecar RPC 边界保留,独立于普通结构化工具失败结果。现有 RPC code、message 和 data 保留,不序列化任意 Error 属性;sidecar 接收端同时提供错误码与 data。 + +部分成功时仍展示成功图片和逐项失败。未配置返回结构化错误及“设置 → 模型”跳转操作。会话重载或重启后同一引用仍可显示。图片结果和配置操作显示在可折叠过程详情之外,展开工具详情不会重复图片列表。 + +生成文件绝对路径(包括 Windows 盘符路径)的 Markdown 图片引用通过现有受限宿主图片读取接口解析。URL 清理仍启用,宿主继续拒绝允许的工作区、scratch 和附件根目录之外的文件。 + +验证:`node scripts/e2e-image-generation.mjs` 覆盖宿主、stdio、HTTP 和存储;`node scripts/e2e-image-generation-ui.mjs` 使用 API 边界夹具覆盖真实 React/Chromium 交互。单元及服务测试覆盖限制、取消、部分失败、认证、超时、不安全路径和受限下载。 + +`node scripts/e2e-image-chat.mjs` 在隔离桌面中使用本地模型/图片 HTTP 夹具,覆盖相邻默认设置、Composer 提交、批量结果、引用生成文件编辑、收起详情和配置跳转。真实接口验证通过 `scripts/test-image-generation-live.mjs` 显式启用,仅限一次生成和一次编辑,不属于默认测试命令。 diff --git a/docs/zh-CN/spec/03-runtime/README.md b/docs/zh-CN/spec/03-runtime/README.md index ba72d2a20d..42e009c298 100644 --- a/docs/zh-CN/spec/03-runtime/README.md +++ b/docs/zh-CN/spec/03-runtime/README.md @@ -25,3 +25,5 @@ | [18-line-anchored-edit-contract.md](/zh-CN/spec/03-runtime/18-line-anchored-edit-contract) | 按行锚定的 Edit 合约 | | [19-remote-agent-control-protocol.md](/zh-CN/spec/03-runtime/19-remote-agent-control-protocol) | 远程 Agent 控制协议 | | [20-speech.md](/zh-CN/spec/03-runtime/20-speech) | 宿主语音(转写/朗读) | + +- [图片生成与编辑](/zh-CN/spec/03-runtime/21-image-generation) diff --git a/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md b/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md index b736bd847f..9a68c4ab5b 100644 --- a/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md +++ b/docs/zh-CN/spec/06-delivery/04-e2e-test-plan.md @@ -8263,6 +8263,36 @@ the latest destination. These assertions measure work counts, not device FPS. default. - **Status:** Contract-covered; no end-to-end driver waits out a real 70s call. +### E2E-IMAGES-desktop-conversation + +- **Preconditions:** Isolated desktop profile and workspace, built image feature, + local chat and OpenAI Images HTTP fixtures; no live provider credentials. +- **Steps:** Open Models settings; verify the conversation and image defaults + share a compact panel. Submit a two-image request through the composer, then + edit the first output through a follow-up message. Collapse tool details. + Clear the binding and follow the visible configuration action back to Models. +- **Expected:** A 12px default-row gap, decoded image previews outside collapsed + process details, multipart source upload for editing, preserved originals, + and no image HTTP request while unconfigured. +- **Settings interactions:** The image summary has no Change/Clear buttons and + matches the default model's provider/model text styles. Select another + provider's image model in Advanced and save; the summary changes while the + chat default stays unchanged. Missing/disabled bindings show only Currently + unavailable. Covered in `scripts/e2e-image-generation-ui.mjs`. +- **Conversation selection:** The selected image provider/model is absent from + default and Composer candidates. Other providers retain same-ID models. An + existing session pinned to the image binding is rejected before inference. +- **Transport contracts:** Real stdio reverse RPC retains a thrown local image + error's stable code in the production ParentHostProxy. Local HTTP tests check + single/multiple binary multipart fields and boundaries, DALL-E `b64_json` + requests, GPT Image parameter omission, bounded responses, and rejection of + more than four references before I/O. These are protocol tests, not official + provider account/live compatibility certification. +- **Status:** Automated in `node scripts/e2e-image-chat.mjs`; optional screenshots + use `PI_IMAGE_CHAT_EVIDENCE_DIR`. The images are deterministic raster fixtures, + not evidence of real-model quality or provider compatibility. + + ### E2E-SCHEDULED-desktop-automation-lifecycle - **前提:** 独立桌面配置、构建后的任务候选版本、本地 SSE 模拟模型;不使用真实 @@ -8398,6 +8428,29 @@ the latest destination. These assertions measure work counts, not device FPS. - **状态:** 运行时层已自动化;无 UI 驱动读取插件页的错误文本。 +### E2E-IMAGE-generation-and-editing + +- **Preconditions:** Built task candidate containing latest origin/main; isolated + host data directory, local image HTTP fixture, no production credentials. +- **Steps:** Choose a model in Advanced, cancel and verify no change; save and + replace it through another provider's Advanced settings. Clear the binding via + the test settings API to verify recovery. Generate same-prompt variants and distinct images, + edit a generated image, inspect partial failures, then restart the host/session. + Attempt the tool in a durable Plan session and verify no HTTP request occurs. +- **Expected:** One image binding persists without changing the chat default; + generated files, edit sources and transcript references survive restart. + Images render in chat; unconfigured errors navigate to Models settings. +- **Specs:** 03-runtime/21-image-generation; 03-runtime/13-model-catalog-and-selection. +- **Acceptance:** Configured image generation/editing, safe cancellation and persistence. +- **Milestone:** Post-MVP. +- **Status:** Automated by `scripts/e2e-image-generation.mjs` and + `scripts/e2e-image-generation-ui.mjs`; backend test uses real Rust/stdio/HTTP/files, + UI test uses real Chromium and production components with API-boundary fixtures. + +| Scenario | Acceptance | Specification | Automation | +| --- | --- | --- | --- | +| E2E-IMAGE-generation-and-editing | Image capability and recovery | 03-runtime/21-image-generation | Host and UI suites above | + #### E2E-CHAT-parenthesized-url:用户消息中的完整网址 - **步骤**:在用户消息中发送 `https://en.wikipedia.org/wiki/React_(software)` 并点击链接,再验证正文用圆括号包裹该网址、网址后跟句号及紧接另一链接或文件引用的情况。 diff --git a/docs/zh-CN/spec/NAV.md b/docs/zh-CN/spec/NAV.md index 80e36552d3..d51b3af11a 100644 --- a/docs/zh-CN/spec/NAV.md +++ b/docs/zh-CN/spec/NAV.md @@ -45,6 +45,7 @@ - [18-line-anchored-edit-contract.md](/zh-CN/spec/03-runtime/18-line-anchored-edit-contract) - [19-remote-agent-control-protocol.md](/zh-CN/spec/03-runtime/19-remote-agent-control-protocol) - [20-speech.md](/zh-CN/spec/03-runtime/20-speech) +- [21-image-generation.md](/zh-CN/spec/03-runtime/21-image-generation) ## 4. 用户体验 - [README.md](/zh-CN/spec/04-ux/README) From 29a89837ad676e164efe5c8ce0baadf8738e1f47 Mon Sep 17 00:00:00 2001 From: Jack <2075649045@qq.com> Date: Mon, 21 Sep 2026 17:45:27 +0800 Subject: [PATCH 8/8] test(images): supply model readiness dependencies in native session tests The Composer readiness expression now checks the configured image binding. Pass the real shared predicate and settings into the existing expression harness so it evaluates the same dependencies as the component. Cover provider-scoped image exclusion while preserving native capability and read-only behavior. --- apps/desktop/test/native-pi-sessions.test.mjs | 13 +++++++++++-- 1 file changed, 11 insertions(+), 2 deletions(-) diff --git a/apps/desktop/test/native-pi-sessions.test.mjs b/apps/desktop/test/native-pi-sessions.test.mjs index cd685953aa..ec355f1cdb 100644 --- a/apps/desktop/test/native-pi-sessions.test.mjs +++ b/apps/desktop/test/native-pi-sessions.test.mjs @@ -46,7 +46,7 @@ test("native compact and session-addressed queue endpoints reject before host or // Real slices/IPC with synthetic state; no Electron process or native home. const { register } = await import("node:module"); register(new URL("./helpers/ts-import-hooks.mjs", import.meta.url)); -const { IPC } = await import("@pi-desktop/shared"); +const { IPC, isImageGenerationModel } = await import("@pi-desktop/shared"); const { registerAgentIpc } = await import("../electron/main/ipc/agent-ipc.ts"); const { searchSessionsAcrossSources } = await import("../electron/main/services/session-search.ts"); const { createEventsSlice } = await import("../src/stores/slices/events-slice.ts"); @@ -157,13 +157,22 @@ test("native prompt only dispatches sidecar and cannot create a host queue entry test("native model readiness never depends on a Desktop provider but read-only fails closed", () => { const expression = composer.match(/const modelReady = ([\s\S]*?);/)[1]; - const ready = new Function("nativeSession", "activeSessionSummary", "provider", "modelId", `return ${expression}`); + const evaluateReady = new Function("isImageGenerationModel", "settings", "nativeSession", "activeSessionSummary", "provider", "modelId", `return ${expression}`); + const ready = (nativeSession, activeSessionSummary, provider, modelId, settings) => + evaluateReady(isImageGenerationModel, settings, nativeSession, activeSessionSummary, provider, modelId); assert.equal(ready(true, { capabilities: { canPrompt: true } }, undefined, undefined), true); assert.equal(ready(true, { capabilities: { canPrompt: false } }, { enabled: true, hasSecret: true }, "model"), false); assert.equal(ready(true, {}, undefined, undefined), false); assert.equal(ready(false, {}, undefined, undefined), false); assert.equal(ready(false, {}, { enabled: true, hasSecret: false }, "model"), false); assert.equal(ready(false, {}, { enabled: true, hasSecret: true }, "model"), true); + const settings = { imageGeneration: { providerId: "images", modelId: "model" } }; + const imageProvider = { id: "images", enabled: true, hasSecret: true }; + assert.equal(ready(false, {}, imageProvider, "model", settings), false); + assert.equal(ready(false, {}, { ...imageProvider, id: "chat" }, "model", settings), true); + assert.equal(ready(false, {}, imageProvider, "other-model", settings), true); + assert.equal(ready(true, { capabilities: { canPrompt: true } }, imageProvider, "model", settings), true); + assert.equal(ready(true, { capabilities: { canPrompt: false } }, imageProvider, "model", settings), false); });