diff --git a/src/skillify/local-source.ts b/src/skillify/local-source.ts index f14b2fa7..dfcab608 100644 --- a/src/skillify/local-source.ts +++ b/src/skillify/local-source.ts @@ -208,6 +208,24 @@ export function pickSessions( return picked; } +/** + * Classify a native `type: "user"` record. Returns the prompt text for a real + * user turn ("" for an image-only prompt), or null when the record is not a + * user turn (tool results, meta injections, empty or unrecognized content). + */ +function userPromptText(obj: any): string | null { + const c = obj?.message?.content; + if (typeof c === "string") return c.trim().length > 0 ? c : null; + if (!Array.isArray(c) || obj?.isMeta === true) return null; + if (c.some((b: any) => b?.type === "tool_result")) return null; + const text = c + .filter((b: any) => b?.type === "text" && typeof b.text === "string") + .map((b: any) => b.text) + .join("\n\n"); + if (text.trim().length > 0) return text; + return c.some((b: any) => b?.type === "image") ? "" : null; +} + /** * Convert a native Claude Code JSONL file into the SessionRow shape that * extractPairs() expects. @@ -218,8 +236,14 @@ export function pickSessions( * { type: "system"|"attachment"|"last-prompt"|... } ← dropped * * Semantics mirror what the production capture hook stores in Deeplake: - * - User: only string-content user messages (the typed prompt). Tool-result - * arrays sent back to the model are dropped. + * - Sidechain records (`isSidechain: true`, subagent traffic) are dropped + * entirely; they are not turns of the main conversation. + * - User: the typed prompt, either string content or the joined text blocks + * of a content array (e.g. text pasted alongside an image). Arrays that + * carry any tool_result block are tool results sent back to the model and + * are dropped, as are isMeta arrays and arrays with no text/image blocks. + * An image-only prompt has no text to emit, but it still ends the previous + * turn; its answer is discarded rather than merged into the prior pair. * - Assistant: only the LAST text-bearing assistant entry per turn — the * same `last_assistant_message` the Stop hook captures. Without this we * would emit every intermediate "Now I'll run X" mini-narration that @@ -235,6 +259,9 @@ export function nativeJsonlToRows(filePath: string, sessionId: string, agent: st // flushed on the next user_message or at EOF. let pendingAsstText: string | undefined; let pendingAsstTs: string | undefined; + // True after a prompt with no emittable text (image-only): its answer has + // no prompt to pair with, so assistant text is dropped until the next one. + let discardAnswer = false; const flushAssistant = (): void => { if (pendingAsstText && pendingAsstText.trim().length > 0) { @@ -254,20 +281,25 @@ export function nativeJsonlToRows(filePath: string, sessionId: string, agent: st if (!line) continue; let obj: any; try { obj = JSON.parse(line); } catch { continue; } + if (obj?.isSidechain === true) continue; const t = obj?.type; const ts: string | undefined = obj?.timestamp ?? obj?.created_at; if (t === "user") { - const c = obj?.message?.content; - if (typeof c === "string" && c.trim().length > 0) { - flushAssistant(); + const prompt = userPromptText(obj); + if (prompt === null) continue; + flushAssistant(); + if (prompt.trim().length > 0) { + discardAnswer = false; rows.push({ type: "user_message", - content: c, + content: prompt, creation_date: ts, session_id: sessionId, agent, }); + } else { + discardAnswer = true; } } else if (t === "assistant") { const c = obj?.message?.content; @@ -276,7 +308,7 @@ export function nativeJsonlToRows(filePath: string, sessionId: string, agent: st .filter((b: any) => b?.type === "text" && typeof b.text === "string") .map((b: any) => b.text) .join("\n\n"); - if (text.trim().length > 0) { + if (text.trim().length > 0 && !discardAnswer) { pendingAsstText = text; pendingAsstTs = ts; } diff --git a/tests/claude-code/local-source.test.ts b/tests/claude-code/local-source.test.ts index 26e632a8..efef7599 100644 --- a/tests/claude-code/local-source.test.ts +++ b/tests/claude-code/local-source.test.ts @@ -11,6 +11,7 @@ import { mkdtempSync, mkdirSync, writeFileSync, rmSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import { pickSessions, nativeJsonlToRows, listLocalSessions, type SessionFile, type AgentInstall } from "../../src/skillify/local-source.js"; +import { extractPairs } from "../../src/skillify/extractors/index.js"; function makeSession(id: string, mtime: number, inCwd: boolean): SessionFile { return { @@ -201,6 +202,146 @@ describe("nativeJsonlToRows", () => { }); }); +describe("nativeJsonlToRows → extractPairs: conversational boundaries", () => { + const tmpDir = mkdtempSync(join(tmpdir(), "mine-local-pairs-")); + afterAll(() => rmSync(tmpDir, { recursive: true, force: true })); + + const img = { type: "image", source: { type: "base64", media_type: "image/png", data: "iVBORw0KGgo=" } }; + const text = (t: string) => ({ type: "text", text: t }); + const asst = (t: string, extra: object = {}) => ({ type: "assistant", message: { content: [text(t)] }, ...extra }); + const toolResult = { type: "user", message: { content: [{ type: "tool_result", tool_use_id: "x", content: "output" }] } }; + + function pairsFor(name: string, lines: object[]) { + const path = writeJsonl(tmpDir, name, lines); + return extractPairs(nativeJsonlToRows(path, "sid", "claude_code")).map(p => ({ prompt: p.prompt, answer: p.answer })); + } + + it("text+image user array is a prompt and does not overwrite the previous answer", () => { + expect(pairsFor("text-image.jsonl", [ + { type: "user", message: { content: "A" } }, + asst("answerA"), + { type: "user", message: { content: [text("B"), img] } }, + asst("answerB"), + ])).toEqual([ + { prompt: "A", answer: "answerA" }, + { prompt: "B", answer: "answerB" }, + ]); + }); + + it("joins multiple text blocks of a user array", () => { + expect(pairsFor("multi-text.jsonl", [ + { type: "user", message: { content: [text("part one"), img, text("part two")] } }, + asst("answer"), + ])).toEqual([{ prompt: "part one\n\npart two", answer: "answer" }]); + }); + + it("image-only user turn delimits the previous answer and its own answer is not paired with the prior prompt", () => { + expect(pairsFor("image-only.jsonl", [ + { type: "user", message: { content: "A" } }, + asst("answerA"), + { type: "user", message: { content: [img] } }, + asst("answer to image"), + { type: "user", message: { content: "C" } }, + asst("answerC"), + ])).toEqual([ + { prompt: "A", answer: "answerA" }, + { prompt: "C", answer: "answerC" }, + ]); + }); + + it("image-only user turn at the end drops its trailing answer", () => { + expect(pairsFor("image-only-eof.jsonl", [ + { type: "user", message: { content: "A" } }, + asst("answerA"), + { type: "user", message: { content: [img, text(" ")] } }, + asst("answer to image"), + ])).toEqual([{ prompt: "A", answer: "answerA" }]); + }); + + it("sidechain records interleaved with the main thread are ignored", () => { + expect(pairsFor("sidechain-interleaved.jsonl", [ + { type: "user", message: { content: "A" }, isSidechain: false }, + { type: "assistant", message: { content: [{ type: "tool_use", id: "t", name: "Task", input: {} }] } }, + { type: "user", message: { content: "subagent task prompt" }, isSidechain: true }, + asst("subagent answer", { isSidechain: true }), + { type: "user", message: { content: [text("subagent array prompt"), img] }, isSidechain: true }, + toolResult, + asst("answerA"), + ])).toEqual([{ prompt: "A", answer: "answerA" }]); + }); + + it("sidechain assistant text after the main answer does not replace it", () => { + expect(pairsFor("sidechain-after.jsonl", [ + { type: "user", message: { content: "A" } }, + asst("answerA"), + asst("late subagent text", { isSidechain: true }), + ])).toEqual([{ prompt: "A", answer: "answerA" }]); + }); + + it("all-sidechain file yields no rows", () => { + const path = writeJsonl(tmpDir, "all-sidechain.jsonl", [ + { type: "user", message: { content: "sub prompt" }, isSidechain: true }, + asst("sub answer", { isSidechain: true }), + { type: "user", message: { content: [text("sub B"), img] }, isSidechain: true }, + asst("sub answer B", { isSidechain: true }), + ]); + expect(nativeJsonlToRows(path, "sid", "claude_code")).toEqual([]); + }); + + it("tool_result user arrays (even with text or image blocks) are not prompts and do not split the turn", () => { + expect(pairsFor("tool-results.jsonl", [ + { type: "user", message: { content: "A" } }, + asst("Let me check"), + toolResult, + { type: "user", message: { content: [{ type: "tool_result", tool_use_id: "y", content: [text("screenshot"), img] }] } }, + { type: "user", message: { content: [{ type: "tool_result", tool_use_id: "z", content: "o" }, text("extra")] } }, + asst("final answerA"), + ])).toEqual([{ prompt: "A", answer: "final answerA" }]); + }); + + it("final assistant accumulation: only the last text-bearing entry of the turn is kept, across tool results", () => { + expect(pairsFor("accumulation.jsonl", [ + { type: "user", message: { content: [text("B"), img] } }, + asst("step one"), + { type: "assistant", message: { content: [{ type: "tool_use", id: "t", name: "Bash", input: {} }] } }, + toolResult, + asst("step two"), + { type: "assistant", message: { content: [{ type: "tool_use", id: "u", name: "Read", input: {} }] } }, + toolResult, + asst("final B"), + { type: "assistant", message: { content: [{ type: "tool_use", id: "v", name: "Read", input: {} }] } }, + ])).toEqual([{ prompt: "B", answer: "final B" }]); + }); + + it("null, meta, and unrecognized user content neither create prompts nor split the turn", () => { + const path = join(tmpDir, "unrecognized.jsonl"); + writeFileSync(path, [ + JSON.stringify({ type: "user", message: { content: "A" } }), + JSON.stringify(asst("early")), + "null", + "42", + JSON.stringify({ type: "user" }), + JSON.stringify({ type: "user", message: null }), + JSON.stringify({ type: "user", message: { content: null } }), + JSON.stringify({ type: "user", message: { content: 7 } }), + JSON.stringify({ type: "user", message: { content: [] } }), + JSON.stringify({ type: "user", message: { content: [null, { type: "document" }, { type: "text", text: 5 }] } }), + JSON.stringify({ type: "user", message: { content: [text(" ")] } }), + JSON.stringify({ type: "user", message: { content: " " } }), + JSON.stringify({ type: "user", isMeta: true, message: { content: [text("injected skill body")] } }), + JSON.stringify({ type: "assistant", message: { content: null } }), + JSON.stringify({ type: "assistant", message: { content: [null, { type: "text", text: 1 }] } }), + JSON.stringify(asst("answerA")), + ].join("\n")); + const rows = nativeJsonlToRows(path, "sid", "claude_code"); + expect(rows.map(r => [r.type, r.content])).toEqual([ + ["user_message", "A"], + ["assistant_message", "answerA"], + ]); + expect(extractPairs(rows).map(p => [p.prompt, p.answer])).toEqual([["A", "answerA"]]); + }); +}); + describe("listLocalSessions", () => { const root = mkdtempSync(join(tmpdir(), "list-local-test-")); afterAll(() => rmSync(root, { recursive: true, force: true }));