diff --git a/README.md b/README.md index 2c713832..df14e929 100644 --- a/README.md +++ b/README.md @@ -140,7 +140,7 @@ That phrase in the feature list is **auto-capture**: after a conversation, a bac ### User profile -The **User Profile** is a separate, cross-project summary of how you like to work (preferences, habits). It is updated on an interval (`userProfileAnalysisInterval`, default every 10 analyzed prompts), shown in the web UI’s profile view, and readable via `memory({ mode: "profile" })`. You do not populate it by hand for normal use — profile learning fills it when a provider is ready. +The **User Profile** is a separate, cross-project summary of how you like to work (preferences, habits). It is updated on an interval (`userProfileAnalysisInterval`, default every 10 analyzed prompts), shown in the web UI’s profile view, and readable via `memory({ mode: "profile" })`. You do not populate it by hand for normal use — profile learning fills it when a provider is ready. Output language follows `autoCaptureLanguage` (default `"auto"`, mirroring the language of your prompts), the same setting used for auto-captured memories. ### Web UI diff --git a/opencode-mem.example.jsonc b/opencode-mem.example.jsonc index b7505c1a..519a043e 100644 --- a/opencode-mem.example.jsonc +++ b/opencode-mem.example.jsonc @@ -249,7 +249,7 @@ // Example for Qwen3 models: { "enable_thinking": false } // "memoryExtraParams": {}, - // Language for auto-capture summaries (default: "auto" for auto-detection) + // Language for auto-capture summaries and user profile learning (default: "auto" for auto-detection) // Options: "auto", "en", "id", "zh", "ja", "es", "fr", "de", "ru", "pt", "ar", "ko" // "autoCaptureLanguage": "auto", diff --git a/src/config.ts b/src/config.ts index 56c3d8df..428dafaf 100644 --- a/src/config.ts +++ b/src/config.ts @@ -507,7 +507,7 @@ const CONFIG_TEMPLATE = `{ // Example for Qwen3 models: { "enable_thinking": false } // "memoryExtraParams": {}, - // Language for auto-capture summaries (default: "auto" for auto-detection) + // Language for auto-capture summaries and user profile learning (default: "auto" for auto-detection) // Options: "auto", "en", "id", "zh", "ja", "es", "fr", "de", "ru", "pt", "ar", "ko" // "autoCaptureLanguage": "auto", diff --git a/src/services/user-memory-learning.ts b/src/services/user-memory-learning.ts index d09ac081..d5d91f57 100644 --- a/src/services/user-memory-learning.ts +++ b/src/services/user-memory-learning.ts @@ -168,8 +168,9 @@ Rules: } const context = buildUserAnalysisContext(prompts, existingProfile, validationPrompt); + const langName = await resolveProfileLanguageName(prompts); - const analysisResult = await analyzeUserProfile(context, existingProfile); + const analysisResult = await analyzeUserProfile(context, existingProfile, langName); log("user-profile-learning: analyze done", { hasResult: !!analysisResult }); @@ -640,10 +641,42 @@ async function applyValidations( return `validated: ${confirmed} confirmed, ${contradicted} contradicted, ${inaccurate} inaccurate, ${oversimplified} oversimplified`; } +/** + * Resolves the language the profile-analysis LLM must write in, honoring the + * same `autoCaptureLanguage` config used by auto-capture (src/services/auto-capture.ts) + * so auto-captured memories and the user profile follow the same language setting. + * + * In "auto" mode, detection runs on the raw prompt text only — never on the + * analysis context, whose English instruction scaffold would bias detection + * toward English for short non-English prompts. + */ +export async function resolveProfileLanguageName(prompts: UserPrompt[]): Promise { + const { detectLanguage, getLanguageName } = await import("./language-detector.js"); + const targetLang = + CONFIG.autoCaptureLanguage === "auto" || !CONFIG.autoCaptureLanguage + ? detectLanguage(prompts.map((p) => p.content).join("\n\n")) + : CONFIG.autoCaptureLanguage; + return getLanguageName(targetLang); +} + +function buildProfileSystemPrompt(existingProfile: UserProfile | null, langName: string): string { + return `You are a user behavior analyst for a coding assistant. + +Your task is to analyze user prompts and ${existingProfile ? "update" : "create"} a comprehensive user profile. + +CRITICAL: You MUST write all descriptions, categories, and text in ${langName}. + +CRITICAL: All JSON string values MUST escape double quotes with backslash. Do NOT use unescaped quotation marks inside string values. + +Use the update_user_profile tool to save the ${existingProfile ? "updated" : "new"} profile.`; +} + async function analyzeUserProfile( context: string, - existingProfile: UserProfile | null + existingProfile: UserProfile | null, + langName: string ): Promise { + const systemPrompt = buildProfileSystemPrompt(existingProfile, langName); log("user-profile-learning: analyze called", { hasProfile: !!existingProfile }); let opencodeProviderError: unknown; if (CONFIG.opencodeProvider && CONFIG.opencodeModel) { @@ -659,16 +692,6 @@ async function analyzeUserProfile( const v2Client = await getOpenCodeClient(); - const systemPrompt = `You are a user behavior analyst for a coding assistant. - -Your task is to analyze user prompts and ${existingProfile ? "update" : "create"} a comprehensive user profile. - -CRITICAL: Detect the language used by the user in their prompts. You MUST output all descriptions, categories, and text in the SAME language as the user's prompts. - -CRITICAL: All JSON string values MUST escape double quotes with backslash. Do NOT use unescaped quotation marks inside string values. - -Use the update_user_profile tool to save the ${existingProfile ? "updated" : "new"} profile.`; - const { z } = await import("zod"); const schema = createUserProfileAnalysisSchema(z); @@ -733,16 +756,6 @@ Use the update_user_profile tool to save the ${existingProfile ? "updated" : "ne const provider = AIProviderFactory.createProvider(CONFIG.memoryProvider, providerConfig); - const systemPrompt = `You are a user behavior analyst for a coding assistant. - -Your task is to analyze user prompts and ${existingProfile ? "update" : "create"} a comprehensive user profile. - -CRITICAL: Detect the language used by the user in their prompts. You MUST output all descriptions, categories, and text in the SAME language as the user's prompts. - -CRITICAL: All JSON string values MUST escape double quotes with backslash. Do NOT use unescaped quotation marks inside string values. - -Use the update_user_profile tool to save the ${existingProfile ? "updated" : "new"} profile.`; - const toolSchema = createUserProfileToolSchema(Boolean(existingProfile)); const result = await provider.executeToolCall( diff --git a/tests/user-profile-learning-language.test.ts b/tests/user-profile-learning-language.test.ts new file mode 100644 index 00000000..50976602 --- /dev/null +++ b/tests/user-profile-learning-language.test.ts @@ -0,0 +1,202 @@ +import { afterEach, describe, expect, it } from "bun:test"; +import { mkdtempSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import { join } from "node:path"; + +// Regression coverage for the profile-learning output language. The structured-output +// call is mocked and only records the system prompt it would have been sent, so we +// can assert which language the model is instructed to write in. + +const tempDirs: string[] = []; + +const learningUrl = new URL("../src/services/user-memory-learning.js", import.meta.url).href; +const configUrl = new URL("../src/config.js", import.meta.url).href; +const tagsUrl = new URL("../src/services/tags.js", import.meta.url).href; +const promptManagerUrl = new URL( + "../src/services/user-prompt/user-prompt-manager.js", + import.meta.url +).href; +const profileManagerUrl = new URL( + "../src/services/user-profile/user-profile-manager.js", + import.meta.url +).href; +const opencodeProviderLoaderUrl = new URL( + "../src/services/ai/opencode-provider-loader.js", + import.meta.url +).href; +const profileLlmClientUrl = new URL("../src/services/ai/profile-llm-client.js", import.meta.url) + .href; +const loggerUrl = new URL("../src/services/logger.js", import.meta.url).href; + +const SHORT_SPANISH_PROMPTS = [ + "Corrige el error en la función de login", + "Añade pruebas para el servicio de usuarios", + "Explica por qué falla la compilación", + "Refactoriza el módulo de pagos", + "Revisa los cambios de esta rama", +]; + +const SHORT_ENGLISH_PROMPTS = [ + "Fix the bug in the login function", + "Add tests for the user service", + "Explain why the build is failing", + "Refactor the payments module", + "Review the changes on this branch", +]; + +function runLanguageScenario(autoCaptureLanguage: string | undefined, promptTexts: string[]) { + const dir = mkdtempSync(join(tmpdir(), "opencode-mem-profile-lang-")); + tempDirs.push(dir); + const scriptPath = join(dir, "scenario.mjs"); + const script = ` +import { mock } from "bun:test"; + +const promptTexts = ${JSON.stringify(promptTexts)}; +const prompts = promptTexts.map((content, i) => ({ + id: \`prompt-\${i}\`, + sessionId: "session-1", + messageId: \`msg-\${i}\`, + projectPath: "/workspace", + content, + createdAt: i + 1, + captured: false, + user_learning_captured: false, + capture_attempts: 0, +})); + +mock.module(${JSON.stringify(configUrl)}, () => ({ + CONFIG: { + autoCaptureProviderStatus: { ready: true, mode: "opencode", issues: [] }, + userProfileAnalysisInterval: prompts.length, + opencodeProvider: "test-provider", + opencodeModel: "test-model", + showUserProfileToasts: false, + autoCaptureLanguage: ${JSON.stringify(autoCaptureLanguage)}, + }, +})); + +mock.module(${JSON.stringify(tagsUrl)}, () => ({ + getTags: () => ({ + user: { + tag: "opencode_user_test", + displayName: "Test User", + userName: "tester", + userEmail: "test@example.com", + }, + }), +})); + +mock.module(${JSON.stringify(promptManagerUrl)}, () => ({ + userPromptManager: { + countUnanalyzedForUserLearning: async () => prompts.length, + getPromptsForUserLearning: async () => prompts, + markMultipleAsUserLearningCaptured: async () => {}, + }, +})); + +mock.module(${JSON.stringify(profileManagerUrl)}, () => ({ + userProfileManager: { + getActiveProfile: async () => null, + createProfile: async () => ({}), + mergeProfileData: async () => ({}), + updateProfile: async () => true, + decayInMemory: (d) => ({ data: d }), + syncConfidence: () => {}, + }, +})); + +mock.module(${JSON.stringify(loggerUrl)}, () => ({ log: () => {} })); + +let capturedSystemPrompt = null; +let capturedUserPrompt = null; + +mock.module(${JSON.stringify(opencodeProviderLoaderUrl)}, () => ({ + loadOpencodeProvider: async () => ({ + generateStructuredOutput: async ({ systemPrompt, userPrompt }) => { + capturedSystemPrompt = systemPrompt; + capturedUserPrompt = userPrompt; + return { preferences: [], patterns: [], workflows: [] }; + }, + }), +})); + +mock.module(${JSON.stringify(profileLlmClientUrl)}, () => ({ + getOpenCodeClient: async () => ({}), +})); + +try { + const { performUserProfileLearning } = await import(${JSON.stringify(learningUrl)}); + await performUserProfileLearning({}, "/workspace"); + console.log(JSON.stringify({ error: null, systemPrompt: capturedSystemPrompt, userPrompt: capturedUserPrompt })); +} catch (e) { + console.log(JSON.stringify({ error: e?.message ?? String(e), systemPrompt: capturedSystemPrompt, userPrompt: capturedUserPrompt })); +} +process.exit(0); +`; + + writeFileSync(scriptPath, script, "utf-8"); + const result = Bun.spawnSync({ + cmd: [process.execPath, scriptPath], + stdout: "pipe", + stderr: "pipe", + }); + const stdout = Buffer.from(result.stdout).toString("utf8").trim(); + const stderr = Buffer.from(result.stderr).toString("utf8").trim(); + const jsonLine = stdout + .split("\n") + .reverse() + .find((line) => line.trim().startsWith("{")); + + return { + exitCode: result.exitCode, + stderr, + parsed: jsonLine + ? (JSON.parse(jsonLine) as { + error: string | null; + systemPrompt: string | null; + userPrompt: string | null; + }) + : null, + }; +} + +afterEach(() => { + while (tempDirs.length > 0) { + const dir = tempDirs.pop(); + if (dir) rmSync(dir, { recursive: true, force: true }); + } +}); + +describe("user profile learning output language", () => { + it("auto: detects short non-English prompts despite the English analysis scaffold", () => { + const result = runLanguageScenario("auto", SHORT_SPANISH_PROMPTS); + + expect(result.exitCode).toBe(0); + expect(result.stderr).toBe(""); + expect(result.parsed?.error).toBeNull(); + // Sanity check: the context sent to the LLM really is wrapped in the English scaffold. + expect(result.parsed?.userPrompt).toContain("# User Profile Analysis"); + expect(result.parsed?.systemPrompt).toContain("text in Spanish."); + }); + + it("unset language behaves like auto", () => { + const result = runLanguageScenario(undefined, SHORT_SPANISH_PROMPTS); + + expect(result.parsed?.error).toBeNull(); + expect(result.parsed?.systemPrompt).toContain("text in Spanish."); + }); + + it("auto: English prompts resolve to English", () => { + const result = runLanguageScenario("auto", SHORT_ENGLISH_PROMPTS); + + expect(result.parsed?.error).toBeNull(); + expect(result.parsed?.systemPrompt).toContain("text in English."); + }); + + it("configured language overrides the language of the prompts", () => { + const result = runLanguageScenario("en", SHORT_SPANISH_PROMPTS); + + expect(result.parsed?.error).toBeNull(); + expect(result.parsed?.systemPrompt).toContain("text in English."); + }); +});