Files
StepForge/tests/unit/text-intel.test.js
T
TylerandClaude Sonnet 4.6 8d645c0aca fix smart title generation from browser window context
The title engine was falling back to "Screen capture" for all browser
captures because it treated the entire browser window title as noise.

Changes:
- stripBrowserNameSuffix: removes "- Google Chrome" / "| Firefox" etc.
  from the end of window titles, leaving just the page title.
  "oracle - Google Search - Google Chrome" → "oracle - Google Search"
  "Oracle | Cloud Applications - Google Chrome" → page title only

- extractSearchQuery: detects "[query] - Google Search" / Bing / etc.
  patterns after stripping the browser suffix, and formats the result
  as "Search for oracle".

- buildCaptureTitle: uses stripped page title + search detection before
  falling through to the "Screen capture" fallback.

- pickBestOcrPhrase: considers the full OCR line (≤80 chars) as a
  candidate with a +35 completeness bonus before splitting on | or ·.
  This preserves "Oracle | Cloud Applications and Cloud Platform" as a
  single phrase instead of breaking it into fragments.

- candidateWords: filters out standalone punctuation tokens (|, ·, •)
  so they don't inflate word-count penalties for compound brand names.

- verbForElementRole: hyperlinks and links now produce "Select" instead
  of "Click"; search box / search field produces "Search for".

Five new unit tests cover: browser title stripping, search query
extraction, full pipe-separated link text, link and search box verbs.

Co-Authored-By: Claude Sonnet 4.6 <[email protected]>
2026-06-24 09:44:36 -05:00

313 lines
9.1 KiB
JavaScript

'use strict';
const test = require('node:test');
const assert = require('node:assert/strict');
const { makeTmpDir, rmrf } = require('./helpers');
const { createStep } = require('../../core/schema');
const {
buildCaptureTitle,
buildAiPrompt,
normalizeAiPatch,
applyAiPatchToStep,
} = require('../../core/text-intel');
const { TextIntelService } = require('../../app/text-intel');
function makeSettings(values = {}) {
const data = {
ai: {
enabled: true,
ollama: {
host: 'http://127.0.0.1:11434',
model: 'llama3.2:1b',
},
},
...values,
};
return {
get(key) {
return key.split('.').reduce((acc, part) => (acc == null ? undefined : acc[part]), data);
},
};
}
test('capture titles prefer semantic metadata before OCR fallback', () => {
const title = buildCaptureTitle({
mode: 'fullscreen',
metadata: { windowTitle: 'Reset user password in admin portal' },
ocrText: 'Save',
});
assert.equal(title, 'Click Save');
});
test('capture titles prefer element metadata before window chrome and OCR', () => {
const title = buildCaptureTitle({
mode: 'window',
metadata: {
elementLabel: 'Open advanced settings',
windowTitle: 'Preferences',
},
ocrText: 'Cancel',
});
assert.equal(title, 'Open advanced settings');
});
test('capture titles ignore browser chrome noise in favor of OCR', () => {
const title = buildCaptureTitle({
mode: 'window',
metadata: {
windowTitle: 'Google Chrome ** PR reviews ** /chrome/tyler/autodoc',
appName: 'Google Chrome',
},
ocrText: 'New tab',
});
assert.equal(title, 'Click New tab');
});
test('tab-like roles use select when OCR identifies a tab label', () => {
const title = buildCaptureTitle({
mode: 'window',
metadata: {
elementLabel: 'New tab',
elementRole: 'tab item',
windowTitle: 'Google Chrome - PR reviews',
},
ocrText: 'New tab',
});
assert.equal(title, 'Select New tab');
});
test('capture titles fall back to OCR when metadata is absent', () => {
const title = buildCaptureTitle({
mode: 'window',
metadata: {},
ocrText: 'Save changes',
});
assert.equal(title, 'Click Save changes');
});
test('browser window title strips browser name and falls back to page title', () => {
// OCR fails; browser window title should give something useful, not "Screen capture".
const title = buildCaptureTitle({
mode: 'fullscreen',
metadata: {
windowTitle: 'Oracle | Cloud Applications and Cloud Platform - Google Chrome',
appName: 'chrome',
},
ocrText: '',
});
// Stripped title "Oracle | Cloud Applications and Cloud Platform" → best fragment
assert.ok(title !== 'Screen capture', `Expected smart title, got: ${title}`);
assert.ok(title.toLowerCase().includes('oracle') || title.toLowerCase().includes('cloud'), `Expected oracle/cloud in title, got: ${title}`);
});
test('search query is extracted from browser window title pattern', () => {
const title = buildCaptureTitle({
mode: 'fullscreen',
metadata: {
windowTitle: 'oracle - Google Search - Google Chrome',
appName: 'chrome',
},
ocrText: '',
});
assert.equal(title, 'Search for Oracle');
});
test('full link text with pipe separator is preserved in OCR phrases', () => {
const title = buildCaptureTitle({
mode: 'fullscreen',
metadata: { elementRole: 'hyperlink' },
ocrText: 'Oracle | Cloud Applications and Cloud Platform',
});
assert.equal(title, 'Select Oracle | Cloud Applications and Cloud Platform');
});
test('link element role uses Select verb', () => {
const title = buildCaptureTitle({
mode: 'fullscreen',
metadata: { elementLabel: 'Sign in', elementRole: 'hyperlink' },
ocrText: '',
});
assert.equal(title, 'Select Sign in');
});
test('search box element role uses Search for verb', () => {
const title = buildCaptureTitle({
mode: 'fullscreen',
metadata: { elementLabel: 'oracle', elementRole: 'search box' },
ocrText: '',
});
assert.equal(title, 'Search for Oracle');
});
test('ai prompts include the deterministic OCR-backed title candidate', () => {
const { prompt } = buildAiPrompt({
captureContext: {
windowTitle: 'Google Chrome ** PR reviews ** /chrome/tyler/autodoc',
appName: 'Google Chrome',
ocrText: 'New tab',
titleCandidate: 'Click New tab',
mode: 'content',
},
});
assert.match(prompt, /Suggested title: Click New tab/);
});
test('ocr crop rectangles clamp to the image bounds', (t) => {
const root = makeTmpDir('text-intel-crop');
t.after(() => rmrf(root));
const service = new TextIntelService({
store: { settingsDir: root },
settings: makeSettings(),
getWindow: () => null,
dataDir: root,
fetchImpl: global.fetch,
});
const frame = {
size: { width: 1000, height: 500 },
display: { bounds: { x: 0, y: 0, width: 1000, height: 500 } },
};
const topLeft = service.cropRectForPoint(frame, { x: 5, y: 5 });
assert.deepEqual(topLeft, { x: 0, y: 0, width: 420, height: 220 });
const bottomRight = service.cropRectForPoint(frame, { x: 995, y: 495 });
assert.deepEqual(bottomRight, { x: 580, y: 280, width: 420, height: 220 });
});
test('ocr failures fall back to empty text instead of crashing', async (t) => {
const root = makeTmpDir('text-intel-ocr-fallback');
t.after(() => rmrf(root));
const service = new TextIntelService({
store: { settingsDir: root },
settings: makeSettings(),
getWindow: () => null,
dataDir: root,
fetchImpl: global.fetch,
});
service.getWorker = async () => {
throw new Error('tesseract missing');
};
const result = await service.ocrAroundClick({
image: {},
size: { width: 100, height: 100 },
display: { bounds: { x: 0, y: 0, width: 100, height: 100 } },
}, { x: 50, y: 50 });
assert.deepEqual(result, { text: '', confidence: null });
});
test('ai response normalization and application keeps fields structured', () => {
const patch = normalizeAiPatch(JSON.stringify({
title: 'Open settings',
description: 'Pick the AI tab.',
blocks: [
{
kind: 'text',
position: 'after-description',
level: 'tip',
title: 'Tip',
body: 'Use the local Ollama model.',
},
{
kind: 'code',
language: 'bash',
code: 'ollama pull llama3.2:1b',
},
{
kind: 'table',
rows: [['Name', 'Value'], ['Host', '127.0.0.1']],
},
],
}));
const step = createStep({
title: 'Old title',
descriptionHtml: '<p>Old text</p>',
textBlocks: [{ id: 'tb1', order: 1, position: 'after-description', level: 'info', title: 'Old tip', descriptionHtml: '<p>Old body</p>' }],
codeBlocks: [{ id: 'cb1', order: 2, language: 'text', code: 'old' }],
tableBlocks: [{ id: 'tbl1', order: 3, rows: [['x']] }],
});
const updated = applyAiPatchToStep(step, patch, { target: 'all' });
assert.equal(updated.title, 'Open settings');
assert.equal(updated.descriptionHtml, '<p>Pick the AI tab.</p>');
assert.equal(updated.textBlocks.length, 1);
assert.equal(updated.textBlocks[0].level, 'success');
assert.equal(updated.textBlocks[0].descriptionHtml, '<p>Use the local Ollama model.</p>');
assert.equal(updated.codeBlocks[0].code, 'ollama pull llama3.2:1b');
assert.deepEqual(updated.tableBlocks[0].rows, [['Name', 'Value'], ['Host', '127.0.0.1']]);
});
test('ollama connection test reports installed models', async (t) => {
const root = makeTmpDir('text-intel-ai');
t.after(() => rmrf(root));
const service = new TextIntelService({
store: { settingsDir: root },
settings: makeSettings(),
getWindow: () => null,
dataDir: root,
fetchImpl: async () => ({
ok: true,
json: async () => ({
models: [
{ name: 'llama3.2:1b' },
{ name: 'qwen3:0.6b' },
],
}),
}),
});
const result = await service.testAiConnection();
assert.equal(result.ok, true);
assert.equal(result.installed, true);
assert.equal(result.model, 'llama3.2:1b');
});
test('invalid ollama output fails safely without saving the step', async (t) => {
const root = makeTmpDir('text-intel-ai-invalid');
t.after(() => rmrf(root));
let saveCalls = 0;
const step = createStep({
title: 'Old title',
descriptionHtml: '<p>Old text</p>',
});
const service = new TextIntelService({
store: {
settingsDir: root,
getGuide: () => ({ guideId: 'g1', title: 'Guide', descriptionHtml: '', stepsOrder: ['s1'] }),
getStep: () => step,
stepImagePath: () => null,
saveStep: () => {
saveCalls += 1;
throw new Error('save should not be called');
},
},
settings: makeSettings(),
getWindow: () => null,
dataDir: root,
fetchImpl: async () => ({
ok: true,
json: async () => ({
message: {
content: 'not json at all',
},
}),
}),
});
const result = await service.generateStepPatch({
guideId: 'g1',
stepId: 's1',
target: 'all',
});
assert.equal(result.ok, false);
assert.match(result.reason, /JSON/i);
assert.equal(saveCalls, 0);
assert.equal(step.title, 'Old title');
});