The title engine was falling back to "Screen capture" for all browser captures because it treated the entire browser window title as noise. Changes: - stripBrowserNameSuffix: removes "- Google Chrome" / "| Firefox" etc. from the end of window titles, leaving just the page title. "oracle - Google Search - Google Chrome" → "oracle - Google Search" "Oracle | Cloud Applications - Google Chrome" → page title only - extractSearchQuery: detects "[query] - Google Search" / Bing / etc. patterns after stripping the browser suffix, and formats the result as "Search for oracle". - buildCaptureTitle: uses stripped page title + search detection before falling through to the "Screen capture" fallback. - pickBestOcrPhrase: considers the full OCR line (≤80 chars) as a candidate with a +35 completeness bonus before splitting on | or ·. This preserves "Oracle | Cloud Applications and Cloud Platform" as a single phrase instead of breaking it into fragments. - candidateWords: filters out standalone punctuation tokens (|, ·, •) so they don't inflate word-count penalties for compound brand names. - verbForElementRole: hyperlinks and links now produce "Select" instead of "Click"; search box / search field produces "Search for". Five new unit tests cover: browser title stripping, search query extraction, full pipe-separated link text, link and search box verbs. Co-Authored-By: Claude Sonnet 4.6 <[email protected]>
313 lines
9.1 KiB
JavaScript
313 lines
9.1 KiB
JavaScript
'use strict';
|
|
|
|
const test = require('node:test');
|
|
const assert = require('node:assert/strict');
|
|
|
|
const { makeTmpDir, rmrf } = require('./helpers');
|
|
const { createStep } = require('../../core/schema');
|
|
const {
|
|
buildCaptureTitle,
|
|
buildAiPrompt,
|
|
normalizeAiPatch,
|
|
applyAiPatchToStep,
|
|
} = require('../../core/text-intel');
|
|
const { TextIntelService } = require('../../app/text-intel');
|
|
|
|
function makeSettings(values = {}) {
|
|
const data = {
|
|
ai: {
|
|
enabled: true,
|
|
ollama: {
|
|
host: 'http://127.0.0.1:11434',
|
|
model: 'llama3.2:1b',
|
|
},
|
|
},
|
|
...values,
|
|
};
|
|
return {
|
|
get(key) {
|
|
return key.split('.').reduce((acc, part) => (acc == null ? undefined : acc[part]), data);
|
|
},
|
|
};
|
|
}
|
|
|
|
test('capture titles prefer semantic metadata before OCR fallback', () => {
|
|
const title = buildCaptureTitle({
|
|
mode: 'fullscreen',
|
|
metadata: { windowTitle: 'Reset user password in admin portal' },
|
|
ocrText: 'Save',
|
|
});
|
|
assert.equal(title, 'Click Save');
|
|
});
|
|
|
|
test('capture titles prefer element metadata before window chrome and OCR', () => {
|
|
const title = buildCaptureTitle({
|
|
mode: 'window',
|
|
metadata: {
|
|
elementLabel: 'Open advanced settings',
|
|
windowTitle: 'Preferences',
|
|
},
|
|
ocrText: 'Cancel',
|
|
});
|
|
assert.equal(title, 'Open advanced settings');
|
|
});
|
|
|
|
test('capture titles ignore browser chrome noise in favor of OCR', () => {
|
|
const title = buildCaptureTitle({
|
|
mode: 'window',
|
|
metadata: {
|
|
windowTitle: 'Google Chrome ** PR reviews ** /chrome/tyler/autodoc',
|
|
appName: 'Google Chrome',
|
|
},
|
|
ocrText: 'New tab',
|
|
});
|
|
assert.equal(title, 'Click New tab');
|
|
});
|
|
|
|
test('tab-like roles use select when OCR identifies a tab label', () => {
|
|
const title = buildCaptureTitle({
|
|
mode: 'window',
|
|
metadata: {
|
|
elementLabel: 'New tab',
|
|
elementRole: 'tab item',
|
|
windowTitle: 'Google Chrome - PR reviews',
|
|
},
|
|
ocrText: 'New tab',
|
|
});
|
|
assert.equal(title, 'Select New tab');
|
|
});
|
|
|
|
test('capture titles fall back to OCR when metadata is absent', () => {
|
|
const title = buildCaptureTitle({
|
|
mode: 'window',
|
|
metadata: {},
|
|
ocrText: 'Save changes',
|
|
});
|
|
assert.equal(title, 'Click Save changes');
|
|
});
|
|
|
|
test('browser window title strips browser name and falls back to page title', () => {
|
|
// OCR fails; browser window title should give something useful, not "Screen capture".
|
|
const title = buildCaptureTitle({
|
|
mode: 'fullscreen',
|
|
metadata: {
|
|
windowTitle: 'Oracle | Cloud Applications and Cloud Platform - Google Chrome',
|
|
appName: 'chrome',
|
|
},
|
|
ocrText: '',
|
|
});
|
|
// Stripped title "Oracle | Cloud Applications and Cloud Platform" → best fragment
|
|
assert.ok(title !== 'Screen capture', `Expected smart title, got: ${title}`);
|
|
assert.ok(title.toLowerCase().includes('oracle') || title.toLowerCase().includes('cloud'), `Expected oracle/cloud in title, got: ${title}`);
|
|
});
|
|
|
|
test('search query is extracted from browser window title pattern', () => {
|
|
const title = buildCaptureTitle({
|
|
mode: 'fullscreen',
|
|
metadata: {
|
|
windowTitle: 'oracle - Google Search - Google Chrome',
|
|
appName: 'chrome',
|
|
},
|
|
ocrText: '',
|
|
});
|
|
assert.equal(title, 'Search for Oracle');
|
|
});
|
|
|
|
test('full link text with pipe separator is preserved in OCR phrases', () => {
|
|
const title = buildCaptureTitle({
|
|
mode: 'fullscreen',
|
|
metadata: { elementRole: 'hyperlink' },
|
|
ocrText: 'Oracle | Cloud Applications and Cloud Platform',
|
|
});
|
|
assert.equal(title, 'Select Oracle | Cloud Applications and Cloud Platform');
|
|
});
|
|
|
|
test('link element role uses Select verb', () => {
|
|
const title = buildCaptureTitle({
|
|
mode: 'fullscreen',
|
|
metadata: { elementLabel: 'Sign in', elementRole: 'hyperlink' },
|
|
ocrText: '',
|
|
});
|
|
assert.equal(title, 'Select Sign in');
|
|
});
|
|
|
|
test('search box element role uses Search for verb', () => {
|
|
const title = buildCaptureTitle({
|
|
mode: 'fullscreen',
|
|
metadata: { elementLabel: 'oracle', elementRole: 'search box' },
|
|
ocrText: '',
|
|
});
|
|
assert.equal(title, 'Search for Oracle');
|
|
});
|
|
|
|
test('ai prompts include the deterministic OCR-backed title candidate', () => {
|
|
const { prompt } = buildAiPrompt({
|
|
captureContext: {
|
|
windowTitle: 'Google Chrome ** PR reviews ** /chrome/tyler/autodoc',
|
|
appName: 'Google Chrome',
|
|
ocrText: 'New tab',
|
|
titleCandidate: 'Click New tab',
|
|
mode: 'content',
|
|
},
|
|
});
|
|
|
|
assert.match(prompt, /Suggested title: Click New tab/);
|
|
});
|
|
|
|
test('ocr crop rectangles clamp to the image bounds', (t) => {
|
|
const root = makeTmpDir('text-intel-crop');
|
|
t.after(() => rmrf(root));
|
|
const service = new TextIntelService({
|
|
store: { settingsDir: root },
|
|
settings: makeSettings(),
|
|
getWindow: () => null,
|
|
dataDir: root,
|
|
fetchImpl: global.fetch,
|
|
});
|
|
|
|
const frame = {
|
|
size: { width: 1000, height: 500 },
|
|
display: { bounds: { x: 0, y: 0, width: 1000, height: 500 } },
|
|
};
|
|
|
|
const topLeft = service.cropRectForPoint(frame, { x: 5, y: 5 });
|
|
assert.deepEqual(topLeft, { x: 0, y: 0, width: 420, height: 220 });
|
|
|
|
const bottomRight = service.cropRectForPoint(frame, { x: 995, y: 495 });
|
|
assert.deepEqual(bottomRight, { x: 580, y: 280, width: 420, height: 220 });
|
|
});
|
|
|
|
test('ocr failures fall back to empty text instead of crashing', async (t) => {
|
|
const root = makeTmpDir('text-intel-ocr-fallback');
|
|
t.after(() => rmrf(root));
|
|
const service = new TextIntelService({
|
|
store: { settingsDir: root },
|
|
settings: makeSettings(),
|
|
getWindow: () => null,
|
|
dataDir: root,
|
|
fetchImpl: global.fetch,
|
|
});
|
|
service.getWorker = async () => {
|
|
throw new Error('tesseract missing');
|
|
};
|
|
|
|
const result = await service.ocrAroundClick({
|
|
image: {},
|
|
size: { width: 100, height: 100 },
|
|
display: { bounds: { x: 0, y: 0, width: 100, height: 100 } },
|
|
}, { x: 50, y: 50 });
|
|
|
|
assert.deepEqual(result, { text: '', confidence: null });
|
|
});
|
|
|
|
test('ai response normalization and application keeps fields structured', () => {
|
|
const patch = normalizeAiPatch(JSON.stringify({
|
|
title: 'Open settings',
|
|
description: 'Pick the AI tab.',
|
|
blocks: [
|
|
{
|
|
kind: 'text',
|
|
position: 'after-description',
|
|
level: 'tip',
|
|
title: 'Tip',
|
|
body: 'Use the local Ollama model.',
|
|
},
|
|
{
|
|
kind: 'code',
|
|
language: 'bash',
|
|
code: 'ollama pull llama3.2:1b',
|
|
},
|
|
{
|
|
kind: 'table',
|
|
rows: [['Name', 'Value'], ['Host', '127.0.0.1']],
|
|
},
|
|
],
|
|
}));
|
|
|
|
const step = createStep({
|
|
title: 'Old title',
|
|
descriptionHtml: '<p>Old text</p>',
|
|
textBlocks: [{ id: 'tb1', order: 1, position: 'after-description', level: 'info', title: 'Old tip', descriptionHtml: '<p>Old body</p>' }],
|
|
codeBlocks: [{ id: 'cb1', order: 2, language: 'text', code: 'old' }],
|
|
tableBlocks: [{ id: 'tbl1', order: 3, rows: [['x']] }],
|
|
});
|
|
|
|
const updated = applyAiPatchToStep(step, patch, { target: 'all' });
|
|
assert.equal(updated.title, 'Open settings');
|
|
assert.equal(updated.descriptionHtml, '<p>Pick the AI tab.</p>');
|
|
assert.equal(updated.textBlocks.length, 1);
|
|
assert.equal(updated.textBlocks[0].level, 'success');
|
|
assert.equal(updated.textBlocks[0].descriptionHtml, '<p>Use the local Ollama model.</p>');
|
|
assert.equal(updated.codeBlocks[0].code, 'ollama pull llama3.2:1b');
|
|
assert.deepEqual(updated.tableBlocks[0].rows, [['Name', 'Value'], ['Host', '127.0.0.1']]);
|
|
});
|
|
|
|
test('ollama connection test reports installed models', async (t) => {
|
|
const root = makeTmpDir('text-intel-ai');
|
|
t.after(() => rmrf(root));
|
|
const service = new TextIntelService({
|
|
store: { settingsDir: root },
|
|
settings: makeSettings(),
|
|
getWindow: () => null,
|
|
dataDir: root,
|
|
fetchImpl: async () => ({
|
|
ok: true,
|
|
json: async () => ({
|
|
models: [
|
|
{ name: 'llama3.2:1b' },
|
|
{ name: 'qwen3:0.6b' },
|
|
],
|
|
}),
|
|
}),
|
|
});
|
|
|
|
const result = await service.testAiConnection();
|
|
assert.equal(result.ok, true);
|
|
assert.equal(result.installed, true);
|
|
assert.equal(result.model, 'llama3.2:1b');
|
|
});
|
|
|
|
test('invalid ollama output fails safely without saving the step', async (t) => {
|
|
const root = makeTmpDir('text-intel-ai-invalid');
|
|
t.after(() => rmrf(root));
|
|
let saveCalls = 0;
|
|
const step = createStep({
|
|
title: 'Old title',
|
|
descriptionHtml: '<p>Old text</p>',
|
|
});
|
|
const service = new TextIntelService({
|
|
store: {
|
|
settingsDir: root,
|
|
getGuide: () => ({ guideId: 'g1', title: 'Guide', descriptionHtml: '', stepsOrder: ['s1'] }),
|
|
getStep: () => step,
|
|
stepImagePath: () => null,
|
|
saveStep: () => {
|
|
saveCalls += 1;
|
|
throw new Error('save should not be called');
|
|
},
|
|
},
|
|
settings: makeSettings(),
|
|
getWindow: () => null,
|
|
dataDir: root,
|
|
fetchImpl: async () => ({
|
|
ok: true,
|
|
json: async () => ({
|
|
message: {
|
|
content: 'not json at all',
|
|
},
|
|
}),
|
|
}),
|
|
});
|
|
|
|
const result = await service.generateStepPatch({
|
|
guideId: 'g1',
|
|
stepId: 's1',
|
|
target: 'all',
|
|
});
|
|
|
|
assert.equal(result.ok, false);
|
|
assert.match(result.reason, /JSON/i);
|
|
assert.equal(saveCalls, 0);
|
|
assert.equal(step.title, 'Old title');
|
|
});
|