Added feature to where local ai models can see the image in each step.
This commit is contained in:
@@ -1 +1 @@
|
|||||||
In the future, planned updates include a re-vamp of the settings panel, text box's (important, tip, warning, and note) are not properly moving to the intended areas and I will look into locking the size (or being able to control the size) of images throughout varias export formats.
|
In the future, planned updates include a re-vamp of the settings panel, text box's (important, tip, warning, and note) are not properly moving to the intended areas and I will look into locking the size (or being able to control the size) of images throughout varias export formats. OCR or local ai would be pretty cool....
|
||||||
|
|||||||
@@ -318,7 +318,7 @@ function showSettingsDialog({
|
|||||||
const aiAutoDoc = el('input', { type: 'checkbox', checked: Boolean(settings.ai?.autoDoc) });
|
const aiAutoDoc = el('input', { type: 'checkbox', checked: Boolean(settings.ai?.autoDoc) });
|
||||||
const ollamaHost = makeInput(settings.ai?.ollama?.host || 'http://127.0.0.1:11434');
|
const ollamaHost = makeInput(settings.ai?.ollama?.host || 'http://127.0.0.1:11434');
|
||||||
const ollamaModel = makeInput(settings.ai?.ollama?.model || 'llama3.2:1b');
|
const ollamaModel = makeInput(settings.ai?.ollama?.model || 'llama3.2:1b');
|
||||||
const aiStatus = el('div', { className: 'muted ai-status' }, 'AI stays local through Ollama. Nothing is sent to the cloud.');
|
const aiStatus = el('div', { className: 'muted ai-status' }, 'AI stays local through Ollama. Vision-capable models can also inspect the screenshot attached to each step.');
|
||||||
const testAiBtn = el('button', { type: 'button' }, 'Test connection');
|
const testAiBtn = el('button', { type: 'button' }, 'Test connection');
|
||||||
|
|
||||||
const updateAiStatus = (message, { error = false } = {}) => {
|
const updateAiStatus = (message, { error = false } = {}) => {
|
||||||
@@ -341,7 +341,9 @@ function showSettingsDialog({
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if (result.installed) {
|
if (result.installed) {
|
||||||
updateAiStatus(`Connected to ${result.host} with ${result.model}.`);
|
updateAiStatus(result.vision
|
||||||
|
? `Connected to ${result.host} with ${result.model}. It can inspect screenshots.`
|
||||||
|
: `Connected to ${result.host} with ${result.model}. This model is text-only, so StepForge will use OCR and metadata only.`);
|
||||||
} else {
|
} else {
|
||||||
updateAiStatus(`Connected to ${result.host}. Model ${result.model} is not installed yet.`, { error: true });
|
updateAiStatus(`Connected to ${result.host}. Model ${result.model} is not installed yet.`, { error: true });
|
||||||
}
|
}
|
||||||
|
|||||||
+74
-2
@@ -35,6 +35,15 @@ function clamp(v, min, max) {
|
|||||||
return Math.min(max, Math.max(min, v));
|
return Math.min(max, Math.max(min, v));
|
||||||
}
|
}
|
||||||
|
|
||||||
|
function modelLooksVisionCapable(model) {
|
||||||
|
const clean = normalizeWhitespace(model).toLowerCase();
|
||||||
|
if (!clean) return false;
|
||||||
|
return clean.includes('vision')
|
||||||
|
|| clean.includes('llava')
|
||||||
|
|| clean.includes('gemma4')
|
||||||
|
|| /(^|[^a-z0-9])qwen[23](?:\.[0-9]+)?vl([^a-z0-9]|$)/.test(clean);
|
||||||
|
}
|
||||||
|
|
||||||
let createWorkerImpl = null;
|
let createWorkerImpl = null;
|
||||||
function loadCreateWorker() {
|
function loadCreateWorker() {
|
||||||
if (createWorkerImpl) return createWorkerImpl;
|
if (createWorkerImpl) return createWorkerImpl;
|
||||||
@@ -64,6 +73,7 @@ class TextIntelService {
|
|||||||
this.workerPromise = null;
|
this.workerPromise = null;
|
||||||
this.workerQueue = Promise.resolve();
|
this.workerQueue = Promise.resolve();
|
||||||
this.ocrDataDir = path.join(dataDir, 'ocr', 'eng');
|
this.ocrDataDir = path.join(dataDir, 'ocr', 'eng');
|
||||||
|
this.modelCapabilityCache = new Map();
|
||||||
}
|
}
|
||||||
|
|
||||||
async shutdown() {
|
async shutdown() {
|
||||||
@@ -372,15 +382,63 @@ public static class Win32 {
|
|||||||
const data = await res.json();
|
const data = await res.json();
|
||||||
const models = Array.isArray(data?.models) ? data.models.map((model) => model.name).filter(Boolean) : [];
|
const models = Array.isArray(data?.models) ? data.models.map((model) => model.name).filter(Boolean) : [];
|
||||||
const installed = config.ollama.model ? models.includes(config.ollama.model) : false;
|
const installed = config.ollama.model ? models.includes(config.ollama.model) : false;
|
||||||
|
const vision = installed ? await this.modelSupportsVision({
|
||||||
|
host: config.ollama.host,
|
||||||
|
model: config.ollama.model,
|
||||||
|
}) : false;
|
||||||
return {
|
return {
|
||||||
ok: true,
|
ok: true,
|
||||||
installed,
|
installed,
|
||||||
|
vision,
|
||||||
models,
|
models,
|
||||||
host: config.ollama.host,
|
host: config.ollama.host,
|
||||||
model: config.ollama.model,
|
model: config.ollama.model,
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
|
async modelCapabilities({ host, model }) {
|
||||||
|
const normalizedHost = normalizeOllamaHost(host);
|
||||||
|
const normalizedModel = normalizeWhitespace(model);
|
||||||
|
if (!normalizedHost || !normalizedModel) return [];
|
||||||
|
const cacheKey = `${normalizedHost}::${normalizedModel}`;
|
||||||
|
if (this.modelCapabilityCache.has(cacheKey)) {
|
||||||
|
return this.modelCapabilityCache.get(cacheKey);
|
||||||
|
}
|
||||||
|
const url = new URL('/api/show', `${normalizedHost.replace(/\/+$/, '')}/`);
|
||||||
|
let capabilities = [];
|
||||||
|
try {
|
||||||
|
const response = await this.fetch(url, {
|
||||||
|
method: 'POST',
|
||||||
|
headers: { 'Content-Type': 'application/json' },
|
||||||
|
body: JSON.stringify({ model: normalizedModel }),
|
||||||
|
});
|
||||||
|
if (response.ok) {
|
||||||
|
const payload = await response.json();
|
||||||
|
capabilities = Array.isArray(payload?.capabilities)
|
||||||
|
? payload.capabilities.map((cap) => normalizeWhitespace(cap).toLowerCase()).filter(Boolean)
|
||||||
|
: [];
|
||||||
|
}
|
||||||
|
} catch {
|
||||||
|
capabilities = [];
|
||||||
|
}
|
||||||
|
if (!capabilities.includes('vision') && modelLooksVisionCapable(normalizedModel)) {
|
||||||
|
capabilities = [...capabilities, 'vision'];
|
||||||
|
}
|
||||||
|
this.modelCapabilityCache.set(cacheKey, capabilities);
|
||||||
|
return capabilities;
|
||||||
|
}
|
||||||
|
|
||||||
|
async modelSupportsVision({ host, model }) {
|
||||||
|
const capabilities = await this.modelCapabilities({ host, model });
|
||||||
|
return capabilities.includes('vision');
|
||||||
|
}
|
||||||
|
|
||||||
|
readStepImageBase64(guideId, stepId) {
|
||||||
|
const imagePath = this.store.stepImagePath(guideId, stepId, 'working') || this.store.stepImagePath(guideId, stepId, 'original');
|
||||||
|
if (!imagePath || !fs.existsSync(imagePath)) return '';
|
||||||
|
return fs.readFileSync(imagePath).toString('base64');
|
||||||
|
}
|
||||||
|
|
||||||
async callOllamaText({ host, model, prompt, systemPrompt }) {
|
async callOllamaText({ host, model, prompt, systemPrompt }) {
|
||||||
const url = new URL('/api/chat', `${host.replace(/\/+$/, '')}/`);
|
const url = new URL('/api/chat', `${host.replace(/\/+$/, '')}/`);
|
||||||
const response = await this.fetch(url, {
|
const response = await this.fetch(url, {
|
||||||
@@ -403,8 +461,12 @@ public static class Win32 {
|
|||||||
return content.trim();
|
return content.trim();
|
||||||
}
|
}
|
||||||
|
|
||||||
async callOllama({ host, model, prompt, systemPrompt }) {
|
async callOllama({ host, model, prompt, systemPrompt, images = [] }) {
|
||||||
const url = new URL('/api/chat', `${host.replace(/\/+$/, '')}/`);
|
const url = new URL('/api/chat', `${host.replace(/\/+$/, '')}/`);
|
||||||
|
const userMessage = { role: 'user', content: prompt };
|
||||||
|
if (Array.isArray(images) && images.length) {
|
||||||
|
userMessage.images = images;
|
||||||
|
}
|
||||||
const response = await this.fetch(url, {
|
const response = await this.fetch(url, {
|
||||||
method: 'POST',
|
method: 'POST',
|
||||||
headers: { 'Content-Type': 'application/json' },
|
headers: { 'Content-Type': 'application/json' },
|
||||||
@@ -414,7 +476,7 @@ public static class Win32 {
|
|||||||
format: 'json',
|
format: 'json',
|
||||||
messages: [
|
messages: [
|
||||||
{ role: 'system', content: systemPrompt },
|
{ role: 'system', content: systemPrompt },
|
||||||
{ role: 'user', content: prompt },
|
userMessage,
|
||||||
],
|
],
|
||||||
options: {
|
options: {
|
||||||
temperature: 0.2,
|
temperature: 0.2,
|
||||||
@@ -460,6 +522,14 @@ public static class Win32 {
|
|||||||
return { ok: false, reason: 'Block not found.' };
|
return { ok: false, reason: 'Block not found.' };
|
||||||
}
|
}
|
||||||
|
|
||||||
|
const screenshotBase64 = step.image ? this.readStepImageBase64(guideId, stepId) : '';
|
||||||
|
const screenshotAttached = Boolean(screenshotBase64)
|
||||||
|
? await this.modelSupportsVision({
|
||||||
|
host: config.ollama.host,
|
||||||
|
model: config.ollama.model,
|
||||||
|
})
|
||||||
|
: false;
|
||||||
|
|
||||||
let captureContext = null;
|
let captureContext = null;
|
||||||
// Use stored capture metadata when available (best context, from capture time).
|
// Use stored capture metadata when available (best context, from capture time).
|
||||||
// Fall back to re-running OCR on the stored image only when metadata is absent.
|
// Fall back to re-running OCR on the stored image only when metadata is absent.
|
||||||
@@ -514,6 +584,7 @@ public static class Win32 {
|
|||||||
step,
|
step,
|
||||||
captureContext,
|
captureContext,
|
||||||
block: currentBlock,
|
block: currentBlock,
|
||||||
|
screenshotAttached,
|
||||||
});
|
});
|
||||||
|
|
||||||
const raw = await this.callOllama({
|
const raw = await this.callOllama({
|
||||||
@@ -521,6 +592,7 @@ public static class Win32 {
|
|||||||
model: config.ollama.model,
|
model: config.ollama.model,
|
||||||
prompt,
|
prompt,
|
||||||
systemPrompt,
|
systemPrompt,
|
||||||
|
images: screenshotAttached ? [screenshotBase64] : [],
|
||||||
});
|
});
|
||||||
const patch = normalizeAiPatch(raw);
|
const patch = normalizeAiPatch(raw);
|
||||||
const updated = applyAiPatchToStep(step, patch, { target, blockId });
|
const updated = applyAiPatchToStep(step, patch, { target, blockId });
|
||||||
|
|||||||
@@ -673,6 +673,7 @@ function buildAiPrompt({
|
|||||||
step = null,
|
step = null,
|
||||||
captureContext = null,
|
captureContext = null,
|
||||||
block = null,
|
block = null,
|
||||||
|
screenshotAttached = false,
|
||||||
} = {}) {
|
} = {}) {
|
||||||
const hasDraftTitle = step && !isPlaceholderTitle(step.title);
|
const hasDraftTitle = step && !isPlaceholderTitle(step.title);
|
||||||
const hasDraftDesc = step && Boolean(htmlToText(step.descriptionHtml || ''));
|
const hasDraftDesc = step && Boolean(htmlToText(step.descriptionHtml || ''));
|
||||||
@@ -717,6 +718,7 @@ function buildAiPrompt({
|
|||||||
(!hasDraftTitle || target === 'description') && captureContext.titleCandidate
|
(!hasDraftTitle || target === 'description') && captureContext.titleCandidate
|
||||||
? `Suggested title: ${captureContext.titleCandidate}` : null,
|
? `Suggested title: ${captureContext.titleCandidate}` : null,
|
||||||
] : []),
|
] : []),
|
||||||
|
screenshotAttached ? 'Screenshot: attached to this request.' : null,
|
||||||
draftTitleLine,
|
draftTitleLine,
|
||||||
draftDescLine,
|
draftDescLine,
|
||||||
].filter(Boolean);
|
].filter(Boolean);
|
||||||
@@ -770,6 +772,9 @@ function buildAiPrompt({
|
|||||||
richContext
|
richContext
|
||||||
? '- Use the OCR text, window title, app name, and element info to make the documentation specific.'
|
? '- Use the OCR text, window title, app name, and element info to make the documentation specific.'
|
||||||
: '- Context is limited. Use the app name or window title if available; generate a reasonable action title.',
|
: '- Context is limited. Use the app name or window title if available; generate a reasonable action title.',
|
||||||
|
screenshotAttached
|
||||||
|
? '- A screenshot is attached. Use it together with the OCR and metadata to resolve visual details, but do not mention the screenshot in the output.'
|
||||||
|
: '- No screenshot is attached. Rely on OCR, the window title, app name, and element info.',
|
||||||
'- Do NOT generate blocks that describe the technical capture process or mention OCR.',
|
'- Do NOT generate blocks that describe the technical capture process or mention OCR.',
|
||||||
'- Do NOT invent details not supported by the capture context.',
|
'- Do NOT invent details not supported by the capture context.',
|
||||||
'- If the target is one block, only rewrite that block.',
|
'- If the target is one block, only rewrite that block.',
|
||||||
|
|||||||
@@ -22,6 +22,14 @@ ollama pull llama3.2:1b
|
|||||||
|
|
||||||
That model is small enough to feel responsive on modest hardware, but still good enough for human-sounding titles and short text blocks.
|
That model is small enough to feel responsive on modest hardware, but still good enough for human-sounding titles and short text blocks.
|
||||||
|
|
||||||
|
If you want StepForge to send the screenshot itself to the model, pull a vision-capable model instead:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
ollama pull llama3.2-vision
|
||||||
|
```
|
||||||
|
|
||||||
|
That model can inspect pictures as well as text, so it is better when you want the AI to read the UI directly from the screenshot.
|
||||||
|
|
||||||
If you need something even smaller, try:
|
If you need something even smaller, try:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
@@ -44,7 +52,7 @@ Set:
|
|||||||
|
|
||||||
* `Enable AI text filling` to on
|
* `Enable AI text filling` to on
|
||||||
* `Ollama host` to your local Ollama server
|
* `Ollama host` to your local Ollama server
|
||||||
* `Ollama model` to `llama3.2:1b` or the smaller model you pulled
|
* `Ollama model` to `llama3.2:1b` for text-only mode, or `llama3.2-vision` if you want screenshot-aware AI
|
||||||
|
|
||||||
The default host is:
|
The default host is:
|
||||||
|
|
||||||
@@ -73,3 +81,4 @@ You can also use `More -> Generate all text fields with AI` to fill the whole st
|
|||||||
* Capture titles are still generated automatically without AI.
|
* Capture titles are still generated automatically without AI.
|
||||||
* AI generation only works when `Enable AI text filling` is turned on.
|
* AI generation only works when `Enable AI text filling` is turned on.
|
||||||
* The app always uses local OCR around the click area first, then local AI only when you ask for it.
|
* The app always uses local OCR around the click area first, then local AI only when you ask for it.
|
||||||
|
* When the selected Ollama model supports vision, StepForge also sends the screenshot to the model so it can cross-check OCR and visual context.
|
||||||
|
|||||||
@@ -1,5 +1,7 @@
|
|||||||
'use strict';
|
'use strict';
|
||||||
|
|
||||||
|
const fs = require('node:fs');
|
||||||
|
const path = require('node:path');
|
||||||
const test = require('node:test');
|
const test = require('node:test');
|
||||||
const assert = require('node:assert/strict');
|
const assert = require('node:assert/strict');
|
||||||
|
|
||||||
@@ -309,21 +311,112 @@ test('ollama connection test reports installed models', async (t) => {
|
|||||||
settings: makeSettings(),
|
settings: makeSettings(),
|
||||||
getWindow: () => null,
|
getWindow: () => null,
|
||||||
dataDir: root,
|
dataDir: root,
|
||||||
fetchImpl: async () => ({
|
fetchImpl: async (url) => {
|
||||||
ok: true,
|
const pathname = new URL(url).pathname;
|
||||||
json: async () => ({
|
if (pathname === '/api/tags') {
|
||||||
models: [
|
return {
|
||||||
{ name: 'llama3.2:1b' },
|
ok: true,
|
||||||
{ name: 'qwen3:0.6b' },
|
json: async () => ({
|
||||||
],
|
models: [
|
||||||
}),
|
{ name: 'llama3.2:1b' },
|
||||||
}),
|
{ name: 'qwen3:0.6b' },
|
||||||
|
],
|
||||||
|
}),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
if (pathname === '/api/show') {
|
||||||
|
return {
|
||||||
|
ok: true,
|
||||||
|
json: async () => ({
|
||||||
|
capabilities: ['completion', 'vision'],
|
||||||
|
}),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
throw new Error(`unexpected fetch: ${pathname}`);
|
||||||
|
},
|
||||||
});
|
});
|
||||||
|
|
||||||
const result = await service.testAiConnection();
|
const result = await service.testAiConnection();
|
||||||
assert.equal(result.ok, true);
|
assert.equal(result.ok, true);
|
||||||
assert.equal(result.installed, true);
|
assert.equal(result.installed, true);
|
||||||
assert.equal(result.model, 'llama3.2:1b');
|
assert.equal(result.model, 'llama3.2:1b');
|
||||||
|
assert.equal(result.vision, true);
|
||||||
|
});
|
||||||
|
|
||||||
|
test('vision-capable models receive the screenshot in the chat request', async (t) => {
|
||||||
|
const root = makeTmpDir('text-intel-ai-vision');
|
||||||
|
t.after(() => rmrf(root));
|
||||||
|
const imagePath = path.join(root, 'step.png');
|
||||||
|
fs.writeFileSync(imagePath, Buffer.from('fake screenshot bytes'));
|
||||||
|
|
||||||
|
const step = createStep({
|
||||||
|
title: 'Old title',
|
||||||
|
descriptionHtml: '<p>Old text</p>',
|
||||||
|
image: {
|
||||||
|
originalPath: 'original.png',
|
||||||
|
workingPath: 'working.png',
|
||||||
|
size: { width: 10, height: 10 },
|
||||||
|
},
|
||||||
|
captureMetadata: {
|
||||||
|
windowTitle: 'Settings',
|
||||||
|
appName: 'chrome',
|
||||||
|
ocrText: 'Open settings',
|
||||||
|
titleCandidate: 'Open settings',
|
||||||
|
mode: 'fullscreen',
|
||||||
|
},
|
||||||
|
});
|
||||||
|
const fetchCalls = [];
|
||||||
|
const service = new TextIntelService({
|
||||||
|
store: {
|
||||||
|
settingsDir: root,
|
||||||
|
getGuide: () => ({ guideId: 'g1', title: 'Guide', descriptionHtml: '', stepsOrder: ['s1'] }),
|
||||||
|
getStep: () => step,
|
||||||
|
stepImagePath: () => imagePath,
|
||||||
|
saveStep: (_, next) => next,
|
||||||
|
},
|
||||||
|
settings: makeSettings(),
|
||||||
|
getWindow: () => null,
|
||||||
|
dataDir: root,
|
||||||
|
fetchImpl: async (url, init = {}) => {
|
||||||
|
const pathname = new URL(url).pathname;
|
||||||
|
fetchCalls.push({ pathname, init });
|
||||||
|
if (pathname === '/api/show') {
|
||||||
|
return {
|
||||||
|
ok: true,
|
||||||
|
json: async () => ({
|
||||||
|
capabilities: ['completion', 'vision'],
|
||||||
|
}),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
if (pathname === '/api/chat') {
|
||||||
|
return {
|
||||||
|
ok: true,
|
||||||
|
json: async () => ({
|
||||||
|
message: {
|
||||||
|
content: JSON.stringify({
|
||||||
|
title: 'Open settings',
|
||||||
|
description: 'Use the AI tab.',
|
||||||
|
}),
|
||||||
|
},
|
||||||
|
}),
|
||||||
|
};
|
||||||
|
}
|
||||||
|
throw new Error(`unexpected fetch: ${pathname}`);
|
||||||
|
},
|
||||||
|
});
|
||||||
|
|
||||||
|
const result = await service.generateStepPatch({
|
||||||
|
guideId: 'g1',
|
||||||
|
stepId: 's1',
|
||||||
|
target: 'all',
|
||||||
|
});
|
||||||
|
|
||||||
|
assert.equal(result.ok, true);
|
||||||
|
const chatCall = fetchCalls.find((call) => call.pathname === '/api/chat');
|
||||||
|
assert.ok(chatCall, 'expected an Ollama chat request');
|
||||||
|
const body = JSON.parse(chatCall.init.body);
|
||||||
|
assert.deepEqual(body.messages[1].images, [fs.readFileSync(imagePath).toString('base64')]);
|
||||||
|
assert.match(body.messages[1].content, /Screenshot: attached/i);
|
||||||
});
|
});
|
||||||
|
|
||||||
test('invalid ollama output fails safely without saving the step', async (t) => {
|
test('invalid ollama output fails safely without saving the step', async (t) => {
|
||||||
|
|||||||
Reference in New Issue
Block a user