Files
autofirmer-expanded/extension/background.js
T
Brandon LiandClaude Opus 5 b748f95372 Add automation framework, typing, and runner for the autobuyer
Turns the autobuyer from a page scraper into something that acts. A dashboard
button queues a run; a desktop process executes it against the real browser.

lib/automations.ts — automations are declarative step lists nested inside the
firm whose site they drive. Steps are click / type / wait / navigate, and they
inherit the firm's tab pattern and URL, so one firm's automation can't act on
another's tab. Adding a button means adding an entry here; the page renders
buttons from the API and the runner receives steps from the server, so neither
needs editing. Runs key on firm:automation — every firm will plausibly have its
own "buy-accounts", and a bare id would resolve to the wrong one.

clicker/runner.py — the daemon behind the buttons. Claims a queued run, works
through the steps, reports each one back for the page's live log. Only one run
executes at a time: two processes driving one physical mouse would interleave
clicks. Heartbeats on its own thread, because a step can block for tens of
seconds and folding the beat into the main loop would show the runner as offline
in the middle of the run it was executing.

clicker/actions.py — one implementation of the safety checks, shared by the CLI
and the runner. Refuses to act when the element is covered by an overlay, when
coordinates fall off-screen, when the browser can't be confirmed frontmost, or
(for type) when the target isn't an editable field.

Typing: uneven human cadence, and the field is read back afterwards and compared
against what was typed — a field that never took focus fails silently and looks
identical to success otherwise. Non-ASCII is rejected because pyautogui skips
those characters without complaint, and newlines because Enter may submit the
form. Typos are deliberately not simulated: a mistyped digit in a trading form
is a real loss, and the correction is the part that can go wrong.

Extension: opens the firm's page when no tab matches, navigates to a specific
page for a navigate step (skipped when already there, so page state survives),
and retries the locate while a freshly loaded React app mounts — `complete` only
means the document loaded.

Staleness reporting, after it cost three debugging rounds: Chrome doesn't reload
an unpacked extension and Python doesn't reload a running process, so both now
report their version. A stale runner gets a red banner naming both versions and
the automation buttons are disabled, rather than failing mid-run on a step type
it predates.

Scale detection is now conservative: a raw OS/browser width ratio is only
trusted when it lands on a real scaling factor. On this multi-monitor desktop
the previous logic would have silently halved every coordinate.

Verified end to end against the live browser: navigate, locate, and a real
click (run #12, all three steps). API round-trips, claim-once semantics, run
cancellation, the heartbeat online/offline lifecycle, motion geometry and
timing, focus activation, and typing verification all pass.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-28 14:00:33 -05:00

417 lines
18 KiB
JavaScript

/* AutoFirmer Capture — MV3 service worker.
*
* Loop: poll GET /api/autobuyer/status. While `enabled` is true, scrape the
* target tab's HTML and POST it to /api/autobuyer/capture. The dashboard's
* AutoBuyer page renders whatever landed there.
*/
const DEFAULTS = {
apiBase: 'http://localhost:3000',
pollSeconds: 3,
// Chrome match pattern, e.g. "https://*.tradovate.com/*". Empty = whichever
// tab is active in the last focused window — which would mean scraping your
// banking or email tab if you switched to one. Default it to the broker.
targetUrlPattern: 'https://*.tradeify.co/*',
// Optional CSS selector — matching elements get their on-screen position
// measured alongside the HTML.
elementSelector: '',
};
let timer = null;
let pendingLocate = 0;
// ── Config / state ──────────────────────────────────────────────────────────
async function getConfig() {
const stored = await chrome.storage.local.get(Object.keys(DEFAULTS));
const cfg = { ...DEFAULTS, ...stored };
cfg.apiBase = String(cfg.apiBase).replace(/\/+$/, '');
return cfg;
}
/** Mirror the worker's status into storage so the popup can render it. */
async function setState(patch) {
const { state: prev } = await chrome.storage.local.get('state');
await chrome.storage.local.set({ state: { ...(prev || {}), ...patch, updatedAt: Date.now() } });
}
function setBadge(connected, enabled) {
chrome.action.setBadgeText({ text: !connected ? '!' : enabled ? 'ON' : '' });
chrome.action.setBadgeBackgroundColor({ color: !connected ? '#ef4444' : '#22c55e' });
}
// ── The injected scraper ────────────────────────────────────────────────────
/** Runs in the page's own world. Must be self-contained — nothing from this
* file's scope is available inside it. */
function pageCapture(selector) {
// getBoundingClientRect() measures from the top-left of the viewport, so
// rect.left / rect.top ARE the element's position relative to the window.
// + scrollX/scrollY -> position within the document
// + screenX/screenY and the chrome height -> absolute desktop coordinates,
// which is what an OS-level clicker needs.
const chromeHeight = window.outerHeight - window.innerHeight;
const elements = selector
? Array.from(document.querySelectorAll(selector)).slice(0, 200).map((el, index) => {
const r = el.getBoundingClientRect();
return {
index,
tag: el.tagName.toLowerCase(),
id: el.id || null,
text: (el.textContent || '').trim().slice(0, 80),
// Off-screen or display:none elements still return a rect — all zeroes.
visible: r.width > 0 && r.height > 0,
inViewport: r.top < window.innerHeight && r.bottom > 0 && r.left < window.innerWidth && r.right > 0,
viewport: { x: r.left, y: r.top, width: r.width, height: r.height },
page: { x: r.left + window.scrollX, y: r.top + window.scrollY },
// Centre of the element in desktop coordinates (CSS pixels —
// multiply by devicePixelRatio for physical pixels on HiDPI).
screen: {
x: window.screenX + r.left + r.width / 2,
y: window.screenY + chromeHeight + r.top + r.height / 2,
},
};
})
: [];
return {
url: location.href,
title: document.title,
html: document.documentElement.outerHTML,
viewport: {
scrollX: window.scrollX,
scrollY: window.scrollY,
innerWidth: window.innerWidth,
innerHeight: window.innerHeight,
outerWidth: window.outerWidth,
outerHeight: window.outerHeight,
screenX: window.screenX,
screenY: window.screenY,
chromeHeight,
devicePixelRatio: window.devicePixelRatio,
},
elements,
};
}
// ── Capture pipeline ────────────────────────────────────────────────────────
async function pickTab(cfg) {
if (cfg.targetUrlPattern) {
try {
const tabs = await chrome.tabs.query({ url: cfg.targetUrlPattern });
return tabs.find((t) => /^https?:/.test(t.url || ''));
} catch {
throw new Error(`Invalid target URL pattern: ${cfg.targetUrlPattern}`);
}
}
const [tab] = await chrome.tabs.query({ active: true, lastFocusedWindow: true });
// chrome://, about:, the Web Store and PDF viewers can't be scripted.
if (!tab || !/^https?:/.test(tab.url || '')) return undefined;
// Skip the dashboard itself, otherwise it just captures its own output.
if (tab.url.startsWith(cfg.apiBase)) return undefined;
return tab;
}
async function captureAndSend(cfg) {
const tab = await pickTab(cfg);
if (!tab) {
await setState({ connected: true, enabled: true, error: null, waiting: 'No eligible tab to capture' });
return;
}
const [injection] = await chrome.scripting.executeScript({
target: { tabId: tab.id },
func: pageCapture,
args: [cfg.elementSelector || ''],
});
const result = injection?.result;
if (!result) throw new Error('Injection returned nothing');
const res = await fetch(`${cfg.apiBase}/api/autobuyer/capture`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({
...result,
// Stamp the build so the dashboard side can tell which version of this
// worker is actually live — Chrome keeps running the old one until the
// extension is reloaded, which is invisible from the server otherwise.
viewport: { ...result.viewport, extVersion: chrome.runtime.getManifest().version },
capturedAt: Date.now(),
}),
});
if (!res.ok) throw new Error(`Capture POST failed: HTTP ${res.status}`);
await setState({
connected: true,
enabled: true,
error: null,
waiting: null,
lastCapture: { url: result.url, title: result.title, bytes: result.html.length, at: Date.now() },
});
}
// ── Locate: turn a CSS selector into desktop coordinates ────────────────────
/** Runs in the page. Scrolls the element into view, then reports where it ended up. */
function pageLocate(selector, index) {
const matches = document.querySelectorAll(selector);
if (matches.length === 0) return { error: `No element matches ${selector}` };
const el = matches[index];
if (!el) return { error: `Selector ${selector} matched ${matches.length} element(s), no index ${index}` };
// Centre it in the viewport first — an element scrolled off-screen has a rect,
// but clicking those coordinates would hit whatever is actually displayed there.
el.scrollIntoView({ block: 'center', inline: 'center', behavior: 'instant' });
const r = el.getBoundingClientRect();
if (r.width === 0 || r.height === 0) {
return { error: `Element ${selector} has zero size (hidden or display:none)` };
}
if (r.bottom <= 0 || r.top >= window.innerHeight || r.right <= 0 || r.left >= window.innerWidth) {
return { error: `Element ${selector} is outside the viewport even after scrolling` };
}
const chromeHeight = window.outerHeight - window.innerHeight;
const cx = r.left + r.width / 2;
const cy = r.top + r.height / 2;
// What sits at that point? If it's not our element (or a descendant), something
// is covering it — a cookie banner, a modal — and the click would hit that instead.
const atPoint = document.elementFromPoint(cx, cy);
const covered = atPoint && atPoint !== el && !el.contains(atPoint);
// Current contents, so the caller can confirm afterwards that what it typed
// actually landed — a field that silently ignored the keystrokes (masked,
// read-only, or never focused) is otherwise indistinguishable from success.
const isField = el.tagName === 'INPUT' || el.tagName === 'TEXTAREA';
const value = isField ? el.value : (el.isContentEditable ? el.innerText : null);
return {
tag: el.tagName.toLowerCase(),
text: (el.textContent || '').trim().slice(0, 80),
matchCount: matches.length,
value,
editable: (isField || el.isContentEditable) && !el.disabled && !el.readOnly,
inputType: isField ? (el.type || null) : null,
covered: !!covered,
coveredBy: covered ? `${atPoint.tagName.toLowerCase()}${atPoint.id ? '#' + atPoint.id : ''}` : null,
viewport: { x: r.left, y: r.top, width: r.width, height: r.height },
// Desktop coordinates of the element's centre, in CSS pixels.
screen: { x: window.screenX + cx, y: window.screenY + chromeHeight + cy },
// Lets the Python side work out whether the OS uses a different pixel
// scale than CSS does (Windows display scaling, some Linux setups).
screenSize: { width: window.screen.width, height: window.screen.height },
devicePixelRatio: window.devicePixelRatio,
chromeHeight,
url: location.href,
};
}
/** Is the tab already showing this page? Compares origin and path only —
* query strings and hashes shouldn't force a reload. */
function alreadyAt(current, target) {
try {
const a = new URL(current);
const b = new URL(target);
return a.origin === b.origin && a.pathname.replace(/\/$/, '') === b.pathname.replace(/\/$/, '');
} catch {
return false;
}
}
/** Resolve after the tab reports `complete`. A tab that has only just been
* created has no DOM to inject into yet. */
function waitForTabLoad(tabId, timeoutMs = 15000) {
return new Promise((resolve, reject) => {
const started = Date.now();
const check = async () => {
let tab;
try {
tab = await chrome.tabs.get(tabId);
} catch {
return reject(new Error('the tab was closed while loading'));
}
if (tab.status === 'complete') return resolve(tab);
if (Date.now() - started > timeoutMs) {
return reject(new Error(`the page did not finish loading within ${timeoutMs / 1000}s`));
}
setTimeout(check, 200);
};
check();
});
}
async function resolveLocateTab(cfg, request) {
const pattern = request.urlPattern || cfg.targetUrlPattern;
if (pattern) {
const tabs = await chrome.tabs.query({ url: pattern });
const tab = tabs.find((t) => /^https?:/.test(t.url || ''));
if (tab) return { tab, opened: false };
// Nothing matching is open. Open it rather than failing the run — but only
// to the URL the firm declared, never to something a request supplied that
// we have no host permission for.
if (request.openUrl) {
const created = await chrome.tabs.create({ url: request.openUrl, active: true });
await waitForTabLoad(created.id);
return { tab: await chrome.tabs.get(created.id), opened: true };
}
throw new Error(`No open tab matches ${pattern}, and no URL is configured to open`);
}
const [tab] = await chrome.tabs.query({ active: true, lastFocusedWindow: true });
if (!tab || !/^https?:/.test(tab.url || '')) throw new Error('No eligible active tab');
return { tab, opened: false };
}
/** Returns true if a request was handled, false if the queue was empty. */
async function serveLocateRequest(cfg) {
const claim = await fetch(`${cfg.apiBase}/api/autobuyer/locate/claim`, { method: 'POST' });
if (!claim.ok) throw new Error(`Claim failed: HTTP ${claim.status}`);
const { request } = await claim.json();
if (!request) return false;
let result = null;
let error = null;
try {
let { tab, opened } = await resolveLocateTab(cfg, request);
// A navigate step points the tab at a specific page first. Skip it when we
// are already there — reloading would throw away page state for nothing,
// and the common case is that the tab is on the right page already.
if (request.navigateUrl && !alreadyAt(tab.url, request.navigateUrl)) {
await chrome.tabs.update(tab.id, { url: request.navigateUrl });
await waitForTabLoad(tab.id);
tab = await chrome.tabs.get(tab.id);
opened = true; // treat as a cold load: the app still has to mount
}
// The reported coordinates are only worth anything if that tab is the one
// actually visible: raise its window and bring the tab to the front.
// Only un-minimise — passing state:'normal' unconditionally would drop a
// maximised or fullscreen window out of that state on every single click.
const win = await chrome.windows.get(tab.windowId);
const focus = win.state === 'minimized'
? { focused: true, state: 'normal' }
: { focused: true };
await chrome.windows.update(tab.windowId, focus);
await chrome.tabs.update(tab.id, { active: true });
await new Promise((r) => setTimeout(r, 250)); // let the OS finish raising it
// `complete` only means the document loaded — a React app still has to
// mount and paint. Retry briefly rather than declaring the element missing,
// with a longer budget when we just opened the page from cold.
const deadline = Date.now() + (opened ? 8000 : 2500);
let out;
for (;;) {
const [injection] = await chrome.scripting.executeScript({
target: { tabId: tab.id },
func: pageLocate,
args: [request.selector, request.index || 0],
});
out = injection?.result;
if (!out) throw new Error('Injection returned nothing');
if (!out.error) break;
if (Date.now() >= deadline) throw new Error(out.error);
await new Promise((r) => setTimeout(r, 350));
}
result = { ...out, tabId: tab.id, windowId: tab.windowId, openedTab: opened };
} catch (err) {
error = err.message;
}
await fetch(`${cfg.apiBase}/api/autobuyer/locate/result`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ id: request.id, result, error }),
});
return true;
}
async function tick() {
const cfg = await getConfig();
let enabled;
try {
const res = await fetch(`${cfg.apiBase}/api/autobuyer/status`, { cache: 'no-store' });
if (!res.ok) throw new Error(`HTTP ${res.status}`);
const status = await res.json();
enabled = !!status.enabled;
pendingLocate = Number(status.pendingLocate) || 0;
} catch (err) {
setBadge(false, false);
await setState({ connected: false, enabled: false, error: `Cannot reach ${cfg.apiBase}${err.message}` });
return;
}
setBadge(true, enabled);
if (!enabled) {
await setState({ connected: true, enabled: false, error: null, waiting: null });
return;
}
// Serve queued clicks before capturing — the Python side is waiting on these,
// and a capture of a megabyte page shouldn't sit in front of them.
// Failures inside a request are reported back to the caller; a failure of the
// queue plumbing itself lands here, and must not take the capture loop down.
try {
for (let i = 0; i < pendingLocate; i++) {
if (!(await serveLocateRequest(cfg))) break;
}
} catch (err) {
await setState({ connected: true, enabled: true, error: `Locate queue: ${err.message}` });
}
try {
await captureAndSend(cfg);
} catch (err) {
await setState({ connected: true, enabled: true, error: err.message });
}
}
// ── Scheduling ──────────────────────────────────────────────────────────────
async function restart() {
if (timer) clearInterval(timer);
const cfg = await getConfig();
// Re-check after the await: restart() is called from several places at startup,
// and without this a concurrent call leaks an untracked interval — two polling
// loops in one worker, capturing everything twice.
if (timer) clearInterval(timer);
const ms = Math.max(1, Number(cfg.pollSeconds) || DEFAULTS.pollSeconds) * 1000;
timer = setInterval(tick, ms);
tick();
}
chrome.runtime.onInstalled.addListener(restart);
chrome.runtime.onStartup.addListener(restart);
// The setInterval above only survives while the worker is alive. Each tick makes
// an extension API call, which resets the idle timer, so in practice the loop
// keeps itself running — this alarm is the safety net that revives it if Chrome
// tears the worker down anyway. 30s is the shortest period Chrome honours.
chrome.alarms.create('keepalive', { periodInMinutes: 0.5 });
chrome.alarms.onAlarm.addListener(() => {
if (timer) tick(); else restart();
});
chrome.storage.onChanged.addListener((changes, area) => {
if (area !== 'local') return;
if (Object.keys(DEFAULTS).some((k) => k in changes)) restart();
});
chrome.runtime.onMessage.addListener((msg, _sender, sendResponse) => {
if (msg?.type !== 'capture-now') return;
getConfig()
.then(captureAndSend)
.then(() => sendResponse({ ok: true }))
.catch((err) => sendResponse({ ok: false, error: err.message }));
return true; // keep the message channel open for the async response
});
restart();