""" run_session.py One command InsightUX session. Instead of (start bridge, open Chrome, paste JS, go fullscreen, run tracker), this hosts the website itself in a pywebview window, injects the AOI capture plus a live gaze overlay automatically, and logs everything from Python. Setup once: pip install pywebview Run: python run_session.py https://www.yoursite.com (or just: python run_session.py and it will prompt for a url) Prerequisite: calibration.pkl must exist. If it does not, run calibrate.py first. Writes: sessions/live/gaze_log.jsonl raw gaze, every frame (for fixation detection) sessions/live/dom_log.jsonl page elements + scroll + url, a few times a second Smoothing: The on screen dot is smoothed with a One Euro filter (heavy smoothing when your eyes are still, light when they move fast). The LOGGED gaze stays raw, because Phase 3 fixation detection wants the unsmoothed signal. """ import os import sys import json import time import math import threading from dataclasses import replace import cv2 import numpy as np import webview import pyautogui from preprocessing.preprocessing_pipeline import ( create_face_mesh, estimate_camera_matrix, estimate_head_pose, compute_iris_radius, step1_normalize, step2_illumination, LEFT_EYE_INDICES, LEFT_EAR_INDICES, LEFT_IRIS_INDICES, RIGHT_EYE_INDICES, RIGHT_EAR_INDICES, RIGHT_IRIS_INDICES, ) from inference_pipeline import InsightUXPipeline from session_logger import GazeLogger # ============================================================================= # CONFIG # ============================================================================= ONNX_PATH = "models/gaze_cnn_v4.onnx" CALIBRATION_PATH = "calibration.pkl" SCREEN_W, SCREEN_H = pyautogui.size() # Which eye patch to feed the CNN. MUST match calibrate.py, and recalibrate after changing. # Settled from training history: the model was trained on the Step 2 "blended" patch # (training ran apply_step2 on every MPIIGaze sample so train and inference match). # So "blended" is correct. "norm_crop" is the mismatch that tracked badly. PATCH_SOURCE = "blended" POSE_SMOOTH = 0.65 # head pose EMA (0 = none, higher = smoother but laggier) MAX_JUMP = 220 # px, clamp wild teleport spikes NO_FACE_RESET = 15 # frames with no face before recentering DOM_EVERY = 6 # read page AOIs every N frames (about 5 per second) # One Euro filter feel. Lower mincutoff = smoother when still. # Lower beta = smoother but a touch laggier when the eyes move fast. EURO_MINCUTOFF = 0.35 EURO_BETA = 0.12 # FIX A / FIX B — MUST match calibrate.py and main_webcam_pipeline.py exactly. # calibration.pkl was fit with pose scaled by /30 going into the model, and # with the RBF trained on head-pitch-compensated pitch. Skipping either of # these here would feed the calibrated model/RBF inputs it was never built # for, even though the model and calibration themselves are now correct. POSE_NORM_SCALE = 30.0 HEAD_PITCH_COMPENSATION = 0.0 def normalize_pose(pitch_deg, yaw_deg, roll_deg): return np.array([ pitch_deg / POSE_NORM_SCALE, yaw_deg / POSE_NORM_SCALE, roll_deg / POSE_NORM_SCALE, ], dtype=np.float32) def compensate_pitch(raw_pitch, head_pitch_deg): return raw_pitch - np.radians(head_pitch_deg) * HEAD_PITCH_COMPENSATION # FIX C — HEAD-YAW COMPENSATION (must match calibrate.py exactly) HEAD_YAW_COMPENSATION = 0.0 # disabled def compensate_yaw(raw_yaw, head_yaw_deg): return raw_yaw - np.radians(head_yaw_deg) * HEAD_YAW_COMPENSATION SESSION_DIR = os.path.join("sessions", "live") os.makedirs(SESSION_DIR, exist_ok=True) DOM_PATH = os.path.join(SESSION_DIR, "dom_log.jsonl") # ============================================================================= # ONE EURO FILTER (display smoothing only) # ============================================================================= class OneEuroFilter: def __init__(self, mincutoff=0.8, beta=0.4, dcutoff=1.0): self.mincutoff = mincutoff self.beta = beta self.dcutoff = dcutoff self.x_prev = None self.dx_prev = 0.0 self.t_prev = None @staticmethod def _alpha(cutoff, dt): tau = 1.0 / (2 * math.pi * cutoff) return 1.0 / (1.0 + tau / dt) def __call__(self, x, t): if self.x_prev is None: self.x_prev, self.t_prev = x, t return x dt = t - self.t_prev if dt <= 0: dt = 1e-3 self.t_prev = t dx = (x - self.x_prev) / dt a_d = self._alpha(self.dcutoff, dt) dx_hat = a_d * dx + (1 - a_d) * self.dx_prev cutoff = self.mincutoff + self.beta * abs(dx_hat) a = self._alpha(cutoff, dt) x_hat = a * x + (1 - a) * self.x_prev self.x_prev, self.dx_prev = x_hat, dx_hat return x_hat def reset(self): self.x_prev = None self.dx_prev = 0.0 self.t_prev = None # ============================================================================= # JS injected into the page: AOI reader + gaze overlay # ============================================================================= JS_SETUP = r""" (function(){ if (window.__insightux) { return; } const state = { aois: [] }; const dwell = { pendLabel: null, pendSince: 0, activeLabel: null, emptySince: 0 }; const DWELL_MS = 420; // gaze must rest in a box this long before it lights up const RELEASE_MS = 1200; // clear the highlight after gaze leaves everything this long // Render loop state: detection (above) runs at Python's update rate // (webcam fps). Drawing runs separately at 60fps via requestAnimationFrame // and eases the dot toward the latest target - this is what makes the // dot look fluid instead of choppy, independent of how often Python // actually sends a new sample. let activeBox = null; const target = { fx: 0.5, fy: 0.5 }; const dot = { x: null, y: null }; const DOT_LERP = 0.07; // lower = smoother/laggier, higher = snappier/twitchier const cv = document.createElement('canvas'); cv.id = '__insightux_canvas'; cv.style.cssText = 'position:fixed;left:0;top:0;width:100vw;height:100vh;pointer-events:none;z-index:2147483647;'; (document.body || document.documentElement).appendChild(cv); const ctx = cv.getContext('2d'); function resize(){ cv.width = window.innerWidth; cv.height = window.innerHeight; } resize(); window.addEventListener('resize', resize, {passive:true}); // Some pages re-render
contents in ways that can detach the canvas. // A periodic check re-attaches it instead of silently losing the overlay. setInterval(function(){ if (!document.body.contains(cv)) document.body.appendChild(cv); }, 1000); const MIN_W=40, MIN_H=24, MAX_AOIS=90, MAX_SCAN=2500; // PAD_PX: forgiveness margin around each AOI when testing for a hit. // validate.py measured ~110-150px average gaze error, well larger than // many real elements (navbar height, thumbnails). Without this, the // strict point-in-rect test misses those elements almost every time // even when the tracker is genuinely pointed at them. The highlight // itself is still drawn at the element's TRUE bounds, only the hit // test is padded, so the UI doesn't look inflated. const PAD_PX = 90; // TALL_WRAPPER_H: a candidate taller than this that also CONTAINS another // matched element is dropped entirely. Without this, a full-page-height // wrapping tags. A selector
// limited to semantic tags finds almost nothing on those pages, leaving
// only a giant outer wrapper to match - which is exactly the "stuck on
// a few elements" symptom.
//
// ALWAYS: meaningful by tag semantics alone, regardless of whether they
// carry their own text (chrome regions, images, headings, buttons).
const ALWAYS_SELECTOR = "nav, header, footer, img, video, iframe, h1, h2, h3, button, figure, [class*='hero'], [class*='banner'], [class*='card']";
// TEXT_CONTAINER: only eligible if the element has its OWN direct text
// node content (not just text inherited from nested children) - this is
// what lets us safely include generic div/span/section without matching
// every layout wrapper on the page.
const TEXT_CONTAINER_SELECTOR = "div, span, section, article, main, aside, p, li";
function hasOwnText(el){
for (const node of el.childNodes){
if (node.nodeType === 3 && node.textContent.trim().length > 2) return true;
}
return false;
}
function labelFor(el){
if (el.dataset && el.dataset.aoi) return el.dataset.aoi.slice(0,40);
const tag = el.tagName.toLowerCase();
if (tag==='nav') return 'navbar';
if (tag==='header') return 'header';
if (tag==='footer') return 'footer';
if (tag==='img'){
const alt=(el.getAttribute('alt')||'').trim();
if (alt) return 'img: '+alt.slice(0,30);
const src=el.getAttribute('src')||'';
const name=src.split('/').pop().split('?')[0];
return 'img: '+(name||'image').slice(0,30);
}
if (tag==='video') return 'video';
if (tag==='iframe') return 'embed';
if (tag==='h1'||tag==='h2'){
const t=(el.innerText||'').trim().replace(/\s+/g,' ');
if (t) return tag+': '+t.slice(0,30);
}
const id=el.id?('#'+el.id):'';
let cls='';
if (el.className && typeof el.className==='string'){
const f=el.className.trim().split(/\s+/)[0];
if (f) cls='.'+f;
}
const txt=(el.innerText||'').trim().replace(/\s+/g,' ').slice(0,24);
const base=id||cls||tag;
return txt?(base+' ('+txt+')'):base;
}
function refresh(){
const vh = window.innerHeight;
const raw = [];
const seen = new Set();
let full = false;
function consider(el){
if (full) return;
const tag = el.tagName.toLowerCase();
const explicit = !!(el.dataset && el.dataset.aoi);
if (!explicit){
const isAlways = (tag==='nav'||tag==='header'||tag==='footer'||tag==='img'||
tag==='video'||tag==='iframe'||
tag==='h1'||tag==='h2'||tag==='h3'||tag==='button'||tag==='figure') ||
(el.className && typeof el.className==='string' &&
/hero|banner|card/i.test(el.className));
// generic containers (div/span/section/etc) only qualify if they
// carry their own direct text - skip pure layout wrappers without
// paying the cost of a layout-forcing getBoundingClientRect() call
if (!isAlways && !hasOwnText(el)) return;
}
const r = el.getBoundingClientRect();
if (r.width