""" run_session.py One command InsightUX session. Instead of (start bridge, open Chrome, paste JS, go fullscreen, run tracker), this hosts the website itself in a pywebview window, injects the AOI capture plus a live gaze overlay automatically, and logs everything from Python. Setup once: pip install pywebview Run: python run_session.py https://www.yoursite.com (or just: python run_session.py and it will prompt for a url) Prerequisite: calibration.pkl must exist. If it does not, run calibrate.py first. Writes: sessions/live/gaze_log.jsonl raw gaze, every frame (for fixation detection) sessions/live/dom_log.jsonl page elements + scroll + url, a few times a second Smoothing: The on screen dot is smoothed with a One Euro filter (heavy smoothing when your eyes are still, light when they move fast). The LOGGED gaze stays raw, because Phase 3 fixation detection wants the unsmoothed signal. """ import os import sys import json import time import math import threading from dataclasses import replace import cv2 import numpy as np import webview import pyautogui from preprocessing.preprocessing_pipeline import ( create_face_mesh, estimate_camera_matrix, estimate_head_pose, compute_iris_radius, step1_normalize, step2_illumination, LEFT_EYE_INDICES, LEFT_EAR_INDICES, LEFT_IRIS_INDICES, RIGHT_EYE_INDICES, RIGHT_EAR_INDICES, RIGHT_IRIS_INDICES, ) from inference_pipeline import InsightUXPipeline from session_logger import GazeLogger # ============================================================================= # CONFIG # ============================================================================= ONNX_PATH = "models/gaze_cnn_v4.onnx" CALIBRATION_PATH = "calibration.pkl" SCREEN_W, SCREEN_H = pyautogui.size() # Which eye patch to feed the CNN. MUST match calibrate.py, and recalibrate after changing. # Settled from training history: the model was trained on the Step 2 "blended" patch # (training ran apply_step2 on every MPIIGaze sample so train and inference match). # So "blended" is correct. "norm_crop" is the mismatch that tracked badly. PATCH_SOURCE = "blended" POSE_SMOOTH = 0.65 # head pose EMA (0 = none, higher = smoother but laggier) MAX_JUMP = 220 # px, clamp wild teleport spikes NO_FACE_RESET = 15 # frames with no face before recentering DOM_EVERY = 6 # read page AOIs every N frames (about 5 per second) # One Euro filter feel. Lower mincutoff = smoother when still. # Lower beta = smoother but a touch laggier when the eyes move fast. EURO_MINCUTOFF = 0.35 EURO_BETA = 0.12 # FIX A / FIX B — MUST match calibrate.py and main_webcam_pipeline.py exactly. # calibration.pkl was fit with pose scaled by /30 going into the model, and # with the RBF trained on head-pitch-compensated pitch. Skipping either of # these here would feed the calibrated model/RBF inputs it was never built # for, even though the model and calibration themselves are now correct. POSE_NORM_SCALE = 30.0 HEAD_PITCH_COMPENSATION = 0.0 def normalize_pose(pitch_deg, yaw_deg, roll_deg): return np.array([ pitch_deg / POSE_NORM_SCALE, yaw_deg / POSE_NORM_SCALE, roll_deg / POSE_NORM_SCALE, ], dtype=np.float32) def compensate_pitch(raw_pitch, head_pitch_deg): return raw_pitch - np.radians(head_pitch_deg) * HEAD_PITCH_COMPENSATION # FIX C — HEAD-YAW COMPENSATION (must match calibrate.py exactly) HEAD_YAW_COMPENSATION = 0.0 # disabled def compensate_yaw(raw_yaw, head_yaw_deg): return raw_yaw - np.radians(head_yaw_deg) * HEAD_YAW_COMPENSATION SESSION_DIR = os.path.join("sessions", "live") os.makedirs(SESSION_DIR, exist_ok=True) DOM_PATH = os.path.join(SESSION_DIR, "dom_log.jsonl") # ============================================================================= # ONE EURO FILTER (display smoothing only) # ============================================================================= class OneEuroFilter: def __init__(self, mincutoff=0.8, beta=0.4, dcutoff=1.0): self.mincutoff = mincutoff self.beta = beta self.dcutoff = dcutoff self.x_prev = None self.dx_prev = 0.0 self.t_prev = None @staticmethod def _alpha(cutoff, dt): tau = 1.0 / (2 * math.pi * cutoff) return 1.0 / (1.0 + tau / dt) def __call__(self, x, t): if self.x_prev is None: self.x_prev, self.t_prev = x, t return x dt = t - self.t_prev if dt <= 0: dt = 1e-3 self.t_prev = t dx = (x - self.x_prev) / dt a_d = self._alpha(self.dcutoff, dt) dx_hat = a_d * dx + (1 - a_d) * self.dx_prev cutoff = self.mincutoff + self.beta * abs(dx_hat) a = self._alpha(cutoff, dt) x_hat = a * x + (1 - a) * self.x_prev self.x_prev, self.dx_prev = x_hat, dx_hat return x_hat def reset(self): self.x_prev = None self.dx_prev = 0.0 self.t_prev = None # ============================================================================= # JS injected into the page: AOI reader + gaze overlay # ============================================================================= JS_SETUP = r""" (function(){ if (window.__insightux) { return; } const state = { aois: [] }; const dwell = { pendLabel: null, pendSince: 0, activeLabel: null, emptySince: 0 }; const DWELL_MS = 420; // gaze must rest in a box this long before it lights up const RELEASE_MS = 1200; // clear the highlight after gaze leaves everything this long // Render loop state: detection (above) runs at Python's update rate // (webcam fps). Drawing runs separately at 60fps via requestAnimationFrame // and eases the dot toward the latest target - this is what makes the // dot look fluid instead of choppy, independent of how often Python // actually sends a new sample. let activeBox = null; const target = { fx: 0.5, fy: 0.5 }; const dot = { x: null, y: null }; const DOT_LERP = 0.07; // lower = smoother/laggier, higher = snappier/twitchier const cv = document.createElement('canvas'); cv.id = '__insightux_canvas'; cv.style.cssText = 'position:fixed;left:0;top:0;width:100vw;height:100vh;pointer-events:none;z-index:2147483647;'; (document.body || document.documentElement).appendChild(cv); const ctx = cv.getContext('2d'); function resize(){ cv.width = window.innerWidth; cv.height = window.innerHeight; } resize(); window.addEventListener('resize', resize, {passive:true}); // Some pages re-render contents in ways that can detach the canvas. // A periodic check re-attaches it instead of silently losing the overlay. setInterval(function(){ if (!document.body.contains(cv)) document.body.appendChild(cv); }, 1000); const MIN_W=40, MIN_H=24, MAX_AOIS=90, MAX_SCAN=2500; // PAD_PX: forgiveness margin around each AOI when testing for a hit. // validate.py measured ~110-150px average gaze error, well larger than // many real elements (navbar height, thumbnails). Without this, the // strict point-in-rect test misses those elements almost every time // even when the tracker is genuinely pointed at them. The highlight // itself is still drawn at the element's TRUE bounds, only the hit // test is padded, so the UI doesn't look inflated. const PAD_PX = 90; // TALL_WRAPPER_H: a candidate taller than this that also CONTAINS another // matched element is dropped entirely. Without this, a full-page-height // wrapping
/
becomes a permanent fallback target whenever // the gaze sits in a gap between paragraphs - it always "contains" the // point, so it wins by elimination even though it's a useless answer. // Compact chrome (navbar/header/footer) stays well under this, so it's // unaffected even if it happens to wrap a logo image. const TALL_WRAPPER_H = 220; // Two tiers of candidate, because real pages (especially page-builder // sites like WPBakery/Visual Composer/Elementor) wrap body text in plain //
s with manual line breaks instead of real

tags. A selector // limited to semantic tags finds almost nothing on those pages, leaving // only a giant outer wrapper to match - which is exactly the "stuck on // a few elements" symptom. // // ALWAYS: meaningful by tag semantics alone, regardless of whether they // carry their own text (chrome regions, images, headings, buttons). const ALWAYS_SELECTOR = "nav, header, footer, img, video, iframe, h1, h2, h3, button, figure, [class*='hero'], [class*='banner'], [class*='card']"; // TEXT_CONTAINER: only eligible if the element has its OWN direct text // node content (not just text inherited from nested children) - this is // what lets us safely include generic div/span/section without matching // every layout wrapper on the page. const TEXT_CONTAINER_SELECTOR = "div, span, section, article, main, aside, p, li"; function hasOwnText(el){ for (const node of el.childNodes){ if (node.nodeType === 3 && node.textContent.trim().length > 2) return true; } return false; } function labelFor(el){ if (el.dataset && el.dataset.aoi) return el.dataset.aoi.slice(0,40); const tag = el.tagName.toLowerCase(); if (tag==='nav') return 'navbar'; if (tag==='header') return 'header'; if (tag==='footer') return 'footer'; if (tag==='img'){ const alt=(el.getAttribute('alt')||'').trim(); if (alt) return 'img: '+alt.slice(0,30); const src=el.getAttribute('src')||''; const name=src.split('/').pop().split('?')[0]; return 'img: '+(name||'image').slice(0,30); } if (tag==='video') return 'video'; if (tag==='iframe') return 'embed'; if (tag==='h1'||tag==='h2'){ const t=(el.innerText||'').trim().replace(/\s+/g,' '); if (t) return tag+': '+t.slice(0,30); } const id=el.id?('#'+el.id):''; let cls=''; if (el.className && typeof el.className==='string'){ const f=el.className.trim().split(/\s+/)[0]; if (f) cls='.'+f; } const txt=(el.innerText||'').trim().replace(/\s+/g,' ').slice(0,24); const base=id||cls||tag; return txt?(base+' ('+txt+')'):base; } function refresh(){ const vh = window.innerHeight; const raw = []; const seen = new Set(); let full = false; function consider(el){ if (full) return; const tag = el.tagName.toLowerCase(); const explicit = !!(el.dataset && el.dataset.aoi); if (!explicit){ const isAlways = (tag==='nav'||tag==='header'||tag==='footer'||tag==='img'|| tag==='video'||tag==='iframe'|| tag==='h1'||tag==='h2'||tag==='h3'||tag==='button'||tag==='figure') || (el.className && typeof el.className==='string' && /hero|banner|card/i.test(el.className)); // generic containers (div/span/section/etc) only qualify if they // carry their own direct text - skip pure layout wrappers without // paying the cost of a layout-forcing getBoundingClientRect() call if (!isAlways && !hasOwnText(el)) return; } const r = el.getBoundingClientRect(); if (r.widthvh) return; const label = labelFor(el); if (!label) return; const key = label+'@'+Math.round(r.left)+','+Math.round(r.top); if (seen.has(key)) return; seen.add(key); const pos = getComputedStyle(el).position; raw.push({el:el, label:label, x:Math.round(r.left), y:Math.round(r.top), w:Math.round(r.width), h:Math.round(r.height), sticky:(pos==='sticky'||pos==='fixed')}); if (raw.length >= MAX_AOIS*2) full = true; } document.querySelectorAll('[data-aoi]').forEach(consider); document.querySelectorAll(ALWAYS_SELECTOR).forEach(consider); const candidates = document.querySelectorAll(TEXT_CONTAINER_SELECTOR); for (let i=0; i { const isTallWrapper = o.h > TALL_WRAPPER_H && raw.some(o2 => o2 !== o && o.el.contains(o2.el)); return !isTallWrapper; }); state.aois = filtered.slice(0, MAX_AOIS).map(o => ({ label:o.label, x:o.x, y:o.y, w:o.w, h:o.h, sticky:o.sticky })); } refresh(); window.addEventListener('scroll', refresh, {passive:true}); setInterval(refresh, 400); window.insightuxAOIs = function(){ return JSON.stringify({ url: location.href, scrollX: Math.round(window.scrollX), scrollY: Math.round(window.scrollY), viewport:{w:window.innerWidth,h:window.innerHeight}, page:{w:document.documentElement.scrollWidth,h:document.documentElement.scrollHeight}, aois: state.aois }); }; window.insightuxUpdate = function(fx, fy){ // Only sets WHERE the dot is heading. The actual hit-test now runs in // renderLoop, against the dot's already-eased position - see below for // why that's the fix for box/dot desync. target.fx = fx; target.fy = fy; }; // Render loop: runs at display refresh rate (~60fps), independent of how // often Python actually calls insightuxUpdate (limited by webcam fps). // The dot eases toward the latest target every frame, which is what makes // it look smooth instead of jumping in steps. // // IMPORTANT: the AOI hit-test and dwell logic run HERE too, against the // dot's EASED position (dot.x/dot.y), not the raw incoming target. They // used to run in insightuxUpdate against the raw target, one step ahead // of where the dot visually was - that's exactly why the box and the dot // could disagree (box reacting to a position the dot hadn't caught up to // yet). Now both are driven by the exact same number every frame, so // they can't visually contradict each other. function renderLoop(){ const w = window.innerWidth, h = window.innerHeight; const tx = target.fx * w, ty = target.fy * h; if (dot.x === null){ dot.x = tx; dot.y = ty; } dot.x += (tx - dot.x) * DOT_LERP; dot.y += (ty - dot.y) * DOT_LERP; const now = performance.now(); const px = dot.x, py = dot.y; // smallest box currently under the EASED dot position (padded by PAD_PX) let cand = null; for (const a of state.aois){ if (px>=a.x-PAD_PX && px<=a.x+a.w+PAD_PX && py>=a.y-PAD_PX && py<=a.y+a.h+PAD_PX){ if (!cand || (a.w*a.h)<(cand.w*cand.h)) cand = a; } } // dwell + hysteresis: only commit to a box the gaze has rested on const candLabel = cand ? cand.label : null; if (candLabel !== dwell.pendLabel){ dwell.pendLabel = candLabel; dwell.pendSince = now; } if (cand){ dwell.emptySince = 0; if (dwell.activeLabel !== cand.label && (now - dwell.pendSince) >= DWELL_MS){ dwell.activeLabel = cand.label; } } else { if (dwell.emptySince === 0) dwell.emptySince = now; if (dwell.activeLabel && (now - dwell.emptySince) >= RELEASE_MS){ dwell.activeLabel = null; } } let active = null; if (dwell.activeLabel){ for (const a of state.aois){ if (a.label === dwell.activeLabel){ active = a; break; } } if (!active) dwell.activeLabel = null; } activeBox = active; ctx.clearRect(0,0,cv.width,cv.height); if (activeBox){ ctx.fillStyle = 'rgba(123,47,190,0.16)'; ctx.fillRect(activeBox.x, activeBox.y, activeBox.w, activeBox.h); ctx.strokeStyle = '#7B2FBE'; ctx.lineWidth = 3; ctx.strokeRect(activeBox.x, activeBox.y, activeBox.w, activeBox.h); const label = activeBox.label; ctx.font = 'bold 13px Arial'; const tw = ctx.measureText(label).width + 16; const ly = Math.max(0, activeBox.y - 22); ctx.fillStyle = '#7B2FBE'; ctx.fillRect(activeBox.x, ly, tw, 20); ctx.fillStyle = '#FFFFFF'; ctx.fillText(label, activeBox.x + 8, ly + 14); } // smooth gaze dot, drawn last so it sits on top of the highlight. // Bigger + white halo underneath, so it stays visible against any // background color or busy content. ctx.beginPath(); ctx.arc(dot.x, dot.y, 13, 0, 2*Math.PI); ctx.fillStyle = 'rgba(255,255,255,0.9)'; ctx.fill(); ctx.beginPath(); ctx.arc(dot.x, dot.y, 9, 0, 2*Math.PI); ctx.fillStyle = '#FF2DF0'; ctx.fill(); ctx.lineWidth = 2; ctx.strokeStyle = 'rgba(0,0,0,0.55)'; ctx.stroke(); requestAnimationFrame(renderLoop); } requestAnimationFrame(renderLoop); document.addEventListener('keydown', function(e){ if (e.key==='Escape' && window.pywebview && window.pywebview.api){ window.pywebview.api.stop(); } }); window.__insightux = true; })(); """ # ============================================================================= # Control object exposed to JS (so Escape can stop the session) # ============================================================================= class Api: def __init__(self): self.running = True def stop(self): self.running = False return True api = Api() # ============================================================================= # GAZE WORKER (runs in a pywebview background thread) # ============================================================================= def inject(window): """(Re)inject the overlay + AOI script. Called on every page load.""" try: window.evaluate_js(JS_SETUP) except Exception as e: print(f"[run_session] inject failed (will retry): {e}") def gaze_worker(window): if not os.path.exists(CALIBRATION_PATH): print(f"[run_session] No {CALIBRATION_PATH}. Run calibrate.py first.") api.running = False window.destroy() return pipeline = InsightUXPipeline(ONNX_PATH, CALIBRATION_PATH) face_mesh = create_face_mesh(static_image_mode=False) cap = cv2.VideoCapture(0) logger = GazeLogger(SCREEN_W, SCREEN_H, session_dir=SESSION_DIR) dom_f = open(DOM_PATH, "w", buffering=1) # re-inject on each navigation, and inject the first page now window.events.loaded += lambda: inject(window) for _ in range(10): inject(window) try: if window.evaluate_js("window.__insightux === true"): break except Exception: pass time.sleep(0.4) print("[run_session] tracking. Look at the site. Press Esc to stop.") cam_matrix = None last_x = SCREEN_W / 2 last_y = SCREEN_H / 2 fil_x = OneEuroFilter(mincutoff=EURO_MINCUTOFF, beta=EURO_BETA) fil_y = OneEuroFilter(mincutoff=EURO_MINCUTOFF, beta=EURO_BETA) sm_pitch = sm_yaw = sm_roll = None no_face_count = 0 frame_i = 0 while api.running: ret, frame = cap.read() if not ret: break if cam_matrix is None: cam_matrix = estimate_camera_matrix(frame.shape) rgb = cv2.cvtColor(frame, cv2.COLOR_BGR2RGB) results = face_mesh.process(rgb) if not results.multi_face_landmarks: no_face_count += 1 if no_face_count >= NO_FACE_RESET: last_x, last_y = SCREEN_W / 2, SCREEN_H / 2 fil_x.reset(); fil_y.reset() sm_pitch = sm_yaw = sm_roll = None no_face_count = 0 continue no_face_count = 0 lms = results.multi_face_landmarks[0].landmark head_pose = estimate_head_pose(lms, frame.shape, cam_matrix) if head_pose is None: continue # smooth the head pose: cuts solvePnP jitter feeding both the warp and the model if sm_pitch is None: sm_pitch, sm_yaw, sm_roll = head_pose.pitch, head_pose.yaw, head_pose.roll else: b = POSE_SMOOTH sm_pitch = b*sm_pitch + (1-b)*head_pose.pitch sm_yaw = b*sm_yaw + (1-b)*head_pose.yaw sm_roll = b*sm_roll + (1-b)*head_pose.roll head_pose = replace(head_pose, pitch=sm_pitch, yaw=sm_yaw, roll=sm_roll) pose_vec = normalize_pose(sm_pitch, sm_yaw, sm_roll) # FIX A def get_patch(eye_idx, ear_idx, iris_idx): s1 = step1_normalize(frame, lms, head_pose, eye_idx, ear_idx, iris_idx) if not s1.is_open: return None if PATCH_SOURCE == "norm": return s1.norm_crop ir = compute_iris_radius(lms, iris_idx, frame.shape) s2 = step2_illumination(s1, ir) return s2.blended if s2.is_usable else None left_patch = get_patch(LEFT_EYE_INDICES, LEFT_EAR_INDICES, LEFT_IRIS_INDICES) right_patch = get_patch(RIGHT_EYE_INDICES, RIGHT_EAR_INDICES, RIGHT_IRIS_INDICES) if left_patch is None and right_patch is None: continue if left_patch is None: left_patch = right_patch if right_patch is None: right_patch = left_patch _, _, raw_pitch, raw_yaw = pipeline.predict_gaze_vector(left_patch, pose_vec, right_patch) pitch = compensate_pitch(raw_pitch, sm_pitch) # FIX B yaw = compensate_yaw(raw_yaw, sm_yaw) # FIX C sx, sy = pipeline.calibration.predict(pitch, yaw) sx = max(0.0, min(sx, SCREEN_W)) sy = max(0.0, min(sy, SCREEN_H)) # guard against wild teleport spikes jump = np.hypot(sx - last_x, sy - last_y) if jump > MAX_JUMP: sx = last_x + (sx - last_x) * 0.3 sy = last_y + (sy - last_y) * 0.3 last_x, last_y = sx, sy # log the RAW gaze (Phase 3 fixation detection wants it unsmoothed) logger.log(sx, sy) # smooth only the DISPLAYED dot with a One Euro filter now = time.time() fx = fil_x(sx / SCREEN_W, now) fy = fil_y(sy / SCREEN_H, now) try: window.evaluate_js(f"window.insightuxUpdate({fx:.5f},{fy:.5f})") except Exception: pass # log the page state a few times a second frame_i += 1 if frame_i % DOM_EVERY == 0: try: raw = window.evaluate_js("window.insightuxAOIs()") if raw: rec = json.loads(raw) rec["type"] = "dom" rec["t"] = round(time.time(), 4) dom_f.write(json.dumps(rec) + "\n") except Exception: pass time.sleep(0.005) cap.release() logger.close() dom_f.close() print(f"[run_session] session saved in {SESSION_DIR}") try: window.destroy() except Exception: pass # ============================================================================= # MAIN # ============================================================================= if __name__ == "__main__": url = sys.argv[1] if len(sys.argv) > 1 else input("Website url to test: ").strip() if not url.startswith("http"): url = "https://" + url window = webview.create_window("InsightUX", url, fullscreen=True, js_api=api) webview.start(gaze_worker, window)