// Web research — search the web and fetch page content. // Pure Node, no external dependencies, multi-OS. const UA = 'Mozilla/5.0 (Macintosh; Intel Mac OS X 14_8) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/17.6 Safari/605.1.15 mona-agent/2.x'; const FETCH_TIMEOUT_MS = 15000; const MAX_TEXT = 10000; async function httpGet(url, headers = {}) { const res = await fetch(url, { headers: { 'user-agent': UA, accept: 'text/html,application/xhtml+xml', ...headers }, signal: AbortSignal.timeout(FETCH_TIMEOUT_MS), }); if (!res.ok) throw new Error(`HTTP ${res.status} for ${url}`); return res.text(); } /** Crude but effective HTML → text extraction (no external deps). */ export function htmlToText(html) { let t = String(html || ''); t = t.replace(//gi, ' ') .replace(//gi, ' ') .replace(//gi, ' '); t = t.replace(/<[^>]+>/g, ' '); t = t.replace(/ /gi, ' ') .replace(/&/gi, '&') .replace(/</gi, '<') .replace(/>/gi, '>') .replace(/"/gi, '"') .replace(/'/gi, "'"); t = t.replace(/\s+/g, ' ').trim(); return t.slice(0, MAX_TEXT); } /** DuckDuckGo HTML search → [{title, url, snippet}] (no API key needed). */ export async function ddgSearch(query, max = 8) { const q = encodeURIComponent(String(query || '').slice(0, 200)); const html = await httpGet(`https://html.duckduckgo.com/html/?q=${q}`); const results = []; const itemRe = /]+class="result__a"[^>]+href="([^"]+)"[^>]*>([\s\S]*?)<\/a>[\s\S]*?]+class="result__snippet"[^>]*>([\s\S]*?)<\/a>/gi; let m; while ((m = itemRe.exec(html)) !== null && results.length < max) { const rawUrl = m[1]; let url = rawUrl; if (rawUrl.startsWith('//')) url = 'https:' + rawUrl; // DuckDuckGo redirect links: extract the real target const uddg = rawUrl.match(/uddg=([^&]+)/); if (uddg) { try { url = decodeURIComponent(uddg[1]); } catch { /* keep raw */ } } if (!url.startsWith('http')) continue; results.push({ title: htmlToText(m[2]).slice(0, 120), url, snippet: htmlToText(m[3]).slice(0, 300), }); } return results; } export const web = { name: 'web', description: 'Web research: search the web (DuckDuckGo) and fetch page content as text.', args: { action: 'string — search | fetch', query: 'string — search terms (for search)', url: 'string — full URL (for fetch)', max: 'number — max results (search, default 8)', }, platform: 'any', async run(args) { const action = String(args.action || 'search').toLowerCase(); if (action === 'search') { const query = String(args.query || '').trim(); if (!query) return { error: 'query required' }; const results = await ddgSearch(query, Math.min(parseInt(args.max || 8, 10) || 8, 10)); if (!results.length) return { results: [], note: 'No results returned.' }; return { results }; } if (action === 'fetch') { const url = String(args.url || '').trim(); if (!/^https?:\/\//i.test(url)) return { error: 'A valid http(s) URL is required' }; const html = await httpGet(url); const text = htmlToText(html); return { url, textLength: text.length, text: text.slice(0, MAX_TEXT) }; } return { error: `Unknown action: ${action} (use search | fetch)` }; }, };