Spaces:
Running
Running
Download src/api/iserv-parse.js from Luca448/APP-Backend: direct link, hf CLI and curl.
- Browser
- Download file 6.9 kB
-
https://huggingface.co/spaces/Luca448/APP-Backend/resolve/main/src/api/iserv-parse.js
- Command line
-
hf download hf://spaces/Luca448/APP-Backend/src/api/iserv-parse.js
-
curl -L -o iserv-parse.js https://huggingface.co/spaces/Luca448/APP-Backend/resolve/main/src/api/iserv-parse.js
6.9 kB
| // Parsers that run INSIDE the browser page: puppeteer's page.evaluate serializes a function, so every function here | |
| // is self-contained (its helpers sit inside it). They get the HTML as the server sent it and read it through | |
| // DOMParser. Page scripts do not run in that document and cannot change what is read: in the live DOM, Bootstrap | |
| // tooltips move the title attribute of the status icon away, so a done task would look open. | |
| // | |
| // Port of the parsers of the IServ filter project (iserv.py: parse_task_list, parse_description, parse_start), | |
| // which are tested against fake pages of the older and the current IServ layout. Keep both in step. | |
| /** | |
| * Task list: { "<task id>": { tags, start, due, done } } | |
| * tags: text of the "Tags" column ('' if empty), start: 'yyyy-mm-dd' (older layouts only), due: 'yyyy-mm-ddThh:mm', | |
| * done: the status icon says done/submitted. Columns are found by their header, never by position (the column | |
| * after the due date is the teacher's feedback). | |
| */ | |
| export function parseListHtml(html) { | |
| const doc = new DOMParser().parseFromString(html, 'text/html'); | |
| // BeautifulSoup's get_text(sep, strip=True): the stripped text nodes, joined | |
| const text = (el, sep = ' ') => { | |
| const parts = []; | |
| const walk = (node) => { | |
| for (const child of node.childNodes) { | |
| if (child.nodeType === 3) { | |
| const t = child.textContent.trim(); | |
| if (t) parts.push(t); | |
| } else if (child.nodeType === 1 && !['SCRIPT', 'STYLE'].includes(child.tagName)) { | |
| walk(child); | |
| } | |
| } | |
| }; | |
| walk(el); | |
| return parts.join(sep); | |
| }; | |
| const headerIndex = (row, names) => { | |
| const table = row.closest('table'); | |
| const head = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('th')) : null; | |
| const texts = head ? [...head.querySelectorAll('th, td')].map((c) => text(c).replace(/:+$/, '')) : []; | |
| return texts.findIndex((t) => names.includes(t)); | |
| }; | |
| const tagsOf = (cell) => { | |
| const all = text(cell); | |
| const italic = [...cell.querySelectorAll('i')].map((i) => text(i)).join(' '); | |
| return all === italic || ['keine', 'none'].includes(all.replace(/^[()]+|[()]+$/g, '').toLowerCase()) ? '' : all; | |
| }; | |
| const doneRe = /erledigt|abgegeben|done|submitted|completed/i; | |
| const out = {}; | |
| for (const a of doc.querySelectorAll('a[href*="/exercise/show/"]')) { | |
| const m = /\/exercise\/show\/(\d+)/.exec(a.getAttribute('href')); | |
| const row = a.closest('tr'); | |
| if (!m || !row || m[1] in out) continue; | |
| const cells = [...row.querySelectorAll('td')]; | |
| let start = null; | |
| let due = null; | |
| for (const td of cells) { | |
| const ds = td.getAttribute('data-sort') || ''; | |
| if (/^\d{8}$/.test(ds) && start === null) { | |
| start = `${ds.slice(0, 4)}-${ds.slice(4, 6)}-${ds.slice(6)}`; | |
| } else if (/^\d{14}$/.test(ds) && due === null) { | |
| due = `${ds.slice(0, 4)}-${ds.slice(4, 6)}-${ds.slice(6, 8)}T${ds.slice(8, 10)}:${ds.slice(10, 12)}`; | |
| } | |
| } | |
| const col = headerIndex(row, ['Tags']); | |
| const titles = [...row.querySelectorAll('[title]')].map((e) => e.getAttribute('title')).join(' '); | |
| out[m[1]] = { tags: col !== -1 && col < cells.length ? tagsOf(cells[col]) : '', start, due, done: doneRe.test(titles) }; | |
| } | |
| return out; | |
| } | |
| /** | |
| * Task page: { description, start } | |
| * description: only the task's own description (never own submission, teacher feedback, creator, participants), | |
| * as text with the paragraphs and line breaks of the page. start: 'yyyy-mm-dd[Thh:mm]' from the "Starttermin" column. | |
| */ | |
| export function parseTaskHtml(html) { | |
| const doc = new DOMParser().parseFromString(html, 'text/html'); | |
| const text = (el, sep = ' ') => { | |
| const parts = []; | |
| const walk = (node) => { | |
| for (const child of node.childNodes) { | |
| if (child.nodeType === 3) { | |
| const t = child.textContent.trim(); | |
| if (t) parts.push(t); | |
| } else if (child.nodeType === 1 && !['SCRIPT', 'STYLE'].includes(child.tagName)) { | |
| walk(child); | |
| } | |
| } | |
| }; | |
| walk(el); | |
| return parts.join(sep); | |
| }; | |
| const th = (names) => [...doc.querySelectorAll('th')].find((t) => names.includes(text(t).replace(/:+$/, ''))); | |
| const headerIndex = (row, names) => { | |
| const table = row.closest('table'); | |
| const head = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('th')) : null; | |
| const texts = head ? [...head.querySelectorAll('th, td')].map((c) => text(c).replace(/:+$/, '')) : []; | |
| return texts.findIndex((t) => names.includes(t)); | |
| }; | |
| const htmlText = (el) => { | |
| const copy = el.cloneNode(true); | |
| copy.querySelectorAll('script, style').forEach((n) => n.remove()); | |
| copy.querySelectorAll('br').forEach((br) => br.replaceWith('\n')); | |
| copy.querySelectorAll('p, div, li, ul, ol, table, tr, td, th, h1, h2, h3, h4, h5, h6, blockquote, pre, dt, dd, hr, section, article') | |
| .forEach((block) => { block.before('\n'); block.after('\n'); }); | |
| return copy.textContent | |
| .split(/\r\n|[\n\r\v\f]/) | |
| .map((line) => line.split(/\s+/).filter(Boolean).join(' ')) | |
| .filter(Boolean) | |
| .join('\n'); | |
| }; | |
| // Older versions: the element after a "Beschreibung:" label. Current versions: the text block in the task panel, | |
| // the .panel that holds the date table. A task without a description yields '' and never another block of the page. | |
| const label = [...doc.querySelectorAll('div, th, dt, h4, h5, strong, label')] | |
| .find((t) => ['Beschreibung', 'Description'].includes(text(t, '').replace(/:+$/, ''))); | |
| let box = label ? label.nextElementSibling : null; | |
| if (!box && label && label.tagName === 'TH') { | |
| for (let s = label.nextElementSibling; s; s = s.nextElementSibling) { | |
| if (s.tagName === 'TD') { box = s; break; } | |
| } | |
| } | |
| const dates = !box ? th(['Starttermin', 'Beginn', 'Start date', 'Abgabetermin', 'Due date']) : null; | |
| if (dates) { | |
| const panel = dates.closest('.panel'); | |
| box = panel ? [...panel.querySelectorAll('div.text-break-word')].find((d) => !d.closest('form')) || null : null; | |
| } else if (!box) { | |
| box = [...doc.querySelectorAll('div.text-break-word.p-3')].find((e) => !e.closest('form[name="submission"]')) || null; | |
| } | |
| let start = null; | |
| const startNames = ['Starttermin', 'Beginn', 'Start date']; | |
| const startTh = th(startNames); | |
| const table = startTh ? startTh.closest('table') : null; | |
| const row = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('td')) : null; | |
| const col = row ? headerIndex(row, startNames) : -1; | |
| const cells = row ? [...row.querySelectorAll('td')] : []; | |
| const m = col !== -1 && col < cells.length | |
| ? /(\d{2})\.(\d{2})\.(\d{4})(?:\D+(\d{2}):(\d{2}))?/.exec(text(cells[col])) | |
| : null; | |
| if (m) start = `${m[3]}-${m[2]}-${m[1]}` + (m[4] ? `T${m[4]}:${m[5]}` : ''); | |
| return { description: box ? htmlText(box) : '', start }; | |
| } | |