File size: 6,901 Bytes
e561127
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
// Parsers that run INSIDE the browser page: puppeteer's page.evaluate serializes a function, so every function here
// is self-contained (its helpers sit inside it). They get the HTML as the server sent it and read it through
// DOMParser. Page scripts do not run in that document and cannot change what is read: in the live DOM, Bootstrap
// tooltips move the title attribute of the status icon away, so a done task would look open.
//
// Port of the parsers of the IServ filter project (iserv.py: parse_task_list, parse_description, parse_start),
// which are tested against fake pages of the older and the current IServ layout. Keep both in step.

/**
 * Task list: { "<task id>": { tags, start, due, done } }
 * tags: text of the "Tags" column ('' if empty), start: 'yyyy-mm-dd' (older layouts only), due: 'yyyy-mm-ddThh:mm',
 * done: the status icon says done/submitted. Columns are found by their header, never by position (the column
 * after the due date is the teacher's feedback).
 */
export function parseListHtml(html) {
  const doc = new DOMParser().parseFromString(html, 'text/html');
  // BeautifulSoup's get_text(sep, strip=True): the stripped text nodes, joined
  const text = (el, sep = ' ') => {
    const parts = [];
    const walk = (node) => {
      for (const child of node.childNodes) {
        if (child.nodeType === 3) {
          const t = child.textContent.trim();
          if (t) parts.push(t);
        } else if (child.nodeType === 1 && !['SCRIPT', 'STYLE'].includes(child.tagName)) {
          walk(child);
        }
      }
    };
    walk(el);
    return parts.join(sep);
  };
  const headerIndex = (row, names) => {
    const table = row.closest('table');
    const head = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('th')) : null;
    const texts = head ? [...head.querySelectorAll('th, td')].map((c) => text(c).replace(/:+$/, '')) : [];
    return texts.findIndex((t) => names.includes(t));
  };
  const tagsOf = (cell) => {
    const all = text(cell);
    const italic = [...cell.querySelectorAll('i')].map((i) => text(i)).join(' ');
    return all === italic || ['keine', 'none'].includes(all.replace(/^[()]+|[()]+$/g, '').toLowerCase()) ? '' : all;
  };
  const doneRe = /erledigt|abgegeben|done|submitted|completed/i;
  const out = {};
  for (const a of doc.querySelectorAll('a[href*="/exercise/show/"]')) {
    const m = /\/exercise\/show\/(\d+)/.exec(a.getAttribute('href'));
    const row = a.closest('tr');
    if (!m || !row || m[1] in out) continue;
    const cells = [...row.querySelectorAll('td')];
    let start = null;
    let due = null;
    for (const td of cells) {
      const ds = td.getAttribute('data-sort') || '';
      if (/^\d{8}$/.test(ds) && start === null) {
        start = `${ds.slice(0, 4)}-${ds.slice(4, 6)}-${ds.slice(6)}`;
      } else if (/^\d{14}$/.test(ds) && due === null) {
        due = `${ds.slice(0, 4)}-${ds.slice(4, 6)}-${ds.slice(6, 8)}T${ds.slice(8, 10)}:${ds.slice(10, 12)}`;
      }
    }
    const col = headerIndex(row, ['Tags']);
    const titles = [...row.querySelectorAll('[title]')].map((e) => e.getAttribute('title')).join(' ');
    out[m[1]] = { tags: col !== -1 && col < cells.length ? tagsOf(cells[col]) : '', start, due, done: doneRe.test(titles) };
  }
  return out;
}

/**
 * Task page: { description, start }
 * description: only the task's own description (never own submission, teacher feedback, creator, participants),
 * as text with the paragraphs and line breaks of the page. start: 'yyyy-mm-dd[Thh:mm]' from the "Starttermin" column.
 */
export function parseTaskHtml(html) {
  const doc = new DOMParser().parseFromString(html, 'text/html');
  const text = (el, sep = ' ') => {
    const parts = [];
    const walk = (node) => {
      for (const child of node.childNodes) {
        if (child.nodeType === 3) {
          const t = child.textContent.trim();
          if (t) parts.push(t);
        } else if (child.nodeType === 1 && !['SCRIPT', 'STYLE'].includes(child.tagName)) {
          walk(child);
        }
      }
    };
    walk(el);
    return parts.join(sep);
  };
  const th = (names) => [...doc.querySelectorAll('th')].find((t) => names.includes(text(t).replace(/:+$/, '')));
  const headerIndex = (row, names) => {
    const table = row.closest('table');
    const head = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('th')) : null;
    const texts = head ? [...head.querySelectorAll('th, td')].map((c) => text(c).replace(/:+$/, '')) : [];
    return texts.findIndex((t) => names.includes(t));
  };
  const htmlText = (el) => {
    const copy = el.cloneNode(true);
    copy.querySelectorAll('script, style').forEach((n) => n.remove());
    copy.querySelectorAll('br').forEach((br) => br.replaceWith('\n'));
    copy.querySelectorAll('p, div, li, ul, ol, table, tr, td, th, h1, h2, h3, h4, h5, h6, blockquote, pre, dt, dd, hr, section, article')
      .forEach((block) => { block.before('\n'); block.after('\n'); });
    return copy.textContent
      .split(/\r\n|[\n\r\v\f]/)
      .map((line) => line.split(/\s+/).filter(Boolean).join(' '))
      .filter(Boolean)
      .join('\n');
  };

  // Older versions: the element after a "Beschreibung:" label. Current versions: the text block in the task panel,
  // the .panel that holds the date table. A task without a description yields '' and never another block of the page.
  const label = [...doc.querySelectorAll('div, th, dt, h4, h5, strong, label')]
    .find((t) => ['Beschreibung', 'Description'].includes(text(t, '').replace(/:+$/, '')));
  let box = label ? label.nextElementSibling : null;
  if (!box && label && label.tagName === 'TH') {
    for (let s = label.nextElementSibling; s; s = s.nextElementSibling) {
      if (s.tagName === 'TD') { box = s; break; }
    }
  }
  const dates = !box ? th(['Starttermin', 'Beginn', 'Start date', 'Abgabetermin', 'Due date']) : null;
  if (dates) {
    const panel = dates.closest('.panel');
    box = panel ? [...panel.querySelectorAll('div.text-break-word')].find((d) => !d.closest('form')) || null : null;
  } else if (!box) {
    box = [...doc.querySelectorAll('div.text-break-word.p-3')].find((e) => !e.closest('form[name="submission"]')) || null;
  }

  let start = null;
  const startNames = ['Starttermin', 'Beginn', 'Start date'];
  const startTh = th(startNames);
  const table = startTh ? startTh.closest('table') : null;
  const row = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('td')) : null;
  const col = row ? headerIndex(row, startNames) : -1;
  const cells = row ? [...row.querySelectorAll('td')] : [];
  const m = col !== -1 && col < cells.length
    ? /(\d{2})\.(\d{2})\.(\d{4})(?:\D+(\d{2}):(\d{2}))?/.exec(text(cells[col]))
    : null;
  if (m) start = `${m[3]}-${m[2]}-${m[1]}` + (m[4] ? `T${m[4]}:${m[5]}` : '');

  return { description: box ? htmlText(box) : '', start };
}