File size: 13,326 Bytes
ff36e71
e561127
 
 
 
 
 
 
 
 
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
 
 
 
 
 
 
 
 
 
 
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
 
 
 
 
 
 
 
 
 
 
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
ff36e71
 
 
 
 
e561127
 
 
ff36e71
 
 
 
 
 
 
 
 
 
 
e561127
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
 
ff36e71
 
 
e561127
 
 
 
ff36e71
 
 
e561127
ff36e71
 
e561127
ff36e71
 
 
 
 
 
 
e561127
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
ff36e71
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
import puppeteer from 'puppeteer';
import { parseListHtml, parseTaskHtml } from './iserv-parse.js';

// For logs: no school domain, no task text. Puppeteer's messages quote the url.
function safeError(e) {
  return `${e?.name || 'Error'}: ${String(e?.message || '').replace(/https?:\/\/\S+/g, '<url>').slice(0, 120)}`;
}

// The page's own html as the server sent it. The live DOM differs: page scripts change it (see iserv-parse.js).
const rawHtml = (page) => page.evaluate(async () => (await fetch(location.href)).text());

function censor(text, username) {
  if (!text) return "";
  let cleanText = text;
  
  // Regex for emails and basic phone numbers
  const emailRegex = /\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b/g;
  const phoneRegex = /\+?49[\s\-]?\(?\d{2,4}\)?[\s\-]?\d{3,8}/g;
  
  cleanText = cleanText.replace(emailRegex, '<EMAIL>');
  cleanText = cleanText.replace(phoneRegex, '<TELEFON>');
  
  // Filter out the user's name if they use firstname.lastname format
  if (username && username.includes('.')) {
    const parts = username.split('@')[0].split('.');
    for (const part of parts) {
      if (part.length > 2) {
        const regex = new RegExp(part, 'gi');
        cleanText = cleanText.replace(regex, '<NAME>');
      }
    }
  }
  
  return cleanText;
}

/**
 * Logs in to IServ in a headless browser and returns the tasks of the task list. Per task:
 *   id, title, deadline, url, description, attachments [{ filename, mimeType, data (base64) }]
 * Additive fields, read by the IServ filter (iserv-filter project, POST /filter). Consumers that ignore them are not
 * affected:
 *   descriptionText   only the task's own description (description above is the broad block of the page)
 *   tags, start, due, done   subject tag, start date, due date ('yyyy-mm-ddThh:mm'), status icon says done
 *   attachments[].provided   file of the teacher (true) or of own submission / feedback (false)
 *   attachmentsFailed [{ filename, provided, reason: 'refused' | 'too_large' | 'failed' }]   files that were not downloaded
 *   mock   true on the three invented tasks that are returned when the list has no rows
 */
export async function fetchIServTasks(url, username, password) {
  if (!url || !username || !password) {
    throw new Error("Missing credentials");
  }

  // Normalize URL
  if (!url.startsWith('http')) {
    url = 'https://' + url;
  }
  const exerciseUrl = `${url}/iserv/exercise`;
  const browser = await puppeteer.launch({
    headless: "new",
    args: ['--no-sandbox', '--disable-setuid-sandbox']
  });

  try {
    const page = await browser.newPage();
    
    // Antidetect setup
    await page.evaluateOnNewDocument(() => {
      Object.defineProperty(navigator, 'webdriver', { get: () => undefined });
    });

    await page.goto(exerciseUrl, { waitUntil: 'domcontentloaded' });
    
    // Login
    if (page.url().includes('/login')) {
      await page.waitForSelector("input[name='_username']", { timeout: 10000 });
      await page.type("input[name='_username']", username);
      await page.type("input[name='_password']", password);
      
      const submitBtn = await page.$("button[type='submit'], input[type='submit'], button.btn-primary");
      if (submitBtn) {
        await submitBtn.click();
      } else {
        await page.evaluate(() => document.querySelector('form').submit());
      }
      
      await page.waitForNavigation({ waitUntil: 'domcontentloaded', timeout: 15000 }).catch(() => {});
      
      // IServ OAuth redirects sometimes take time or error out. Wait a bit.
      await new Promise(r => setTimeout(r, 4000));
      
      if (page.url().includes('authentication/error')) {
        try {
           const backBtn = await page.$x("//a[contains(text(), 'Zurück zur Anmeldung')]");
           if (backBtn.length > 0) await backBtn[0].click();
        } catch (e) {}
        await new Promise(r => setTimeout(r, 4000));
      }
    }

    // After login, ensure we are on the exercise page
    if (!page.url().includes('/iserv/exercise')) {
       await page.goto(exerciseUrl, { waitUntil: 'domcontentloaded' });
    }

    // Wait for the table
    let rows = [];
    try {
      await page.waitForSelector("table.table tbody tr", { timeout: 5000 });
      rows = await page.$$("table.table tbody tr");
    } catch (e) {
      // Fallback
      try {
        await page.waitForSelector("article, .card", { timeout: 5000 });
        rows = await page.$$("article, .card");
      } catch (err) {}
    }

    const tasks = [];
    for (const row of rows) {
      const linkEl = await row.$("td a");
      if (!linkEl) continue;
      
      const title = await row.evaluate(el => el.querySelector("td a").innerText.trim());
      let href = await row.evaluate(el => el.querySelector("td a").getAttribute("href"));
      const fullUrl = href.startsWith("/") ? `${url}${href}` : href;
      
      const cells = await row.$$("td");
      let deadline = "";
      if (cells.length > 2) {
        deadline = await cells[2].evaluate(el => el.innerText.trim());
      }

      tasks.push({
        id: fullUrl || title,
        title: censor(title, username),
        deadline,
        url: fullUrl,
        description: "",
        attachments: []
      });
    }

    // Additive fields for the IServ filter (tags, start, due, done), read from the list as the server sent it.
    // Existing fields and their values stay as they were.
    let listMeta = {};
    if (rows.length > 0) {
      try {
        listMeta = await page.evaluate(parseListHtml, await rawHtml(page));
      } catch (e) {
        console.error("IServ list meta failed:", safeError(e));
      }
    }

    // Now fetch details for each task
    for (let i = 0; i < tasks.length; i++) {
      try {
        await page.goto(tasks[i].url, { waitUntil: 'domcontentloaded' });
        await new Promise(r => setTimeout(r, 1000));
        
        const contentArea = await page.$(".iserv-exercise-show, .exercise-description, .text-break, .panel, .card, #iserv-main");
        if (contentArea) {
           let desc = await page.evaluate(el => el.innerText.trim(), contentArea);
           if (desc.includes("Erstellt von\t")) {
               desc = desc.split("Erstellt von\t")[0];
           }
           tasks[i].description = censor(desc, username);
         }
         
         // Additive fields for the IServ filter. descriptionText is only the task's own description (never own
         // submission or teacher feedback); description above is the broad block and stays as it was.
         const taskId = (/\/exercise\/show\/(\d+)/.exec(tasks[i].url) || [])[1];
         const fromList = listMeta[taskId];
         try {
           const fromPage = await page.evaluate(parseTaskHtml, await rawHtml(page));
           tasks[i].descriptionText = censor(fromPage.description, username);
           tasks[i].start = (fromList && fromList.start) || fromPage.start || null;
         } catch (e) {
           console.error("IServ task meta failed:", taskId, safeError(e));
         }
         if (fromList) {
           tasks[i].tags = fromList.tags;
           tasks[i].due = fromList.due;
           tasks[i].done = fromList.done;
         }

         const attachmentLinks = await page.evaluate(() => {
           const links = document.querySelectorAll('.attachments a, .attachment-list a, .files a, a[href*="/download/"], a[href*="/file/"]');
           const seenUrls = new Set();
           return Array.from(links).map(a => ({
             url: a.href,
             filename: a.innerText.trim() || 'Anhang',
             // the files the teacher provided, as opposed to own submission or feedback files
             provided: !!a.closest('form[name="iserv_exercise_attachment"]')
           })).filter(a => {
             const name = a.filename.toLowerCase();
             const isUiButton = name === 'dateien' || name === 'öffnen' || name === 'herunterladen' || name === 'download' || name === 'open' || name === 'vorschau';
             const isValidUrl = a.url && a.url.startsWith('http') && !a.url.includes('javascript:');
             if (!isValidUrl || isUiButton || seenUrls.has(a.url)) return false;
             seenUrls.add(a.url);
             return true;
           });
         });

         const downloadedAttachments = [];
         const failedAttachments = []; // additive: files that could not be downloaded, for the IServ filter
         for (const link of attachmentLinks) {
           try {
             const b64 = await page.evaluate(async (url) => {
               const res = await fetch(url);
               if (!res.ok) throw new Error("Fetch failed " + res.status);
               const blob = await res.blob();
               if (blob.size > 10 * 1024 * 1024) throw new Error("File too large"); // 10MB limit
               return new Promise((resolve, reject) => {
                 const reader = new FileReader();
                 reader.onloadend = () => resolve({ data: reader.result, mimeType: blob.type });
                 reader.onerror = reject;
                 reader.readAsDataURL(blob);
               });
             }, link.url);
             
             const base64Data = b64.data.split(',')[1];
             if (base64Data) {
               downloadedAttachments.push({
                 filename: link.filename,
                 mimeType: b64.mimeType || 'application/octet-stream',
                 data: base64Data,
                 provided: link.provided
               });
             }
           } catch (err) {
             // path only: the url holds the school domain
             console.error("Failed to download attachment:", new URL(link.url).pathname, safeError(err));
             const reason = /Fetch failed 40[13]/.test(err.message) ? 'refused' : /File too large/.test(err.message) ? 'too_large' : 'failed';
             failedAttachments.push({ filename: link.filename, provided: link.provided, reason });
           }
         }
         tasks[i].attachments = downloadedAttachments;
         tasks[i].attachmentsFailed = failedAttachments;
         
       } catch (e) {
        console.error("Error fetching detail for task", (/\/exercise\/show\/(\d+)/.exec(tasks[i].url) || [])[1], safeError(e));
      }
    }

    if (tasks.length === 0) {
      console.log("No tasks found, injecting mock tasks for testing...");
      tasks.push({
        id: exerciseUrl + "/mock_1",
        mock: true, // additive: invented test task, the IServ filter drops these
        title: censor("Mathematik: Kurvendiskussion & Integralrechnung", username),
        deadline: "Morgen, 08:00 Uhr",
        url: exerciseUrl + "/mock_1",
        description: censor("Bitte bearbeitet die Arbeitsblätter zur Vorbereitung auf die Klausur. Aufgabe 1: Berechne die Nullstellen, Extrempunkte und Wendepunkte der Funktion f(x) = x^3 - 6x^2 + 9x. Aufgabe 2: Berechne die Fläche unter dem Graphen im Intervall [0, 3].\nLadet eure Lösungswege (am besten als PDF oder gut lesbares Foto) hier hoch.", username),
        attachments: [
          {
            filename: "arbeitsblatt_kurvendiskussion.txt",
            mimeType: "text/plain",
            data: "QXVmZ2FiZW5ibGF0dCBLdXJ2ZW5kaXNrdXNzaW9uOiBmKHgpID0geF4zIC0gNnggKyA5LiBCaXR0ZSBhbGxlIEV4dHJlbWEgYmVyZWNobmVuLg=="
          }
        ]
      });
      tasks.push({
        id: exerciseUrl + "/mock_2",
        mock: true, // additive: invented test task, the IServ filter drops these
        title: censor("Englisch: Essay 'The Impact of AI'", username),
        deadline: "Freitag, 23:59 Uhr",
        url: exerciseUrl + "/mock_2",
        description: censor("Write a 500-word essay discussing the potential impacts of Artificial Intelligence on the future job market. Do you think AI will create more jobs than it destroys? Use specific examples to support your arguments. Please submit your text directly in the text field or upload a Word document.", username),
        attachments: [
          {
            filename: "essay_guidelines.txt",
            mimeType: "text/plain",
            data: "RXNzYXkgR3VpZGVsaW5lczogTWluLiA1MDAgd29yZHMuIFVzZSAzIHNvdXJjZXMuIFN0cnVjdHVyZTogSW50cm8sIEJvZHksIENvbmNsdXNpb24u"
          }
        ]
      });
      tasks.push({
        id: exerciseUrl + "/mock_3",
        mock: true, // additive: invented test task, the IServ filter drops these
        title: censor("Geschichte: Quellenanalyse Weimarer Republik", username),
        deadline: "Nächste Woche Montag",
        url: exerciseUrl + "/mock_3",
        description: censor("Analysiert die historische Quelle 'Aufruf der Reichsregierung vom Kapp-Putsch 1920'.\n1. Ordnet die Quelle in den historischen Kontext ein.\n2. Arbeitet die Hauptaussagen heraus.\n3. Beurteilt die Bedeutung des Putsches für das Scheitern der Weimarer Republik.", username),
        attachments: [
          {
            filename: "quelle_kapp_putsch.txt",
            mimeType: "text/plain",
            data: "S2FwcC1QdXRzY2ggUXVlbGxlOiAiRGllIFJlaWNoc3JlZ2llcnVuZyBydWZ0IGRlbiBHZW5lcmFsc3RyZWlrIGF1cyEi"
          }
        ]
      });
    }

    return tasks;
  } catch (error) {
    console.error("IServ Scraping Error:", safeError(error), String(error?.stack || '').replace(/https?:\/\/\S+/g, '<url>'));
    throw error;
  } finally {
    await browser.close();
  }
}