File size: 17,833 Bytes
ff36e71
866e5c2
e561127
 
 
 
 
 
 
 
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
866e5c2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
 
 
 
 
 
 
 
 
 
 
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
866e5c2
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
 
 
 
 
 
 
 
 
 
 
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
ff36e71
 
 
 
 
e561127
 
 
ff36e71
 
 
 
 
 
 
 
 
 
 
e561127
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
 
ff36e71
 
 
e561127
 
 
 
ff36e71
 
 
e561127
ff36e71
 
e561127
ff36e71
 
 
 
 
 
 
e561127
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
ff36e71
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e561127
ff36e71
 
 
 
 
866e5c2
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
import puppeteer from 'puppeteer';
import { isoFromText, parseAttachmentLinks, parseListHtml, parseTaskHtml } from './iserv-parse.js';

// For logs: no school domain, no task text. Puppeteer's messages quote the url.
function safeError(e) {
  return `${e?.name || 'Error'}: ${String(e?.message || '').replace(/https?:\/\/\S+/g, '<url>').slice(0, 120)}`;
}

// The page's own html as the server sent it. The live DOM differs: page scripts change it (see iserv-parse.js).
const rawHtml = (page) => page.evaluate(async () => (await fetch(location.href)).text());

function censor(text, username) {
  if (!text) return "";
  let cleanText = text;
  
  // Regex for emails and basic phone numbers
  const emailRegex = /\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b/g;
  const phoneRegex = /\+?49[\s\-]?\(?\d{2,4}\)?[\s\-]?\d{3,8}/g;
  
  cleanText = cleanText.replace(emailRegex, '<EMAIL>');
  cleanText = cleanText.replace(phoneRegex, '<TELEFON>');
  
  // Filter out the user's name if they use firstname.lastname format
  if (username && username.includes('.')) {
    const parts = username.split('@')[0].split('.');
    for (const part of parts) {
      if (part.length > 2) {
        const regex = new RegExp(part, 'gi');
        cleanText = cleanText.replace(regex, '<NAME>');
      }
    }
  }
  
  return cleanText;
}

async function openExercisePage(browser, url, username, password) {
  const exerciseUrl = `${url}/iserv/exercise`;
  const page = await browser.newPage();
  
  // Antidetect setup
  await page.evaluateOnNewDocument(() => {
    Object.defineProperty(navigator, 'webdriver', { get: () => undefined });
  });

  await page.goto(exerciseUrl, { waitUntil: 'domcontentloaded' });
  
  // Login
  if (page.url().includes('/login')) {
    await page.waitForSelector("input[name='_username']", { timeout: 10000 });
    await page.type("input[name='_username']", username);
    await page.type("input[name='_password']", password);
    
    const submitBtn = await page.$("button[type='submit'], input[type='submit'], button.btn-primary");
    if (submitBtn) {
      await submitBtn.click();
    } else {
      await page.evaluate(() => document.querySelector('form').submit());
    }
    
    await page.waitForNavigation({ waitUntil: 'domcontentloaded', timeout: 15000 }).catch(() => {});
    
    // IServ OAuth redirects sometimes take time or error out. Wait a bit.
    await new Promise(r => setTimeout(r, 4000));
    
    if (page.url().includes('authentication/error')) {
      try {
         const backBtn = await page.$x("//a[contains(text(), 'Zurück zur Anmeldung')]");
         if (backBtn.length > 0) await backBtn[0].click();
      } catch (e) {}
      await new Promise(r => setTimeout(r, 4000));
    }
  }

  // After login, ensure we are on the exercise page
  if (!page.url().includes('/iserv/exercise')) {
     await page.goto(exerciseUrl, { waitUntil: 'domcontentloaded' });
  }
  return page;
}

/**
 * Logs in to IServ in a headless browser and returns the tasks of the task list. Per task:
 *   id, title, deadline, url, description, attachments [{ filename, mimeType, data (base64) }]
 * Additive fields, read by the IServ filter (iserv-filter project, POST /filter). Consumers that ignore them are not
 * affected:
 *   descriptionText   only the task's own description (description above is the broad block of the page)
 *   tags, start, due, done   subject tag, start date, due date ('yyyy-mm-ddThh:mm'), status icon says done
 *   attachments[].provided   file of the teacher (true) or of own submission / feedback (false)
 *   attachmentsFailed [{ filename, provided, reason: 'refused' | 'too_large' | 'failed' }]   files that were not downloaded
 *   mock   true on the three invented tasks that are returned when the list has no rows
 */
export async function fetchIServTasks(url, username, password) {
  if (!url || !username || !password) {
    throw new Error("Missing credentials");
  }

  // Normalize URL
  if (!url.startsWith('http')) {
    url = 'https://' + url;
  }
  const exerciseUrl = `${url}/iserv/exercise`;
  const browser = await puppeteer.launch({
    headless: "new",
    args: ['--no-sandbox', '--disable-setuid-sandbox']
  });

  try {
    const page = await openExercisePage(browser, url, username, password);

    // Wait for the table
    let rows = [];
    try {
      await page.waitForSelector("table.table tbody tr", { timeout: 5000 });
      rows = await page.$$("table.table tbody tr");
    } catch (e) {
      // Fallback
      try {
        await page.waitForSelector("article, .card", { timeout: 5000 });
        rows = await page.$$("article, .card");
      } catch (err) {}
    }

    const tasks = [];
    for (const row of rows) {
      const linkEl = await row.$("td a");
      if (!linkEl) continue;
      
      const title = await row.evaluate(el => el.querySelector("td a").innerText.trim());
      let href = await row.evaluate(el => el.querySelector("td a").getAttribute("href"));
      const fullUrl = href.startsWith("/") ? `${url}${href}` : href;
      
      const cells = await row.$$("td");
      let deadline = "";
      if (cells.length > 2) {
        deadline = await cells[2].evaluate(el => el.innerText.trim());
      }

      tasks.push({
        id: fullUrl || title,
        title: censor(title, username),
        deadline,
        url: fullUrl,
        description: "",
        attachments: []
      });
    }

    // Additive fields for the IServ filter (tags, start, due, done), read from the list as the server sent it.
    // Existing fields and their values stay as they were.
    let listMeta = {};
    if (rows.length > 0) {
      try {
        listMeta = await page.evaluate(parseListHtml, await rawHtml(page));
      } catch (e) {
        console.error("IServ list meta failed:", safeError(e));
      }
    }

    // Now fetch details for each task
    for (let i = 0; i < tasks.length; i++) {
      try {
        await page.goto(tasks[i].url, { waitUntil: 'domcontentloaded' });
        await new Promise(r => setTimeout(r, 1000));
        
        const contentArea = await page.$(".iserv-exercise-show, .exercise-description, .text-break, .panel, .card, #iserv-main");
        if (contentArea) {
           let desc = await page.evaluate(el => el.innerText.trim(), contentArea);
           if (desc.includes("Erstellt von\t")) {
               desc = desc.split("Erstellt von\t")[0];
           }
           tasks[i].description = censor(desc, username);
         }
         
         // Additive fields for the IServ filter. descriptionText is only the task's own description (never own
         // submission or teacher feedback); description above is the broad block and stays as it was.
         const taskId = (/\/exercise\/show\/(\d+)/.exec(tasks[i].url) || [])[1];
         const fromList = listMeta[taskId];
         try {
           const fromPage = await page.evaluate(parseTaskHtml, await rawHtml(page));
           tasks[i].descriptionText = censor(fromPage.description, username);
           tasks[i].start = (fromList && fromList.start) || fromPage.start || null;
         } catch (e) {
           console.error("IServ task meta failed:", taskId, safeError(e));
         }
         if (fromList) {
           tasks[i].tags = fromList.tags;
           tasks[i].due = fromList.due;
           tasks[i].done = fromList.done;
         }

         const attachmentLinks = await page.evaluate(() => {
           const links = document.querySelectorAll('.attachments a, .attachment-list a, .files a, a[href*="/download/"], a[href*="/file/"]');
           const seenUrls = new Set();
           return Array.from(links).map(a => ({
             url: a.href,
             filename: a.innerText.trim() || 'Anhang',
             // the files the teacher provided, as opposed to own submission or feedback files
             provided: !!a.closest('form[name="iserv_exercise_attachment"]')
           })).filter(a => {
             const name = a.filename.toLowerCase();
             const isUiButton = name === 'dateien' || name === 'öffnen' || name === 'herunterladen' || name === 'download' || name === 'open' || name === 'vorschau';
             const isValidUrl = a.url && a.url.startsWith('http') && !a.url.includes('javascript:');
             if (!isValidUrl || isUiButton || seenUrls.has(a.url)) return false;
             seenUrls.add(a.url);
             return true;
           });
         });

         const downloadedAttachments = [];
         const failedAttachments = []; // additive: files that could not be downloaded, for the IServ filter
         for (const link of attachmentLinks) {
           try {
             const b64 = await page.evaluate(async (url) => {
               const res = await fetch(url);
               if (!res.ok) throw new Error("Fetch failed " + res.status);
               const blob = await res.blob();
               if (blob.size > 10 * 1024 * 1024) throw new Error("File too large"); // 10MB limit
               return new Promise((resolve, reject) => {
                 const reader = new FileReader();
                 reader.onloadend = () => resolve({ data: reader.result, mimeType: blob.type });
                 reader.onerror = reject;
                 reader.readAsDataURL(blob);
               });
             }, link.url);
             
             const base64Data = b64.data.split(',')[1];
             if (base64Data) {
               downloadedAttachments.push({
                 filename: link.filename,
                 mimeType: b64.mimeType || 'application/octet-stream',
                 data: base64Data,
                 provided: link.provided
               });
             }
           } catch (err) {
             // path only: the url holds the school domain
             console.error("Failed to download attachment:", new URL(link.url).pathname, safeError(err));
             const reason = /Fetch failed 40[13]/.test(err.message) ? 'refused' : /File too large/.test(err.message) ? 'too_large' : 'failed';
             failedAttachments.push({ filename: link.filename, provided: link.provided, reason });
           }
         }
         tasks[i].attachments = downloadedAttachments;
         tasks[i].attachmentsFailed = failedAttachments;
         
       } catch (e) {
        console.error("Error fetching detail for task", (/\/exercise\/show\/(\d+)/.exec(tasks[i].url) || [])[1], safeError(e));
      }
    }

    if (tasks.length === 0) {
      console.log("No tasks found, injecting mock tasks for testing...");
      tasks.push({
        id: exerciseUrl + "/mock_1",
        mock: true, // additive: invented test task, the IServ filter drops these
        title: censor("Mathematik: Kurvendiskussion & Integralrechnung", username),
        deadline: "Morgen, 08:00 Uhr",
        url: exerciseUrl + "/mock_1",
        description: censor("Bitte bearbeitet die Arbeitsblätter zur Vorbereitung auf die Klausur. Aufgabe 1: Berechne die Nullstellen, Extrempunkte und Wendepunkte der Funktion f(x) = x^3 - 6x^2 + 9x. Aufgabe 2: Berechne die Fläche unter dem Graphen im Intervall [0, 3].\nLadet eure Lösungswege (am besten als PDF oder gut lesbares Foto) hier hoch.", username),
        attachments: [
          {
            filename: "arbeitsblatt_kurvendiskussion.txt",
            mimeType: "text/plain",
            data: "QXVmZ2FiZW5ibGF0dCBLdXJ2ZW5kaXNrdXNzaW9uOiBmKHgpID0geF4zIC0gNnggKyA5LiBCaXR0ZSBhbGxlIEV4dHJlbWEgYmVyZWNobmVuLg=="
          }
        ]
      });
      tasks.push({
        id: exerciseUrl + "/mock_2",
        mock: true, // additive: invented test task, the IServ filter drops these
        title: censor("Englisch: Essay 'The Impact of AI'", username),
        deadline: "Freitag, 23:59 Uhr",
        url: exerciseUrl + "/mock_2",
        description: censor("Write a 500-word essay discussing the potential impacts of Artificial Intelligence on the future job market. Do you think AI will create more jobs than it destroys? Use specific examples to support your arguments. Please submit your text directly in the text field or upload a Word document.", username),
        attachments: [
          {
            filename: "essay_guidelines.txt",
            mimeType: "text/plain",
            data: "RXNzYXkgR3VpZGVsaW5lczogTWluLiA1MDAgd29yZHMuIFVzZSAzIHNvdXJjZXMuIFN0cnVjdHVyZTogSW50cm8sIEJvZHksIENvbmNsdXNpb24u"
          }
        ]
      });
      tasks.push({
        id: exerciseUrl + "/mock_3",
        mock: true, // additive: invented test task, the IServ filter drops these
        title: censor("Geschichte: Quellenanalyse Weimarer Republik", username),
        deadline: "Nächste Woche Montag",
        url: exerciseUrl + "/mock_3",
        description: censor("Analysiert die historische Quelle 'Aufruf der Reichsregierung vom Kapp-Putsch 1920'.\n1. Ordnet die Quelle in den historischen Kontext ein.\n2. Arbeitet die Hauptaussagen heraus.\n3. Beurteilt die Bedeutung des Putsches für das Scheitern der Weimarer Republik.", username),
        attachments: [
          {
            filename: "quelle_kapp_putsch.txt",
            mimeType: "text/plain",
            data: "S2FwcC1QdXRzY2ggUXVlbGxlOiAiRGllIFJlaWNoc3JlZ2llcnVuZyBydWZ0IGRlbiBHZW5lcmFsc3RyZWlrIGF1cyEi"
          }
        ]
      });
    }

    return tasks;
  } catch (error) {
    console.error("IServ Scraping Error:", safeError(error), String(error?.stack || '').replace(/https?:\/\/\S+/g, '<url>'));
    throw error;
  } finally {
    await browser.close();
  }
}

const DETAIL_PARALLEL = 4;
const FILE_LIMIT = 25 * 1024 * 1024;
const TASK_ID = /\/exercise\/show\/(\d+)/;

const baseOf = (url) => (url.startsWith('http') ? url : 'https://' + url).replace(/\/+$/, '');
const launch = () => puppeteer.launch({ headless: 'new', args: ['--no-sandbox', '--disable-setuid-sandbox'] });

/**
 * Overview for the notes app (its own client, so nothing is censored): every task of the list with status, and for
 * every open one the description and the provided files (links only, no downloads). Done and expired tasks get no
 * task page. Per task: id, url, title, tags, due ('yyyy-mm-dd[Thh:mm]' | null), done, expired, description,
 * attachments [{ filename, path }]. No mock tasks.
 */
export async function fetchIServOverview(url, username, password) {
  if (!url || !username || !password) throw new Error('Missing credentials');
  const base = baseOf(url);
  const browser = await launch();
  try {
    const page = await openExercisePage(browser, base, username, password);
    const rows = await page.evaluate(() =>
      [...document.querySelectorAll('a[href*="/exercise/show/"]')].map((a) => ({
        href: a.getAttribute('href'),
        title: a.innerText.trim(),
        deadline: (a.closest('tr')?.querySelectorAll('td')[2]?.innerText || '').trim(),
      })),
    );
    const meta = rows.length ? await page.evaluate(parseListHtml, await rawHtml(page)) : {};
    const tasks = [];
    const seen = new Set();
    for (const row of rows) {
      const id = TASK_ID.exec(row.href || '')?.[1];
      if (!id || seen.has(id)) continue;
      seen.add(id);
      const listed = meta[id] || {};
      tasks.push({
        id: Number(id),
        url: `${base}/iserv/exercise/show/${id}`,
        title: row.title,
        tags: listed.tags || '',
        due: listed.due || isoFromText(row.deadline),
        done: listed.done === true,
        expired: listed.expired === true,
        description: '',
        attachments: [],
      });
    }

    const fill = async (task) => {
      try {
        const html = await page.evaluate(async (u) => (await fetch(u)).text(), task.url);
        task.description = (await page.evaluate(parseTaskHtml, html)).description;
        for (const link of await page.evaluate(parseAttachmentLinks, html)) {
          const target = new URL(link.href, base);
          if (target.host === new URL(base).host) {
            task.attachments.push({ filename: link.filename, path: target.pathname + target.search });
          }
        }
      } catch (e) {
        console.error('IServ overview detail failed:', task.id, safeError(e));
      }
    };
    const queue = tasks.filter((task) => !task.done && !task.expired);
    await Promise.all(
      Array.from({ length: DETAIL_PARALLEL }, async () => {
        for (let task = queue.shift(); task; task = queue.shift()) await fill(task);
      }),
    );
    return tasks;
  } finally {
    await browser.close();
  }
}

/** One file of the school server, fetched inside the logged-in tab. Throws Error with code refused | too_large | failed. */
export async function downloadIServFile(url, username, password, path) {
  if (!url || !username || !password) throw new Error('Missing credentials');
  const base = baseOf(url);
  const browser = await launch();
  try {
    const page = await openExercisePage(browser, base, username, password);
    const result = await page.evaluate(
      async (target, limit) => {
        const res = await fetch(target);
        if (!res.ok) return { error: res.status === 401 || res.status === 403 ? 'refused' : 'failed' };
        const blob = await res.blob();
        if (blob.size > limit) return { error: 'too_large' };
        const data = await new Promise((resolve, reject) => {
          const reader = new FileReader();
          reader.onloadend = () => resolve(reader.result);
          reader.onerror = reject;
          reader.readAsDataURL(blob);
        });
        return { contentType: blob.type || 'application/octet-stream', data: String(data).split(',')[1] || '' };
      },
      `${base}${path}`,
      FILE_LIMIT,
    );
    if (result.error) throw Object.assign(new Error(result.error), { code: result.error });
    return { contentType: result.contentType, buffer: Buffer.from(result.data, 'base64') };
  } finally {
    await browser.close();
  }
}