File size: 8,664 Bytes
e561127
 
 
 
 
 
 
 
 
b12cb35
e561127
b12cb35
 
 
e561127
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b12cb35
e561127
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b12cb35
 
 
 
 
 
 
e561127
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
b12cb35
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
// Parsers that run INSIDE the browser page: puppeteer's page.evaluate serializes a function, so every function here
// is self-contained (its helpers sit inside it). They get the HTML as the server sent it and read it through
// DOMParser. Page scripts do not run in that document and cannot change what is read: in the live DOM, Bootstrap
// tooltips move the title attribute of the status icon away, so a done task would look open.
//
// Port of the parsers of the IServ filter project (iserv.py: parse_task_list, parse_description, parse_start),
// which are tested against fake pages of the older and the current IServ layout. Keep both in step.

/**
 * Task list: { "<task id>": { tags, start, due, done, expired } }
 * tags: text of the "Tags" column ('' if empty), start: 'yyyy-mm-dd' (older layouts only), due: 'yyyy-mm-ddThh:mm',
 * done: the status icon says done/submitted, expired: it says expired (guessed pattern, only "Erledigt" was seen
 * live). Columns are found by their header, never by position (the column after the due date is the teacher's
 * feedback).
 */
export function parseListHtml(html) {
  const doc = new DOMParser().parseFromString(html, 'text/html');
  // BeautifulSoup's get_text(sep, strip=True): the stripped text nodes, joined
  const text = (el, sep = ' ') => {
    const parts = [];
    const walk = (node) => {
      for (const child of node.childNodes) {
        if (child.nodeType === 3) {
          const t = child.textContent.trim();
          if (t) parts.push(t);
        } else if (child.nodeType === 1 && !['SCRIPT', 'STYLE'].includes(child.tagName)) {
          walk(child);
        }
      }
    };
    walk(el);
    return parts.join(sep);
  };
  const headerIndex = (row, names) => {
    const table = row.closest('table');
    const head = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('th')) : null;
    const texts = head ? [...head.querySelectorAll('th, td')].map((c) => text(c).replace(/:+$/, '')) : [];
    return texts.findIndex((t) => names.includes(t));
  };
  const tagsOf = (cell) => {
    const all = text(cell);
    const italic = [...cell.querySelectorAll('i')].map((i) => text(i)).join(' ');
    return all === italic || ['keine', 'none'].includes(all.replace(/^[()]+|[()]+$/g, '').toLowerCase()) ? '' : all;
  };
  const doneRe = /erledigt|abgegeben|done|submitted|completed/i;
  const expiredRe = /abgelaufen|expired/i;
  const out = {};
  for (const a of doc.querySelectorAll('a[href*="/exercise/show/"]')) {
    const m = /\/exercise\/show\/(\d+)/.exec(a.getAttribute('href'));
    const row = a.closest('tr');
    if (!m || !row || m[1] in out) continue;
    const cells = [...row.querySelectorAll('td')];
    let start = null;
    let due = null;
    for (const td of cells) {
      const ds = td.getAttribute('data-sort') || '';
      if (/^\d{8}$/.test(ds) && start === null) {
        start = `${ds.slice(0, 4)}-${ds.slice(4, 6)}-${ds.slice(6)}`;
      } else if (/^\d{14}$/.test(ds) && due === null) {
        due = `${ds.slice(0, 4)}-${ds.slice(4, 6)}-${ds.slice(6, 8)}T${ds.slice(8, 10)}:${ds.slice(10, 12)}`;
      }
    }
    const col = headerIndex(row, ['Tags']);
    const titles = [...row.querySelectorAll('[title]')].map((e) => e.getAttribute('title')).join(' ');
    out[m[1]] = {
      tags: col !== -1 && col < cells.length ? tagsOf(cells[col]) : '',
      start,
      due,
      done: doneRe.test(titles),
      expired: expiredRe.test(titles),
    };
  }
  return out;
}

/**
 * Task page: { description, start }
 * description: only the task's own description (never own submission, teacher feedback, creator, participants),
 * as text with the paragraphs and line breaks of the page. start: 'yyyy-mm-dd[Thh:mm]' from the "Starttermin" column.
 */
export function parseTaskHtml(html) {
  const doc = new DOMParser().parseFromString(html, 'text/html');
  const text = (el, sep = ' ') => {
    const parts = [];
    const walk = (node) => {
      for (const child of node.childNodes) {
        if (child.nodeType === 3) {
          const t = child.textContent.trim();
          if (t) parts.push(t);
        } else if (child.nodeType === 1 && !['SCRIPT', 'STYLE'].includes(child.tagName)) {
          walk(child);
        }
      }
    };
    walk(el);
    return parts.join(sep);
  };
  const th = (names) => [...doc.querySelectorAll('th')].find((t) => names.includes(text(t).replace(/:+$/, '')));
  const headerIndex = (row, names) => {
    const table = row.closest('table');
    const head = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('th')) : null;
    const texts = head ? [...head.querySelectorAll('th, td')].map((c) => text(c).replace(/:+$/, '')) : [];
    return texts.findIndex((t) => names.includes(t));
  };
  const htmlText = (el) => {
    const copy = el.cloneNode(true);
    copy.querySelectorAll('script, style').forEach((n) => n.remove());
    copy.querySelectorAll('br').forEach((br) => br.replaceWith('\n'));
    copy.querySelectorAll('p, div, li, ul, ol, table, tr, td, th, h1, h2, h3, h4, h5, h6, blockquote, pre, dt, dd, hr, section, article')
      .forEach((block) => { block.before('\n'); block.after('\n'); });
    return copy.textContent
      .split(/\r\n|[\n\r\v\f]/)
      .map((line) => line.split(/\s+/).filter(Boolean).join(' '))
      .filter(Boolean)
      .join('\n');
  };

  // Older versions: the element after a "Beschreibung:" label. Current versions: the text block in the task panel,
  // the .panel that holds the date table. A task without a description yields '' and never another block of the page.
  const label = [...doc.querySelectorAll('div, th, dt, h4, h5, strong, label')]
    .find((t) => ['Beschreibung', 'Description'].includes(text(t, '').replace(/:+$/, '')));
  let box = label ? label.nextElementSibling : null;
  if (!box && label && label.tagName === 'TH') {
    for (let s = label.nextElementSibling; s; s = s.nextElementSibling) {
      if (s.tagName === 'TD') { box = s; break; }
    }
  }
  const dates = !box ? th(['Starttermin', 'Beginn', 'Start date', 'Abgabetermin', 'Due date']) : null;
  if (dates) {
    const panel = dates.closest('.panel');
    box = panel ? [...panel.querySelectorAll('div.text-break-word')].find((d) => !d.closest('form')) || null : null;
  } else if (!box) {
    box = [...doc.querySelectorAll('div.text-break-word.p-3')].find((e) => !e.closest('form[name="submission"]')) || null;
  }

  let start = null;
  const startNames = ['Starttermin', 'Beginn', 'Start date'];
  const startTh = th(startNames);
  const table = startTh ? startTh.closest('table') : null;
  const row = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('td')) : null;
  const col = row ? headerIndex(row, startNames) : -1;
  const cells = row ? [...row.querySelectorAll('td')] : [];
  const m = col !== -1 && col < cells.length
    ? /(\d{2})\.(\d{2})\.(\d{4})(?:\D+(\d{2}):(\d{2}))?/.exec(text(cells[col]))
    : null;
  if (m) start = `${m[3]}-${m[2]}-${m[1]}` + (m[4] ? `T${m[4]}:${m[5]}` : '');

  return { description: box ? htmlText(box) : '', start };
}

/**
 * Provided files of a task page: [{ href, filename }]. Only the "provided files" form, never own submission or
 * teacher feedback. Per row: the download link if there is one (current versions also put a view link and a menu
 * toggle in each row), the file name from the row's name link. Port of parse_attachment_links in iserv.py.
 */
export function parseAttachmentLinks(html) {
  const doc = new DOMParser().parseFromString(html, 'text/html');
  const form = doc.querySelector('form[name="iserv_exercise_attachment"]');
  const out = [];
  for (const row of form ? form.querySelectorAll('tr') : []) {
    const anchors = [...row.querySelectorAll('a[href]')].filter(
      (a) => !['/', '#'].includes(a.getAttribute('href')) && !a.hasAttribute('data-toggle'),
    );
    if (anchors.length === 0) continue;
    const pick = anchors.find((a) => a.getAttribute('href').includes('/download/')) || anchors[0];
    const named = row.querySelector('a.text-break-word') || pick;
    out.push({ href: pick.getAttribute('href'), filename: (named.textContent || '').replace(/\s+/g, ' ').trim() });
  }
  return out;
}

/** 'dd.mm.yyyy[ hh:mm]' somewhere in a text to 'yyyy-mm-dd[Thh:mm]', null if there is none. Plain Node, not for page.evaluate. */
export function isoFromText(text) {
  const m = /(\d{1,2})\.(\d{1,2})\.(\d{4})(?:\D+(\d{1,2}):(\d{2}))?/.exec(String(text || ''));
  if (!m) return null;
  const [, d, mo, y, h, mi] = m;
  const date = `${y}-${mo.padStart(2, '0')}-${d.padStart(2, '0')}`;
  return h ? `${date}T${h.padStart(2, '0')}:${mi}` : date;
}