Spaces:
Paused
Paused
File size: 8,664 Bytes
e561127 b12cb35 e561127 b12cb35 e561127 b12cb35 e561127 b12cb35 e561127 b12cb35 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 150 151 152 153 154 155 156 157 158 159 160 161 162 163 164 165 166 167 168 169 170 171 172 173 174 175 176 177 178 179 180 | // Parsers that run INSIDE the browser page: puppeteer's page.evaluate serializes a function, so every function here
// is self-contained (its helpers sit inside it). They get the HTML as the server sent it and read it through
// DOMParser. Page scripts do not run in that document and cannot change what is read: in the live DOM, Bootstrap
// tooltips move the title attribute of the status icon away, so a done task would look open.
//
// Port of the parsers of the IServ filter project (iserv.py: parse_task_list, parse_description, parse_start),
// which are tested against fake pages of the older and the current IServ layout. Keep both in step.
/**
* Task list: { "<task id>": { tags, start, due, done, expired } }
* tags: text of the "Tags" column ('' if empty), start: 'yyyy-mm-dd' (older layouts only), due: 'yyyy-mm-ddThh:mm',
* done: the status icon says done/submitted, expired: it says expired (guessed pattern, only "Erledigt" was seen
* live). Columns are found by their header, never by position (the column after the due date is the teacher's
* feedback).
*/
export function parseListHtml(html) {
const doc = new DOMParser().parseFromString(html, 'text/html');
// BeautifulSoup's get_text(sep, strip=True): the stripped text nodes, joined
const text = (el, sep = ' ') => {
const parts = [];
const walk = (node) => {
for (const child of node.childNodes) {
if (child.nodeType === 3) {
const t = child.textContent.trim();
if (t) parts.push(t);
} else if (child.nodeType === 1 && !['SCRIPT', 'STYLE'].includes(child.tagName)) {
walk(child);
}
}
};
walk(el);
return parts.join(sep);
};
const headerIndex = (row, names) => {
const table = row.closest('table');
const head = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('th')) : null;
const texts = head ? [...head.querySelectorAll('th, td')].map((c) => text(c).replace(/:+$/, '')) : [];
return texts.findIndex((t) => names.includes(t));
};
const tagsOf = (cell) => {
const all = text(cell);
const italic = [...cell.querySelectorAll('i')].map((i) => text(i)).join(' ');
return all === italic || ['keine', 'none'].includes(all.replace(/^[()]+|[()]+$/g, '').toLowerCase()) ? '' : all;
};
const doneRe = /erledigt|abgegeben|done|submitted|completed/i;
const expiredRe = /abgelaufen|expired/i;
const out = {};
for (const a of doc.querySelectorAll('a[href*="/exercise/show/"]')) {
const m = /\/exercise\/show\/(\d+)/.exec(a.getAttribute('href'));
const row = a.closest('tr');
if (!m || !row || m[1] in out) continue;
const cells = [...row.querySelectorAll('td')];
let start = null;
let due = null;
for (const td of cells) {
const ds = td.getAttribute('data-sort') || '';
if (/^\d{8}$/.test(ds) && start === null) {
start = `${ds.slice(0, 4)}-${ds.slice(4, 6)}-${ds.slice(6)}`;
} else if (/^\d{14}$/.test(ds) && due === null) {
due = `${ds.slice(0, 4)}-${ds.slice(4, 6)}-${ds.slice(6, 8)}T${ds.slice(8, 10)}:${ds.slice(10, 12)}`;
}
}
const col = headerIndex(row, ['Tags']);
const titles = [...row.querySelectorAll('[title]')].map((e) => e.getAttribute('title')).join(' ');
out[m[1]] = {
tags: col !== -1 && col < cells.length ? tagsOf(cells[col]) : '',
start,
due,
done: doneRe.test(titles),
expired: expiredRe.test(titles),
};
}
return out;
}
/**
* Task page: { description, start }
* description: only the task's own description (never own submission, teacher feedback, creator, participants),
* as text with the paragraphs and line breaks of the page. start: 'yyyy-mm-dd[Thh:mm]' from the "Starttermin" column.
*/
export function parseTaskHtml(html) {
const doc = new DOMParser().parseFromString(html, 'text/html');
const text = (el, sep = ' ') => {
const parts = [];
const walk = (node) => {
for (const child of node.childNodes) {
if (child.nodeType === 3) {
const t = child.textContent.trim();
if (t) parts.push(t);
} else if (child.nodeType === 1 && !['SCRIPT', 'STYLE'].includes(child.tagName)) {
walk(child);
}
}
};
walk(el);
return parts.join(sep);
};
const th = (names) => [...doc.querySelectorAll('th')].find((t) => names.includes(text(t).replace(/:+$/, '')));
const headerIndex = (row, names) => {
const table = row.closest('table');
const head = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('th')) : null;
const texts = head ? [...head.querySelectorAll('th, td')].map((c) => text(c).replace(/:+$/, '')) : [];
return texts.findIndex((t) => names.includes(t));
};
const htmlText = (el) => {
const copy = el.cloneNode(true);
copy.querySelectorAll('script, style').forEach((n) => n.remove());
copy.querySelectorAll('br').forEach((br) => br.replaceWith('\n'));
copy.querySelectorAll('p, div, li, ul, ol, table, tr, td, th, h1, h2, h3, h4, h5, h6, blockquote, pre, dt, dd, hr, section, article')
.forEach((block) => { block.before('\n'); block.after('\n'); });
return copy.textContent
.split(/\r\n|[\n\r\v\f]/)
.map((line) => line.split(/\s+/).filter(Boolean).join(' '))
.filter(Boolean)
.join('\n');
};
// Older versions: the element after a "Beschreibung:" label. Current versions: the text block in the task panel,
// the .panel that holds the date table. A task without a description yields '' and never another block of the page.
const label = [...doc.querySelectorAll('div, th, dt, h4, h5, strong, label')]
.find((t) => ['Beschreibung', 'Description'].includes(text(t, '').replace(/:+$/, '')));
let box = label ? label.nextElementSibling : null;
if (!box && label && label.tagName === 'TH') {
for (let s = label.nextElementSibling; s; s = s.nextElementSibling) {
if (s.tagName === 'TD') { box = s; break; }
}
}
const dates = !box ? th(['Starttermin', 'Beginn', 'Start date', 'Abgabetermin', 'Due date']) : null;
if (dates) {
const panel = dates.closest('.panel');
box = panel ? [...panel.querySelectorAll('div.text-break-word')].find((d) => !d.closest('form')) || null : null;
} else if (!box) {
box = [...doc.querySelectorAll('div.text-break-word.p-3')].find((e) => !e.closest('form[name="submission"]')) || null;
}
let start = null;
const startNames = ['Starttermin', 'Beginn', 'Start date'];
const startTh = th(startNames);
const table = startTh ? startTh.closest('table') : null;
const row = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('td')) : null;
const col = row ? headerIndex(row, startNames) : -1;
const cells = row ? [...row.querySelectorAll('td')] : [];
const m = col !== -1 && col < cells.length
? /(\d{2})\.(\d{2})\.(\d{4})(?:\D+(\d{2}):(\d{2}))?/.exec(text(cells[col]))
: null;
if (m) start = `${m[3]}-${m[2]}-${m[1]}` + (m[4] ? `T${m[4]}:${m[5]}` : '');
return { description: box ? htmlText(box) : '', start };
}
/**
* Provided files of a task page: [{ href, filename }]. Only the "provided files" form, never own submission or
* teacher feedback. Per row: the download link if there is one (current versions also put a view link and a menu
* toggle in each row), the file name from the row's name link. Port of parse_attachment_links in iserv.py.
*/
export function parseAttachmentLinks(html) {
const doc = new DOMParser().parseFromString(html, 'text/html');
const form = doc.querySelector('form[name="iserv_exercise_attachment"]');
const out = [];
for (const row of form ? form.querySelectorAll('tr') : []) {
const anchors = [...row.querySelectorAll('a[href]')].filter(
(a) => !['/', '#'].includes(a.getAttribute('href')) && !a.hasAttribute('data-toggle'),
);
if (anchors.length === 0) continue;
const pick = anchors.find((a) => a.getAttribute('href').includes('/download/')) || anchors[0];
const named = row.querySelector('a.text-break-word') || pick;
out.push({ href: pick.getAttribute('href'), filename: (named.textContent || '').replace(/\s+/g, ' ').trim() });
}
return out;
}
/** 'dd.mm.yyyy[ hh:mm]' somewhere in a text to 'yyyy-mm-dd[Thh:mm]', null if there is none. Plain Node, not for page.evaluate. */
export function isoFromText(text) {
const m = /(\d{1,2})\.(\d{1,2})\.(\d{4})(?:\D+(\d{1,2}):(\d{2}))?/.exec(String(text || ''));
if (!m) return null;
const [, d, mo, y, h, mi] = m;
const date = `${y}-${mo.padStart(2, '0')}-${d.padStart(2, '0')}`;
return h ? `${date}T${h.padStart(2, '0')}:${mi}` : date;
}
|