Spaces:
Paused
Paused
Download src/api/iserv-parse.js from Luca448/APP-Backend: direct link, hf CLI and curl.
- Browser
- Download file 8.66 kB
-
https://huggingface.co/spaces/Luca448/APP-Backend/resolve/main/src/api/iserv-parse.js
- Command line
-
hf download hf://spaces/Luca448/APP-Backend/src/api/iserv-parse.js
-
curl -L -o iserv-parse.js https://huggingface.co/spaces/Luca448/APP-Backend/resolve/main/src/api/iserv-parse.js
8.66 kB
| // Parsers that run INSIDE the browser page: puppeteer's page.evaluate serializes a function, so every function here | |
| // is self-contained (its helpers sit inside it). They get the HTML as the server sent it and read it through | |
| // DOMParser. Page scripts do not run in that document and cannot change what is read: in the live DOM, Bootstrap | |
| // tooltips move the title attribute of the status icon away, so a done task would look open. | |
| // | |
| // Port of the parsers of the IServ filter project (iserv.py: parse_task_list, parse_description, parse_start), | |
| // which are tested against fake pages of the older and the current IServ layout. Keep both in step. | |
| /** | |
| * Task list: { "<task id>": { tags, start, due, done, expired } } | |
| * tags: text of the "Tags" column ('' if empty), start: 'yyyy-mm-dd' (older layouts only), due: 'yyyy-mm-ddThh:mm', | |
| * done: the status icon says done/submitted, expired: it says expired (guessed pattern, only "Erledigt" was seen | |
| * live). Columns are found by their header, never by position (the column after the due date is the teacher's | |
| * feedback). | |
| */ | |
| export function parseListHtml(html) { | |
| const doc = new DOMParser().parseFromString(html, 'text/html'); | |
| // BeautifulSoup's get_text(sep, strip=True): the stripped text nodes, joined | |
| const text = (el, sep = ' ') => { | |
| const parts = []; | |
| const walk = (node) => { | |
| for (const child of node.childNodes) { | |
| if (child.nodeType === 3) { | |
| const t = child.textContent.trim(); | |
| if (t) parts.push(t); | |
| } else if (child.nodeType === 1 && !['SCRIPT', 'STYLE'].includes(child.tagName)) { | |
| walk(child); | |
| } | |
| } | |
| }; | |
| walk(el); | |
| return parts.join(sep); | |
| }; | |
| const headerIndex = (row, names) => { | |
| const table = row.closest('table'); | |
| const head = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('th')) : null; | |
| const texts = head ? [...head.querySelectorAll('th, td')].map((c) => text(c).replace(/:+$/, '')) : []; | |
| return texts.findIndex((t) => names.includes(t)); | |
| }; | |
| const tagsOf = (cell) => { | |
| const all = text(cell); | |
| const italic = [...cell.querySelectorAll('i')].map((i) => text(i)).join(' '); | |
| return all === italic || ['keine', 'none'].includes(all.replace(/^[()]+|[()]+$/g, '').toLowerCase()) ? '' : all; | |
| }; | |
| const doneRe = /erledigt|abgegeben|done|submitted|completed/i; | |
| const expiredRe = /abgelaufen|expired/i; | |
| const out = {}; | |
| for (const a of doc.querySelectorAll('a[href*="/exercise/show/"]')) { | |
| const m = /\/exercise\/show\/(\d+)/.exec(a.getAttribute('href')); | |
| const row = a.closest('tr'); | |
| if (!m || !row || m[1] in out) continue; | |
| const cells = [...row.querySelectorAll('td')]; | |
| let start = null; | |
| let due = null; | |
| for (const td of cells) { | |
| const ds = td.getAttribute('data-sort') || ''; | |
| if (/^\d{8}$/.test(ds) && start === null) { | |
| start = `${ds.slice(0, 4)}-${ds.slice(4, 6)}-${ds.slice(6)}`; | |
| } else if (/^\d{14}$/.test(ds) && due === null) { | |
| due = `${ds.slice(0, 4)}-${ds.slice(4, 6)}-${ds.slice(6, 8)}T${ds.slice(8, 10)}:${ds.slice(10, 12)}`; | |
| } | |
| } | |
| const col = headerIndex(row, ['Tags']); | |
| const titles = [...row.querySelectorAll('[title]')].map((e) => e.getAttribute('title')).join(' '); | |
| out[m[1]] = { | |
| tags: col !== -1 && col < cells.length ? tagsOf(cells[col]) : '', | |
| start, | |
| due, | |
| done: doneRe.test(titles), | |
| expired: expiredRe.test(titles), | |
| }; | |
| } | |
| return out; | |
| } | |
| /** | |
| * Task page: { description, start } | |
| * description: only the task's own description (never own submission, teacher feedback, creator, participants), | |
| * as text with the paragraphs and line breaks of the page. start: 'yyyy-mm-dd[Thh:mm]' from the "Starttermin" column. | |
| */ | |
| export function parseTaskHtml(html) { | |
| const doc = new DOMParser().parseFromString(html, 'text/html'); | |
| const text = (el, sep = ' ') => { | |
| const parts = []; | |
| const walk = (node) => { | |
| for (const child of node.childNodes) { | |
| if (child.nodeType === 3) { | |
| const t = child.textContent.trim(); | |
| if (t) parts.push(t); | |
| } else if (child.nodeType === 1 && !['SCRIPT', 'STYLE'].includes(child.tagName)) { | |
| walk(child); | |
| } | |
| } | |
| }; | |
| walk(el); | |
| return parts.join(sep); | |
| }; | |
| const th = (names) => [...doc.querySelectorAll('th')].find((t) => names.includes(text(t).replace(/:+$/, ''))); | |
| const headerIndex = (row, names) => { | |
| const table = row.closest('table'); | |
| const head = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('th')) : null; | |
| const texts = head ? [...head.querySelectorAll('th, td')].map((c) => text(c).replace(/:+$/, '')) : []; | |
| return texts.findIndex((t) => names.includes(t)); | |
| }; | |
| const htmlText = (el) => { | |
| const copy = el.cloneNode(true); | |
| copy.querySelectorAll('script, style').forEach((n) => n.remove()); | |
| copy.querySelectorAll('br').forEach((br) => br.replaceWith('\n')); | |
| copy.querySelectorAll('p, div, li, ul, ol, table, tr, td, th, h1, h2, h3, h4, h5, h6, blockquote, pre, dt, dd, hr, section, article') | |
| .forEach((block) => { block.before('\n'); block.after('\n'); }); | |
| return copy.textContent | |
| .split(/\r\n|[\n\r\v\f]/) | |
| .map((line) => line.split(/\s+/).filter(Boolean).join(' ')) | |
| .filter(Boolean) | |
| .join('\n'); | |
| }; | |
| // Older versions: the element after a "Beschreibung:" label. Current versions: the text block in the task panel, | |
| // the .panel that holds the date table. A task without a description yields '' and never another block of the page. | |
| const label = [...doc.querySelectorAll('div, th, dt, h4, h5, strong, label')] | |
| .find((t) => ['Beschreibung', 'Description'].includes(text(t, '').replace(/:+$/, ''))); | |
| let box = label ? label.nextElementSibling : null; | |
| if (!box && label && label.tagName === 'TH') { | |
| for (let s = label.nextElementSibling; s; s = s.nextElementSibling) { | |
| if (s.tagName === 'TD') { box = s; break; } | |
| } | |
| } | |
| const dates = !box ? th(['Starttermin', 'Beginn', 'Start date', 'Abgabetermin', 'Due date']) : null; | |
| if (dates) { | |
| const panel = dates.closest('.panel'); | |
| box = panel ? [...panel.querySelectorAll('div.text-break-word')].find((d) => !d.closest('form')) || null : null; | |
| } else if (!box) { | |
| box = [...doc.querySelectorAll('div.text-break-word.p-3')].find((e) => !e.closest('form[name="submission"]')) || null; | |
| } | |
| let start = null; | |
| const startNames = ['Starttermin', 'Beginn', 'Start date']; | |
| const startTh = th(startNames); | |
| const table = startTh ? startTh.closest('table') : null; | |
| const row = table ? [...table.querySelectorAll('tr')].find((tr) => tr.querySelector('td')) : null; | |
| const col = row ? headerIndex(row, startNames) : -1; | |
| const cells = row ? [...row.querySelectorAll('td')] : []; | |
| const m = col !== -1 && col < cells.length | |
| ? /(\d{2})\.(\d{2})\.(\d{4})(?:\D+(\d{2}):(\d{2}))?/.exec(text(cells[col])) | |
| : null; | |
| if (m) start = `${m[3]}-${m[2]}-${m[1]}` + (m[4] ? `T${m[4]}:${m[5]}` : ''); | |
| return { description: box ? htmlText(box) : '', start }; | |
| } | |
| /** | |
| * Provided files of a task page: [{ href, filename }]. Only the "provided files" form, never own submission or | |
| * teacher feedback. Per row: the download link if there is one (current versions also put a view link and a menu | |
| * toggle in each row), the file name from the row's name link. Port of parse_attachment_links in iserv.py. | |
| */ | |
| export function parseAttachmentLinks(html) { | |
| const doc = new DOMParser().parseFromString(html, 'text/html'); | |
| const form = doc.querySelector('form[name="iserv_exercise_attachment"]'); | |
| const out = []; | |
| for (const row of form ? form.querySelectorAll('tr') : []) { | |
| const anchors = [...row.querySelectorAll('a[href]')].filter( | |
| (a) => !['/', '#'].includes(a.getAttribute('href')) && !a.hasAttribute('data-toggle'), | |
| ); | |
| if (anchors.length === 0) continue; | |
| const pick = anchors.find((a) => a.getAttribute('href').includes('/download/')) || anchors[0]; | |
| const named = row.querySelector('a.text-break-word') || pick; | |
| out.push({ href: pick.getAttribute('href'), filename: (named.textContent || '').replace(/\s+/g, ' ').trim() }); | |
| } | |
| return out; | |
| } | |
| /** 'dd.mm.yyyy[ hh:mm]' somewhere in a text to 'yyyy-mm-dd[Thh:mm]', null if there is none. Plain Node, not for page.evaluate. */ | |
| export function isoFromText(text) { | |
| const m = /(\d{1,2})\.(\d{1,2})\.(\d{4})(?:\D+(\d{1,2}):(\d{2}))?/.exec(String(text || '')); | |
| if (!m) return null; | |
| const [, d, mo, y, h, mi] = m; | |
| const date = `${y}-${mo.padStart(2, '0')}-${d.padStart(2, '0')}`; | |
| return h ? `${date}T${h.padStart(2, '0')}:${mi}` : date; | |
| } | |