| const { v4 } = require("uuid"); |
| const { |
| createdDate, |
| trashFile, |
| writeToServerDocuments, |
| } = require("../../../utils/files"); |
| const { tokenizeString } = require("../../../utils/tokenizer"); |
| const { default: slugify } = require("slugify"); |
| const PDFLoader = require("./PDFLoader"); |
| const OCRLoader = require("../../../utils/OCRLoader"); |
|
|
| async function asPdf({ |
| fullFilePath = "", |
| filename = "", |
| options = {}, |
| metadata = {}, |
| }) { |
| const pdfLoader = new PDFLoader(fullFilePath, { |
| splitPages: true, |
| }); |
|
|
| console.log(`-- Working ${filename} --`); |
| const pageContent = []; |
| let docs = await pdfLoader.load(); |
|
|
| if (docs.length === 0) { |
| console.log( |
| `[asPDF] No text content found for ${filename}. Will attempt OCR parse.` |
| ); |
| docs = await new OCRLoader({ |
| targetLanguages: options?.ocr?.langList, |
| }).ocrPDF(fullFilePath); |
| } |
|
|
| for (const doc of docs) { |
| console.log( |
| `-- Parsing content from pg ${ |
| doc.metadata?.loc?.pageNumber || "unknown" |
| } --` |
| ); |
| if (!doc.pageContent || !doc.pageContent.length) continue; |
| pageContent.push(doc.pageContent); |
| } |
|
|
| if (!pageContent.length) { |
| console.error(`[asPDF] Resulting text content was empty for ${filename}.`); |
| trashFile(fullFilePath); |
| return { |
| success: false, |
| reason: `No text content found in ${filename}.`, |
| documents: [], |
| }; |
| } |
|
|
| const content = pageContent.join(""); |
| const data = { |
| id: v4(), |
| url: "file://" + fullFilePath, |
| title: metadata.title || filename, |
| docAuthor: |
| metadata.docAuthor || |
| docs[0]?.metadata?.pdf?.info?.Creator || |
| "no author found", |
| description: |
| metadata.description || |
| docs[0]?.metadata?.pdf?.info?.Title || |
| "No description found.", |
| docSource: metadata.docSource || "pdf file uploaded by the user.", |
| chunkSource: metadata.chunkSource || "", |
| published: createdDate(fullFilePath), |
| wordCount: content.split(" ").length, |
| pageContent: content, |
| token_count_estimate: tokenizeString(content), |
| }; |
|
|
| const document = writeToServerDocuments({ |
| data, |
| filename: `${slugify(filename)}-${data.id}`, |
| options: { parseOnly: options.parseOnly }, |
| }); |
| trashFile(fullFilePath); |
| console.log(`[SUCCESS]: ${filename} converted & ready for embedding.\n`); |
| return { success: true, reason: null, documents: [document] }; |
| } |
|
|
| module.exports = asPdf; |
|
|