| const { v4 } = require("uuid"); |
| const { tokenizeString } = require("../../utils/tokenizer"); |
| const { |
| createdDate, |
| trashFile, |
| writeToServerDocuments, |
| } = require("../../utils/files"); |
| const OCRLoader = require("../../utils/OCRLoader"); |
| const { default: slugify } = require("slugify"); |
|
|
| async function asImage({ |
| fullFilePath = "", |
| filename = "", |
| options = {}, |
| metadata = {}, |
| }) { |
| let content = await new OCRLoader({ |
| targetLanguages: options?.ocr?.langList, |
| }).ocrImage(fullFilePath); |
|
|
| if (!content?.length) { |
| console.error(`Resulting text content was empty for ${filename}.`); |
| trashFile(fullFilePath); |
| return { |
| success: false, |
| reason: `No text content found in ${filename}.`, |
| documents: [], |
| }; |
| } |
|
|
| console.log(`-- Working ${filename} --`); |
| const data = { |
| id: v4(), |
| url: "file://" + fullFilePath, |
| title: metadata.title || filename, |
| docAuthor: metadata.docAuthor || "Unknown", |
| description: metadata.description || "Unknown", |
| docSource: metadata.docSource || "image file uploaded by the user.", |
| chunkSource: metadata.chunkSource || "", |
| published: createdDate(fullFilePath), |
| wordCount: content.split(" ").length, |
| pageContent: content, |
| token_count_estimate: tokenizeString(content), |
| }; |
|
|
| const document = writeToServerDocuments({ |
| data, |
| filename: `${slugify(filename)}-${data.id}`, |
| options: { parseOnly: options.parseOnly }, |
| }); |
| trashFile(fullFilePath); |
| console.log(`[SUCCESS]: ${filename} converted & ready for embedding.\n`); |
| return { success: true, reason: null, documents: [document] }; |
| } |
|
|
| module.exports = asImage; |
|
|