From c5a8e9c5e3ab637e9fd18c3d1697a05ad7a8e406 Mon Sep 17 00:00:00 2001 From: Jonas Braus Date: Tue, 10 Dec 2024 19:41:01 +0100 Subject: [PATCH] =?UTF-8?q?blubbbl=C3=BCl=C3=BCll=C3=BC?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- lib.py | 22 ++++++++++++++++++++++ main.py | 31 +++++++++++++++++++++++++++++++ 2 files changed, 53 insertions(+) create mode 100644 lib.py create mode 100644 main.py diff --git a/lib.py b/lib.py new file mode 100644 index 0000000..f81a156 --- /dev/null +++ b/lib.py @@ -0,0 +1,22 @@ +import os, pymupdf + +def read_files(path, output, filetypes=None): + if filetypes is None: + filetypes = ["pdf", "txt", "png", "jpg", "mp3"] + + for root, dirs, files in os.walk(path): + for file in files: + if (file_type := file.split(".")[-1]) in filetypes: + output.append((file_type, os.path.join(path, file))) + + for folder in dirs: + read_files(os.path.join(path, folder), output) + + break + +def extract_pdf_content(pdf_datei): + doc = pymupdf.open(pdf_datei) + a = "" + for page in doc: + a += page.get_text() + return a \ No newline at end of file diff --git a/main.py b/main.py new file mode 100644 index 0000000..26e9928 --- /dev/null +++ b/main.py @@ -0,0 +1,31 @@ +import pymupdf +import os +import lib +import json + +def retrieve_file_contents(path): + file_extraction_functions = { + "pdf": lambda path: lib.extract_pdf_content(path), + "jpg": lambda path: "", + "png": lambda path: "", + "txt": lambda path: lib.extract_pdf_content(path), + "mp3": lambda path: "test mp3" + } + + lib.read_files(path, files := []) + + contents = [] + + for file in files: + content = file_extraction_functions[file[0]](file[1]) + contents.append({ + "type": file[0], + "path": file[1], + "content": content + }) + + return contents + +contents = retrieve_file_contents("C:\\SoftwareEng") + +print(json.dumps(contents, indent="\t")) \ No newline at end of file