Files
datenbanken-llm/lib.py
T
2024-12-11 23:39:46 +01:00

83 lines
2.2 KiB
Python

import os, pymupdf, whisper, json
import cv2
import psycopg2
import pymongo
import pytesseract
import ollama
from aiohttp.web_response import json_response
import ollama_conn
def read_files(path, output, filetypes=None):
if filetypes is None:
filetypes = ["pdf", "txt", "png", "jpg", "mp3"]
for root, dirs, files in os.walk(path):
for file in files:
if (file_type := file.split(".")[-1]) in filetypes:
output.append((file_type, os.path.join(path, file), file))
for folder in dirs:
read_files(os.path.join(path, folder), output)
break
def extract_pdf_content(pdf_datei):
doc = pymupdf.open(pdf_datei)
a = ""
for page in doc:
a += page.get_text()
return a
def extract_mp3_content(mp3_datei):
model = whisper.load_model('tiny')
result = model.transcribe(str(mp3_datei), language='de', verbose=True)
# with open('transcript.json', "w") as f:
# json.dump(result['text'], f, indent=4)
return result["text"]
def extract_image_content(path):
pytesseract.pytesseract.tesseract_cmd = r'C:\Program Files\Tesseract-OCR\tesseract.exe'
img = cv2.imread(path)
return pytesseract.image_to_string(img)
def prompt_embedding(prompt):
embedding = ollama.embeddings(model="mxbai-embed-large", prompt=prompt)
conn = psycopg2.connect(
dbname="embeddings",
user="python",
password="PasswordPassword123",
host="localhost",
port="5555"
)
cur = conn.cursor()
cur.execute(f"select id, filepath, embedding <-> %s::vector as distance from dbtable order by distance limit 3;", (embedding["embedding"],))
result = cur.fetchall()
res = get_data_from_mongo(result)
print(ollama_conn.ask_ollama(prompt, res))
conn.close()
def get_data_from_mongo(dataset):
result_string = ""
client = pymongo.MongoClient('mongodb://python:PasswordPassword123@localhost:27017/')
mongodb = client['document_table']
collection = mongodb['documents']
for set in dataset:
json_data = collection.find_one({"doc_id": set[0]})
result_string += json_data["chunk_content"]
return result_string