rag-tool

Sleeping

App Files Files Community

Chris4K commited on Jan 20

Commit

c33d1d0

•

1 Parent(s): 229718e

Update app.py

Browse files

Files changed (1) hide show

app.py +9 -27

app.py CHANGED Viewed

@@ -1,21 +1,18 @@
-import os
-#!pip install -q gradio langchain pypdf chromadb
 import gradio as gr
 from dotenv import load_dotenv
-from PyPDF2 import PdfReader
-from langchain.vectorstores import Chroma
-from langchain.vectorstores import FAISS
 from langchain.document_loaders import PyPDFLoader
 from langchain.text_splitter import CharacterTextSplitter
 from langchain.embeddings import HuggingFaceInferenceAPIEmbeddings
 from langchain.embeddings import HuggingFaceBgeEmbeddings
-from langchain.memory import ConversationBufferMemory
-from langchain.chains import ConversationalRetrievalChain
-from langchain.llms import HuggingFaceHub
 # Use Hugging Face Inference API embeddings
-inference_api_key = os.environ['HF']
 api_hf_embeddings = HuggingFaceInferenceAPIEmbeddings(
     api_key=inference_api_key,
     model_name="sentence-transformers/all-MiniLM-l6-v2"
@@ -24,17 +21,11 @@ api_hf_embeddings = HuggingFaceInferenceAPIEmbeddings(
 # Load and process the PDF files
 loader = PyPDFLoader("./new_papers/ALiBi.pdf")
 documents = loader.load()
-print("-----------")
-print(documents[0])
-print("-----------")
 # Split the documents into chunks and embed them using the HfApiEmbeddingTool
 text_splitter = CharacterTextSplitter(chunk_size=100, chunk_overlap=0)
 vdocuments = text_splitter.split_documents(documents)
 model = "BAAI/bge-base-en-v1.5"
 encode_kwargs = {
     "normalize_embeddings": True
@@ -42,17 +33,9 @@ encode_kwargs = {
 embeddings = HuggingFaceBgeEmbeddings(
     model_name=model, encode_kwargs=encode_kwargs, model_kwargs={"device": "cpu"}
 )
-api_db = FAISS.from_texts(texts=vdocuments, embedding=embeddings)
-api_db.as_retriever.similarity("What is ICD?")
-# Extract the embedding arrays from the PDF documents
-#embeddings = []
-#for doc in vdocuments:
-#    embeddings.extend(api_hf_embeddings.get_embeddings(doc))
-# Create Chroma vector store for API embeddings
-#api_db = Chroma.from_documents(vdocuments, HfApiEmbeddingRetriever, collection_name="api-collection")
 # Define the PDF retrieval function
 def pdf_retrieval(query):
@@ -60,7 +43,6 @@ def pdf_retrieval(query):
     response = api_db.similarity_search(query)
     return response
-# Create Gradio interface for the API retriever
 # Create Gradio interface for the API retriever
 api_tool = gr.Interface(
     fn=pdf_retrieval,
@@ -72,4 +54,4 @@ api_tool = gr.Interface(
 )
 # Launch the Gradio interface
-api_tool.launch()

+import os
 import gradio as gr
 from dotenv import load_dotenv
+from langchain.vectorstores.faiss import FAISS  # Import FAISS
+from langchain.vectorstores.chroma import Chroma  # Import Chroma
 from langchain.document_loaders import PyPDFLoader
 from langchain.text_splitter import CharacterTextSplitter
 from langchain.embeddings import HuggingFaceInferenceAPIEmbeddings
 from langchain.embeddings import HuggingFaceBgeEmbeddings
+# Load environment variables
+load_dotenv()
 # Use Hugging Face Inference API embeddings
+inference_api_key = os.getenv('HF')  # Use getenv to retrieve environment variable
 api_hf_embeddings = HuggingFaceInferenceAPIEmbeddings(
     api_key=inference_api_key,
     model_name="sentence-transformers/all-MiniLM-l6-v2"
 # Load and process the PDF files
 loader = PyPDFLoader("./new_papers/ALiBi.pdf")
 documents = loader.load()
 # Split the documents into chunks and embed them using the HfApiEmbeddingTool
 text_splitter = CharacterTextSplitter(chunk_size=100, chunk_overlap=0)
 vdocuments = text_splitter.split_documents(documents)
 model = "BAAI/bge-base-en-v1.5"
 encode_kwargs = {
     "normalize_embeddings": True
 embeddings = HuggingFaceBgeEmbeddings(
     model_name=model, encode_kwargs=encode_kwargs, model_kwargs={"device": "cpu"}
 )
+# Create FAISS vector store for API embeddings
+api_db = FAISS.from_texts(texts=vdocuments, embedding=embeddings)
 # Define the PDF retrieval function
 def pdf_retrieval(query):
     response = api_db.similarity_search(query)
     return response
 # Create Gradio interface for the API retriever
 api_tool = gr.Interface(
     fn=pdf_retrieval,
 )
 # Launch the Gradio interface
+api_tool.launch()