Downloads · 30 days
0
Maqboolzia/Leagalbase-Chatbot
Leagalbase-Chatbot is a machine learning model from Maqboolzia. Use it for the machine learning task on the model card, and read the license before you ship it in a product.
import numpy as np linear algebra import pandas as pd data processing, CSV file I/O (e.g. pd.readcsv)
Downloads · 30 days
0
Access
Public
Updated Jan 3, 2025
Repo size
—
Likes
0
Public
Click a slice to open those files.
.md5 KB · 77%
From the Hugging Face model README
import numpy as np # linear algebra import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)
import os for dirname, _, filenames in os.walk('/kaggle/input'): for filename in filenames: print(os.path.join(dirname, filename))
pip install --quiet PyPDF2 gradio groq langchain langchain-community langchain-huggingface sentence-transformers transformers faiss-gpu import gradio as gr from langchain.prompts import PromptTemplate from langchain_community.vectorstores import FAISS from langchain_huggingface import HuggingFaceEmbeddings from PyPDF2 import PdfReader from groq import Groq import os import json # For handling JSON data from PyPDF2 import PdfReader from langchain.embeddings import HuggingFaceEmbeddings from langchain.vectorstores import FAISS import gradio as gr
pdf_path = "/kaggle/input/ai-assignment/pdf_data.json" # Updated path try: with open(pdf_path, "r") as f: data = json.load(f)
# Handle different possible structures of the JSON file
if isinstance(data, list):
# Assuming the text data is in the first element of the list
document_text = data[0].get("text", "") if isinstance(data[0], dict) else ""
elif isinstance(data, dict):
document_text = data.get("text", "")
else:
raise ValueError("Unexpected JSON structure: expected a list or dict.")
except Exception as e: raise FileNotFoundError(f"Error reading JSON file: {e}")
if not document_text.strip(): raise ValueError("The document text is empty. Please check the file content.")
sections = document_text.split("\n\n") if not sections: raise ValueError("No sections found in the document text. Please check the formatting.")
embedding_model_name = "sentence-transformers/all-MiniLM-L6-v2" try: embeddings_model = HuggingFaceEmbeddings(model_name=embedding_model_name) except Exception as e: raise RuntimeError(f"Error loading embedding model: {e}")
try: vector_store = FAISS.from_texts( texts=sections, embedding=embeddings_model ) except Exception as e: raise RuntimeError(f"Error creating FAISS vector store: {e}")
prompt = """ You are a chatbot designed to answer questions about my following CV: {retrieved_data}
User Query: {user_query} """
try: from groq.api import Groq client = Groq(api_key="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") except Exception as e: raise RuntimeError(f"Error initializing Groq client: {e}")
def process_query(user_query): try: retrieved_docs = vector_store.similarity_search(user_query, k=3) retrieved_data = "\n".join([doc.page_content for doc in retrieved_docs]) formatted_prompt = prompt.format(retrieved_data=retrieved_data, user_query=user_query)
completion = client.chat.completions.create(
model="llama3-70b-8192",
messages=[{"role": "user", "content": formatted_prompt}],
temperature=0.5,
max_tokens=500,
top_p=0.85,
stream=False,
stop=None,
)
response = completion.choices[0].message.content
return response
except Exception as e:
return f"Error processing query: {e}"
with gr.Blocks() as app: gr.Markdown("""<h1>CV Chatbot</h1> Use this chatbot to ask questions about my CV. """)
# CV Chatbot
query_input = gr.Textbox(
label="Ask a Question",
placeholder="Enter your question about the CV...",
lines=3
)
query_output = gr.Textbox(
label="Response",
placeholder="The chatbot's response will appear here...",
lines=10,
interactive=False
)
ask_button = gr.Button("Ask")
ask_button.click(
fn=process_query,
inputs=query_input,
outputs=query_output
)
gr.Markdown("""<h3>Disclaimer</h3>
The chatbot provides responses based on my CV content. Verify critical details independently.
""")
app.launch()