Spaces:

Moha782
/

gen-ai-project

Sleeping

App Files Files Community

Moha782 commited on Jun 26, 2024

Commit

34e179e

verified ·

1 Parent(s): 6c6bd03

Update app.py

Browse files

Files changed (1) hide show

app.py +18 -13

app.py CHANGED Viewed

@@ -2,7 +2,6 @@ import gradio as gr
 from huggingface_hub import InferenceClient
 from langchain_community.vectorstores.faiss import FAISS
 from langchain.chains import RetrievalQA
-from langchain_huggingface import HuggingFacePipeline
 # Load the vector store from the saved index files
 vector_store = FAISS.load_local("db.index", embeddings=None, allow_dangerous_deserialization=True)
@@ -10,17 +9,17 @@ vector_store = FAISS.load_local("db.index", embeddings=None, allow_dangerous_des
 # Load the model using InferenceClient
 client = InferenceClient("HuggingFaceH4/zephyr-7b-beta")
-# Initialize the HuggingFacePipeline LLM
-llm = HuggingFacePipeline(client=client, model_kwargs={"temperature": None, "top_p": None})
 # Initialize the RetrievalQA chain
-qa = RetrievalQA.from_chain_type(llm=llm, chain_type="stuff", retriever=vector_store.as_retriever())
-def respond(message, history, system_message, max_tokens, temperature, top_p):
-    # Update the temperature and top_p values for the LLM
-    llm.model_kwargs["temperature"] = temperature
-    llm.model_kwargs["top_p"] = top_p
     messages = [{"role": "system", "content": system_message}]
     for val in history:
@@ -43,10 +42,16 @@ For information on how to customize the ChatInterface, peruse the gradio docs: h
 demo = gr.ChatInterface(
     respond,
     additional_inputs=[
-        gr.Textbox(value="You are a helpful car configuration assistant, specifically you are the assistant for Apex Customs (https://www.apexcustoms.com/). Given the user's input, provide suggestions for car models, colors, and customization options. Be creative and conversational in your responses. You should remember the user car model and tailor your answers accordingly. (You must not generate the next question of the user yourself, you only have to answer.) \n\nUser: ", label="System message"),
         gr.Slider(minimum=1, maximum=2048, value=512, step=1, label="Max new tokens"),
         gr.Slider(minimum=0.1, maximum=4.0, value=0.7, step=0.1, label="Temperature"),
-        gr.Slider(minimum=0.1, maximum=1.0, value=0.95, step=0.05, label="Top-p (nucleus sampling)"),
     ],
 )

 from huggingface_hub import InferenceClient
 from langchain_community.vectorstores.faiss import FAISS
 from langchain.chains import RetrievalQA
 # Load the vector store from the saved index files
 vector_store = FAISS.load_local("db.index", embeddings=None, allow_dangerous_deserialization=True)
 # Load the model using InferenceClient
 client = InferenceClient("HuggingFaceH4/zephyr-7b-beta")
 # Initialize the RetrievalQA chain
+qa = RetrievalQA.from_chain_type(client=client, chain_type="stuff", retriever=vector_store.as_retriever())
+def respond(
+    message,
+    history: list[tuple[str, str]],
+    system_message,
+    max_tokens,
+    temperature,
+    top_p,
+):
     messages = [{"role": "system", "content": system_message}]
     for val in history:
 demo = gr.ChatInterface(
     respond,
     additional_inputs=[
+        gr.Textbox(value="You are a helpful car configuration assistant, specifically you are the assistant for Apex Customs (https://www.apexcustoms.com/). Given the user's input, provide suggestions for car models, colors, and customization options. Be creative and conversational in your responses. You should remember the user car model and tailor your answers accordingly. \n\nUser: ", label="System message"),
         gr.Slider(minimum=1, maximum=2048, value=512, step=1, label="Max new tokens"),
         gr.Slider(minimum=0.1, maximum=4.0, value=0.7, step=0.1, label="Temperature"),
+        gr.Slider(
+            minimum=0.1,
+            maximum=1.0,
+            value=0.95,
+            step=0.05,
+            label="Top-p (nucleus sampling)",
+        ),
     ],
 )