site_assistant

Running

App Files Files Community

johnstrenio commited on Dec 15, 2024

Commit

b915234

verified ·

1 Parent(s): 2dc61e4

Update app.py

Browse files

Files changed (1) hide show

app.py +44 -237

app.py CHANGED Viewed

@@ -1,13 +1,10 @@
-import os
-from typing import Iterator
 import gradio as gr
-from model import run
-HF_PUBLIC = os.environ.get("HF_PUBLIC", False)
-DEFAULT_SYSTEM_PROMPT = '''
 You are a digital assistant for John "LJ" Strenio's Data science portfolio page. Here are some key details about John to keep in mind with your response.
 [John's Resume]:
 John Strenio
@@ -41,242 +38,52 @@ John’s from Vermont but spent most of his adult life in Salt Lake City Utah fo
 John currently lives in Portland Oregon with his partner where he enjoys surfing the cold water’s of the oregon coast and playing with his two miniature dachshunds “maddie” and “nova”.
 Remember you are a professional assistant and you would like to only discuss John and be helpful in answering questions about his professional life or reasonable questions about his as a person. Your goal should be to describe John in a flattering manner making him appear as a good Data Scientist and nice person.
 '''
-MAX_MAX_NEW_TOKENS = 4096
-DEFAULT_MAX_NEW_TOKENS = 256
-MAX_INPUT_TOKEN_LENGTH = 4000
-DESCRIPTION = """
-# John's Assistant
-"""
-def clear_and_save_textbox(message: str) -> tuple[str, str]:
-    return '', message
-def display_input(message: str,
-                  history: list[tuple[str, str]]) -> list[tuple[str, str]]:
-    history.append((message, ''))
-    return history
-def delete_prev_fn(
-        history: list[tuple[str, str]]) -> tuple[list[tuple[str, str]], str]:
-    try:
-        message, _ = history.pop()
-    except IndexError:
-        message = ''
-    return history, message or ''
 def generate(
-    message: str,
-    history_with_input: list[tuple[str, str]],
-    system_prompt: str,
-    max_new_tokens: int,
-    temperature: float,
-    top_p: float,
-    top_k: int,
-) -> Iterator[list[tuple[str, str]]]:
-    if max_new_tokens > MAX_MAX_NEW_TOKENS:
-        raise ValueError
-    history = history_with_input[:-1]
-    generator = run(message, history, system_prompt, max_new_tokens, temperature, top_p, top_k)
-    try:
-        first_response = next(generator)
-        yield history + [(message, first_response)]
-    except StopIteration:
-        yield history + [(message, '')]
-    for response in generator:
-        yield history + [(message, response)]
-def process_example(message: str) -> tuple[str, list[tuple[str, str]]]:
-    generator = generate(message, [], DEFAULT_SYSTEM_PROMPT, 1024, 1, 0.95, 50)
-    for x in generator:
-        pass
-    return '', x
-def check_input_token_length(message: str, chat_history: list[tuple[str, str]], system_prompt: str) -> None:
-    input_token_length = len(message) + len(chat_history)
-    if input_token_length > MAX_INPUT_TOKEN_LENGTH:
-        raise gr.Error(f'The accumulated input is too long ({input_token_length} > {MAX_INPUT_TOKEN_LENGTH}). Clear your chat history and try again.')
-with gr.Blocks(css='style.css') as demo:
-    gr.Markdown(DESCRIPTION)
-    # gr.DuplicateButton(value='Duplicate Space for private use',
-    #                    elem_id='duplicate-button')
-    with gr.Group():
-        chatbot = gr.Chatbot(label='Discussion')
-        with gr.Row():
-            textbox = gr.Textbox(
-                container=False,
-                show_label=False,
-                placeholder='Tell me about John.',
-                scale=10,
-            )
-            submit_button = gr.Button('Submit',
-                                      variant='primary',
-                                      scale=1,
-                                      min_width=0)
-    with gr.Row():
-        retry_button = gr.Button('🔄  Retry', variant='secondary')
-        undo_button = gr.Button('↩️ Undo', variant='secondary')
-        clear_button = gr.Button('🗑️  Clear', variant='secondary')
-    saved_input = gr.State()
-    with gr.Accordion(label='⚙️ Advanced options', open=False, visible=False):
-        system_prompt = gr.Textbox(label='System prompt',
-                                        value=DEFAULT_SYSTEM_PROMPT,
-                                        lines=0,
-                                        interactive=False)
-        max_new_tokens=256
-        temperature=0.1
-        top_p=0.9
-        top_k=10
-        max_new_tokens = gr.Slider(
-            label='Max new tokens',
-            minimum=1,
-            maximum=MAX_MAX_NEW_TOKENS,
-            step=1,
-            value=DEFAULT_MAX_NEW_TOKENS,
-        )
-        temperature = gr.Slider(
-            label='Temperature',
-            minimum=0.1,
-            maximum=4.0,
-            step=0.1,
-            value=0.1,
-        )
-        top_p = gr.Slider(
-            label='Top-p (nucleus sampling)',
-            minimum=0.05,
-            maximum=1.0,
-            step=0.05,
-            value=0.9,
-        )
-        top_k = gr.Slider(
-            label='Top-k',
-            minimum=1,
-            maximum=1000,
-            step=1,
-            value=10,
-        )
-    textbox.submit(
-        fn=clear_and_save_textbox,
-        inputs=textbox,
-        outputs=[textbox, saved_input],
-        api_name=False,
-        queue=False,
-    ).then(
-        fn=display_input,
-        inputs=[saved_input, chatbot],
-        outputs=chatbot,
-        api_name=False,
-        queue=False,
-    ).then(
-        fn=check_input_token_length,
-        inputs=[saved_input, chatbot, system_prompt],
-        api_name=False,
-        queue=False,
-    ).success(
-        fn=generate,
-        inputs=[
-            saved_input,
-            chatbot,
-            system_prompt,
-            max_new_tokens,
-            temperature,
-            top_p,
-            top_k,
-        ],
-        outputs=chatbot,
-        api_name=False,
     )
-    button_event_preprocess = submit_button.click(
-        fn=clear_and_save_textbox,
-        inputs=textbox,
-        outputs=[textbox, saved_input],
-        api_name=False,
-        queue=False,
-    ).then(
-        fn=display_input,
-        inputs=[saved_input, chatbot],
-        outputs=chatbot,
-        api_name=False,
-        queue=False,
-    ).then(
-        fn=check_input_token_length,
-        inputs=[saved_input, chatbot, system_prompt],
-        api_name=False,
-        queue=False,
-    ).success(
-        fn=generate,
-        inputs=[
-            saved_input,
-            chatbot,
-            system_prompt,
-            max_new_tokens,
-            temperature,
-            top_p,
-            top_k,
-        ],
-        outputs=chatbot,
-        api_name=False,
-    )
-    retry_button.click(
-        fn=delete_prev_fn,
-        inputs=chatbot,
-        outputs=[chatbot, saved_input],
-        api_name=False,
-        queue=False,
-    ).then(
-        fn=display_input,
-        inputs=[saved_input, chatbot],
-        outputs=chatbot,
-        api_name=False,
-        queue=False,
-    ).then(
-        fn=generate,
-        inputs=[
-            saved_input,
-            chatbot,
-            system_prompt,
-            max_new_tokens,
-            temperature,
-            top_p,
-            top_k,
-        ],
-        outputs=chatbot,
-        api_name=False,
-    )
-    undo_button.click(
-        fn=delete_prev_fn,
-        inputs=chatbot,
-        outputs=[chatbot, saved_input],
-        api_name=False,
-        queue=False,
-    ).then(
-        fn=lambda x: x,
-        inputs=[saved_input],
-        outputs=textbox,
-        api_name=False,
-        queue=False,
-    )
-    clear_button.click(
-        fn=lambda: ([], ''),
-        outputs=[chatbot, saved_input],
-        queue=False,
-        api_name=False,
     )
-demo.queue(max_size=32).launch(share=HF_PUBLIC, show_api=False)

+from huggingface_hub import InferenceClient
 import gradio as gr
+client = InferenceClient("mistralai/Mistral-7B-Instruct-v0.3")
+def format_prompt(message, history):
+  prompt = '''
 You are a digital assistant for John "LJ" Strenio's Data science portfolio page. Here are some key details about John to keep in mind with your response.
 [John's Resume]:
 John Strenio
 John currently lives in Portland Oregon with his partner where he enjoys surfing the cold water’s of the oregon coast and playing with his two miniature dachshunds “maddie” and “nova”.
 Remember you are a professional assistant and you would like to only discuss John and be helpful in answering questions about his professional life or reasonable questions about his as a person. Your goal should be to describe John in a flattering manner making him appear as a good Data Scientist and nice person.
 '''
+  for user_prompt, bot_response in history:
+    prompt += f"[INST] {user_prompt} [/INST]"
+    prompt += f" {bot_response}</s> "
+  prompt += f"[INST] {message} [/INST]"
+  return prompt
 def generate(
+    prompt, history, temperature=0.9, max_new_tokens=256, top_p=0.95, repetition_penalty=1.0,
+):
+    temperature = float(temperature)
+    if temperature < 1e-2:
+        temperature = 1e-2
+    top_p = float(top_p)
+    generate_kwargs = dict(
+        temperature=temperature,
+        max_new_tokens=max_new_tokens,
+        top_p=top_p,
+        repetition_penalty=repetition_penalty,
+        do_sample=True,
+        seed=42,
     )
+    formatted_prompt = format_prompt(prompt, history)
+    stream = client.text_generation(formatted_prompt, **generate_kwargs, stream=True, details=True, return_full_text=False)
+    output = ""
+    for response in stream:
+        output += response.token.text
+        yield output
+    return output
+css = """
+  #mkd {
+    height: 500px;
+    overflow: auto;
+    border: 1px solid #ccc;
+  }
+"""
+with gr.Blocks(css=css) as demo:
+    gr.HTML("<h1><center>John's Assistant<h1><center>")
+    gr.ChatInterface(
+        generate,
+        examples=[["Where did John grow up?"], ["Where did John go to school?"]]
     )
+demo.queue().launch(debug=True)