hossamdaoud commited on
Commit
14d0c87
·
verified ·
1 Parent(s): d622bed

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +89 -1
app.py CHANGED
@@ -1,3 +1,91 @@
 
1
  import gradio as gr
2
 
3
- gr.load("models/mistralai/Mistral-7B-v0.1").launch()
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from huggingface_hub import InferenceClient
2
  import gradio as gr
3
 
4
+ client = InferenceClient("mistralai/mistralai/Mistral-7B-v0.1")
5
+ def format_prompt(message, history):
6
+ prompt = "<s>"
7
+ for user_prompt, bot_response in history:
8
+ prompt += f"[INST] {user_prompt} [/INST]"
9
+ prompt += f" {bot_response}</s> "
10
+ prompt += f"[INST] {message} [/INST]"
11
+ return prompt
12
+
13
+ def generate(
14
+ prompt, history, system_prompt, temperature=0.2, max_new_tokens=512, top_p=0.1, repetition_penalty=1.0,
15
+ ):
16
+ temperature = float(temperature)
17
+ if temperature < 1e-2:
18
+ temperature = 1e-2
19
+ top_p = float(top_p)
20
+
21
+ generate_kwargs = dict(
22
+ temperature=temperature,
23
+ max_new_tokens=max_new_tokens,
24
+ top_p=top_p,
25
+ repetition_penalty=repetition_penalty,
26
+ do_sample=True,
27
+ seed=42,
28
+ )
29
+
30
+ formatted_prompt = format_prompt(f"{system_prompt}, {prompt}", history)
31
+ stream = client.text_generation(formatted_prompt, **generate_kwargs, stream=True, details=True, return_full_text=False)
32
+ output = ""
33
+
34
+ for response in stream:
35
+ output += response.token.text
36
+ yield output
37
+ return output
38
+
39
+
40
+ additional_inputs=[
41
+ gr.Textbox(
42
+ label="System Prompt",
43
+ max_lines=1,
44
+ interactive=True,
45
+ ),
46
+ gr.Slider(
47
+ label="Temperature",
48
+ value=0.2,
49
+ minimum=0.0,
50
+ maximum=1.0,
51
+ step=0.05,
52
+ interactive=True,
53
+ info="Higher values produce more diverse outputs",
54
+ ),
55
+ gr.Slider(
56
+ label="Max new tokens",
57
+ value=512,
58
+ minimum=0,
59
+ maximum=1048,
60
+ step=64,
61
+ interactive=True,
62
+ info="The maximum numbers of new tokens",
63
+ ),
64
+ gr.Slider(
65
+ label="Top-p (nucleus sampling)",
66
+ value=0.1,
67
+ minimum=0.0,
68
+ maximum=1,
69
+ step=0.05,
70
+ interactive=True,
71
+ info="Higher values sample more low-probability tokens",
72
+ ),
73
+ gr.Slider(
74
+ label="Repetition penalty",
75
+ value=1.2,
76
+ minimum=1.0,
77
+ maximum=2.0,
78
+ step=0.05,
79
+ interactive=True,
80
+ info="Penalize repeated tokens",
81
+ )
82
+ ]
83
+
84
+
85
+ gr.ChatInterface(
86
+ fn=generate,
87
+ chatbot=gr.Chatbot(show_label=False, show_share_button=False, show_copy_button=True, likeable=True, layout="panel"),
88
+ additional_inputs=additional_inputs,
89
+ title="Exam Advisor for MBA",
90
+ concurrency_limit=20,
91
+ ).launch(show_api=False)