Spaces:
Sleeping
Sleeping
File size: 4,164 Bytes
5c90aba 1184664 5c90aba 58e4560 5c90aba 58e4560 5c90aba 58e4560 a1d3e5f 5c90aba 6531b64 de7122c 5c90aba f2c5032 5c90aba |
1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 |
from huggingface_hub import InferenceClient
import gradio as gr
import random
client = InferenceClient("mistralai/Mixtral-8x7B-Instruct-v0.1")
from prompts import GAME_MASTER, COMPRESS_HISTORY
def format_prompt(message, history):
prompt=""
'''
prompt = "<s>"
for user_prompt, bot_response in history:
prompt += f"[INST] {user_prompt} [/INST]"
prompt += f" {bot_response}</s> "
'''
prompt += f"[INST] {message} [/INST]"
return prompt
def compress_history(history,temperature=0.9, max_new_tokens=256, top_p=0.95, repetition_penalty=1.0):
formatted_prompt=f"{COMPRESS_HISTORY.format(history=history)}"
generate_kwargs = dict(
temperature=temperature,
max_new_tokens=max_new_tokens,
top_p=top_p,
repetition_penalty=repetition_penalty,
do_sample=True,
seed=random.randint(1,99999999999)
#seed=42,
)
stream = client.text_generation(formatted_prompt, **generate_kwargs, stream=True, details=True, return_full_text=False)
output = ""
for response in stream:
output += response.token.text
return output
MAX_HISTORY=100
def generate(
prompt, history, system_prompt, temperature=0.9, max_new_tokens=256, top_p=0.95, repetition_penalty=1.0,
):
temperature = float(temperature)
if temperature < 1e-2:
temperature = 1e-2
top_p = float(top_p)
generate_kwargs = dict(
temperature=temperature,
max_new_tokens=max_new_tokens,
top_p=top_p,
repetition_penalty=repetition_penalty,
do_sample=True,
seed=random.randint(1,99999999999)
#seed=42,
)
cnt=0
for ea in history:
print (ea)
for l in ea:
print (l)
cnt+=len(l.split("\n"))
print(f'cnt:: {cnt}')
if cnt > MAX_HISTORY:
history = compress_history(history, temperature, max_new_tokens, top_p, repetition_penalty)
formatted_prompt = format_prompt(f"{GAME_MASTER.format(history=history)}, {prompt}", history)
stream = client.text_generation(formatted_prompt, **generate_kwargs, stream=True, details=True, return_full_text=False)
output = ""
for response in stream:
output += response.token.text
yield output
lines = output.strip().strip("\n").split("\n")
#history=""
for i,line in enumerate(lines):
if line.startswith("1. "):
print(line)
if line.startswith("2. "):
print(line)
if line.startswith("3. "):
print(line)
if line.startswith("4. "):
print(line)
if line.startswith("5. "):
print(line)
return output
additional_inputs=[
gr.Textbox(
label="System Prompt",
max_lines=1,
interactive=True,
),
gr.Slider(
label="Temperature",
value=0.9,
minimum=0.0,
maximum=1.0,
step=0.05,
interactive=True,
info="Higher values produce more diverse outputs",
),
gr.Slider(
label="Max new tokens",
value=1048,
minimum=0,
maximum=1048*10,
step=64,
interactive=True,
info="The maximum numbers of new tokens",
),
gr.Slider(
label="Top-p (nucleus sampling)",
value=0.90,
minimum=0.0,
maximum=1,
step=0.05,
interactive=True,
info="Higher values sample more low-probability tokens",
),
gr.Slider(
label="Repetition penalty",
value=1.2,
minimum=1.0,
maximum=2.0,
step=0.05,
interactive=True,
info="Penalize repeated tokens",
)
]
examples=[["Start the Game", None, None, None, None, None, ],
["Start a Game based in the year 1322", None, None, None, None, None,],
]
gr.ChatInterface(
fn=generate,
chatbot=gr.Chatbot(show_label=False, show_share_button=False, show_copy_button=True, likeable=True, layout="panel"),
additional_inputs=additional_inputs,
title="Mixtral RPG Game Master",
examples=examples,
concurrency_limit=20,
).launch(share=True,show_api=True) |