File size: 11,079 Bytes
e34a93e
 
 
 
 
 
 
 
 
 
a095bcc
 
 
 
 
 
e34a93e
9ef7776
 
7c0f8d5
 
 
 
 
 
9ef7776
a095bcc
 
 
 
e34a93e
 
 
9ef7776
 
 
 
 
e34a93e
9ef7776
e34a93e
 
 
 
 
 
 
 
 
 
 
 
 
 
5186af2
62dc511
e34a93e
 
5186af2
e34a93e
7dba37b
5186af2
62dc511
5186af2
 
e34a93e
 
 
9ef7776
e34a93e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
7dba37b
e34a93e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
e62fecd
e34a93e
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
a095bcc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
4945599
a095bcc
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289


import openai
import gradio as gr

from os import getenv
from typing import Any, Dict, Generator, List

from huggingface_hub import InferenceClient
from transformers import AutoTokenizer
import google.generativeai as genai
import os
import PIL.Image
import gradio as gr
#from gradio_multimodalchatbot import MultimodalChatbot
from gradio.data_classes import FileData

#tokenizer = AutoTokenizer.from_pretrained("mistralai/Mistral-7B-Instruct-v0.1")
tokenizer = AutoTokenizer.from_pretrained("mistralai/Mixtral-8x7B-Instruct-v0.1")
# temperature = 0.2
# #top_p = 0.6
# repetition_penalty = 1.0
temperature = 0.5
top_p = 0.7
repetition_penalty = 1.2


# Fetch an environment variable.
GOOGLE_API_KEY = os.environ.get('GOOGLE_API_KEY')
genai.configure(api_key=GOOGLE_API_KEY)
OPENAI_KEY = getenv("OPENAI_API_KEY")
HF_TOKEN = getenv("HUGGING_FACE_HUB_TOKEN")

# hf_client = InferenceClient(
#         "mistralai/Mistral-7B-Instruct-v0.1",
#         token=HF_TOKEN
#         )

hf_client = InferenceClient(
        "mistralai/Mixtral-8x7B-Instruct-v0.1",
        token=HF_TOKEN
        )
def format_prompt(message: str, api_kind: str):
    """
    Formats the given message using a chat template.

    Args:
        message (str): The user message to be formatted.

    Returns:
        str: Formatted message after applying the chat template.
    """

    # Create a list of message dictionaries with role and content
    messages1: List[Dict[str, Any]] = [{'role': 'user', 'content': message}]
    messages2: List[Dict[str, Any]] = [{'role': 'user', 'parts': message}]

    if api_kind == "openai":
        return messages1
    elif api_kind == "hf":
        return tokenizer.apply_chat_template(messages1, tokenize=False)
    elif api_kind=="gemini":
        print(messages2)
        return messages2
    else:
        raise ValueError("API is not supported")


def generate_hf(prompt: str, history: str, temperature: float = 0.9, max_new_tokens: int = 4000,
             top_p: float = 0.95, repetition_penalty: float = 1.0) -> Generator[str, None, str]:
    """
    Generate a sequence of tokens based on a given prompt and history using Mistral client.

    Args:
        prompt (str): The initial prompt for the text generation.
        history (str): Context or history for the text generation.
        temperature (float, optional): The softmax temperature for sampling. Defaults to 0.9.
        max_new_tokens (int, optional): Maximum number of tokens to be generated. Defaults to 256.
        top_p (float, optional): Nucleus sampling probability. Defaults to 0.95.
        repetition_penalty (float, optional): Penalty for repeated tokens. Defaults to 1.0.

    Returns:
        Generator[str, None, str]: A generator yielding chunks of generated text.
                                   Returns a final string if an error occurs.
    """

    temperature = max(float(temperature), 1e-2)  # Ensure temperature isn't too low
    top_p = float(top_p)

    generate_kwargs = {
        'temperature': temperature,
        'max_new_tokens': max_new_tokens,
        'top_p': top_p,
        'repetition_penalty': repetition_penalty,
        'do_sample': True,
        'seed': 42,
        }
    
    formatted_prompt = format_prompt(prompt, "hf")
    print('formatted_prompt ', formatted_prompt )

    try:
        stream = hf_client.text_generation(formatted_prompt, **generate_kwargs,
                                            stream=True, details=True, return_full_text=False)
        output = ""
        for response in stream:
            output += response.token.text
            yield output

    except Exception as e:
        if "Too Many Requests" in str(e):
            print("ERROR: Too many requests on Mistral client")
            gr.Warning("Unfortunately Mistral is unable to process")
            return "Unfortunately, I am not able to process your request now."
        elif "Authorization header is invalid" in str(e):
            print("Authetification error:", str(e))
            gr.Warning("Authentication error: HF token was either not provided or incorrect")
            return "Authentication error"
        else:
            print("Unhandled Exception:", str(e))
            gr.Warning("Unfortunately Mistral is unable to process")
            return "I do not know what happened, but I couldn't understand you."


def generate_openai(prompt: str, history: str, temperature: float = 0.9, max_new_tokens: int = 256,
             top_p: float = 0.95, repetition_penalty: float = 1.0) -> Generator[str, None, str]:
    """
    Generate a sequence of tokens based on a given prompt and history using Mistral client.

    Args:
        prompt (str): The initial prompt for the text generation.
        history (str): Context or history for the text generation.
        temperature (float, optional): The softmax temperature for sampling. Defaults to 0.9.
        max_new_tokens (int, optional): Maximum number of tokens to be generated. Defaults to 256.
        top_p (float, optional): Nucleus sampling probability. Defaults to 0.95.
        repetition_penalty (float, optional): Penalty for repeated tokens. Defaults to 1.0.

    Returns:
        Generator[str, None, str]: A generator yielding chunks of generated text.
                                   Returns a final string if an error occurs.
    """

    temperature = max(float(temperature), 1e-2)  # Ensure temperature isn't too low
    top_p = float(top_p)
    
    generate_kwargs = {
        'temperature': temperature,
        'max_tokens': max_new_tokens,
        'top_p': top_p,
        'frequency_penalty': max(-2., min(repetition_penalty, 2.)),
        }

    formatted_prompt = format_prompt(prompt, "hf")

    try:
        stream = openai.ChatCompletion.create(model="gpt-3.5-turbo-0301",
                                                messages=formatted_prompt, 
                                                **generate_kwargs, 
                                                stream=True)
        output = ""
        for chunk in stream:
            output += chunk.choices[0].delta.get("content", "")
            yield output

    except Exception as e:
        if "Too Many Requests" in str(e):
            print("ERROR: Too many requests on OpenAI client")
            gr.Warning("Unfortunately OpenAI is unable to process")
            return "Unfortunately, I am not able to process your request now."
        elif "You didn't provide an API key" in str(e):
            print("Authetification error:", str(e))
            gr.Warning("Authentication error: OpenAI key was either not provided or incorrect")
            return "Authentication error"
        else:
            print("Unhandled Exception:", str(e))
            gr.Warning("Unfortunately OpenAI is unable to process")
            return "I do not know what happened, but I couldn't understand you."


def generate_gemini(prompt: str, history: str, temperature: float = 0.9, max_new_tokens: int = 4000,
             top_p: float = 0.95, repetition_penalty: float = 1.0):
    
    # For better security practices, retrieve sensitive information like API keys from environment variables.
 
    
    # Initialize genai models
    model = genai.GenerativeModel('gemini-pro')
    api_key = os.environ.get("GOOGEL_API_KEY")
    genai.configure(api_key=api_key)
    #model = genai.GenerativeModel('gemini-pro')
    #chat = model.start_chat(history=[])


   
    candidate_count=1
    max_output_tokens=max_new_tokens
    temperature=1
    top_p=top_p

    
    formatted_prompt = format_prompt(prompt, "gemini")

    try:
        stream = model.generate_content(formatted_prompt,generation_config=genai.GenerationConfig(temperature=temperature,candidate_count=1 ,max_output_tokens=max_new_tokens,top_p=top_p),
                                            stream=True)
        output = ""
        for response in stream:
            output += response.text
            yield output

    except Exception as e:
        if "Too Many Requests" in str(e):
            print("ERROR: Too many requests on Mistral client")
            gr.Warning("Unfortunately Mistral is unable to process")
            return "Unfortunately, I am not able to process your request now."
        elif "Authorization header is invalid" in str(e):
            print("Authetification error:", str(e))
            gr.Warning("Authentication error: HF token was either not provided or incorrect")
            return "Authentication error"
        else:
            print("Unhandled Exception:", str(e))
            gr.Warning("Unfortunately Mistral is unable to process")
            return "I do not know what happened, but I couldn't understand you."

    

    
    # def gemini(input, file, chatbot=[]):
    #     """
    #     Function to handle gemini model and gemini vision model interactions.
    #     Parameters:
    #     input (str): The input text.
    #     file (File): An optional file object for image processing.
    #     chatbot (list): A list to keep track of chatbot interactions.
    #     Returns:
    #     tuple: Updated chatbot interaction list, an empty string, and None.
    #     """
    
    #     messages = []
    #     print(chatbot)
    
    #     # Process previous chatbot messages if present
    #     if len(chatbot) != 0:
    #         for messages_dict in chatbot:
    #             user_text = messages_dict[0]['text']
    #             bot_text = messages_dict[1]['text']
    #             messages.extend([
    #                 {'role': 'user', 'parts': [user_text]},
    #                 {'role': 'model', 'parts': [bot_text]}
    #             ])
    #         messages.append({'role': 'user', 'parts': [input]})
    #     else:
    #         messages.append({'role': 'user', 'parts': [input]})
    
    #     try:
    #         response = model.generate_content(messages)
    #         gemini_resp = response.text
    #         # Construct list of messages in the required format
    #         user_msg = {"text": input, "files": []}
    #         bot_msg = {"text": gemini_resp, "files": []}
    #         chatbot.append([user_msg, bot_msg])
       
    #     except Exception as e:
    #         # Handling exceptions and raising error to the modal
    #         print(f"An error occurred: {e}")
    #         raise gr.Error(e)
    
    #     return chatbot, "", None
    
    # # Define the Gradio Blocks interface
    # with gr.Blocks() as demo:
    #     # Add a centered header using HTML
    #     gr.HTML("<center><h1>Gemini Chat PRO API</h1></center>")
    
    #     # Initialize the MultimodalChatbot component
    #     multi = MultimodalChatbot(value=[], height=800)
    
    #     with gr.Row():
    #         # Textbox for user input with increased scale for better visibility
    #         tb = gr.Textbox(scale=4, placeholder='Input text and press Enter')
    
           
    
    #     # Define the behavior on text submission
    #     tb.submit(gemini, [tb, multi], [multi, tb])
    
    
    # # Launch the demo with a queue to handle multiple users
    # demo.queue().launch()