Spaces:

JeCabrera
/

copywriter

Running

App Files Files Community

JeCabrera commited on Jan 18

Commit

03a628f

verified ·

1 Parent(s): 4ad1ff2

Update app.py

Browse files

Files changed (1) hide show

app.py +91 -131

app.py CHANGED Viewed

@@ -1,162 +1,122 @@
-TITLE = """<h1 align="center">Gemini Playground ✨</h1>"""
-SUBTITLE = """<h2 align="center">Play with Gemini Pro and Gemini Pro Vision</h2>"""
 import os
 import time
 import uuid
 from typing import List, Tuple, Optional, Union
 import google.generativeai as genai
 import gradio as gr
-from PIL import Image
 from dotenv import load_dotenv
 # Cargar las variables de entorno desde el archivo .env
 load_dotenv()
-print("google-generativeai:", genai.__version__)
-# Obtener la clave de la API de las variables de entorno
-GOOGLE_API_KEY = os.getenv("GOOGLE_API_KEY")
-# Verificar que la clave de la API esté configurada
-if not GOOGLE_API_KEY:
-    raise ValueError("GOOGLE_API_KEY is not set in environment variables.")
 IMAGE_CACHE_DIRECTORY = "/tmp"
 IMAGE_WIDTH = 512
 CHAT_HISTORY = List[Tuple[Optional[Union[Tuple[str], str]], Optional[str]]]
 def preprocess_image(image: Image.Image) -> Optional[Image.Image]:
     if image:
         image_height = int(image.height * IMAGE_WIDTH / image.width)
         return image.resize((IMAGE_WIDTH, image_height))
 def cache_pil_image(image: Image.Image) -> str:
     image_filename = f"{uuid.uuid4()}.jpeg"
     os.makedirs(IMAGE_CACHE_DIRECTORY, exist_ok=True)
     image_path = os.path.join(IMAGE_CACHE_DIRECTORY, image_filename)
     image.save(image_path, "JPEG")
     return image_path
-def upload(files: Optional[List[str]], chatbot: CHAT_HISTORY) -> CHAT_HISTORY:
-    for file in files:
-        image = Image.open(file).convert('RGB')
-        image_preview = preprocess_image(image)
-        if image_preview:
-            gr.Image(image_preview).render()
-        image_path = cache_pil_image(image)
-        chatbot.append(((image_path,), None))
-    return chatbot
-def user(text_prompt: str, chatbot: CHAT_HISTORY):
-    if text_prompt:
-        chatbot.append((text_prompt, None))
-    return "", chatbot
-def bot(
-    files: Optional[List[str]],
-    model_choice: str,
-    system_instruction: Optional[str],  # Sistema de instrucciones opcional
-    chatbot: CHAT_HISTORY
-):
-    if not GOOGLE_API_KEY:
-        raise ValueError("GOOGLE_API_KEY is not set.")
-    genai.configure(api_key=GOOGLE_API_KEY)
-    generation_config = genai.types.GenerationConfig(
-        temperature=0.7,
-        max_output_tokens=8192,
-        top_k=10,
-        top_p=0.9
-    )
-    # Usar el valor por defecto para system_instruction si está vacío
-    if not system_instruction:
-        system_instruction = "1"  # O puedes poner un valor predeterminado como "No system instruction provided."
-    text_prompt = [chatbot[-1][0]] if chatbot and chatbot[-1][0] and isinstance(chatbot[-1][0], str) else []
-    image_prompt = [preprocess_image(Image.open(file).convert('RGB')) for file in files] if files else []
-    model = genai.GenerativeModel(
-        model_name=model_choice,
-        generation_config=generation_config,
-        system_instruction=system_instruction  # Usar el valor por defecto si está vacío
-    )
-    response = model.generate_content(text_prompt + image_prompt, stream=True, generation_config=generation_config)
-    chatbot[-1][1] = ""
-    for chunk in response:
-        for i in range(0, len(chunk.text), 10):
-            section = chunk.text[i:i + 10]
-            chatbot[-1][1] += section
-            time.sleep(0.01)
-            yield chatbot
-# Componente para el acordeón que contiene el cuadro de texto para la instrucción del sistema
-system_instruction_component = gr.Textbox(
-    placeholder="Enter system instruction...",
-    show_label=True,
-    scale=8
 )
-# Definir los componentes de entrada y salida
-chatbot_component = gr.Chatbot(label='Gemini', bubble_full_width=False, scale=2, height=300)
-text_prompt_component = gr.Textbox(placeholder="Message...", show_label=False, autofocus=True, scale=8)
-upload_button_component = gr.UploadButton(label="Upload Images", file_count="multiple", file_types=["image"], scale=1)
-run_button_component = gr.Button(value="Run", variant="primary", scale=1)
-model_choice_component = gr.Dropdown(
-    choices=["gemini-1.5-flash", "gemini-2.0-flash-exp", "gemini-1.5-pro"],
-    value="gemini-1.5-flash",
-    label="Select Model",
-    scale=2
-)
-user_inputs = [text_prompt_component, chatbot_component]
-bot_inputs = [upload_button_component, model_choice_component, system_instruction_component, chatbot_component]
-# Definir la interfaz de usuario
-with gr.Blocks() as demo:
-    gr.HTML(TITLE)
-    gr.HTML(SUBTITLE)
-    with gr.Column():
-        # Campo de selección de modelo arriba
-        model_choice_component.render()
-        chatbot_component.render()
-        with gr.Row():
-            text_prompt_component.render()
-            upload_button_component.render()
-            run_button_component.render()
-        # Crear el acordeón para la instrucción del sistema al final
-        with gr.Accordion("System Instruction", open=False):  # Acordeón cerrado por defecto
-            system_instruction_component.render()
-    run_button_component.click(
-        fn=user,
-        inputs=user_inputs,
-        outputs=[text_prompt_component, chatbot_component],
-        queue=False
-    ).then(
-        fn=bot, inputs=bot_inputs, outputs=[chatbot_component],
-    )
-    text_prompt_component.submit(
-        fn=user,
-        inputs=user_inputs,
-        outputs=[text_prompt_component, chatbot_component],
-        queue=False
-    ).then(
-        fn=bot, inputs=bot_inputs, outputs=[chatbot_component],
-    )
-    upload_button_component.upload(
-        fn=upload,
-        inputs=[upload_button_component, chatbot_component],
-        outputs=[chatbot_component],
-        queue=False
-    )
 # Lanzar la aplicación
-demo.queue(max_size=99).launch(debug=False, show_error=True)

 import os
 import time
 import uuid
 from typing import List, Tuple, Optional, Union
+from PIL import Image
 import google.generativeai as genai
 import gradio as gr
 from dotenv import load_dotenv
 # Cargar las variables de entorno desde el archivo .env
 load_dotenv()
+API_KEY = os.getenv("GOOGLE_API_KEY")
+if not API_KEY:
+    raise ValueError("La clave de API 'GOOGLE_API_KEY' no está configurada en el archivo .env")
+# Configuración del modelo Gemini
+genai.configure(api_key=API_KEY)
+generation_config = {
+    "temperature": 0.7,
+    "top_p": 0.9,
+    "top_k": 40,
+    "max_output_tokens": 8192,
+    "response_mime_type": "text/plain",
+}
+model = genai.GenerativeModel(
+    model_name="gemini-1.5-flash",
+    generation_config=generation_config,
+)
+# Inicializar la sesión de chat
+chat = model.start_chat(history=[])
+# Constantes para el manejo de imágenes
 IMAGE_CACHE_DIRECTORY = "/tmp"
 IMAGE_WIDTH = 512
 CHAT_HISTORY = List[Tuple[Optional[Union[Tuple[str], str]], Optional[str]]]
+# Función para preprocesar una imagen
 def preprocess_image(image: Image.Image) -> Optional[Image.Image]:
+    """Redimensiona una imagen manteniendo la relación de aspecto."""
     if image:
         image_height = int(image.height * IMAGE_WIDTH / image.width)
         return image.resize((IMAGE_WIDTH, image_height))
+# Función para almacenar una imagen en caché
 def cache_pil_image(image: Image.Image) -> str:
+    """Guarda la imagen como archivo JPEG en un directorio temporal."""
     image_filename = f"{uuid.uuid4()}.jpeg"
     os.makedirs(IMAGE_CACHE_DIRECTORY, exist_ok=True)
     image_path = os.path.join(IMAGE_CACHE_DIRECTORY, image_filename)
     image.save(image_path, "JPEG")
     return image_path
+# Función para transformar el historial de Gradio al formato de Gemini
+def transform_history(history):
+    """Transforma el historial del formato de Gradio al formato que Gemini espera."""
+    new_history = []
+    for chat in history:
+        if chat[0]:  # Mensaje del usuario
+            new_history.append({"parts": [{"text": chat[0]}], "role": "user"})
+        if chat[1]:  # Respuesta del modelo
+            new_history.append({"parts": [{"text": chat[1]}], "role": "model"})
+    return new_history
+# Función principal para manejar las respuestas del chat
+def response(message, history):
+    """Maneja la interacción multimodal y envía texto e imágenes al modelo."""
+    global chat
+    # Transformar el historial al formato esperado por Gemini
+    chat.history = transform_history(history)
+    # Obtener el texto del mensaje y las imágenes cargadas
+    text_prompt = message["text"]
+    files = message["files"]
+    # Procesar imágenes cargadas
+    image_prompts = [preprocess_image(Image.open(file).convert('RGB')) for file in files] if files else []
+    if files:
+        for file in files:
+            image = Image.open(file).convert('RGB')
+            image_preview = preprocess_image(image)
+            if image_preview:
+                # Guardar la imagen y obtener la ruta
+                image_path = cache_pil_image(image)
+                # Leer la imagen en formato binario para enviarla como Blob
+                with open(image_path, "rb") as img_file:
+                    img_data = img_file.read()
+                # Crear un diccionario con los datos binarios y su tipo MIME
+                image_prompt = {
+                    "mime_type": "image/jpeg",
+                    "data": img_data
+                }
+                image_prompts.append(image_prompt)
+    # Combinar texto e imágenes para el modelo
+    prompts = [text_prompt] + image_prompts
+    response = chat.send_message(prompts)
+    response.resolve()
+    # Generar respuesta carácter por carácter para una experiencia más fluida
+    for i in range(len(response.text)):
+        time.sleep(0.01)
+        yield response.text[: i + 1]
+# Crear la interfaz de usuario
+demo = gr.ChatInterface(
+    response,
+    examples=[{"text": "Describe the image:", "files": []}],
+    multimodal=True,
+    textbox=gr.MultimodalTextbox(
+        file_count="multiple",
+        file_types=["image"],
+        sources=["upload", "microphone"],
+    ),
 )
 # Lanzar la aplicación
+if __name__ == "__main__":
+    demo.launch(debug=True, show_error=True)