ColPali-demo

Sleeping

App Files Files Community

hantech commited on Jan 3

Commit

c8a9051

verified ·

1 Parent(s): f7fe59b

Update app.py

Browse files

Files changed (1) hide show

app.py +16 -22

app.py CHANGED Viewed

@@ -7,21 +7,20 @@ from pdf2image import convert_from_path
 from PIL import Image
 from torch.utils.data import DataLoader
 from tqdm import tqdm
-import os
 if not os.path.exists('/tmp/gradio'):
     os.makedirs('/tmp/gradio')
-from colpali_engine.models import ColQwen2, ColQwen2Processor
 device = "cuda:0" if torch.cuda.is_available() else "cpu"
 model = ColQwen2.from_pretrained(
-        "manu/colqwen2-v1.0-alpha",
-        torch_dtype=torch.bfloat16,
-        device_map=device,  # or "mps" if on Apple Silicon
-        # attn_implementation="flash_attention_2", # should work on A100
-    ).eval()
 processor = ColQwen2Processor.from_pretrained("manu/colqwen2-v1.0-alpha")
 def search(query: str, ds, images, k):
@@ -49,14 +48,16 @@ def search(query: str, ds, images, k):
     return results
 def index(files, ds):
     print("Converting files")
     images = convert_files(files)
     print(f"Files converted with {len(images)} images.")
-    return index_gpu(images, ds)
 def convert_files(files):
     images = []
@@ -67,15 +68,11 @@ def convert_files(files):
         raise gr.Error("The number of images in the dataset should be less than 150.")
     return images
 def index_gpu(images, ds):
-    """Example script to run inference with ColPali (ColQwen2)"""
     device = "cuda:0" if torch.cuda.is_available() else "cpu"
     if device != model.device:
         model.to(device)
-    # run inference - docs
     dataloader = DataLoader(
         images,
         batch_size=4,
@@ -90,17 +87,15 @@ def index_gpu(images, ds):
         ds.extend(list(torch.unbind(embeddings_doc.to("cpu"))))
     return f"Uploaded and converted {len(images)} pages", ds, images
 with gr.Blocks(theme=gr.themes.Soft()) as demo:
     gr.Markdown("# ColPali: Efficient Document Retrieval with Vision Language Models (ColQwen2) 📚")
     gr.Markdown("""Demo to test ColQwen2 (ColPali) on PDF documents.
-    ColPali is model implemented from the [ColPali paper](https://arxiv.org/abs/2407.01449).
     This demo allows you to upload PDF files and search for the most relevant pages based on your query.
     Refresh the page if you change documents !
-    ⚠️ This demo uses a model trained exclusively on A4 PDFs in portrait mode, containing english text. Performance is expected to drop for other page formats and languages.
     Other models will be released with better robustness towards different languages and document formats !
     """)
     with gr.Row():
@@ -118,12 +113,11 @@ with gr.Blocks(theme=gr.themes.Soft()) as demo:
             query = gr.Textbox(placeholder="Enter your query here", label="Query")
             k = gr.Slider(minimum=1, maximum=10, step=1, label="Number of results", value=5)
     # Define the actions
     search_button = gr.Button("🔍 Search", variant="primary")
     output_gallery = gr.Gallery(label="Retrieved Documents", height=600, show_label=True)
-    convert_button.click(index, inputs=[file, embeds], outputs=[message, embeds, imgs])
     search_button.click(search, inputs=[query, embeds, imgs, k], outputs=[output_gallery])
 if __name__ == "__main__":

 from PIL import Image
 from torch.utils.data import DataLoader
 from tqdm import tqdm
+from colpali_engine.models import ColQwen2, ColQwen2Processor
+# Ensure the temporary directory exists
 if not os.path.exists('/tmp/gradio'):
     os.makedirs('/tmp/gradio')
 device = "cuda:0" if torch.cuda.is_available() else "cpu"
 model = ColQwen2.from_pretrained(
+    "manu/colqwen2-v1.0-alpha",
+    torch_dtype=torch.bfloat16,
+    device_map=device,
+).eval()
 processor = ColQwen2Processor.from_pretrained("manu/colqwen2-v1.0-alpha")
 def search(query: str, ds, images, k):
     return results
 def index(files, ds):
+    if not files:
+        return gr.Error("No files uploaded. Please upload PDF files to index."), ds, []
     print("Converting files")
     images = convert_files(files)
     print(f"Files converted with {len(images)} images.")
+    status_message, ds, images = index_gpu(images, ds)
+    print(f"Indexed {len(ds)} embeddings.")
+    return status_message, ds, images
 def convert_files(files):
     images = []
         raise gr.Error("The number of images in the dataset should be less than 150.")
     return images
 def index_gpu(images, ds):
     device = "cuda:0" if torch.cuda.is_available() else "cpu"
     if device != model.device:
         model.to(device)
     dataloader = DataLoader(
         images,
         batch_size=4,
         ds.extend(list(torch.unbind(embeddings_doc.to("cpu"))))
     return f"Uploaded and converted {len(images)} pages", ds, images
 with gr.Blocks(theme=gr.themes.Soft()) as demo:
     gr.Markdown("# ColPali: Efficient Document Retrieval with Vision Language Models (ColQwen2) 📚")
     gr.Markdown("""Demo to test ColQwen2 (ColPali) on PDF documents.
+    ColPali is a model implemented from the [ColPali paper](https://arxiv.org/abs/2407.01449).
     This demo allows you to upload PDF files and search for the most relevant pages based on your query.
     Refresh the page if you change documents !
+    ⚠️ This demo uses a model trained exclusively on A4 PDFs in portrait mode, containing English text. Performance is expected to drop for other page formats and languages.
     Other models will be released with better robustness towards different languages and document formats !
     """)
     with gr.Row():
             query = gr.Textbox(placeholder="Enter your query here", label="Query")
             k = gr.Slider(minimum=1, maximum=10, step=1, label="Number of results", value=5)
     # Define the actions
+    convert_button.click(index, inputs=[file, embeds], outputs=[message, embeds, imgs])
     search_button = gr.Button("🔍 Search", variant="primary")
     output_gallery = gr.Gallery(label="Retrieved Documents", height=600, show_label=True)
     search_button.click(search, inputs=[query, embeds, imgs, k], outputs=[output_gallery])
 if __name__ == "__main__":