import gi, ollama, asyncio, threading gi.require_version("Gtk", "4.0") from gi.repository import GLib, Gtk, Gdk from prompt import PromptView from chat import ChatView from preferences import ModelPreferenceView import util class PyLlamaWindow(Gtk.ApplicationWindow): def __init__(self, application=None, title=None): super().__init__(application=application, title=title) self.set_default_size(800, 600) self.model_store = Gtk.StringList() self.model_dropdown = Gtk.DropDown(tooltip_text="Select model") self.model_dropdown.set_model(self.model_store) model_preferences = Gtk.Button(icon_name="view-more-symbolic", tooltip_text="Model preferences") model_preferences.connect("clicked", self.on_preferences_clicked) self.model_prefs = ModelPreferenceView(position=Gtk.PositionType.BOTTOM, on_gpu_change=self.on_gpu_changed) self.model_prefs.set_parent(model_preferences) header = Gtk.HeaderBar(show_title_buttons=True) header.pack_start(self.model_dropdown) header.pack_start(model_preferences) self.set_titlebar(header) vbox = Gtk.Box(orientation=Gtk.Orientation.VERTICAL, spacing=6) util.set_margin_all(vbox, 8) self.chat = ChatView(vexpand=True) vbox.append(self.chat) self.prompt = PromptView(spacing=6) self.prompt.on_send = self.on_send self.prompt.on_cancel = self.on_cancel vbox.append(self.prompt) self.set_child(vbox) self.populate_models() def populate_models(self): try: models = ollama.list() names = [m.model for m in models["models"]] except Exception as e: print ("Failed to fetch models: ", e) names = [] self.model_store.splice(0, len(self.model_store), names) def on_gpu_changed(self, value): self.get_application().num_gpu = value def on_preferences_clicked(self, button): self.model_prefs.popup() def on_send(self, *args): text = self.prompt.get_text() if not text: return self.prompt.clear() self.chat.add_message(text, sender=True) self.chat.start_response() app = self.get_application() model = self.model_dropdown.get_selected_item().get_string() async def task(): await app.send_prompt(model, text) if app.current_async_task and not app.current_async_task.done(): app.current_async_task.cancel() app.current_async_task = asyncio.run_coroutine_threadsafe(task(), app.async_loop) def on_cancel(self, *args): app = self.get_application() app.stop_generation() class PyLlama(Gtk.Application): def __init__(self): super().__init__(application_id="eur.cfpi-fpsi.pyllama") GLib.set_application_name("PyLlama") self.num_gpu = 15 self.client = ollama.AsyncClient() self.async_loop = asyncio.new_event_loop() self.current_async_task = None threading.Thread(target=self._run_asyncio_loop, daemon=True).start() def do_activate(self): self.window = PyLlamaWindow(application=self, title="PyLlama") self.window.present() def _run_asyncio_loop(self): asyncio.set_event_loop(self.async_loop) self.async_loop.run_forever() return True def stop_generation(self): if self.current_async_task and not self.current_async_task.done(): self.current_async_task.cancel() async def send_prompt(self, model, text=""): messages = [{ "role": "user", "content": text }] GLib.idle_add(self.window.prompt.hold, True) try: async for part in await self.client.chat(model=model, messages=messages, stream=True, options={ "num_gpu": self.num_gpu }): GLib.idle_add(self.window.chat.add_response_part, part.message.content) except asyncio.CancelledError: raise finally: GLib.idle_add(self.window.prompt.unhold, True) if __name__ == "__main__": app = PyLlama() app.run()