Major speed improvement, split the project into several files

This commit is contained in:
2026-03-21 17:52:23 +01:00
parent 75b909cbc3
commit 8f089af0d6
10 changed files with 264 additions and 174 deletions
+127
View File
@@ -0,0 +1,127 @@
import gi, ollama, asyncio, threading
gi.require_version("Gtk", "4.0")
from gi.repository import GLib, Gtk, Gdk
from prompt import PromptView
from chat import ChatView
from preferences import ModelPreferenceView
import util
class PyLlamaWindow(Gtk.ApplicationWindow):
def __init__(self, application=None, title=None):
super().__init__(application=application, title=title)
self.set_default_size(800, 600)
self.model_store = Gtk.StringList()
self.model_dropdown = Gtk.DropDown(tooltip_text="Select model")
self.model_dropdown.set_model(self.model_store)
model_preferences = Gtk.Button(icon_name="view-more-symbolic",
tooltip_text="Model preferences")
model_preferences.connect("clicked", self.on_preferences_clicked)
self.model_prefs = ModelPreferenceView(position=Gtk.PositionType.BOTTOM,
on_gpu_change=self.on_gpu_changed)
self.model_prefs.set_parent(model_preferences)
header = Gtk.HeaderBar(show_title_buttons=True)
header.pack_start(self.model_dropdown)
header.pack_start(model_preferences)
self.set_titlebar(header)
vbox = Gtk.Box(orientation=Gtk.Orientation.VERTICAL, spacing=6)
util.set_margin_all(vbox, 8)
self.chat = ChatView(vexpand=True)
vbox.append(self.chat)
self.prompt = PromptView(spacing=6)
self.prompt.on_send = self.on_send
self.prompt.on_cancel = self.on_cancel
vbox.append(self.prompt)
self.set_child(vbox)
self.populate_models()
def populate_models(self):
try:
models = ollama.list()
names = [m.model for m in models["models"]]
except Exception as e:
print ("Failed to fetch models: ", e)
names = []
self.model_store.splice(0, len(self.model_store), names)
def on_gpu_changed(self, value):
self.get_application().num_gpu = value
def on_preferences_clicked(self, button):
self.model_prefs.popup()
def on_send(self, *args):
text = self.prompt.get_text()
if not text:
return
self.prompt.clear()
self.chat.add_message(text, sender=True)
self.chat.start_response()
app = self.get_application()
model = self.model_dropdown.get_selected_item().get_string()
async def task():
await app.send_prompt(model, text)
if app.current_async_task and not app.current_async_task.done():
app.current_async_task.cancel()
app.current_async_task = asyncio.run_coroutine_threadsafe(task(), app.async_loop)
def on_cancel(self, *args):
app = self.get_application()
app.stop_generation()
class PyLlama(Gtk.Application):
def __init__(self):
super().__init__(application_id="eur.cfpi-fpsi.pyllama")
GLib.set_application_name("PyLlama")
self.num_gpu = 15
self.client = ollama.AsyncClient()
self.async_loop = asyncio.new_event_loop()
self.current_async_task = None
threading.Thread(target=self._run_asyncio_loop, daemon=True).start()
def do_activate(self):
self.window = PyLlamaWindow(application=self, title="PyLlama")
self.window.present()
def _run_asyncio_loop(self):
asyncio.set_event_loop(self.async_loop)
self.async_loop.run_forever()
return True
def stop_generation(self):
if self.current_async_task and not self.current_async_task.done():
self.current_async_task.cancel()
async def send_prompt(self, model, text=""):
messages = [{ "role": "user", "content": text }]
GLib.idle_add(self.window.prompt.hold, True)
try:
async for part in await self.client.chat(model=model, messages=messages, stream=True, options={ "num_gpu": self.num_gpu }):
GLib.idle_add(self.window.chat.add_response_part, part.message.content)
except asyncio.CancelledError:
raise
finally:
GLib.idle_add(self.window.prompt.unhold, True)
if __name__ == "__main__":
app = PyLlama()
app.run()