Major speed improvement, split the project into several files
This commit is contained in:
@@ -0,0 +1,127 @@
|
||||
import gi, ollama, asyncio, threading
|
||||
gi.require_version("Gtk", "4.0")
|
||||
from gi.repository import GLib, Gtk, Gdk
|
||||
|
||||
from prompt import PromptView
|
||||
from chat import ChatView
|
||||
from preferences import ModelPreferenceView
|
||||
import util
|
||||
|
||||
class PyLlamaWindow(Gtk.ApplicationWindow):
|
||||
def __init__(self, application=None, title=None):
|
||||
super().__init__(application=application, title=title)
|
||||
|
||||
self.set_default_size(800, 600)
|
||||
|
||||
self.model_store = Gtk.StringList()
|
||||
self.model_dropdown = Gtk.DropDown(tooltip_text="Select model")
|
||||
self.model_dropdown.set_model(self.model_store)
|
||||
|
||||
model_preferences = Gtk.Button(icon_name="view-more-symbolic",
|
||||
tooltip_text="Model preferences")
|
||||
model_preferences.connect("clicked", self.on_preferences_clicked)
|
||||
|
||||
self.model_prefs = ModelPreferenceView(position=Gtk.PositionType.BOTTOM,
|
||||
on_gpu_change=self.on_gpu_changed)
|
||||
self.model_prefs.set_parent(model_preferences)
|
||||
|
||||
header = Gtk.HeaderBar(show_title_buttons=True)
|
||||
header.pack_start(self.model_dropdown)
|
||||
header.pack_start(model_preferences)
|
||||
|
||||
self.set_titlebar(header)
|
||||
|
||||
vbox = Gtk.Box(orientation=Gtk.Orientation.VERTICAL, spacing=6)
|
||||
util.set_margin_all(vbox, 8)
|
||||
|
||||
self.chat = ChatView(vexpand=True)
|
||||
vbox.append(self.chat)
|
||||
|
||||
self.prompt = PromptView(spacing=6)
|
||||
self.prompt.on_send = self.on_send
|
||||
self.prompt.on_cancel = self.on_cancel
|
||||
vbox.append(self.prompt)
|
||||
|
||||
self.set_child(vbox)
|
||||
self.populate_models()
|
||||
|
||||
def populate_models(self):
|
||||
try:
|
||||
models = ollama.list()
|
||||
names = [m.model for m in models["models"]]
|
||||
except Exception as e:
|
||||
print ("Failed to fetch models: ", e)
|
||||
names = []
|
||||
|
||||
self.model_store.splice(0, len(self.model_store), names)
|
||||
|
||||
def on_gpu_changed(self, value):
|
||||
self.get_application().num_gpu = value
|
||||
|
||||
def on_preferences_clicked(self, button):
|
||||
self.model_prefs.popup()
|
||||
|
||||
def on_send(self, *args):
|
||||
text = self.prompt.get_text()
|
||||
if not text:
|
||||
return
|
||||
|
||||
self.prompt.clear()
|
||||
self.chat.add_message(text, sender=True)
|
||||
self.chat.start_response()
|
||||
|
||||
app = self.get_application()
|
||||
model = self.model_dropdown.get_selected_item().get_string()
|
||||
|
||||
async def task():
|
||||
await app.send_prompt(model, text)
|
||||
|
||||
if app.current_async_task and not app.current_async_task.done():
|
||||
app.current_async_task.cancel()
|
||||
|
||||
app.current_async_task = asyncio.run_coroutine_threadsafe(task(), app.async_loop)
|
||||
|
||||
def on_cancel(self, *args):
|
||||
app = self.get_application()
|
||||
app.stop_generation()
|
||||
|
||||
class PyLlama(Gtk.Application):
|
||||
def __init__(self):
|
||||
super().__init__(application_id="eur.cfpi-fpsi.pyllama")
|
||||
GLib.set_application_name("PyLlama")
|
||||
|
||||
self.num_gpu = 15
|
||||
|
||||
self.client = ollama.AsyncClient()
|
||||
self.async_loop = asyncio.new_event_loop()
|
||||
self.current_async_task = None
|
||||
threading.Thread(target=self._run_asyncio_loop, daemon=True).start()
|
||||
|
||||
def do_activate(self):
|
||||
self.window = PyLlamaWindow(application=self, title="PyLlama")
|
||||
self.window.present()
|
||||
|
||||
def _run_asyncio_loop(self):
|
||||
asyncio.set_event_loop(self.async_loop)
|
||||
self.async_loop.run_forever()
|
||||
return True
|
||||
|
||||
def stop_generation(self):
|
||||
if self.current_async_task and not self.current_async_task.done():
|
||||
self.current_async_task.cancel()
|
||||
|
||||
async def send_prompt(self, model, text=""):
|
||||
messages = [{ "role": "user", "content": text }]
|
||||
GLib.idle_add(self.window.prompt.hold, True)
|
||||
|
||||
try:
|
||||
async for part in await self.client.chat(model=model, messages=messages, stream=True, options={ "num_gpu": self.num_gpu }):
|
||||
GLib.idle_add(self.window.chat.add_response_part, part.message.content)
|
||||
except asyncio.CancelledError:
|
||||
raise
|
||||
finally:
|
||||
GLib.idle_add(self.window.prompt.unhold, True)
|
||||
|
||||
if __name__ == "__main__":
|
||||
app = PyLlama()
|
||||
app.run()
|
||||
Reference in New Issue
Block a user