feat: Add streaming chat + scroll persistence; improve markdown & links
Backend
- /chat: support streaming via StreamingResponse; save full reply after stream ends. Non-stream path unchanged.
- ChatRequest: add stream flag (default false).
- GenerateTitleRequest: add model and use it instead of hardcoded llama3.
- ollama_client.chat_stream(): new async generator parsing Ollama streaming JSON (both formats).
- Remove response_model from /chat to allow streaming; non-stream still returns { reply }.
Electron
- Open external links in system browser (setWindowOpenHandler, shell.openExternal).
- New IPC: update-settings, open-external-link.
- Set minimum window size; preload exposes updateSettings and openExternalLink.
Frontend (React)
- Streaming UI with live chunking; sticky-bottom only when user at bottom.
- Per-session scroll persistence and robust restore.
- New message tip to jump to latest reply when scrolled up.
- Disable Send while sending; spinner.
- General Settings: stream output toggle; propagate model/stream changes.
- Apply color scheme at boot; extract colorSchemes helper.
- Sidebar UX tweaks and unread badges.
Markdown/rendering
- Code blocks: language title bar and wrapper.
- Tables: GitHub-style parsing, per-cell borders, rounded wrapper, spacing, alignment.
- Headings: remove blank line after h1-h4.
- <hr>: handle after tables; strip following whitespace.
- Links: target=_blank with icon and URL tooltip.
Styles
- Add styles for code/table wrappers, new-message tip, toggle, spinner; hover/active vars; narrower sidebar.
API notes / breaking changes
- /chat accepts stream=true and returns text/plain streamed chunks.
- generate-title now requires a model.
- Non-stream /chat response shape unchanged.
This commit is contained in:
@@ -1,10 +1,11 @@
|
||||
from fastapi import FastAPI, Depends, HTTPException
|
||||
from fastapi.middleware.cors import CORSMiddleware
|
||||
from fastapi.responses import StreamingResponse
|
||||
from sqlalchemy.orm import Session
|
||||
from typing import List
|
||||
from . import models, schemas
|
||||
from .database import Base, engine, SessionLocal
|
||||
from .ollama_client import list_models as ollama_list, chat as ollama_chat
|
||||
from .ollama_client import list_models as ollama_list, chat as ollama_chat, chat_stream as ollama_chat_stream
|
||||
|
||||
# Create tables
|
||||
Base.metadata.create_all(bind=engine)
|
||||
@@ -61,7 +62,7 @@ def history(session_id: str, db: Session = Depends(get_db)):
|
||||
msgs = [{"role": r.role, "content": r.content} for r in rows]
|
||||
return {"messages": msgs}
|
||||
|
||||
@app.post("/chat", response_model=schemas.ChatResponse)
|
||||
@app.post("/chat")
|
||||
async def chat(req: schemas.ChatRequest, db: Session = Depends(get_db)):
|
||||
# Find or create session
|
||||
session = db.query(models.ChatSession).filter(models.ChatSession.session_id == req.session_id).first()
|
||||
@@ -81,17 +82,35 @@ async def chat(req: schemas.ChatRequest, db: Session = Depends(get_db)):
|
||||
|
||||
messages = [{"role": m.role, "content": m.content} for m in last_msgs]
|
||||
|
||||
try:
|
||||
reply = await ollama_chat(req.model, messages)
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=502, detail=f"Ollama error: {e}")
|
||||
if req.stream:
|
||||
async def stream_generator():
|
||||
full_reply = ""
|
||||
try:
|
||||
async for chunk in ollama_chat_stream(req.model, messages):
|
||||
full_reply += chunk
|
||||
yield chunk
|
||||
except Exception as e:
|
||||
# How to handle errors in a stream? Could yield an error message.
|
||||
yield f"Ollama error: {e}"
|
||||
|
||||
# Save assistant reply
|
||||
as_row = models.ChatMessage(session_pk=session.id, role='assistant', content=reply)
|
||||
db.add(as_row)
|
||||
db.commit()
|
||||
# Save full reply after stream is complete
|
||||
as_row = models.ChatMessage(session_pk=session.id, role='assistant', content=full_reply)
|
||||
db.add(as_row)
|
||||
db.commit()
|
||||
|
||||
return StreamingResponse(stream_generator(), media_type="text/plain")
|
||||
else:
|
||||
try:
|
||||
reply = await ollama_chat(req.model, messages)
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=502, detail=f"Ollama error: {e}")
|
||||
|
||||
return {"reply": reply}
|
||||
# Save assistant reply
|
||||
as_row = models.ChatMessage(session_pk=session.id, role='assistant', content=reply)
|
||||
db.add(as_row)
|
||||
db.commit()
|
||||
|
||||
return {"reply": reply}
|
||||
|
||||
@app.post("/generate-title", response_model=schemas.GenerateTitleResponse)
|
||||
async def generate_title(req: schemas.GenerateTitleRequest, db: Session = Depends(get_db)):
|
||||
@@ -102,7 +121,7 @@ async def generate_title(req: schemas.GenerateTitleRequest, db: Session = Depend
|
||||
prompt = f"Generate a very short, concise title (5 words or less) for a chat conversation that begins with this user message: \"{req.message}\". Do not use quotation marks in the title."
|
||||
|
||||
try:
|
||||
title = await ollama_chat("llama3", [{"role": "user", "content": prompt}])
|
||||
title = await ollama_chat(req.model, [{"role": "user", "content": prompt}])
|
||||
except Exception as e:
|
||||
raise HTTPException(status_code=502, detail=f"Ollama error: {e}")
|
||||
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
|
||||
import httpx
|
||||
from typing import Dict, Any, List
|
||||
import json
|
||||
from typing import Dict, Any, List, AsyncGenerator
|
||||
|
||||
OLLAMA_URL = "http://127.0.0.1:11434"
|
||||
|
||||
@@ -32,3 +33,23 @@ async def chat(model: str, messages: List[Dict[str, str]]) -> str:
|
||||
if msgs:
|
||||
return msgs[-1].get("content", "")
|
||||
return data.get("content", "")
|
||||
|
||||
async def chat_stream(model: str, messages: List[Dict[str, str]]) -> AsyncGenerator[str, None]:
|
||||
payload = {
|
||||
"model": model,
|
||||
"messages": messages,
|
||||
"stream": True
|
||||
}
|
||||
async with httpx.AsyncClient(timeout=600.0) as client:
|
||||
async with client.stream("POST", f"{OLLAMA_URL}/api/chat", json=payload) as r:
|
||||
r.raise_for_status()
|
||||
async for line in r.aiter_lines():
|
||||
if line:
|
||||
try:
|
||||
chunk = json.loads(line)
|
||||
if "content" in chunk: # Newer Ollama format
|
||||
yield chunk["content"]
|
||||
elif "message" in chunk and "content" in chunk["message"]: # Older format
|
||||
yield chunk["message"]["content"]
|
||||
except json.JSONDecodeError:
|
||||
pass # Ignore invalid JSON lines
|
||||
|
||||
@@ -10,6 +10,7 @@ class ChatRequest(BaseModel):
|
||||
session_id: str
|
||||
model: str
|
||||
message: str
|
||||
stream: Optional[bool] = False
|
||||
|
||||
class ChatResponse(BaseModel):
|
||||
reply: str
|
||||
@@ -20,6 +21,7 @@ class HistoryResponse(BaseModel):
|
||||
class GenerateTitleRequest(BaseModel):
|
||||
session_id: str
|
||||
message: str
|
||||
model: str
|
||||
|
||||
class GenerateTitleResponse(BaseModel):
|
||||
title: str
|
||||
|
||||
Reference in New Issue
Block a user