batch4: chats-in-sessions, force-send, /compact, right-rail file browser

Session 1:N Chat data model with backfill. Workspace switches to client-side multi-tab pane management. Right-rail file browser with float-over viewer and click-drag line selection replaces FileBrowserPane. Adds /compact streaming summarizer (respects compact markers in context builder), force-send (cancels in-flight, persists partial as 'cancelled', awaits cancellation completion via deferred Promise + 5s timeout), message queue, stop generation, chat auto-rename, session archive/unarchive with Closed Sessions section on repo landing page. CHECK constraints on sessions.status, messages.role, messages.status with KEEP IN SYNC comments tying to MESSAGE_ROLES / MESSAGE_STATUSES const arrays. Deletes dead pane routes/hook and the api.panes.* client block. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-05-15 20:39:48 +00:00
parent 6d9515b8a5
commit c35ec65fc4
37 changed files with 3290 additions and 1012 deletions
--- a/apps/server/src/services/auto_name.ts
+++ b/apps/server/src/services/auto_name.ts
@@ -5,31 +5,12 @@ const NAMING_SYSTEM_PROMPT =

 const MAX_TITLE_CHARS = 60;

-// QWEN3 NON-STREAMING UTILITY-CALL PATTERN
-// ----------------------------------------
-// Qwen3-family chat templates default to chain-of-thought reasoning: the
-// model emits a long <think>…</think> block into `reasoning_content` and
-// only finalizes a real reply in `content`. For short utility calls
-// (naming, classification, routing, summarization) with a tight token
-// budget, the model burns the entire budget on reasoning and returns:
-//   - content: ""
-//   - reasoning_content: "Thinking Process: 1. ..." (mid-thought, truncated)
-//   - finish_reason: "length"
-// Fix: pass `chat_template_kwargs: { enable_thinking: false }` to skip the
-// thinking block, and keep `max_tokens` low (~30 is plenty for a 4-word
-// title). The kwarg is a no-op for non-Qwen chat templates, so it's safe
-// to apply unconditionally for any short non-streaming model call.
-// Apply this same pattern to: fork-message (planned), agent-routing
-// (planned), web-search summarization (planned).
-
 function cleanTitle(raw: string): string {
  let name = raw.trim();
-  // Strip surrounding straight or smart quotes (one layer).
  const quotes = ['"', "'", '`', '‘', '’', '“', '”'];
  while (name.length >= 2 && quotes.includes(name[0]!) && quotes.includes(name[name.length - 1]!)) {
    name = name.slice(1, -1).trim();
  }
-  // Drop a leading "Title:" prefix if the model added one despite instructions.
  name = name.replace(/^title\s*:\s*/i, '').trim();
  if (name.length > MAX_TITLE_CHARS) {
    name = name.slice(0, MAX_TITLE_CHARS).trim();
@@ -46,13 +27,10 @@ interface NamingResponse {
  }>;
 }

-// Some Qwen-family models emit "thinking" tokens into reasoning_content and
-// only finalize a real reply in content. Pull a sensible candidate string.
 function pickTitleSource(data: NamingResponse): string {
  const choice = data.choices?.[0]?.message;
  if (!choice) return '';
  if (choice.content && choice.content.trim().length > 0) return choice.content;
-  // Fallback: try to extract a last-line title from reasoning, if present.
  const reasoning = choice.reasoning_content ?? '';
  if (reasoning.length === 0) return '';
  const lines = reasoning
@@ -62,38 +40,44 @@ function pickTitleSource(data: NamingResponse): string {
  return lines[lines.length - 1] ?? '';
 }

-export async function maybeAutoNameSession(
+export async function maybeAutoNameChat(
  ctx: InferenceContext,
+  chatId: string,
  sessionId: string
 ): Promise<void> {
  const counts = await ctx.sql<{ n: number }[]>`
    SELECT COUNT(*)::int AS n
    FROM messages
-    WHERE session_id = ${sessionId}
+    WHERE chat_id = ${chatId}
      AND role = 'assistant'
      AND status = 'complete'
  `;
  if (counts[0]?.n !== 1) return;

-  const sessionRows = await ctx.sql<
-    { id: string; name: string; model: string }[]
+  const chatRows = await ctx.sql<
+    { id: string; name: string | null; session_id: string }[]
  >`
-    SELECT id, name, model FROM sessions WHERE id = ${sessionId}
+    SELECT id, name, session_id FROM chats WHERE id = ${chatId}
  `;
-  const session = sessionRows[0];
-  if (!session) return;
-  const existingName = session.name ?? '';
-  if (existingName !== '' && existingName !== 'New session') return;
+  const chat = chatRows[0];
+  if (!chat) return;
+  if (chat.name !== null && chat.name !== '') return;
+
+  const sessionRows = await ctx.sql<{ model: string }[]>`
+    SELECT model FROM sessions WHERE id = ${sessionId}
+  `;
+  const model = sessionRows[0]?.model;
+  if (!model) return;

  const userMsg = await ctx.sql<{ content: string }[]>`
    SELECT content FROM messages
-    WHERE session_id = ${sessionId} AND role = 'user'
+    WHERE chat_id = ${chatId} AND role = 'user'
    ORDER BY created_at ASC
    LIMIT 1
  `;
  const assistantMsg = await ctx.sql<{ content: string }[]>`
    SELECT content FROM messages
-    WHERE session_id = ${sessionId}
+    WHERE chat_id = ${chatId}
      AND role = 'assistant'
      AND status = 'complete'
    ORDER BY created_at ASC
@@ -105,7 +89,7 @@ export async function maybeAutoNameSession(
  const assistantText = assistantMsg[0].content.slice(0, 2000);

  const body = {
-    model: session.model,
+    model,
    messages: [
      { role: 'system', content: NAMING_SYSTEM_PROMPT },
      {
@@ -116,9 +100,6 @@ export async function maybeAutoNameSession(
    max_tokens: 30,
    temperature: 0.3,
    stream: false,
-    // Qwen-family models default to chain-of-thought; this template kwarg
-    // tells llama.cpp's chat template renderer to skip the thinking block.
-    // Harmless for non-Qwen models.
    chat_template_kwargs: { enable_thinking: false },
  };

@@ -135,23 +116,30 @@ export async function maybeAutoNameSession(
  const raw = pickTitleSource(data);
  const name = cleanTitle(raw);
  if (!name) {
-    ctx.log.warn({ sessionId, raw }, 'auto-name: empty title from model');
+    ctx.log.warn({ chatId, raw }, 'auto-name: empty title from model');
    return;
  }

-  const updated = await ctx.sql<{ id: string; name: string }[]>`
-    UPDATE sessions
-    SET name = ${name}, updated_at = NOW()
-    WHERE id = ${sessionId}
-      AND (name IS NULL OR name = '' OR name = 'New session')
-    RETURNING id, name
+  const updated = await ctx.sql<{ id: string; name: string; session_id: string; updated_at: string }[]>`
+    UPDATE chats
+    SET name = ${name}, updated_at = clock_timestamp()
+    WHERE id = ${chatId}
+      AND (name IS NULL OR name = '')
+    RETURNING id, name, session_id, updated_at
  `;
  if (updated.length === 0) return;

  ctx.publish(sessionId, {
-    type: 'session_renamed',
-    session_id: sessionId,
+    type: 'chat_renamed',
+    chat_id: chatId,
    name,
  });
-  ctx.log.info({ sessionId, name }, 'session auto-named');
+  ctx.publishUser({
+    type: 'chat_updated',
+    chat_id: chatId,
+    session_id: sessionId,
+    name,
+    updated_at: updated[0]!.updated_at,
+  });
+  ctx.log.info({ chatId, name }, 'chat auto-named');
 }