batch4: chats-in-sessions, force-send, /compact, right-rail file browser
Session 1:N Chat data model with backfill. Workspace switches to client-side multi-tab pane management. Right-rail file browser with float-over viewer and click-drag line selection replaces FileBrowserPane. Adds /compact streaming summarizer (respects compact markers in context builder), force-send (cancels in-flight, persists partial as 'cancelled', awaits cancellation completion via deferred Promise + 5s timeout), message queue, stop generation, chat auto-rename, session archive/unarchive with Closed Sessions section on repo landing page. CHECK constraints on sessions.status, messages.role, messages.status with KEEP IN SYNC comments tying to MESSAGE_ROLES / MESSAGE_STATUSES const arrays. Deletes dead pane routes/hook and the api.panes.* client block. Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -5,31 +5,12 @@ const NAMING_SYSTEM_PROMPT =
|
||||
|
||||
const MAX_TITLE_CHARS = 60;
|
||||
|
||||
// QWEN3 NON-STREAMING UTILITY-CALL PATTERN
|
||||
// ----------------------------------------
|
||||
// Qwen3-family chat templates default to chain-of-thought reasoning: the
|
||||
// model emits a long <think>…</think> block into `reasoning_content` and
|
||||
// only finalizes a real reply in `content`. For short utility calls
|
||||
// (naming, classification, routing, summarization) with a tight token
|
||||
// budget, the model burns the entire budget on reasoning and returns:
|
||||
// - content: ""
|
||||
// - reasoning_content: "Thinking Process: 1. ..." (mid-thought, truncated)
|
||||
// - finish_reason: "length"
|
||||
// Fix: pass `chat_template_kwargs: { enable_thinking: false }` to skip the
|
||||
// thinking block, and keep `max_tokens` low (~30 is plenty for a 4-word
|
||||
// title). The kwarg is a no-op for non-Qwen chat templates, so it's safe
|
||||
// to apply unconditionally for any short non-streaming model call.
|
||||
// Apply this same pattern to: fork-message (planned), agent-routing
|
||||
// (planned), web-search summarization (planned).
|
||||
|
||||
function cleanTitle(raw: string): string {
|
||||
let name = raw.trim();
|
||||
// Strip surrounding straight or smart quotes (one layer).
|
||||
const quotes = ['"', "'", '`', '‘', '’', '“', '”'];
|
||||
while (name.length >= 2 && quotes.includes(name[0]!) && quotes.includes(name[name.length - 1]!)) {
|
||||
name = name.slice(1, -1).trim();
|
||||
}
|
||||
// Drop a leading "Title:" prefix if the model added one despite instructions.
|
||||
name = name.replace(/^title\s*:\s*/i, '').trim();
|
||||
if (name.length > MAX_TITLE_CHARS) {
|
||||
name = name.slice(0, MAX_TITLE_CHARS).trim();
|
||||
@@ -46,13 +27,10 @@ interface NamingResponse {
|
||||
}>;
|
||||
}
|
||||
|
||||
// Some Qwen-family models emit "thinking" tokens into reasoning_content and
|
||||
// only finalize a real reply in content. Pull a sensible candidate string.
|
||||
function pickTitleSource(data: NamingResponse): string {
|
||||
const choice = data.choices?.[0]?.message;
|
||||
if (!choice) return '';
|
||||
if (choice.content && choice.content.trim().length > 0) return choice.content;
|
||||
// Fallback: try to extract a last-line title from reasoning, if present.
|
||||
const reasoning = choice.reasoning_content ?? '';
|
||||
if (reasoning.length === 0) return '';
|
||||
const lines = reasoning
|
||||
@@ -62,38 +40,44 @@ function pickTitleSource(data: NamingResponse): string {
|
||||
return lines[lines.length - 1] ?? '';
|
||||
}
|
||||
|
||||
export async function maybeAutoNameSession(
|
||||
export async function maybeAutoNameChat(
|
||||
ctx: InferenceContext,
|
||||
chatId: string,
|
||||
sessionId: string
|
||||
): Promise<void> {
|
||||
const counts = await ctx.sql<{ n: number }[]>`
|
||||
SELECT COUNT(*)::int AS n
|
||||
FROM messages
|
||||
WHERE session_id = ${sessionId}
|
||||
WHERE chat_id = ${chatId}
|
||||
AND role = 'assistant'
|
||||
AND status = 'complete'
|
||||
`;
|
||||
if (counts[0]?.n !== 1) return;
|
||||
|
||||
const sessionRows = await ctx.sql<
|
||||
{ id: string; name: string; model: string }[]
|
||||
const chatRows = await ctx.sql<
|
||||
{ id: string; name: string | null; session_id: string }[]
|
||||
>`
|
||||
SELECT id, name, model FROM sessions WHERE id = ${sessionId}
|
||||
SELECT id, name, session_id FROM chats WHERE id = ${chatId}
|
||||
`;
|
||||
const session = sessionRows[0];
|
||||
if (!session) return;
|
||||
const existingName = session.name ?? '';
|
||||
if (existingName !== '' && existingName !== 'New session') return;
|
||||
const chat = chatRows[0];
|
||||
if (!chat) return;
|
||||
if (chat.name !== null && chat.name !== '') return;
|
||||
|
||||
const sessionRows = await ctx.sql<{ model: string }[]>`
|
||||
SELECT model FROM sessions WHERE id = ${sessionId}
|
||||
`;
|
||||
const model = sessionRows[0]?.model;
|
||||
if (!model) return;
|
||||
|
||||
const userMsg = await ctx.sql<{ content: string }[]>`
|
||||
SELECT content FROM messages
|
||||
WHERE session_id = ${sessionId} AND role = 'user'
|
||||
WHERE chat_id = ${chatId} AND role = 'user'
|
||||
ORDER BY created_at ASC
|
||||
LIMIT 1
|
||||
`;
|
||||
const assistantMsg = await ctx.sql<{ content: string }[]>`
|
||||
SELECT content FROM messages
|
||||
WHERE session_id = ${sessionId}
|
||||
WHERE chat_id = ${chatId}
|
||||
AND role = 'assistant'
|
||||
AND status = 'complete'
|
||||
ORDER BY created_at ASC
|
||||
@@ -105,7 +89,7 @@ export async function maybeAutoNameSession(
|
||||
const assistantText = assistantMsg[0].content.slice(0, 2000);
|
||||
|
||||
const body = {
|
||||
model: session.model,
|
||||
model,
|
||||
messages: [
|
||||
{ role: 'system', content: NAMING_SYSTEM_PROMPT },
|
||||
{
|
||||
@@ -116,9 +100,6 @@ export async function maybeAutoNameSession(
|
||||
max_tokens: 30,
|
||||
temperature: 0.3,
|
||||
stream: false,
|
||||
// Qwen-family models default to chain-of-thought; this template kwarg
|
||||
// tells llama.cpp's chat template renderer to skip the thinking block.
|
||||
// Harmless for non-Qwen models.
|
||||
chat_template_kwargs: { enable_thinking: false },
|
||||
};
|
||||
|
||||
@@ -135,23 +116,30 @@ export async function maybeAutoNameSession(
|
||||
const raw = pickTitleSource(data);
|
||||
const name = cleanTitle(raw);
|
||||
if (!name) {
|
||||
ctx.log.warn({ sessionId, raw }, 'auto-name: empty title from model');
|
||||
ctx.log.warn({ chatId, raw }, 'auto-name: empty title from model');
|
||||
return;
|
||||
}
|
||||
|
||||
const updated = await ctx.sql<{ id: string; name: string }[]>`
|
||||
UPDATE sessions
|
||||
SET name = ${name}, updated_at = NOW()
|
||||
WHERE id = ${sessionId}
|
||||
AND (name IS NULL OR name = '' OR name = 'New session')
|
||||
RETURNING id, name
|
||||
const updated = await ctx.sql<{ id: string; name: string; session_id: string; updated_at: string }[]>`
|
||||
UPDATE chats
|
||||
SET name = ${name}, updated_at = clock_timestamp()
|
||||
WHERE id = ${chatId}
|
||||
AND (name IS NULL OR name = '')
|
||||
RETURNING id, name, session_id, updated_at
|
||||
`;
|
||||
if (updated.length === 0) return;
|
||||
|
||||
ctx.publish(sessionId, {
|
||||
type: 'session_renamed',
|
||||
session_id: sessionId,
|
||||
type: 'chat_renamed',
|
||||
chat_id: chatId,
|
||||
name,
|
||||
});
|
||||
ctx.log.info({ sessionId, name }, 'session auto-named');
|
||||
ctx.publishUser({
|
||||
type: 'chat_updated',
|
||||
chat_id: chatId,
|
||||
session_id: sessionId,
|
||||
name,
|
||||
updated_at: updated[0]!.updated_at,
|
||||
});
|
||||
ctx.log.info({ chatId, name }, 'chat auto-named');
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user