anvilsign in

collin/strudel-claude

main / server / index.js
1import express from 'express';
2import Anthropic from '@anthropic-ai/sdk';
3import http from 'node:http';
4import path from 'node:path';
5import fs from 'node:fs';
6import { fileURLToPath } from 'node:url';
7import { attachBroadcast } from './broadcast.js';
8
9const __dirname = path.dirname(fileURLToPath(import.meta.url));
10const DIST_DIR = path.join(__dirname, '..', 'dist');
11const PROMPT_PATH = path.join(__dirname, '..', 'PROMPT.md');
12const TUI_PATH = path.join(__dirname, '..', 'tui', 'strudel-tui.py');
13
14const app = express();
15app.use(express.json({ limit: '1mb' }));
16
17// Resolves ANTHROPIC_API_KEY from the environment (or an `ant auth login`
18// profile). The key stays server-side and never reaches the browser. It is
19// optional: users can bring their own key from the app instead, in which case
20// it arrives per request in the x-anthropic-key header and is used for that
21// request only — never logged, never stored.
22let serverClient = null;
23try {
24 serverClient = new Anthropic();
25} catch {
26 console.log(' No server-side key — the app will ask each user for their own.');
27}
28
29const clientFor = (key) => (key ? new Anthropic({ apiKey: key }) : serverClient);
30
31// The models this app is allowed to use. The client chooses by id from this
32// list and nothing else: an id that isn't here is refused rather than quietly
33// swapped for a default, so a hand-edited request can't point the backend at an
34// arbitrary (or expensive) model. Adding one is a deploy, not a client change.
35//
36// `blurb` is user-facing — it goes in the picker. `thinks` and `effort` are the
37// per-model API differences, not preferences; see where the request is built.
38const MODELS = [
39 {
40 id: 'claude-opus-4-8',
41 label: 'Opus 4.8',
42 blurb: 'The default. Richest, most musical patterns.',
43 },
44 {
45 id: 'claude-opus-5',
46 label: 'Opus 5',
47 blurb: 'Newest and strongest, but it thinks first — expect a longer wait.',
48 // Thinking is on by default on this one and is billed against max_tokens,
49 // so it needs headroom for the reasoning as well as the pattern.
50 thinks: true,
51 },
52 {
53 id: 'claude-sonnet-5',
54 label: 'Sonnet 5',
55 blurb: 'Close behind Opus and noticeably quicker.',
56 },
57 {
58 id: 'claude-haiku-4-5',
59 label: 'Haiku 4.5',
60 blurb: 'Fastest. Good for small tweaks to a pattern that already works.',
61 // Haiku 4.5 rejects output_config.effort outright — sending it is a 400.
62 effort: false,
63 // Its minimum cacheable prefix is 4096 tokens. The PROMPT.md system prompt
64 // clears that on its own, so unlike the old inline primer it does get cached
65 // on this model too.
66 },
67];
68
69const DEFAULT_MODEL = MODELS[0].id;
70const modelById = new Map(MODELS.map((m) => [m.id, m]));
71
72// Lets the app tell the user whether a key of their own is required, and which
73// models it may offer. Only the display fields go out — the capability flags are
74// the server's business.
75app.get('/api/config', (_req, res) => {
76 res.json({
77 hasServerKey: Boolean(serverClient),
78 models: MODELS.map(({ id, label, blurb }) => ({ id, label, blurb })),
79 defaultModel: DEFAULT_MODEL,
80 });
81});
82
83// The default system prompt, so the app can show it and offer it back as the
84// "reset" state. A user's own edits live in their browser and ride along with
85// the request — the file on disk is the server's, and stays untouched.
86app.get('/api/prompt', (_req, res) => {
87 res.json({ prompt: SYSTEM });
88});
89
90// The terminal visualiser, served as source. It is a single stdlib-only Python
91// file on purpose: `curl <host>/tui.py | less` to read it, `> strudel-tui.py`
92// to keep it, `python3 strudel-tui.py <host>` to run it. text/plain so a
93// browser shows it instead of downloading it.
94app.get(['/tui.py', '/strudel-tui.py'], (_req, res) => {
95 fs.readFile(TUI_PATH, 'utf8', (err, src) => {
96 if (err) return res.status(404).type('text/plain').send('# tui/strudel-tui.py not found\n');
97 res.type('text/plain; charset=utf-8').send(src);
98 });
99});
100
101// Thinking is off on this model unless asked for, but `effort` still governs how
102// much the model deliberates and how many tokens it spends. The default is
103// "high"; a four-bar drum pattern does not need it, and dropping a level is the
104// cheapest latency win available.
105const EFFORT = process.env.STRUDEL_EFFORT || 'medium';
106
107// The system prompt lives in PROMPT.md so it can be edited — and diffed — without
108// touching server code. It is read once at boot: the prompt is cached upstream
109// (see the cache_control on the system block), and re-reading per request would
110// let the file change mid-session and silently invalidate that cache.
111//
112// The sound inventory in that file is not decorative. It has to match what
113// src/strudel.js actually prebakes, because the prompt tells the model not to
114// invent names — and a name that isn't loaded plays silence rather than erroring.
115// If you change the prebake, change section 3.
116let SYSTEM;
117try {
118 SYSTEM = fs.readFileSync(PROMPT_PATH, 'utf8').trim();
119} catch (err) {
120 console.error(`Could not read the system prompt at ${PROMPT_PATH}: ${err.message}`);
121 process.exit(1);
122}
123if (!SYSTEM) {
124 console.error(`The system prompt at ${PROMPT_PATH} is empty.`);
125 process.exit(1);
126}
127
128const OUTPUT_SCHEMA = {
129 type: 'object',
130 properties: {
131 message: {
132 type: 'string',
133 description: 'A short, friendly reply to the user (1-2 sentences).',
134 },
135 code: {
136 type: 'string',
137 description:
138 'The full runnable Strudel pattern, or an empty string if no music change is needed.',
139 },
140 },
141 required: ['message', 'code'],
142 additionalProperties: false,
143};
144
145// The reply is spec'd as 1-2 sentences. When a session runs long and repetitive
146// ("more twinkles", "more sparkle"), sampling can collapse into a loop that
147// splices its own phrasings together and drifts into stray HTML (<br>, </p>).
148// Never let that reach the chat — or the history, where it would feed the loop.
149const MAX_MESSAGE_LEN = 400;
150
151function cleanMessage(raw) {
152 if (typeof raw !== 'string') return '';
153 const looksLikeMarkup = /<\/?(br|p|div|span)\b[^>]*>/i.test(raw);
154
155 const text = raw.replace(/<[^>]*>/g, ' ').replace(/\s+/g, ' ').trim();
156
157 // A phrase repeated three times over is the signature of the loop, not style.
158 const sentences = text.split(/(?<=[.!?])\s+/).map((s) => s.toLowerCase().trim()).filter(Boolean);
159 const repeats = sentences.length - new Set(sentences).size;
160
161 if (looksLikeMarkup || repeats >= 2) return 'Updated the pattern.';
162 if (text.length <= MAX_MESSAGE_LEN) return text;
163 return text.slice(0, MAX_MESSAGE_LEN).replace(/\s+\S*$/, '') + '…';
164}
165
166// Cap how much transcript goes back to the API; the client keeps the full record
167// for display and revert regardless. Trim to a user message so the first turn
168// sent is a user turn, as the API requires.
169//
170// Trimming happens in chunks rather than one turn at a time on purpose. Prompt
171// caching is a prefix match, so dropping the oldest turn on every request would
172// move the start of the prompt each time and throw the whole cache away. Cutting
173// back to HISTORY_KEEP only when HISTORY_LIMIT is passed means one cold request
174// every eight turns instead of every single one.
175// A user may edit the system prompt in the app; their version arrives with each
176// request. Ceiling is generosity, not policy — PROMPT.md is ~23 KB and an edited
177// copy should have room to grow, but a request body isn't a place to park a
178// novel. Anything blank or oversized falls back to the file on disk.
179const MAX_SYSTEM_LEN = 200_000;
180
181function systemFor(override) {
182 if (typeof override !== 'string') return SYSTEM;
183 const text = override.trim();
184 if (!text || text.length > MAX_SYSTEM_LEN) return SYSTEM;
185 return text;
186}
187
188const HISTORY_LIMIT = 40;
189const HISTORY_KEEP = 24;
190
191function trimHistory(messages) {
192 if (messages.length <= HISTORY_LIMIT) return messages;
193 const tail = messages.slice(-HISTORY_KEEP);
194 const firstUser = tail.findIndex((m) => m.role === 'user');
195 return firstUser <= 0 ? tail : tail.slice(firstUser);
196}
197
198// How a stored turn is rendered for the API. Assistant turns carry the pattern
199// they produced, in the same JSON shape the model emits — without it a long
200// session shows the model nothing but its own content-free replies ("Added more
201// sparkle.") with no record of what it actually wrote, and the patterns drift.
202//
203// This rendering must be stable: once a turn has been sent one way it has to
204// keep being sent that way, or the cached prefix breaks on the next request.
205const renderTurn = (m) =>
206 m.role === 'assistant' && m.code
207 ? JSON.stringify({ message: m.content || '', code: m.code })
208 : m.content || '';
209
210app.post('/api/generate', async (req, res) => {
211 try {
212 const {
213 messages = [],
214 currentCode = '',
215 model: requestedModel,
216 system: systemOverride,
217 } = req.body;
218 if (!Array.isArray(messages) || messages.length === 0) {
219 return res.status(400).json({ error: 'messages array required' });
220 }
221
222 // Refuse an unknown id outright. Falling back to the default would be
223 // friendlier to a stale client but would also mean the picker could say
224 // "Opus" while something else answered — the app resets an unavailable
225 // choice at load instead, so this should only fire on a forged request.
226 const model = modelById.get(requestedModel || DEFAULT_MODEL);
227 if (!model) {
228 return res.status(400).json({
229 error: `unknown model — this server offers ${MODELS.map((m) => m.label).join(', ')}`,
230 });
231 }
232
233 const client = clientFor(req.get('x-anthropic-key')?.trim());
234 if (!client) {
235 return res.status(401).json({
236 error: 'no API key — add your Anthropic key with the 🔑 button, or set ANTHROPIC_API_KEY on the server',
237 });
238 }
239
240 // Only role/content go to the API — the client also carries per-message
241 // bookkeeping (e.g. codeBefore for revert) that the API would reject.
242 const recent = trimHistory(messages);
243 const withContext = recent.map((m) => ({
244 role: m.role,
245 content: [{ type: 'text', text: renderTurn(m) }],
246 }));
247
248 // Everything up to and including the user's request is byte-identical to
249 // what the next request will send, so it is worth caching; the editor
250 // contents are not, and go after the breakpoint as their own turn. Merging
251 // them into the user's message instead (as this used to) rewrites that turn
252 // every time the pattern changes and costs a turn's worth of cache each go.
253 if (withContext.length) {
254 withContext[withContext.length - 1].content[0].cache_control = { type: 'ephemeral' };
255 }
256 withContext.push({
257 role: 'user',
258 content: [
259 {
260 type: 'text',
261 text: `Current pattern in the editor:\n\`\`\`\n${currentCode || '(empty)'}\n\`\`\``,
262 },
263 ],
264 });
265
266 const response = await client.messages.create({
267 model: model.id,
268 // Headroom so a big pattern + message can't truncate the JSON. Models that
269 // think spend the same budget on reasoning first, so they get more.
270 max_tokens: model.thinks ? 8192 : 4096,
271 // Still cached when the user brought their own: a hand-edited prompt is
272 // stable across a session too, it just warms a different cache entry.
273 system: [
274 { type: 'text', text: systemFor(systemOverride), cache_control: { type: 'ephemeral' } },
275 ],
276 messages: withContext,
277 output_config: {
278 ...(model.effort === false ? {} : { effort: EFFORT }),
279 format: { type: 'json_schema', schema: OUTPUT_SCHEMA },
280 },
281 });
282
283 if (process.env.STRUDEL_DEBUG_USAGE) {
284 const u = response.usage;
285 console.log(
286 `usage[${model.id}]: in=${u.input_tokens} cache_read=${u.cache_read_input_tokens} ` +
287 `cache_write=${u.cache_creation_input_tokens} out=${u.output_tokens}`,
288 );
289 }
290
291 // If the model hit the token cap the JSON is incomplete — never ship a
292 // half-parsed pattern to the editor.
293 if (response.stop_reason === 'max_tokens') {
294 return res.json({
295 message: 'That got a bit long and I ran out of room — try again, maybe a touch simpler.',
296 code: '',
297 });
298 }
299
300 const text = response.content.find((b) => b.type === 'text')?.text ?? '{}';
301 let parsed;
302 try {
303 parsed = JSON.parse(text);
304 } catch {
305 return res.json({
306 message: "I couldn't format that cleanly — mind trying again?",
307 code: '',
308 });
309 }
310 // Only pass through a string code field; anything else means "no change".
311 let code = typeof parsed.code === 'string' ? parsed.code : '';
312 // Defensively strip a markdown code fence if the model added one — a stray
313 // ``` at char 0 is exactly the "Unexpected token (1:0)" Strudel parse error.
314 code = code
315 .replace(/^\s*```[a-zA-Z]*\s*\n?/, '')
316 .replace(/\n?```\s*$/, '')
317 .trim();
318 res.json({ message: cleanMessage(parsed.message), code });
319 } catch (err) {
320 console.error('generate error:', err.status, err.message);
321 // Point auth failures at the key rather than at a generic server error.
322 if (err.status === 401 || err.status === 403) {
323 return res.status(401).json({ error: 'that API key was rejected — check it with the 🔑 button' });
324 }
325 if (err.status === 429) {
326 return res.status(429).json({ error: 'rate limited by the API — give it a moment and try again' });
327 }
328 res.status(500).json({ error: err.message || 'generation failed' });
329 }
330});
331
332// Serve the built frontend from the same origin (so one port / one ngrok
333// tunnel serves both the app and the API). In dev, dist/ may not exist yet —
334// Vite serves the frontend and proxies /api here instead.
335if (fs.existsSync(DIST_DIR)) {
336 app.use(express.static(DIST_DIR));
337 // SPA fallback for any non-/api GET (Express 5-safe: middleware, not a wildcard route).
338 app.use((req, res, next) => {
339 if (req.method === 'GET' && !req.path.startsWith('/api')) {
340 return res.sendFile(path.join(DIST_DIR, 'index.html'));
341 }
342 next();
343 });
344}
345
346const PORT = process.env.PORT || 8787;
347// An explicit http.Server so the WebSocket hub can share the port (and any
348// tunnel) with the app and the API.
349const server = http.createServer(app);
350attachBroadcast(server);
351server.listen(PORT, () => {
352 console.log(`🎛 Strudel×Claude backend on http://localhost:${PORT}`);
353 console.log(` Prompt: PROMPT.md (${(SYSTEM.length / 1024).toFixed(1)} KB)`);
354 console.log(` Timings: ws://localhost:${PORT}/ws · terminal view: curl -s http://localhost:${PORT}/tui.py | python3 -`);
355 if (!process.env.ANTHROPIC_API_KEY) {
356 console.log(
357 ' Note: ANTHROPIC_API_KEY not set — relying on an `ant auth login` profile if present.',
358 );
359 }
360});