| 1 | import express from 'express'; |
| 2 | import Anthropic from '@anthropic-ai/sdk'; |
| 3 | import http from 'node:http'; |
| 4 | import path from 'node:path'; |
| 5 | import fs from 'node:fs'; |
| 6 | import { fileURLToPath } from 'node:url'; |
| 7 | import { attachBroadcast } from './broadcast.js'; |
| 8 | |
| 9 | const __dirname = path.dirname(fileURLToPath(import.meta.url)); |
| 10 | const DIST_DIR = path.join(__dirname, '..', 'dist'); |
| 11 | const PROMPT_PATH = path.join(__dirname, '..', 'PROMPT.md'); |
| 12 | const TUI_PATH = path.join(__dirname, '..', 'tui', 'strudel-tui.py'); |
| 13 | |
| 14 | const app = express(); |
| 15 | app.use(express.json({ limit: '1mb' })); |
| 16 | |
| 17 | // Resolves ANTHROPIC_API_KEY from the environment (or an `ant auth login` |
| 18 | // profile). The key stays server-side and never reaches the browser. It is |
| 19 | // optional: users can bring their own key from the app instead, in which case |
| 20 | // it arrives per request in the x-anthropic-key header and is used for that |
| 21 | // request only — never logged, never stored. |
| 22 | let serverClient = null; |
| 23 | try { |
| 24 | serverClient = new Anthropic(); |
| 25 | } catch { |
| 26 | console.log(' No server-side key — the app will ask each user for their own.'); |
| 27 | } |
| 28 | |
| 29 | const clientFor = (key) => (key ? new Anthropic({ apiKey: key }) : serverClient); |
| 30 | |
| 31 | // The models this app is allowed to use. The client chooses by id from this |
| 32 | // list and nothing else: an id that isn't here is refused rather than quietly |
| 33 | // swapped for a default, so a hand-edited request can't point the backend at an |
| 34 | // arbitrary (or expensive) model. Adding one is a deploy, not a client change. |
| 35 | // |
| 36 | // `blurb` is user-facing — it goes in the picker. `thinks` and `effort` are the |
| 37 | // per-model API differences, not preferences; see where the request is built. |
| 38 | const MODELS = [ |
| 39 | { |
| 40 | id: 'claude-opus-4-8', |
| 41 | label: 'Opus 4.8', |
| 42 | blurb: 'The default. Richest, most musical patterns.', |
| 43 | }, |
| 44 | { |
| 45 | id: 'claude-opus-5', |
| 46 | label: 'Opus 5', |
| 47 | blurb: 'Newest and strongest, but it thinks first — expect a longer wait.', |
| 48 | // Thinking is on by default on this one and is billed against max_tokens, |
| 49 | // so it needs headroom for the reasoning as well as the pattern. |
| 50 | thinks: true, |
| 51 | }, |
| 52 | { |
| 53 | id: 'claude-sonnet-5', |
| 54 | label: 'Sonnet 5', |
| 55 | blurb: 'Close behind Opus and noticeably quicker.', |
| 56 | }, |
| 57 | { |
| 58 | id: 'claude-haiku-4-5', |
| 59 | label: 'Haiku 4.5', |
| 60 | blurb: 'Fastest. Good for small tweaks to a pattern that already works.', |
| 61 | // Haiku 4.5 rejects output_config.effort outright — sending it is a 400. |
| 62 | effort: false, |
| 63 | // Its minimum cacheable prefix is 4096 tokens. The PROMPT.md system prompt |
| 64 | // clears that on its own, so unlike the old inline primer it does get cached |
| 65 | // on this model too. |
| 66 | }, |
| 67 | ]; |
| 68 | |
| 69 | const DEFAULT_MODEL = MODELS[0].id; |
| 70 | const modelById = new Map(MODELS.map((m) => [m.id, m])); |
| 71 | |
| 72 | // Lets the app tell the user whether a key of their own is required, and which |
| 73 | // models it may offer. Only the display fields go out — the capability flags are |
| 74 | // the server's business. |
| 75 | app.get('/api/config', (_req, res) => { |
| 76 | res.json({ |
| 77 | hasServerKey: Boolean(serverClient), |
| 78 | models: MODELS.map(({ id, label, blurb }) => ({ id, label, blurb })), |
| 79 | defaultModel: DEFAULT_MODEL, |
| 80 | }); |
| 81 | }); |
| 82 | |
| 83 | // The default system prompt, so the app can show it and offer it back as the |
| 84 | // "reset" state. A user's own edits live in their browser and ride along with |
| 85 | // the request — the file on disk is the server's, and stays untouched. |
| 86 | app.get('/api/prompt', (_req, res) => { |
| 87 | res.json({ prompt: SYSTEM }); |
| 88 | }); |
| 89 | |
| 90 | // The terminal visualiser, served as source. It is a single stdlib-only Python |
| 91 | // file on purpose: `curl <host>/tui.py | less` to read it, `> strudel-tui.py` |
| 92 | // to keep it, `python3 strudel-tui.py <host>` to run it. text/plain so a |
| 93 | // browser shows it instead of downloading it. |
| 94 | app.get(['/tui.py', '/strudel-tui.py'], (_req, res) => { |
| 95 | fs.readFile(TUI_PATH, 'utf8', (err, src) => { |
| 96 | if (err) return res.status(404).type('text/plain').send('# tui/strudel-tui.py not found\n'); |
| 97 | res.type('text/plain; charset=utf-8').send(src); |
| 98 | }); |
| 99 | }); |
| 100 | |
| 101 | // Thinking is off on this model unless asked for, but `effort` still governs how |
| 102 | // much the model deliberates and how many tokens it spends. The default is |
| 103 | // "high"; a four-bar drum pattern does not need it, and dropping a level is the |
| 104 | // cheapest latency win available. |
| 105 | const EFFORT = process.env.STRUDEL_EFFORT || 'medium'; |
| 106 | |
| 107 | // The system prompt lives in PROMPT.md so it can be edited — and diffed — without |
| 108 | // touching server code. It is read once at boot: the prompt is cached upstream |
| 109 | // (see the cache_control on the system block), and re-reading per request would |
| 110 | // let the file change mid-session and silently invalidate that cache. |
| 111 | // |
| 112 | // The sound inventory in that file is not decorative. It has to match what |
| 113 | // src/strudel.js actually prebakes, because the prompt tells the model not to |
| 114 | // invent names — and a name that isn't loaded plays silence rather than erroring. |
| 115 | // If you change the prebake, change section 3. |
| 116 | let SYSTEM; |
| 117 | try { |
| 118 | SYSTEM = fs.readFileSync(PROMPT_PATH, 'utf8').trim(); |
| 119 | } catch (err) { |
| 120 | console.error(`Could not read the system prompt at ${PROMPT_PATH}: ${err.message}`); |
| 121 | process.exit(1); |
| 122 | } |
| 123 | if (!SYSTEM) { |
| 124 | console.error(`The system prompt at ${PROMPT_PATH} is empty.`); |
| 125 | process.exit(1); |
| 126 | } |
| 127 | |
| 128 | const OUTPUT_SCHEMA = { |
| 129 | type: 'object', |
| 130 | properties: { |
| 131 | message: { |
| 132 | type: 'string', |
| 133 | description: 'A short, friendly reply to the user (1-2 sentences).', |
| 134 | }, |
| 135 | code: { |
| 136 | type: 'string', |
| 137 | description: |
| 138 | 'The full runnable Strudel pattern, or an empty string if no music change is needed.', |
| 139 | }, |
| 140 | }, |
| 141 | required: ['message', 'code'], |
| 142 | additionalProperties: false, |
| 143 | }; |
| 144 | |
| 145 | // The reply is spec'd as 1-2 sentences. When a session runs long and repetitive |
| 146 | // ("more twinkles", "more sparkle"), sampling can collapse into a loop that |
| 147 | // splices its own phrasings together and drifts into stray HTML (<br>, </p>). |
| 148 | // Never let that reach the chat — or the history, where it would feed the loop. |
| 149 | const MAX_MESSAGE_LEN = 400; |
| 150 | |
| 151 | function cleanMessage(raw) { |
| 152 | if (typeof raw !== 'string') return ''; |
| 153 | const looksLikeMarkup = /<\/?(br|p|div|span)\b[^>]*>/i.test(raw); |
| 154 | |
| 155 | const text = raw.replace(/<[^>]*>/g, ' ').replace(/\s+/g, ' ').trim(); |
| 156 | |
| 157 | // A phrase repeated three times over is the signature of the loop, not style. |
| 158 | const sentences = text.split(/(?<=[.!?])\s+/).map((s) => s.toLowerCase().trim()).filter(Boolean); |
| 159 | const repeats = sentences.length - new Set(sentences).size; |
| 160 | |
| 161 | if (looksLikeMarkup || repeats >= 2) return 'Updated the pattern.'; |
| 162 | if (text.length <= MAX_MESSAGE_LEN) return text; |
| 163 | return text.slice(0, MAX_MESSAGE_LEN).replace(/\s+\S*$/, '') + '…'; |
| 164 | } |
| 165 | |
| 166 | // Cap how much transcript goes back to the API; the client keeps the full record |
| 167 | // for display and revert regardless. Trim to a user message so the first turn |
| 168 | // sent is a user turn, as the API requires. |
| 169 | // |
| 170 | // Trimming happens in chunks rather than one turn at a time on purpose. Prompt |
| 171 | // caching is a prefix match, so dropping the oldest turn on every request would |
| 172 | // move the start of the prompt each time and throw the whole cache away. Cutting |
| 173 | // back to HISTORY_KEEP only when HISTORY_LIMIT is passed means one cold request |
| 174 | // every eight turns instead of every single one. |
| 175 | // A user may edit the system prompt in the app; their version arrives with each |
| 176 | // request. Ceiling is generosity, not policy — PROMPT.md is ~23 KB and an edited |
| 177 | // copy should have room to grow, but a request body isn't a place to park a |
| 178 | // novel. Anything blank or oversized falls back to the file on disk. |
| 179 | const MAX_SYSTEM_LEN = 200_000; |
| 180 | |
| 181 | function systemFor(override) { |
| 182 | if (typeof override !== 'string') return SYSTEM; |
| 183 | const text = override.trim(); |
| 184 | if (!text || text.length > MAX_SYSTEM_LEN) return SYSTEM; |
| 185 | return text; |
| 186 | } |
| 187 | |
| 188 | const HISTORY_LIMIT = 40; |
| 189 | const HISTORY_KEEP = 24; |
| 190 | |
| 191 | function trimHistory(messages) { |
| 192 | if (messages.length <= HISTORY_LIMIT) return messages; |
| 193 | const tail = messages.slice(-HISTORY_KEEP); |
| 194 | const firstUser = tail.findIndex((m) => m.role === 'user'); |
| 195 | return firstUser <= 0 ? tail : tail.slice(firstUser); |
| 196 | } |
| 197 | |
| 198 | // How a stored turn is rendered for the API. Assistant turns carry the pattern |
| 199 | // they produced, in the same JSON shape the model emits — without it a long |
| 200 | // session shows the model nothing but its own content-free replies ("Added more |
| 201 | // sparkle.") with no record of what it actually wrote, and the patterns drift. |
| 202 | // |
| 203 | // This rendering must be stable: once a turn has been sent one way it has to |
| 204 | // keep being sent that way, or the cached prefix breaks on the next request. |
| 205 | const renderTurn = (m) => |
| 206 | m.role === 'assistant' && m.code |
| 207 | ? JSON.stringify({ message: m.content || '', code: m.code }) |
| 208 | : m.content || ''; |
| 209 | |
| 210 | app.post('/api/generate', async (req, res) => { |
| 211 | try { |
| 212 | const { |
| 213 | messages = [], |
| 214 | currentCode = '', |
| 215 | model: requestedModel, |
| 216 | system: systemOverride, |
| 217 | } = req.body; |
| 218 | if (!Array.isArray(messages) || messages.length === 0) { |
| 219 | return res.status(400).json({ error: 'messages array required' }); |
| 220 | } |
| 221 | |
| 222 | // Refuse an unknown id outright. Falling back to the default would be |
| 223 | // friendlier to a stale client but would also mean the picker could say |
| 224 | // "Opus" while something else answered — the app resets an unavailable |
| 225 | // choice at load instead, so this should only fire on a forged request. |
| 226 | const model = modelById.get(requestedModel || DEFAULT_MODEL); |
| 227 | if (!model) { |
| 228 | return res.status(400).json({ |
| 229 | error: `unknown model — this server offers ${MODELS.map((m) => m.label).join(', ')}`, |
| 230 | }); |
| 231 | } |
| 232 | |
| 233 | const client = clientFor(req.get('x-anthropic-key')?.trim()); |
| 234 | if (!client) { |
| 235 | return res.status(401).json({ |
| 236 | error: 'no API key — add your Anthropic key with the 🔑 button, or set ANTHROPIC_API_KEY on the server', |
| 237 | }); |
| 238 | } |
| 239 | |
| 240 | // Only role/content go to the API — the client also carries per-message |
| 241 | // bookkeeping (e.g. codeBefore for revert) that the API would reject. |
| 242 | const recent = trimHistory(messages); |
| 243 | const withContext = recent.map((m) => ({ |
| 244 | role: m.role, |
| 245 | content: [{ type: 'text', text: renderTurn(m) }], |
| 246 | })); |
| 247 | |
| 248 | // Everything up to and including the user's request is byte-identical to |
| 249 | // what the next request will send, so it is worth caching; the editor |
| 250 | // contents are not, and go after the breakpoint as their own turn. Merging |
| 251 | // them into the user's message instead (as this used to) rewrites that turn |
| 252 | // every time the pattern changes and costs a turn's worth of cache each go. |
| 253 | if (withContext.length) { |
| 254 | withContext[withContext.length - 1].content[0].cache_control = { type: 'ephemeral' }; |
| 255 | } |
| 256 | withContext.push({ |
| 257 | role: 'user', |
| 258 | content: [ |
| 259 | { |
| 260 | type: 'text', |
| 261 | text: `Current pattern in the editor:\n\`\`\`\n${currentCode || '(empty)'}\n\`\`\``, |
| 262 | }, |
| 263 | ], |
| 264 | }); |
| 265 | |
| 266 | const response = await client.messages.create({ |
| 267 | model: model.id, |
| 268 | // Headroom so a big pattern + message can't truncate the JSON. Models that |
| 269 | // think spend the same budget on reasoning first, so they get more. |
| 270 | max_tokens: model.thinks ? 8192 : 4096, |
| 271 | // Still cached when the user brought their own: a hand-edited prompt is |
| 272 | // stable across a session too, it just warms a different cache entry. |
| 273 | system: [ |
| 274 | { type: 'text', text: systemFor(systemOverride), cache_control: { type: 'ephemeral' } }, |
| 275 | ], |
| 276 | messages: withContext, |
| 277 | output_config: { |
| 278 | ...(model.effort === false ? {} : { effort: EFFORT }), |
| 279 | format: { type: 'json_schema', schema: OUTPUT_SCHEMA }, |
| 280 | }, |
| 281 | }); |
| 282 | |
| 283 | if (process.env.STRUDEL_DEBUG_USAGE) { |
| 284 | const u = response.usage; |
| 285 | console.log( |
| 286 | `usage[${model.id}]: in=${u.input_tokens} cache_read=${u.cache_read_input_tokens} ` + |
| 287 | `cache_write=${u.cache_creation_input_tokens} out=${u.output_tokens}`, |
| 288 | ); |
| 289 | } |
| 290 | |
| 291 | // If the model hit the token cap the JSON is incomplete — never ship a |
| 292 | // half-parsed pattern to the editor. |
| 293 | if (response.stop_reason === 'max_tokens') { |
| 294 | return res.json({ |
| 295 | message: 'That got a bit long and I ran out of room — try again, maybe a touch simpler.', |
| 296 | code: '', |
| 297 | }); |
| 298 | } |
| 299 | |
| 300 | const text = response.content.find((b) => b.type === 'text')?.text ?? '{}'; |
| 301 | let parsed; |
| 302 | try { |
| 303 | parsed = JSON.parse(text); |
| 304 | } catch { |
| 305 | return res.json({ |
| 306 | message: "I couldn't format that cleanly — mind trying again?", |
| 307 | code: '', |
| 308 | }); |
| 309 | } |
| 310 | // Only pass through a string code field; anything else means "no change". |
| 311 | let code = typeof parsed.code === 'string' ? parsed.code : ''; |
| 312 | // Defensively strip a markdown code fence if the model added one — a stray |
| 313 | // ``` at char 0 is exactly the "Unexpected token (1:0)" Strudel parse error. |
| 314 | code = code |
| 315 | .replace(/^\s*```[a-zA-Z]*\s*\n?/, '') |
| 316 | .replace(/\n?```\s*$/, '') |
| 317 | .trim(); |
| 318 | res.json({ message: cleanMessage(parsed.message), code }); |
| 319 | } catch (err) { |
| 320 | console.error('generate error:', err.status, err.message); |
| 321 | // Point auth failures at the key rather than at a generic server error. |
| 322 | if (err.status === 401 || err.status === 403) { |
| 323 | return res.status(401).json({ error: 'that API key was rejected — check it with the 🔑 button' }); |
| 324 | } |
| 325 | if (err.status === 429) { |
| 326 | return res.status(429).json({ error: 'rate limited by the API — give it a moment and try again' }); |
| 327 | } |
| 328 | res.status(500).json({ error: err.message || 'generation failed' }); |
| 329 | } |
| 330 | }); |
| 331 | |
| 332 | // Serve the built frontend from the same origin (so one port / one ngrok |
| 333 | // tunnel serves both the app and the API). In dev, dist/ may not exist yet — |
| 334 | // Vite serves the frontend and proxies /api here instead. |
| 335 | if (fs.existsSync(DIST_DIR)) { |
| 336 | app.use(express.static(DIST_DIR)); |
| 337 | // SPA fallback for any non-/api GET (Express 5-safe: middleware, not a wildcard route). |
| 338 | app.use((req, res, next) => { |
| 339 | if (req.method === 'GET' && !req.path.startsWith('/api')) { |
| 340 | return res.sendFile(path.join(DIST_DIR, 'index.html')); |
| 341 | } |
| 342 | next(); |
| 343 | }); |
| 344 | } |
| 345 | |
| 346 | const PORT = process.env.PORT || 8787; |
| 347 | // An explicit http.Server so the WebSocket hub can share the port (and any |
| 348 | // tunnel) with the app and the API. |
| 349 | const server = http.createServer(app); |
| 350 | attachBroadcast(server); |
| 351 | server.listen(PORT, () => { |
| 352 | console.log(`🎛 Strudel×Claude backend on http://localhost:${PORT}`); |
| 353 | console.log(` Prompt: PROMPT.md (${(SYSTEM.length / 1024).toFixed(1)} KB)`); |
| 354 | console.log(` Timings: ws://localhost:${PORT}/ws · terminal view: curl -s http://localhost:${PORT}/tui.py | python3 -`); |
| 355 | if (!process.env.ANTHROPIC_API_KEY) { |
| 356 | console.log( |
| 357 | ' Note: ANTHROPIC_API_KEY not set — relying on an `ant auth login` profile if present.', |
| 358 | ); |
| 359 | } |
| 360 | }); |