| 1 | #!/usr/bin/env node |
| 2 | /** |
| 3 | * Renders the voice pack in `public/voice/`. |
| 4 | * |
| 5 | * A mahjong table only ever says about fifty things — the 42 tile names and a |
| 6 | * handful of calls — so rather than depend on whatever speech synthesis a |
| 7 | * browser happens to have installed (on Linux, usually espeak-ng, which sounds |
| 8 | * like a modem), the whole vocabulary is rendered once, here, with a proper |
| 9 | * neural voice and shipped as audio. Playback is then instant, identical on |
| 10 | * every machine, and needs nothing but the Web Audio graph the chimes already |
| 11 | * use. |
| 12 | * |
| 13 | * This is an authoring step, not part of `npm run build`. Re-run it only to |
| 14 | * change the voice or the wording: |
| 15 | * |
| 16 | * piper --version # https://github.com/OHF-Voice/piper1-gpl |
| 17 | * node scripts/voice.mjs # writes public/voice/*.mp3 |
| 18 | * |
| 19 | * Set PIPER_MODEL to use a different voice. |
| 20 | */ |
| 21 | import { execFileSync } from 'node:child_process'; |
| 22 | import { mkdirSync, rmSync, writeFileSync } from 'node:fs'; |
| 23 | import { homedir } from 'node:os'; |
| 24 | import { join } from 'node:path'; |
| 25 | |
| 26 | const MODEL = |
| 27 | process.env.PIPER_MODEL ?? |
| 28 | join(homedir(), '.local/share/piper-voices/zh_CN/zh_CN-huayan-medium.onnx'); |
| 29 | const OUT = 'public/voice'; |
| 30 | const TMP = join(process.env.TMPDIR ?? '/tmp', 'mahjong-voice'); |
| 31 | |
| 32 | const DIGITS = ['一', '二', '三', '四', '五', '六', '七', '八', '九']; |
| 33 | |
| 34 | /** |
| 35 | * What each clip says. Tile clips are named by tile code so the runtime can |
| 36 | * find them without knowing any of this. |
| 37 | * |
| 38 | * Several tiles are deliberately not read as the bare character: 東 alone is a |
| 39 | * direction, 東風 is the tile, and the dragons are 紅中 / 發財 / 白板 the way |
| 40 | * they are actually called. It also gives the phonemiser enough context to get |
| 41 | * the tone right — 中 on its own is as likely to come out zhòng. |
| 42 | */ |
| 43 | const CLIPS = {}; |
| 44 | for (let i = 0; i < 9; i++) { |
| 45 | CLIPS[`t${i}`] = `${DIGITS[i]}萬`; |
| 46 | CLIPS[`t${i + 9}`] = `${DIGITS[i]}條`; |
| 47 | CLIPS[`t${i + 18}`] = `${DIGITS[i]}筒`; |
| 48 | } |
| 49 | ['東風', '南風', '西風', '北風', '紅中', '發財', '白板'].forEach((w, i) => { |
| 50 | CLIPS[`t${27 + i}`] = w; |
| 51 | }); |
| 52 | ['春', '夏', '秋', '冬', '梅', '蘭', '菊', '竹'].forEach((w, i) => { |
| 53 | CLIPS[`t${34 + i}`] = w; |
| 54 | }); |
| 55 | |
| 56 | // The calls, as they are shouted. |
| 57 | Object.assign(CLIPS, { |
| 58 | pung: '碰', |
| 59 | chow: '吃', |
| 60 | kong: '槓', |
| 61 | hu: '胡了', |
| 62 | selfDraw: '自摸', |
| 63 | drawGame: '流局', |
| 64 | flower: '補花', |
| 65 | // Said to whoever has just put a tile in your lap. Eight of them, so a rough |
| 66 | // table does not sound like a parrot — the first two straight, the rest the |
| 67 | // way it would actually come out at a table in Taipei. The particles (啦, 喔, |
| 68 | // 欸) are most of what makes a line sound like a person rather than a sign. |
| 69 | watchIt: '喂,小心點', |
| 70 | watchIt2: '輕一點啦', |
| 71 | watchIt3: '是在哈囉', |
| 72 | watchIt4: '你是在丟飛鏢喔', |
| 73 | watchIt5: '這是麻將,不是棒球', |
| 74 | watchIt6: '差一點就打到我了', |
| 75 | watchIt7: '你牌品很差欸', |
| 76 | watchIt8: '手下留情啦', |
| 77 | }); |
| 78 | |
| 79 | mkdirSync(OUT, { recursive: true }); |
| 80 | mkdirSync(TMP, { recursive: true }); |
| 81 | |
| 82 | for (const [name, text] of Object.entries(CLIPS)) { |
| 83 | const wav = join(TMP, `${name}.wav`); |
| 84 | execFileSync('piper', ['-m', MODEL, '-f', wav], { input: text }); |
| 85 | // Piper pads both ends with silence; a call has to land on the beat, so trim |
| 86 | // it off and normalise so no clip is noticeably louder than its neighbours. |
| 87 | execFileSync( |
| 88 | 'ffmpeg', |
| 89 | [ |
| 90 | '-hide_banner', '-loglevel', 'error', '-y', '-i', wav, |
| 91 | '-af', |
| 92 | 'silenceremove=start_periods=1:start_threshold=-45dB:start_silence=0.02,' + |
| 93 | 'areverse,silenceremove=start_periods=1:start_threshold=-45dB:start_silence=0.04,areverse,' + |
| 94 | 'loudnorm=I=-18:TP=-2:LRA=7', |
| 95 | '-ac', '1', '-ar', '22050', '-b:a', '48k', |
| 96 | join(OUT, `${name}.mp3`), |
| 97 | ], |
| 98 | { stdio: ['ignore', 'ignore', 'inherit'] }, |
| 99 | ); |
| 100 | process.stdout.write(`${name} ${text}\n`); |
| 101 | } |
| 102 | |
| 103 | // Shipped alongside so the runtime knows what it can ask for without a 404. |
| 104 | writeFileSync(join(OUT, 'manifest.json'), `${JSON.stringify(Object.keys(CLIPS))}\n`); |
| 105 | rmSync(TMP, { recursive: true, force: true }); |
| 106 | console.log(`\n${Object.keys(CLIPS).length} clips → ${OUT}`); |