From 7593f0bd7f287b60d99c2519d87c655d4f56c7de Mon Sep 17 00:00:00 2001 From: Danial Kalbasi Date: Thu, 23 Apr 2026 00:53:31 +0200 Subject: [PATCH] Remove Piper TTS; expand settings, main flow, and styles - Drop piper backend; adjust OpenAI and ElevenLabs TTS integration - Rework settings UI and plugin entry points; update player and styles - Add VOICES.md; refresh README, manifest, package metadata, and .gitignore Made-with: Cursor --- .gitignore | 32 +++- README.md | 48 ++--- VOICES.md | 37 ++++ esbuild.config.mjs | 2 +- manifest.json | 4 +- package-lock.json | 4 +- package.json | 13 +- src/main.ts | 247 +++++++++++++++++++----- src/player.ts | 44 +++-- src/settings.ts | 436 ++++++++++++++++++++++++------------------ src/tts/backend.ts | 11 +- src/tts/elevenlabs.ts | 11 +- src/tts/openai.ts | 14 +- src/tts/piper.ts | 78 -------- styles.css | 105 +++++++++- 15 files changed, 689 insertions(+), 397 deletions(-) create mode 100644 VOICES.md delete mode 100644 src/tts/piper.ts diff --git a/.gitignore b/.gitignore index 1b4ecbd..84f5b44 100644 --- a/.gitignore +++ b/.gitignore @@ -1,6 +1,32 @@ -node_modules/ +# Build output main.js -*.log -.DS_Store +main.js.map +*.js.map + +# Dependencies +node_modules/ + +# Obsidian plugin data data.json cache/ + +# Environment / secrets +.env +.env.* + +# OS +.DS_Store +Thumbs.db + +# Logs +*.log +npm-debug.log* + +# Editor +.vscode/ +.idea/ +*.swp +*.swo + +# TypeScript +*.tsbuildinfo diff --git a/README.md b/README.md index b8245fc..ae4d4c5 100644 --- a/README.md +++ b/README.md @@ -1,20 +1,20 @@ # Vox -Vox reads your Obsidian notes aloud with neural text-to-speech. Different folders can have different voices, so your philosophy notes can sound different from your dream journal. +Vox reads your Obsidian notes aloud with neural text-to-speech. Different folders can have different voices, so your philosophy notes can sound different from your journal. ## Features -- **Ribbon button** — morphs between speaker / pause / play while audio is active; a separate stop button appears during playback -- **Hover voice picker** — hover the ribbon icon to pick a voice on the fly without touching settings -- **Status bar indicator** — clickable pill shows "Reading" / "Paused"; hidden when idle -- **Command palette** — Read active note / Stop / Toggle play–pause -- **File menu** — right-click any note → Vox: read aloud -- **Three TTS backends** — Browser (free, offline), OpenAI, ElevenLabs -- **Per-engine speed control** — each backend has its own range -- **Per-folder persona voices** — longest prefix wins; stored per-engine +- **Ribbon button**: morphs between speaker / pause / play while audio is active; a separate stop button appears during playback +- **Hover voice picker**: hover the ribbon icon to pick a voice on the fly without touching settings +- **Status bar indicator**: clickable pill shows "Reading" / "Paused"; hidden when idle +- **Command palette**: Read active note / Stop / Toggle play-pause +- **File menu**: right-click any note -> Vox: read aloud +- **Three TTS backends**: Browser (free, offline), OpenAI, ElevenLabs +- **Per-engine speed control**: each backend has its own range +- **Per-folder persona voices**: longest prefix wins; stored per-engine - **Per-note voice override** via frontmatter `voice: ` -- **Tone control** (OpenAI) — preset delivery styles (calm, conversational, storytelling, …) -- **Voice library** (ElevenLabs) — save name + ID pairs as clickable chips; click to set default, × to remove +- **Tone control** (OpenAI): preset delivery styles (calm, conversational, storytelling, ...) +- **Voice library** (ElevenLabs): save name + ID pairs as clickable chips; click to set default, x to remove ## Backends @@ -28,19 +28,19 @@ Vox reads your Obsidian notes aloud with neural text-to-speech. Different folder alloy · ash · ballad · cedar · coral · echo · fable · marin · nova · onyx · sage · shimmer · verse ### ElevenLabs models -- **Turbo v2.5** — low latency, English-first -- **Multilingual v2** — broader language support +- **Turbo v2.5**: low latency, English-first +- **Multilingual v2**: broader language support ## Configuration -**Settings → Vox:** +**Settings -> Vox:** -- **Engine** — browser / openai / elevenlabs -- **Speed** — slider (range varies by engine) -- **Voice / API key** — shown only for the active engine -- **Tone** (OpenAI) — delivery style preset -- **Voices** (ElevenLabs) — add name + voice ID pairs -- **Folder voices** — map a folder prefix to a voice id for the active engine +- **Engine**: browser / openai / elevenlabs +- **Speed**: slider (range varies by engine) +- **Voice / API key**: shown only for the active engine +- **Tone** (OpenAI): delivery style preset +- **Voices** (ElevenLabs): add name + voice ID pairs +- **Folder voices**: map a folder prefix to a voice id for the active engine Per-note override via YAML frontmatter: @@ -57,7 +57,7 @@ npm install npm run dev ``` -The plugin should live inside your vault at `.obsidian/plugins/vox/`. Enable it under **Settings → Community plugins**. +The plugin should live inside your vault at `.obsidian/plugins/vox/`. Enable it under **Settings -> Community plugins**. Build for release: @@ -73,6 +73,6 @@ npm run typecheck ## Roadmap -- **Audio caching** — hash-keyed cache so unchanged notes don't re-bill cloud APIs -- **Highlight as reads** — colour the current sentence during playback -- **Section-level playback** — click a paragraph to start reading from there +- **Audio caching**: hash-keyed cache so unchanged notes don't re-bill cloud APIs +- **Highlight as reads**: colour the current sentence during playback +- **Section-level playback**: click a paragraph to start reading from there diff --git a/VOICES.md b/VOICES.md new file mode 100644 index 0000000..4a0d449 --- /dev/null +++ b/VOICES.md @@ -0,0 +1,37 @@ +# Voice Prompts + +A collection of voice design prompts for use with ElevenLabs Voice Design and similar tools. +Paste these into the voice generation prompt field to create a voice that fits your reading style. + +--- + +## Epictetus + +> An older male voice, weathered and direct. Speaks slowly and with weight, as if each sentence has been earned through hardship. No softness, no flattery — just plain truth delivered with patience and quiet authority. A slight roughness to the voice, grounded and still. + +--- + +## Nassim Taleb + +> A deep, accented male voice — Lebanese, educated in France and the US. Unhurried and self-assured, as if the idea is already settled in his mind and he is simply letting you catch up. Slightly gravelly. Speaks in bursts — a long pause, then a dense sentence. Amused rather than angry. The tone of someone who has stopped trying to convince people and is now just stating things. + +--- + +## Tony Robbins + +> A deep, powerful male voice with tremendous presence. Speaks with conviction and urgency, building energy through each sentence. Warm but commanding — like someone who genuinely believes what they are saying and wants you to feel it too. Full chest, no hesitation. + +--- + +## David Attenborough + +> A refined older British male voice, soft and full of quiet wonder. Unhurried, almost reverential. Pauses briefly to let ideas breathe. Feels like someone who has seen extraordinary things and is moved by all of them. Gentle authority — never loud, always compelling. + +--- + +## Tips + +- On **ElevenLabs**, paste the prompt into Voice Design → Description field. +- Adjust **stability** (higher = more consistent) and **clarity** (higher = more distinct) after generating. +- A good starting point: stability 0.5, similarity boost 0.75. +- Try the same prompt with different base voices — the description shapes personality, the base voice shapes timbre. diff --git a/esbuild.config.mjs b/esbuild.config.mjs index 7495e86..d7e02bc 100644 --- a/esbuild.config.mjs +++ b/esbuild.config.mjs @@ -3,7 +3,7 @@ import process from "process"; import builtins from "builtin-modules"; const banner = `/* -Rhapsode — reads your notes aloud. +Vox — reads your notes aloud. Bundled by esbuild; do not edit main.js directly. */`; diff --git a/manifest.json b/manifest.json index eaeef6e..2240235 100644 --- a/manifest.json +++ b/manifest.json @@ -1,6 +1,6 @@ { - "id": "rhapsode", - "name": "Rhapsode", + "id": "vox", + "name": "Vox", "version": "0.1.0", "minAppVersion": "1.4.0", "description": "Reads your notes aloud with neural text-to-speech. Supports per-folder persona voices, local Piper TTS, and streaming cloud voices (ElevenLabs, OpenAI).", diff --git a/package-lock.json b/package-lock.json index d34f3d7..c5d56e6 100644 --- a/package-lock.json +++ b/package-lock.json @@ -1,11 +1,11 @@ { - "name": "rhapsode", + "name": "obsidian-vox", "version": "0.1.0", "lockfileVersion": 3, "requires": true, "packages": { "": { - "name": "rhapsode", + "name": "obsidian-vox", "version": "0.1.0", "license": "MIT", "devDependencies": { diff --git a/package.json b/package.json index 3beb654..671d3c6 100644 --- a/package.json +++ b/package.json @@ -1,5 +1,5 @@ { - "name": "rhapsode", + "name": "obsidian-vox", "version": "0.1.0", "description": "Reads your Obsidian notes aloud with neural text-to-speech.", "main": "main.js", @@ -8,7 +8,13 @@ "build": "tsc -noEmit -skipLibCheck && node esbuild.config.mjs production", "typecheck": "tsc -noEmit -skipLibCheck" }, - "keywords": ["obsidian", "tts", "text-to-speech", "voice", "piper", "elevenlabs"], + "keywords": [ + "obsidian", + "tts", + "text-to-speech", + "voice", + "elevenlabs" + ], "author": "danialkalbasi", "license": "MIT", "devDependencies": { @@ -18,5 +24,6 @@ "obsidian": "^1.7.2", "tslib": "^2.6.2", "typescript": "^5.4.0" - } + }, + "dependencies": {} } diff --git a/src/main.ts b/src/main.ts index 94d5edd..6b69cb4 100644 --- a/src/main.ts +++ b/src/main.ts @@ -6,62 +6,66 @@ import { addIcon, setIcon, } from "obsidian"; -import { DEFAULT_SETTINGS, RhapsodeSettings, RhapsodeSettingTab } from "./settings"; +import { DEFAULT_SETTINGS, VoxSettings, VoxSettingTab } from "./settings"; import { Player, PlayerState } from "./player"; import { stripMarkdown } from "./markdown"; import { createBackend, TtsBackend } from "./tts/backend"; /** - * Rhapsode plugin entry point. + * Vox plugin entry point. * * Responsibilities: * 1. Register UI affordances: read / pause-resume / stop ribbon icons, * editor menu item, command palette commands. * 2. Hold the singleton TTS `Player` and a status-bar indicator that * mirrors its state. - * 3. Load/save `RhapsodeSettings`, including per-folder voice overrides. + * 3. Load/save `VoxSettings`, including per-folder voice overrides. * * The heavy lifting lives in `player.ts`, `markdown.ts`, and `tts/`. * This file just wires things together. */ -const ICON_READ = "rhapsode-read"; -const ICON_PAUSE = "rhapsode-pause"; -const ICON_PLAY = "rhapsode-play"; -const ICON_STOP = "rhapsode-stop"; +const ICON_READ = "vox-read"; +const ICON_PAUSE = "vox-pause"; +const ICON_PLAY = "vox-play"; +const ICON_STOP = "vox-stop"; const ICONS: Record = { [ICON_READ]: ` - - - + + + + `, [ICON_PAUSE]: ` - - - + + + `, [ICON_PLAY]: ` - - + + `, [ICON_STOP]: ` - - + + + `, }; -export default class RhapsodePlugin extends Plugin { - settings!: RhapsodeSettings; +export default class VoxPlugin extends Plugin { + settings!: VoxSettings; player!: Player; private backend!: TtsBackend; // UI elements whose appearance depends on player state. Held as // fields so the state listener can mutate them without re-querying. private statusEl: HTMLElement | null = null; - private pauseRibbonEl: HTMLElement | null = null; + private readRibbonEl: HTMLElement | null = null; private stopRibbonEl: HTMLElement | null = null; private unsubscribeState: (() => void) | null = null; + private pendingVoice: string | null = null; + private voicePickerEl: HTMLElement | null = null; async onload() { await this.loadSettings(); @@ -74,32 +78,31 @@ export default class RhapsodePlugin extends Plugin { this.player = new Player(this.backend); // ── Ribbon icons ──────────────────────────────────────────── - this.addRibbonIcon(ICON_READ, "Rhapsode: read current note", async () => { - await this.readActiveNote(); + // Single morphing icon: speaker (idle) → pause (playing) → play (paused). + // A separate stop icon appears only while active. + this.readRibbonEl = this.addRibbonIcon(ICON_READ, "Vox: read current note", () => { + if (this.player.getState() === "idle") { + this.readActiveNote(); + } else { + this.player.togglePause(); + } }); - - this.pauseRibbonEl = this.addRibbonIcon( - ICON_PAUSE, - "Rhapsode: pause / resume", - () => this.player.togglePause(), - ); - // Hidden until playback starts — these controls have no meaning - // when the player is idle, and an always-visible pause button in - // the ribbon is visual clutter. - this.pauseRibbonEl.style.display = "none"; + this.readRibbonEl.addClass("vox-ribbon"); + this.registerVoicePicker(this.readRibbonEl); this.stopRibbonEl = this.addRibbonIcon( ICON_STOP, - "Rhapsode: stop reading", + "Vox: stop reading", () => this.player.stop(), ); + this.stopRibbonEl.addClass("vox-ribbon"); this.stopRibbonEl.style.display = "none"; // ── Status bar ────────────────────────────────────────────── // Single clickable pill that shows current playback state. // Click behaviour: toggle pause when playing/paused, no-op idle. this.statusEl = this.addStatusBarItem(); - this.statusEl.addClass("rhapsode-status"); + this.statusEl.addClass("vox-status"); this.statusEl.addEventListener("click", () => { if (this.player.getState() !== "idle") this.player.togglePause(); }); @@ -131,7 +134,7 @@ export default class RhapsodePlugin extends Plugin { if (!(file instanceof TFile) || file.extension !== "md") return; menu.addItem((item) => { item - .setTitle("Rhapsode: read aloud") + .setTitle("Vox: read aloud") .setIcon(ICON_READ) .onClick(async () => { await this.readFile(file); @@ -145,7 +148,7 @@ export default class RhapsodePlugin extends Plugin { this.renderState(s), ); - this.addSettingTab(new RhapsodeSettingTab(this.app, this)); + this.addSettingTab(new VoxSettingTab(this.app, this)); } async onunload() { @@ -160,10 +163,11 @@ export default class RhapsodePlugin extends Plugin { private renderState(state: PlayerState) { const active = state !== "idle"; - if (this.pauseRibbonEl) { - this.pauseRibbonEl.style.display = active ? "" : "none"; - // Swap pause↔play icon so the button's meaning is obvious. - setIcon(this.pauseRibbonEl, state === "paused" ? ICON_PLAY : ICON_PAUSE); + if (this.readRibbonEl) { + const icon = state === "idle" ? ICON_READ : state === "playing" ? ICON_PAUSE : ICON_PLAY; + const label = state === "idle" ? "Vox: read current note" : state === "playing" ? "Vox: pause" : "Vox: resume"; + setIcon(this.readRibbonEl, icon); + this.readRibbonEl.setAttribute("aria-label", label); } if (this.stopRibbonEl) { @@ -178,10 +182,10 @@ export default class RhapsodePlugin extends Plugin { return; } this.statusEl.style.display = ""; - const icon = this.statusEl.createSpan({ cls: "rhapsode-status-icon" }); + const icon = this.statusEl.createSpan({ cls: "vox-status-icon" }); setIcon(icon, state === "playing" ? ICON_PAUSE : ICON_PLAY); this.statusEl.createSpan({ - cls: "rhapsode-status-text", + cls: "vox-status-text", text: state === "playing" ? " Reading" : " Paused", }); } @@ -191,7 +195,7 @@ export default class RhapsodePlugin extends Plugin { const view = this.app.workspace.getActiveViewOfType(MarkdownView); const file = view?.file; if (!file) { - new Notice("Rhapsode: no active note to read."); + new Notice("Vox: no active note to read."); return; } await this.readFile(file); @@ -203,28 +207,103 @@ export default class RhapsodePlugin extends Plugin { const raw = await this.app.vault.cachedRead(file); const text = stripMarkdown(raw); if (!text.trim()) { - new Notice("Rhapsode: note is empty after stripping markdown."); + new Notice("Vox: note is empty after stripping markdown."); return; } - const voice = this.resolveVoice(file); + const voice = this.pendingVoice ?? this.resolveVoice(file); + this.pendingVoice = null; try { + this.player.setRate(this.settings.rate); await this.player.play(text, voice); - new Notice(`Rhapsode: reading “${file.basename}”…`); + new Notice(`Vox: reading "${file.basename}"...`); } catch (err) { - console.error("Rhapsode playback failed:", err); - new Notice(`Rhapsode: ${(err as Error).message}`); + console.error("Vox playback failed:", err); + new Notice(`Vox: ${(err as Error).message}`); } } + private registerVoicePicker(anchor: HTMLElement) { + let closeTimer: number; + + const getVoices = (): { label: string; id: string }[] => { + const { engine, elevenlabsVoices } = this.settings; + if (engine === "elevenlabs") { + return elevenlabsVoices.map((v) => ({ label: v.name, id: v.id })); + } + if (engine === "openai") { + return ["alloy","ash","ballad","cedar","coral","echo","fable","marin","nova","onyx","sage","shimmer","verse"] + .map((v) => ({ label: v, id: v })); + } + return []; + }; + + const closePicker = () => { + this.voicePickerEl?.remove(); + this.voicePickerEl = null; + const label = anchor.getAttribute("data-vox-label"); + if (label) anchor.setAttribute("aria-label", label); + }; + + const openPicker = () => { + if (this.player.getState() !== "idle") return; + const voices = getVoices(); + if (voices.length === 0) return; + + // Stash the current aria-label and clear it so Obsidian's native + // tooltip doesn't appear alongside the picker. + if (!anchor.getAttribute("data-vox-label")) { + anchor.setAttribute("data-vox-label", anchor.getAttribute("aria-label") ?? ""); + } + anchor.removeAttribute("aria-label"); + + closePicker(); + const picker = document.body.createEl("div", { cls: "vox-voice-picker" }); + this.voicePickerEl = picker; + + const activeId = this.pendingVoice ?? this.settings.voiceElevenlabs ?? this.settings.voiceOpenai; + for (const voice of voices) { + const item = picker.createEl("div", { + cls: "vox-voice-picker-item" + (voice.id === activeId ? " vox-voice-picker-item--active" : ""), + text: voice.label, + }); + item.addEventListener("mousedown", (e) => { + e.preventDefault(); + this.pendingVoice = voice.id; + closePicker(); + this.readActiveNote(); + }); + } + + const rect = anchor.getBoundingClientRect(); + picker.style.left = `${rect.right + 4}px`; + picker.style.top = `${rect.top}px`; + + picker.addEventListener("mouseenter", () => clearTimeout(closeTimer)); + picker.addEventListener("mouseleave", () => { + closeTimer = window.setTimeout(closePicker, 150); + }); + }; + + anchor.addEventListener("mouseenter", () => { + clearTimeout(closeTimer); + openPicker(); + }); + anchor.addEventListener("mouseleave", () => { + closeTimer = window.setTimeout(closePicker, 150); + }); + } + /** * Persona voice selection: folder prefix (longest wins) → frontmatter * `voice:` key → global default. Centralised here so backends stay * voice-agnostic. */ private resolveVoice(file: TFile): string { - const overrides = Object.entries(this.settings.folderVoices).sort( + const overrides = Object.entries( + this.settings.folderVoicesByEngine[this.settings.engine] ?? {}, + ).sort( (a, b) => b[0].length - a[0].length, ); for (const [prefix, voice] of overrides) { @@ -235,18 +314,80 @@ export default class RhapsodePlugin extends Plugin { const fmVoice = fm?.voice; if (typeof fmVoice === "string" && fmVoice.trim()) return fmVoice.trim(); - return this.settings.defaultVoice; + switch (this.settings.engine) { + case "browser": + return this.settings.voiceBrowser; + case "openai": + return this.settings.voiceOpenai; + case "elevenlabs": + return this.settings.voiceElevenlabs; + } } async loadSettings() { - this.settings = Object.assign({}, DEFAULT_SETTINGS, await this.loadData()); + const raw = (await this.loadData()) as + | (Record & { + defaultVoice?: string; + folderVoices?: Record; + folderVoicesByEngine?: Partial< + Record< + VoxSettings["engine"], + Record + > + >; + }) + | null; + const merged = Object.assign( + {}, + DEFAULT_SETTINGS, + raw ?? {}, + ) as VoxSettings & { + defaultVoice?: string; + folderVoices?: Record; + }; + + merged.folderVoicesByEngine = { + ...DEFAULT_SETTINGS.folderVoicesByEngine, + ...(raw?.folderVoicesByEngine ?? {}), + }; + + if ( + typeof merged.defaultVoice === "string" && + merged.defaultVoice.length > 0 + ) { + const eng = merged.engine; + const leg = merged.defaultVoice; + const isEmpty = (v: string | undefined) => !v || v.length === 0; + if (eng === "browser" && isEmpty(merged.voiceBrowser)) + merged.voiceBrowser = leg; + if (eng === "openai" && isEmpty(merged.voiceOpenai)) + merged.voiceOpenai = leg; + if (eng === "elevenlabs" && isEmpty(merged.voiceElevenlabs)) + merged.voiceElevenlabs = leg; + } + delete merged.defaultVoice; + + if ( + raw?.folderVoices && + typeof raw.folderVoices === "object" && + Object.keys(raw.folderVoices).length > 0 + ) { + const engine = merged.engine; + const existing = merged.folderVoicesByEngine[engine] ?? {}; + merged.folderVoicesByEngine[engine] = { + ...raw.folderVoices, + ...existing, + }; + } + delete merged.folderVoices; + + this.settings = merged; } async saveSettings() { await this.saveData(this.settings); - // Rebuild backend so the live Player picks up engine/API-key - // changes without a full plugin reload. this.backend = createBackend(this.settings, this); this.player.setBackend(this.backend); + this.player.setRate(this.settings.rate); } } diff --git a/src/player.ts b/src/player.ts index bb918ac..d6ee110 100644 --- a/src/player.ts +++ b/src/player.ts @@ -21,8 +21,9 @@ export class Player { private queue: string[] = []; private cursor = 0; private audio: HTMLAudioElement | null = null; - private cancelled = false; private rate = 1.0; + private tokenCounter = 0; + private activeToken = 0; private state: PlayerState = "idle"; private listeners: Set = new Set(); @@ -37,7 +38,6 @@ export class Player { setRate(rate: number) { this.rate = rate; - if (this.audio) this.audio.playbackRate = rate; } getState(): PlayerState { @@ -60,60 +60,63 @@ export class Player { for (const l of this.listeners) l(next); } + private isActive(token: number): boolean { + return token !== 0 && token === this.activeToken; + } + /** * Start playing the given text with the given voice. Stops any * currently-playing audio first. */ async play(text: string, voice: string): Promise { this.stop(); - this.cancelled = false; + const token = ++this.tokenCounter; + this.activeToken = token; this.queue = splitIntoSentences(text); this.cursor = 0; if (this.queue.length === 0) return; - this.setState("playing"); - if (this.backend.kind === "synth") { + this.setState("playing"); // Browser SpeechSynthesis queues utterances internally. We hook // the last one's `onend` to transition back to `idle` on natural // completion; the backend will emit that callback for us. await this.backend.speakAll(this.queue, voice, this.rate, () => { - if (!this.cancelled) this.setState("idle"); + if (this.isActive(token)) this.setState("idle"); }); return; } this.audio = new Audio(); - this.audio.playbackRate = this.rate; // Keep state in sync with underlying element events: pausing the //