Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
67 changes: 67 additions & 0 deletions AntennaHead/Services/AntennaHeadHTTPServer.swift
Original file line number Diff line number Diff line change
Expand Up @@ -765,6 +765,13 @@ final class AntennaHeadHTTPServer {
sequence: sequence, repeatForever: repeatForever)
return okResponse()

case "/speaktextbuttonclicked.html":
// Body: {text, voice}. The Text to Speech page's Speak Text
// section; `voice` "" = the Text to Speech voice setting.
let so = jsonObject(fromBody: request.body)
sdrController?.speakText(so.string("text"), voiceIdentifier: so.string("voice"))
return okResponse()

case "/playaudiofileslistenbuttonclicked.html":
// Same payload shape as Text to Speech, plus an optional `playlist`
// name: {sequence, repeat, files, playlist}. A non-empty playlist
Expand Down Expand Up @@ -2459,9 +2466,69 @@ final class AntennaHeadHTTPServer {
s += "onclick=\"textToSpeechListenButtonClicked(getElementById('textToSpeechForm'));\" "
s += "title='Synthesize the selected folder's text files and stream them through the live audio pipeline.'>"
s += "</form><br>&nbsp;<br>"
s += speakTextFormHTML()
return s
}

/// The SSML example the Speak Text box starts with.
nonisolated static let speakTextExample =
#"<speak>It is four oh nine <break time="700ms"/> <prosody rate="80%">on AntennaHead Radio.</prosody></speak>"#

/// "Speak Text" section at the bottom of the Text to Speech page: type
/// text, pick a voice, Speak → `/speaktextbuttonclicked.html` →
/// `SDRController.speakText`. Text starting with `<speak` is SSML. The
/// markup guide below reflects what was measured with the Premium Ava
/// voice on macOS 27 (rendered clip length/content), not just Apple's docs.
@MainActor private func speakTextFormHTML() -> String {
var s = "<form class='speak_text_form' id='speakTextForm' onsubmit='event.preventDefault(); return false;' method='POST'>"
s += "<label for='tts_speak_text'>Speak Text</label>"
s += "<p>Type text and speak it once through the live audio pipeline. Text that starts with "
s += "<code>&lt;speak&gt;</code> is read as SSML markup (see below).</p>"
s += "<textarea id='tts_speak_text' name='tts_speak_text' class='u-full-width' rows='6' "
s += "\(Self.verbatimInputAttributes)>\(htmlText(Self.speakTextExample))</textarea>"
s += "<label for='tts_speak_voice'>Voice</label>"
s += installedVoicesSelectOptionsHTML(selectID: "tts_speak_voice")
s += "<br><br><input class='twelve columns button button-primary' type='button' value='Speak' "
s += "onclick=\"speakTextButtonClicked(getElementById('speakTextForm'));\" "
s += "title='Synthesize this text with the chosen voice and stream it through the live audio pipeline.'>"
s += "</form><br>"
s += Self.speechMarkupHelpHTML
s += "<br>&nbsp;<br>"
return s
}

nonisolated static let speechMarkupHelpHTML = """
<details class='speech-markup-help'><summary>Controlling the voice: pauses, speed, pronunciation</summary>
<p>There are two ways to control how the text is spoken, and they depend on the voice.</p>
<p><strong>Modern voices</strong> (Ava, Samantha, Siri and other Premium/Enhanced voices): start the text with
<code>&lt;speak&gt;</code> and end it with <code>&lt;/speak&gt;</code> to use SSML markup. Measured with the Premium Ava voice:</p>
<table class='u-full-width'>
<thead><tr><th>Markup</th><th>Effect</th></tr></thead>
<tbody>
<tr><td><code>&lt;prosody rate="50%"&gt;&hellip;&lt;/prosody&gt;</code> (also <code>"150%"</code>, <code>"slow"</code>, <code>"fast"</code>)</td><td>✅ Speed. 50% took 3.7&nbsp;s where normal took 2.8&nbsp;s.</td></tr>
<tr><td><code>&lt;break time="1500ms"/&gt;</code></td><td>✅ A pause of that length.</td></tr>
<tr><td><code>&lt;prosody volume="x-soft"&gt;&hellip;&lt;/prosody&gt;</code></td><td>✅ Volume. <code>x-soft</code> is about a quarter as loud.</td></tr>
<tr><td><code>&lt;say-as interpret-as="characters"&gt;KARK&lt;/say-as&gt;</code></td><td>✅ Spells it out letter by letter.</td></tr>
<tr><td><code>&lt;phoneme alphabet="ipa" ph="&hellip;"&gt;word&lt;/phoneme&gt;</code></td><td>✅ Fixes a pronunciation, written in the IPA phonetic alphabet.</td></tr>
<tr><td><code>&lt;prosody pitch="+40%"&gt;&hellip;&lt;/prosody&gt;</code></td><td>⚠️ Changes the audio only slightly; the pitch barely moves.</td></tr>
<tr><td><code>&lt;emphasis level="strong"&gt;&hellip;&lt;/emphasis&gt;</code></td><td>❌ Ignored.</td></tr>
<tr><td><code>&lt;sub alias="North Little Rock"&gt;NLR&lt;/sub&gt;</code></td><td>❌ Ignored: still says the letters. Type the words out instead.</td></tr>
</tbody></table>
<p>If the markup can&rsquo;t be parsed (for example, a closing tag that doesn&rsquo;t match its opening tag),
nothing is spoken at all and the log shows &ldquo;invalid SSML&rdquo;. A <code>&lt;voice name="&hellip;"&gt;</code> tag inside the markup overrides the Voice menu.</p>
<p><strong>Classic voices</strong> (names in the Voice menu such as Alex, Albert and Fred, whose identifiers start with
<code>com.apple.speech.synthesis.voice.</code>) don&rsquo;t read SSML. Instead, put commands in double brackets
in plain text:</p>
<ul>
<li><code>[[slnc 1500]]</code> &mdash; pause 1.5 seconds</li>
<li><code>[[rate 120]]</code> &mdash; speed in words per minute</li>
<li><code>[[pbas 40]]</code> &mdash; base pitch; <code>[[pmod 60]]</code> &mdash; how much the pitch varies</li>
</ul>
<p>Don&rsquo;t mix the two: modern voices read <code>[[&hellip;]]</code> commands out loud, and classic voices read SSML tags out loud.
The Voice menu&rsquo;s &ldquo;Default voice&rdquo; is the Text to Speech voice chosen in AntennaHead&rsquo;s Configuration tab.</p>
</details>
"""

/// Listing of the `.txt` files in the configured Text to Speech folder, in
/// the same name order Listen speaks them chronologically — each with a
/// checkbox (checked by default) so the user can leave out specific files,
Expand Down
37 changes: 34 additions & 3 deletions AntennaHead/Services/SDRController.swift
Original file line number Diff line number Diff line change
Expand Up @@ -1819,7 +1819,35 @@ final class SDRController {
LogStore.shared.log(.error, source: "SDRController", "Text to Speech: nothing to speak")
return
}
startSpeechSynthPipeline(text: combined, repeatForever: repeatForever, voiceIdentifier: nil, ssml: false, dying: dying)
}

/// Speaks one piece of text typed on the web UI's Text to Speech page
/// ("Speak Text" section) once, through the same pipeline as the folder
/// Listen. `voiceIdentifier` nil/empty = the Text to Speech voice setting.
/// Text starting with `<speak` is passed as SSML (`--ssml`), the same rule
/// StationDirector's announcer uses.
func speakText(_ text: String, voiceIdentifier: String?) {
let trimmed = text.trimmingCharacters(in: .whitespacesAndNewlines)
guard !trimmed.isEmpty else {
lastError = SDRError.notImplemented("Text to Speech — no text to speak")
LogStore.shared.log(.error, source: "SDRController", "Speak Text: nothing to speak")
return
}
let dying = radioTaskPipelineManager.taskItems.compactMap { $0.process }.filter { $0.isRunning }
Self.sweepOrphanedHelpers()
stopFillerForNewSource()
radioTaskPipelineManager.terminate()
startSpeechSynthPipeline(text: trimmed, repeatForever: false,
voiceIdentifier: voiceIdentifier.flatMap { $0.isEmpty ? nil : $0 },
ssml: trimmed.hasPrefix("<speak"), dying: dying)
}

/// The shared tail of `startTextToSpeech` and `speakText`: stages `text`
/// and launches PCMSpeechSynth → sox → PCMUDPSender. The caller has
/// already stopped the previous pipeline (`dying` are its processes).
private func startSpeechSynthPipeline(text combined: String, repeatForever: Bool,
voiceIdentifier: String?, ssml: Bool, dying: [Process]) {
// Stage the combined text in the app's own temp dir (inside the sandbox
// container, so the PCMSpeechSynth child can read it) rather than passing
// it as a --text argument, which would risk ARG_MAX for a large folder.
Expand Down Expand Up @@ -1850,7 +1878,8 @@ final class SDRController {
directSamplingQBranch = false
lastError = nil

guard let synth = makeSpeechSynthTaskItem(textFileURL: textFileURL, repeatForever: repeatForever),
guard let synth = makeSpeechSynthTaskItem(textFileURL: textFileURL, repeatForever: repeatForever,
voiceIdentifier: voiceIdentifier, ssml: ssml),
let resample = makeResampleTaskItem(inputRate: Self.speechSynthSampleRate,
inputChannels: 1,
audioOutputFilter: "vol 1"),
Expand All @@ -1868,7 +1897,8 @@ final class SDRController {
launchCurrentPipeline(dying: dying)
}

private func makeSpeechSynthTaskItem(textFileURL: URL, repeatForever: Bool) -> TaskItem? {
private func makeSpeechSynthTaskItem(textFileURL: URL, repeatForever: Bool,
voiceIdentifier: String? = nil, ssml: Bool = false) -> TaskItem? {
let path = helperPath("PCMSpeechSynth")
guard FileManager.default.isExecutableFile(atPath: path) else {
lastError = SDRError.notImplemented("PCMSpeechSynth helper missing at \(path)")
Expand All @@ -1880,10 +1910,11 @@ final class SDRController {
item.addArgument("--rate"); item.addArgument(Self.speechSynthSampleRate)
// Without --voice the helper would use the built-in default (compact
// Samantha) and ignore the Speech settings entirely.
if let voice = SpeechVoicePreference.resolvedIdentifier(
if let voice = voiceIdentifier ?? SpeechVoicePreference.resolvedIdentifier(
overrideKey: SpeechVoicePreference.textToSpeechVoiceKey, sqlite: sqliteController) {
item.addArgument("--voice"); item.addArgument(voice)
}
if ssml { item.addArgument("--ssml") }
if repeatForever {
item.addArgument("--repeat")
item.addArgument("--gap"); item.addArgument(2)
Expand Down
30 changes: 30 additions & 0 deletions Web/js/antennahead.js
Original file line number Diff line number Diff line change
Expand Up @@ -2300,6 +2300,36 @@ function textToSpeechListenButtonClicked(form)
}


// Speak Text section (bottom of the Text to Speech page): speaks the text
// box once with the chosen voice ("" = the Text to Speech voice setting).
// Text starting with <speak> is SSML — the server decides.
function speakTextButtonClicked(form)
{
var textArea = form.querySelector("#tts_speak_text");
var voiceSelect = form.querySelector("#tts_speak_voice");
var payload = {
text: textArea ? textArea.value : "",
voice: voiceSelect ? voiceSelect.value : ""
};

var getUrl = window.location;
var baseUrl = getUrl.protocol + "//" + getUrl.host + "/";
var xhttp = new XMLHttpRequest();
xhttp.onreadystatechange = function() {
if (this.readyState == 4 && this.status == 200) {
window.top.nowPlayingTitle = window.document.getElementById("listen_title");
}
};
xhttp.open("POST", baseUrl + "speaktextbuttonclicked.html", true);
xhttp.setRequestHeader("Content-Type", "application/json");
xhttp.send(JSON.stringify(payload));

showUpNextInNavBar();

window.top.postMessage("startaudio", "*");
}


// ---- Play Audio Files (Audio Devices page) ------------------------------
//
// Same pattern as Text to Speech above: the folder is chosen in
Expand Down
Loading