From a5932f00252e83d15cc1b7975c367c2c452b77ca Mon Sep 17 00:00:00 2001 From: Carlo Baratto Date: Sat, 12 Sep 2026 12:42:06 +0200 Subject: [PATCH] =?UTF-8?q?refactor(config):=20niente=20configurazione=20c?= =?UTF-8?q?ablata=20=E2=80=94=20tutto=20parametrizzato=20(v0.3.0)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Prima di pubblicare: eliminati dai sorgenti tutti i valori "cablati". - settings: default neutri (URL, voce, engine vuoti = "decide il server"); nuove chiavi ttsVoices/ttsEngines (preset dei selettori) e vadMaxMs (durata massima registrazione, prima fissa a 60 s nel codice) - ApiClient: guard su baseUrl vuoto (checkAuth/login si comportano senza rete), voce/engine omessi nella richiesta TTS se vuoti, Content-Type audio coerente col formato reale del file (ogg/wav/mp3/m4a/opus) - SettingsPage: selettori voce/engine costruiti dai preset configurabili (Repeater dentro ContextMenu = pattern provato su tide) + campi per modificare i preset; LoginPage richiede indirizzo server non vuoto - VoiceRecorder: tetto di registrazione da configurazione - README: sezione "Configuration" completa; docs neutralizzate (niente URL/valori personali: repo public-friendly) --- README.md | 33 +++++++--- docs/BUILD.md | 8 +-- docs/PROTOCOL.md | 2 +- qml/pages/LoginPage.qml | 2 +- qml/pages/SettingsPage.qml | 121 +++++++++++++++++++++++++------------ rpm/harbour-hermes.spec | 8 ++- src/apiclient.cpp | 32 +++++++++- src/main.cpp | 4 +- src/settings.cpp | 62 +++++++++++++++++-- src/settings.h | 20 ++++++ src/voicerecorder.cpp | 4 +- 11 files changed, 232 insertions(+), 64 deletions(-) diff --git a/README.md b/README.md index cfc7518..943b2a1 100644 --- a/README.md +++ b/README.md @@ -9,18 +9,37 @@ chat with your Hermes agent from the phone — and use it with your **voice**. - Sessions list, new session, open any session - Chat with **live streaming** replies (SSE token deltas) - **Voice dictation**: tap Mic, speak, tap Stop → server-side Whisper - transcription lands in the composer -- **Read aloud**: server-side TTS (edge engine, Italian voices) played through + transcription lands in the composer (optionally sent right away) +- **Read aloud**: server-side TTS (voice/engine configurable) played through QMediaPlayer; optional auto-read of every completed reply +- **Voice dialog (hands-free)**: continuous listen → transcribe → send → + read-aloud loop, auto-stopping on speech pauses (tunable) - Cancel a running turn, resync from server state, cover page ## Requirements - Sailfish OS 4.4+ (developed against Sailfish 5.1) -- A running `hermes-webui` instance reachable over HTTPS (default: - `https://hermes.hackatoniclife.com`) +- A running `hermes-webui` instance reachable over HTTPS (the address is + configured in the app — nothing is compiled in) - Sailjail permissions `Internet;Audio;Microphone` (declared in the .desktop) +## Configuration + +No configuration is compiled into the binary: everything lives in the app's +Settings (persisted in `~/.config/harbour/hermes.conf`, sandbox-persistent). +Defaults are neutral — empty means "the server decides" — and the login screen +requires a server address. + +| Setting | Meaning | +|---|---| +| Server | Hermes Web UI address, e.g. `https://hermes.example.com` (required) | +| Read-aloud voice / TTS engine | empty = server defaults | +| Voice/engine presets | comma-separated lists used to fill the pickers | +| Read replies aloud | auto-play every reply (off by default) | +| Send right after dictation | auto-send when dictation lands (on by default) | +| Auto-stop pause / Mic sensitivity | voice-dialog tuning (1.5 s / medium) | +| Max recording time | voice-dialog safety cap (60 s) | + ## Build See [docs/BUILD.md](docs/BUILD.md) (Sailfish SDK, `sfdk build` / `sfdk deploy`) @@ -29,6 +48,6 @@ speaks (validated live against a real instance). ## Status -Version 0.2.1 — protocol and C++ core validated against a real `hermes-webui` -(login, sessions, streaming chat, TTS, STT); on-device testing (audio paths, -Sailjail prompts, keyboard handling) still pending. +Version 0.3.0 — protocol and C++ core validated against a real `hermes-webui` +(login, sessions, streaming chat, TTS, STT, voice dialog); on-device testing +still ongoing. diff --git a/docs/BUILD.md b/docs/BUILD.md index a51d4f6..1e1bb3f 100644 --- a/docs/BUILD.md +++ b/docs/BUILD.md @@ -75,13 +75,13 @@ richiede il probe WebUI su :8899). Esito 12/09/2026: unit 9/9, live completo OK. ## 4. Tarball per la build da Qt Creator ```bash -cd /home/kaneda/workspace -tar czf harbour-hermes-0.2.1.tar.gz \ - --transform 's,^harbour-hermes,harbour-hermes-0.2.1,' \ +cd ~/workspace +tar czf harbour-hermes-0.3.0.tar.gz \ + --transform 's,^harbour-hermes,harbour-hermes-0.3.0,' \ --exclude='.git' harbour-hermes ``` -Lo spec si aspetta la directory `harbour-hermes-0.2.1/` (pattern degli altri +Lo spec si aspetta la directory `harbour-hermes-0.3.0/` (pattern degli altri progetti: cercato in modo robusto anche per i sorgenti live di sfdk). ## 5. Dipendenze lato server (WebUI) diff --git a/docs/PROTOCOL.md b/docs/PROTOCOL.md index 4f14332..a387e4d 100644 --- a/docs/PROTOCOL.md +++ b/docs/PROTOCOL.md @@ -10,7 +10,7 @@ Documentazione del protocollo HTTP/SSE usato da harbour-hermes. > al client. Server di riferimento: in ascolto su `127.0.0.1:8787` dietro reverse proxy -(`https://hermes.hackatoniclife.com`). TLS obbligatorio in produzione. +(URL configurabile nell'app). TLS obbligatorio in produzione. ## 1. Autenticazione diff --git a/qml/pages/LoginPage.qml b/qml/pages/LoginPage.qml index 6db3f4d..d3fcf8b 100644 --- a/qml/pages/LoginPage.qml +++ b/qml/pages/LoginPage.qml @@ -45,7 +45,7 @@ Page { Button { anchors.horizontalCenter: parent.horizontalCenter text: api.busy ? qsTr("Signing in…") : qsTr("Sign in") - enabled: !api.busy && passField.text.length > 0 + enabled: !api.busy && passField.text.length > 0 && urlField.text.trim().length > 0 onClicked: page.doLogin() } diff --git a/qml/pages/SettingsPage.qml b/qml/pages/SettingsPage.qml index 8391f92..31777bc 100644 --- a/qml/pages/SettingsPage.qml +++ b/qml/pages/SettingsPage.qml @@ -4,14 +4,10 @@ import Sailfish.Silica 1.0 Page { id: page - property var voiceNames: [ - "it-IT-ElsaNeural", - "it-IT-DiegoNeural", - "it-IT-IsabellaNeural", - "it-IT-GiuseppeMultilingualNeural", - "en-US-AriaNeural" - ] - property var engineNames: ["edge", "openai", "elevenlabs"] + // Nessun elenco cablato: le voci/gli engine dei selettori vengono dalle + // impostazioni (ttsVoices/ttsEngines) e il valore corrente è sempre + // incluso. loading evita di salvare durante il caricamento iniziale. + property bool loading: true SilicaFlickable { anchors.fill: parent @@ -42,51 +38,61 @@ Page { ComboBox { label: qsTr("Read-aloud voice") - currentIndex: Math.max(0, page.voiceNames.indexOf(appSettings.ttsVoice)) - // MenuItems espliciti (no Repeater: pattern non provato sul kit 5.1) + currentIndex: page.voiceIndex() + // Repeater dentro ContextMenu = pattern provato sul device (tide) menu: ContextMenu { MenuItem { - text: "it-IT-ElsaNeural" - onClicked: appSettings.ttsVoice = "it-IT-ElsaNeural" + text: qsTr("(server default)") + onClicked: appSettings.ttsVoice = "" } - MenuItem { - text: "it-IT-DiegoNeural" - onClicked: appSettings.ttsVoice = "it-IT-DiegoNeural" - } - MenuItem { - text: "it-IT-IsabellaNeural" - onClicked: appSettings.ttsVoice = "it-IT-IsabellaNeural" - } - MenuItem { - text: "it-IT-GiuseppeMultilingualNeural" - onClicked: appSettings.ttsVoice = "it-IT-GiuseppeMultilingualNeural" - } - MenuItem { - text: "en-US-AriaNeural" - onClicked: appSettings.ttsVoice = "en-US-AriaNeural" + Repeater { + model: page.voiceChoices().length + delegate: MenuItem { + text: page.voiceChoices()[index] + onClicked: appSettings.ttsVoice = page.voiceChoices()[index] + } } } } + TextField { + id: voicesField + width: parent.width + label: qsTr("Voice presets (comma-separated)") + onTextChanged: { + if (!page.loading) + appSettings.ttsVoices = text.split(",") + } + } + ComboBox { label: qsTr("TTS engine") - currentIndex: Math.max(0, page.engineNames.indexOf(appSettings.ttsEngine)) + currentIndex: page.engineIndex() menu: ContextMenu { MenuItem { - text: "edge" - onClicked: appSettings.ttsEngine = "edge" + text: qsTr("(server default)") + onClicked: appSettings.ttsEngine = "" } - MenuItem { - text: "openai" - onClicked: appSettings.ttsEngine = "openai" - } - MenuItem { - text: "elevenlabs" - onClicked: appSettings.ttsEngine = "elevenlabs" + Repeater { + model: page.engineChoices().length + delegate: MenuItem { + text: page.engineChoices()[index] + onClicked: appSettings.ttsEngine = page.engineChoices()[index] + } } } } + TextField { + id: enginesField + width: parent.width + label: qsTr("Engine presets (comma-separated)") + onTextChanged: { + if (!page.loading) + appSettings.ttsEngines = text.split(",") + } + } + TextSwitch { text: qsTr("Read replies aloud") description: qsTr("Speak every completed reply automatically") @@ -159,5 +165,44 @@ Page { pageStack.replace(Qt.resolvedUrl("LoginPage.qml")) } - Component.onCompleted: urlField.text = appSettings.baseUrl + // Elenchi dei selettori: preset configurati + valore corrente (così resta + // selezionabile anche se non è tra i preset). Indice 0 = "(server default)". + function voiceChoices() { + var out = appSettings.ttsVoices.slice() + var cur = appSettings.ttsVoice + if (cur.length > 0 && out.indexOf(cur) < 0) + out.push(cur) + return out + } + + function voiceIndex() { + var cur = appSettings.ttsVoice + if (cur.length === 0) + return 0 + var i = page.voiceChoices().indexOf(cur) + return i < 0 ? 0 : i + 1 + } + + function engineChoices() { + var out = appSettings.ttsEngines.slice() + var cur = appSettings.ttsEngine + if (cur.length > 0 && out.indexOf(cur) < 0) + out.push(cur) + return out + } + + function engineIndex() { + var cur = appSettings.ttsEngine + if (cur.length === 0) + return 0 + var i = page.engineChoices().indexOf(cur) + return i < 0 ? 0 : i + 1 + } + + Component.onCompleted: { + urlField.text = appSettings.baseUrl + voicesField.text = appSettings.ttsVoices.join(", ") + enginesField.text = appSettings.ttsEngines.join(", ") + page.loading = false + } } diff --git a/rpm/harbour-hermes.spec b/rpm/harbour-hermes.spec index c51aefb..9f2b9a8 100644 --- a/rpm/harbour-hermes.spec +++ b/rpm/harbour-hermes.spec @@ -1,6 +1,6 @@ Name: harbour-hermes Summary: Client for the Hermes Web UI with voice -Version: 0.2.1 +Version: 0.3.0 Release: 1 Group: Qt/Qt License: MIT @@ -24,6 +24,12 @@ Sessions list, chat with live streaming replies, voice dictation (server-side TTS), plus a hands-free conversation mode. %changelog +* Sat Sep 12 2026 Carlo Baratto - 0.3.0-1 +- Nessuna configurazione cablata nel binario: indirizzo server, voce, + engine e preset sono impostazioni dell'app (default neutri; vuoto = + decide il server); selettori voce/engine costruiti dai preset + configurabili; Content-Type audio coerente col formato reale; durata + massima registrazione configurabile (vadMaxMs) * Sat Sep 12 2026 Carlo Baratto - 0.2.1-1 - Fix build: rimosso il warning -Wreorder (ordine dell'inizializzazione di m_sessionsModel nel costruttore di ApiClient) diff --git a/src/apiclient.cpp b/src/apiclient.cpp index f2694ea..8e54db6 100644 --- a/src/apiclient.cpp +++ b/src/apiclient.cpp @@ -120,6 +120,13 @@ void ApiClient::clearError() void ApiClient::checkAuth() { + // Senza indirizzo configurato non c'e' nulla da verificare: si parte + // dalla pagina di login, che lo richiede. + if (m_settings->baseUrl().isEmpty()) { + setLoggedIn(false); + emit authChecked(false); + return; + } QNetworkReply *reply = m_nam.get(jsonRequest(QStringLiteral("/api/auth/status"))); connect(reply, &QNetworkReply::finished, this, [this, reply]() { reply->deleteLater(); @@ -144,6 +151,10 @@ void ApiClient::checkAuth() void ApiClient::login(const QString &password) { + if (m_settings->baseUrl().isEmpty()) { + setLastError(QStringLiteral("Server address not configured")); + return; + } setLastError(QString()); setBusy(true); QJsonObject body; @@ -429,7 +440,19 @@ void ApiClient::transcribeFile(const QString &path) QHttpMultiPart *multiPart = new QHttpMultiPart(QHttpMultiPart::FormDataType); QHttpPart filePart; - filePart.setHeader(QNetworkRequest::ContentTypeHeader, QVariant(QStringLiteral("audio/ogg"))); + // Content-Type coerente col formato reale (ogg = dettatura manuale, + // wav = modalita' dialogo, ...). + QString mime = QStringLiteral("audio/ogg"); + const QString suffix = QFileInfo(path).suffix().toLower(); + if (suffix == QLatin1String("wav")) + mime = QStringLiteral("audio/wav"); + else if (suffix == QLatin1String("mp3")) + mime = QStringLiteral("audio/mpeg"); + else if (suffix == QLatin1String("m4a")) + mime = QStringLiteral("audio/mp4"); + else if (suffix == QLatin1String("opus")) + mime = QStringLiteral("audio/opus"); + filePart.setHeader(QNetworkRequest::ContentTypeHeader, QVariant(mime)); filePart.setHeader(QNetworkRequest::ContentDispositionHeader, QVariant(QStringLiteral("form-data; name=\"file\"; filename=\"%1\"") .arg(QFileInfo(path).fileName()))); @@ -477,8 +500,11 @@ void ApiClient::speak(const QString &text) setLastError(QString()); QJsonObject body; body.insert(QStringLiteral("text"), spoken); - body.insert(QStringLiteral("voice"), m_settings->ttsVoice()); - body.insert(QStringLiteral("engine"), m_settings->ttsEngine()); + // Voce/engine vuoti = default del server: i parametri si omettono. + if (!m_settings->ttsVoice().isEmpty()) + body.insert(QStringLiteral("voice"), m_settings->ttsVoice()); + if (!m_settings->ttsEngine().isEmpty()) + body.insert(QStringLiteral("engine"), m_settings->ttsEngine()); QNetworkReply *reply = m_nam.post(jsonRequest(QStringLiteral("/api/tts")), QJsonDocument(body).toJson(QJsonDocument::Compact)); connect(reply, &QNetworkReply::finished, this, [this, reply]() { diff --git a/src/main.cpp b/src/main.cpp index 3248022..119d53f 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -19,9 +19,9 @@ Q_DECL_EXPORT int main(int argc, char *argv[]) // AppConfigLocation = ~/.config/harbour/hermes (zona persistente della sandbox). app->setOrganizationName(QStringLiteral("harbour")); app->setApplicationName(QStringLiteral("hermes")); - app->setApplicationVersion(QStringLiteral("0.2.1")); + app->setApplicationVersion(QStringLiteral("0.3.0")); - qDebug() << "harbour-hermes v0.2.1 build" << __DATE__ << __TIME__; + qDebug() << "harbour-hermes v0.3.0 build" << __DATE__ << __TIME__; Settings settings; ApiClient api(&settings); diff --git a/src/settings.cpp b/src/settings.cpp index bfdd94d..fed9845 100644 --- a/src/settings.cpp +++ b/src/settings.cpp @@ -1,10 +1,11 @@ #include "settings.h" -// Default: l'istanza pubblica del WebUI usa questo host (HTTPS, cert valido). -// L'utente puo' cambiarlo dalle impostazioni (LAN, VPN, altro dominio). -static const char *kDefaultBaseUrl = "https://hermes.hackatoniclife.com"; -static const char *kDefaultTtsVoice = "it-IT-ElsaNeural"; -static const char *kDefaultTtsEngine = "edge"; +// Nessuna configurazione personale e' compilata nel binario: i default sono +// neutri e TUTTO si imposta dalle Impostazioni dell'app (QSettings in +// ~/.config/harbour/hermes.conf). Vuoto = "decide il server". +static const char *kDefaultBaseUrl = ""; +static const char *kDefaultTtsVoice = ""; +static const char *kDefaultTtsEngine = ""; Settings::Settings(QObject *parent) : QObject(parent) @@ -109,6 +110,57 @@ void Settings::setVadThreshold(int threshold) emit vadThresholdChanged(); } +int Settings::vadMaxMs() const +{ + return m_settings.value(QStringLiteral("vadMaxMs"), 60000).toInt(); +} + +void Settings::setVadMaxMs(int ms) +{ + if (ms == vadMaxMs()) + return; + m_settings.setValue(QStringLiteral("vadMaxMs"), ms); + emit vadMaxMsChanged(); +} + +QStringList Settings::ttsVoices() const +{ + return m_settings.value(QStringLiteral("ttsVoices")).toStringList(); +} + +void Settings::setTtsVoices(const QStringList &list) +{ + QStringList clean; + for (const QString &v : list) { + const QString t = v.trimmed(); + if (!t.isEmpty() && !clean.contains(t)) + clean << t; + } + if (clean == ttsVoices()) + return; + m_settings.setValue(QStringLiteral("ttsVoices"), clean); + emit ttsVoicesChanged(); +} + +QStringList Settings::ttsEngines() const +{ + return m_settings.value(QStringLiteral("ttsEngines")).toStringList(); +} + +void Settings::setTtsEngines(const QStringList &list) +{ + QStringList clean; + for (const QString &v : list) { + const QString t = v.trimmed(); + if (!t.isEmpty() && !clean.contains(t)) + clean << t; + } + if (clean == ttsEngines()) + return; + m_settings.setValue(QStringLiteral("ttsEngines"), clean); + emit ttsEnginesChanged(); +} + QString Settings::defaultWorkspace() const { return m_settings.value(QStringLiteral("defaultWorkspace"), QString()).toString(); diff --git a/src/settings.h b/src/settings.h index 74a05bd..3d5fa41 100644 --- a/src/settings.h +++ b/src/settings.h @@ -4,6 +4,7 @@ #include #include #include +#include // Impostazioni persistenti dell'app. // QSettings con OrganizationName "harbour" / ApplicationName "hermes" (impostati @@ -25,6 +26,12 @@ class Settings : public QObject NOTIFY vadSilenceMsChanged) Q_PROPERTY(int vadThreshold READ vadThreshold WRITE setVadThreshold NOTIFY vadThresholdChanged) + Q_PROPERTY(int vadMaxMs READ vadMaxMs WRITE setVadMaxMs + NOTIFY vadMaxMsChanged) + Q_PROPERTY(QStringList ttsVoices READ ttsVoices WRITE setTtsVoices + NOTIFY ttsVoicesChanged) + Q_PROPERTY(QStringList ttsEngines READ ttsEngines WRITE setTtsEngines + NOTIFY ttsEnginesChanged) Q_PROPERTY(QString defaultWorkspace READ defaultWorkspace WRITE setDefaultWorkspace NOTIFY defaultWorkspaceChanged) @@ -52,6 +59,16 @@ public: int vadThreshold() const; void setVadThreshold(int threshold); + int vadMaxMs() const; + void setVadMaxMs(int ms); + + // Preset per i selettori voce/engine (liste configurabili, default vuote). + QStringList ttsVoices() const; + void setTtsVoices(const QStringList &list); + + QStringList ttsEngines() const; + void setTtsEngines(const QStringList &list); + QString defaultWorkspace() const; void setDefaultWorkspace(const QString &ws); @@ -63,6 +80,9 @@ signals: void dictationAutoSendChanged(); void vadSilenceMsChanged(); void vadThresholdChanged(); + void vadMaxMsChanged(); + void ttsVoicesChanged(); + void ttsEnginesChanged(); void defaultWorkspaceChanged(); private: diff --git a/src/voicerecorder.cpp b/src/voicerecorder.cpp index 1a47b5d..c9ed3fd 100644 --- a/src/voicerecorder.cpp +++ b/src/voicerecorder.cpp @@ -174,8 +174,8 @@ void VoiceRecorder::onSilenceCheck() return; } - // Tetto di sicurezza anti-registrazione infinita. - if (elapsed >= 60000) { + // Tetto di sicurezza anti-registrazione infinita (configurabile: vadMaxMs). + if (elapsed >= qint64(m_settings->vadMaxMs())) { qDebug() << "[voice] stop automatico: durata massima raggiunta"; finalize(); }