From 9deeedf5b3e565fdf9cc78b5260877554c56f342 Mon Sep 17 00:00:00 2001 From: Carlo Baratto Date: Sat, 12 Sep 2026 09:14:25 +0200 Subject: [PATCH] =?UTF-8?q?feat(voice):=20modalit=C3=A0=20dialogo=20?= =?UTF-8?q?=E2=80=94=20voce=20continua=20hands-free=20(v0.2.0)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - Nuovo VoiceRecorder (QAudioInput, PCM 16 bit mono): analisi del livello in tempo reale e stop automatico dopo il silenzio (pausa e soglia configurabili dalle impostazioni); produce WAV per /api/transcribe; emette failed("no-speech") se non rileva voce (cattura scartata) - ChatPage: modalità dialogo (menu "Voice mode"): ascolto -> trascrizione -> invio -> lettura della risposta -> riascolto in ciclo; tocco sul testo di stato per uscire; limite di 3 errori consecutivi poi esce - Dettatura manuale: invio automatico dopo la trascrizione (opzione, default attivo, disattivabile) - Settings: sezione "Voice dialog" (auto-invio, pausa auto-stop, sensibilità microfono) - Il percorso di dettatura manuale (Recorder OGG) resta invariato --- README.md | 2 +- docs/BUILD.md | 6 +- harbour-hermes.pro | 2 + qml/pages/ChatPage.qml | 184 ++++++++++++++++++++++++-- qml/pages/SettingsPage.qml | 34 +++++ rpm/harbour-hermes.spec | 7 +- src/main.cpp | 7 +- src/settings.cpp | 39 ++++++ src/settings.h | 18 +++ src/voicerecorder.cpp | 264 +++++++++++++++++++++++++++++++++++++ src/voicerecorder.h | 73 ++++++++++ tests/core_test.pro | 2 + 12 files changed, 620 insertions(+), 18 deletions(-) create mode 100644 src/voicerecorder.cpp create mode 100644 src/voicerecorder.h diff --git a/README.md b/README.md index ac77096..6385c68 100644 --- a/README.md +++ b/README.md @@ -29,6 +29,6 @@ speaks (validated live against a real instance). ## Status -Version 0.1.2 — protocol and C++ core validated against a real `hermes-webui` +Version 0.2.0 — protocol and C++ core validated against a real `hermes-webui` (login, sessions, streaming chat, TTS, STT); on-device testing (audio paths, Sailjail prompts, keyboard handling) still pending. diff --git a/docs/BUILD.md b/docs/BUILD.md index 4fd45e9..157060a 100644 --- a/docs/BUILD.md +++ b/docs/BUILD.md @@ -76,12 +76,12 @@ richiede il probe WebUI su :8899). Esito 12/09/2026: unit 9/9, live completo OK. ```bash cd /home/kaneda/workspace -tar czf harbour-hermes-0.1.2.tar.gz \ - --transform 's,^harbour-hermes,harbour-hermes-0.1.2,' \ +tar czf harbour-hermes-0.2.0.tar.gz \ + --transform 's,^harbour-hermes,harbour-hermes-0.2.0,' \ --exclude='.git' harbour-hermes ``` -Lo spec si aspetta la directory `harbour-hermes-0.1.2/` (pattern degli altri +Lo spec si aspetta la directory `harbour-hermes-0.2.0/` (pattern degli altri progetti: cercato in modo robusto anche per i sorgenti live di sfdk). ## 5. Dipendenze lato server (WebUI) diff --git a/harbour-hermes.pro b/harbour-hermes.pro index c60eaa4..1cd481e 100644 --- a/harbour-hermes.pro +++ b/harbour-hermes.pro @@ -16,6 +16,7 @@ SOURCES += \ src/chatmodel.cpp \ src/sessionsmodel.cpp \ src/recorder.cpp \ + src/voicerecorder.cpp \ src/player.cpp HEADERS += \ @@ -26,6 +27,7 @@ HEADERS += \ src/chatmodel.h \ src/sessionsmodel.h \ src/recorder.h \ + src/voicerecorder.h \ src/player.h OTHER_FILES += \ diff --git a/qml/pages/ChatPage.qml b/qml/pages/ChatPage.qml index d343421..0b75987 100644 --- a/qml/pages/ChatPage.qml +++ b/qml/pages/ChatPage.qml @@ -7,6 +7,11 @@ Page { property string sessionId property string noticeText: "" + // Modalità dialogo (voce continua): ascolto → trascrizione → invio → + // lettura della risposta → riascolto (vedi toggleVoiceMode()). + property bool voiceMode: false + property string voiceState: "idle" // listening|transcribing|sending|streaming|speaking + property int voiceErrors: 0 // Altezza tastiera: alza il composer quando la tastiera e' aperta. property real kbHeight: Qt.inputMethod.visible ? Math.max(0, page.height - Qt.inputMethod.keyboardRectangle.y) @@ -49,6 +54,10 @@ Page { } PullDownMenu { + MenuItem { + text: page.voiceMode ? qsTr("Voice mode: on") : qsTr("Voice mode: off") + onClicked: page.toggleVoiceMode() + } MenuItem { text: qsTr("Read last reply aloud") onClicked: page.readLast() @@ -119,18 +128,28 @@ Page { spacing: Theme.paddingSmall / 2 Label { + id: statusLbl x: Theme.horizontalPageMargin width: parent.width - 2 * Theme.horizontalPageMargin font.pixelSize: Theme.fontSizeExtraSmall - color: recorder.recording ? Theme.highlightColor - : Theme.secondaryHighlightColor - visible: recorder.recording || api.transcribing || page.noticeText.length > 0 + color: page.voiceMode ? Theme.highlightColor + : (recorder.recording ? Theme.highlightColor + : Theme.secondaryHighlightColor) + visible: page.voiceMode || recorder.recording || api.transcribing || page.noticeText.length > 0 truncationMode: TruncationMode.Fade - text: recorder.recording - ? qsTr("Recording… %1 — tap Stop to transcribe").arg(page.formatMs(recorder.durationMs)) - : api.transcribing - ? qsTr("Transcribing…") - : page.noticeText + text: page.voiceMode + ? page.voiceStatusText() + : (recorder.recording + ? qsTr("Recording… %1 — tap Stop to transcribe").arg(page.formatMs(recorder.durationMs)) + : (api.transcribing ? qsTr("Transcribing…") : page.noticeText)) + + // In modalità dialogo: un tocco sul testo esce (MouseArea + // interna al Label: nessuna interferenza con la Column). + MouseArea { + anchors.fill: parent + enabled: page.voiceMode + onClicked: page.exitVoiceMode() + } } Row { @@ -142,7 +161,7 @@ Page { id: micBtn text: recorder.recording ? qsTr("Stop") : qsTr("Mic") // Button non ha "highlighted": lo stato è nel testo (Mic/Stop) - enabled: !api.transcribing && !chat.busy || recorder.recording + enabled: !page.voiceMode && (!api.transcribing && !chat.busy || recorder.recording) onClicked: { if (recorder.recording) recorder.stop() @@ -184,11 +203,30 @@ Page { onTriggered: page.noticeText = "" } + Timer { + id: voiceRetry + interval: 1200 + onTriggered: { + if (page.voiceMode) + page.startListening() + } + } + Connections { target: chat onRowCountChanged: scrollDebounce.restart() onStreamTick: scrollDebounce.restart() onAssistantFinished: { + if (page.voiceMode) { + var t = chat.lastAssistantText() + if (t.length > 0) { + page.voiceState = "speaking" + api.speak(t) + } else { + voiceRetry.restart() + } + return + } if (appSettings.ttsAutoRead) page.readLast() } @@ -198,11 +236,39 @@ Page { Connections { target: api onTranscriptionReady: { + page.voiceErrors = 0 + if (page.voiceMode) { + if (text.trim().length === 0) { + voiceRetry.restart() + return + } + input.text = text + page.voiceState = "sending" + page.doSend() + return + } input.text = text input.focus = true + if (appSettings.dictationAutoSend) + page.doSend() + } + onTranscriptionFailed: { + page.showNotice(error) + if (page.voiceMode) { + page.voiceErrors += 1 + if (page.voiceErrors >= 3) { + page.exitVoiceMode() + } else { + voiceRetry.restart() + } + } + } + onTtsFailed: { + if (page.voiceMode) + voiceRetry.restart() + else + page.showNotice(error) } - onTranscriptionFailed: page.showNotice(error) - onTtsFailed: page.showNotice(error) } Component.onCompleted: { @@ -210,6 +276,102 @@ Page { chat.openSession(sessionId) } + Component.onDestruction: { + if (page.voiceMode) { + page.voiceMode = false + voiceRec.stop() + player.stop() + } + } + + Connections { + target: player + onFinished: { + if (page.voiceMode) + page.startListening() + } + } + + Connections { + target: voiceRec + onListening: { + if (page.voiceMode) + page.voiceState = "listening" + } + onFinished: { + if (!page.voiceMode) + return + if (wavPath.length === 0) + return + page.voiceState = "transcribing" + api.transcribeFile(wavPath) + } + onFailed: { + if (error === "no-speech") { + if (page.voiceMode) + voiceRetry.restart() + return + } + page.showNotice(error) + if (page.voiceMode) { + page.voiceErrors += 1 + if (page.voiceErrors >= 3) { + page.exitVoiceMode() + page.showNotice(qsTr("Voice mode off after repeated errors")) + } else { + voiceRetry.restart() + } + } + } + } + + function toggleVoiceMode() { + if (page.voiceMode) + page.exitVoiceMode() + else + page.enterVoiceMode() + } + + function enterVoiceMode() { + if (chat.busy || recorder.recording) { + page.showNotice(qsTr("Wait for the current reply first")) + return + } + player.stop() + page.voiceErrors = 0 + page.voiceMode = true + page.voiceState = "listening" + voiceRec.start() + } + + function exitVoiceMode() { + page.voiceMode = false + page.voiceState = "idle" + voiceRec.stop() + player.stop() + } + + function startListening() { + if (!page.voiceMode) + return + page.voiceState = "listening" + voiceRec.start() + } + + function voiceStatusText() { + if (page.voiceState === "listening") + return qsTr("● Listening — speak, I stop at the pause (tap to exit)") + if (page.voiceState === "transcribing") + return qsTr("Transcribing…") + if (page.voiceState === "sending") + return qsTr("Sending…") + if (page.voiceState === "streaming") + return qsTr("Hermes is writing…") + if (page.voiceState === "speaking") + return qsTr("Speaking…") + return qsTr("Voice mode on") + } + function doSend() { var t = input.text if (t.trim().length === 0) diff --git a/qml/pages/SettingsPage.qml b/qml/pages/SettingsPage.qml index d9e1e56..8391f92 100644 --- a/qml/pages/SettingsPage.qml +++ b/qml/pages/SettingsPage.qml @@ -94,6 +94,40 @@ Page { onCheckedChanged: appSettings.ttsAutoRead = checked } + SectionHeader { text: qsTr("Voice dialog") } + + TextSwitch { + text: qsTr("Send right after dictation") + description: qsTr("Send the transcribed text automatically, without tapping Send") + checked: appSettings.dictationAutoSend + onCheckedChanged: appSettings.dictationAutoSend = checked + } + + ComboBox { + label: qsTr("Auto-stop pause") + currentIndex: { + var v = appSettings.vadSilenceMs + return v <= 1000 ? 0 : (v <= 1500 ? 1 : (v <= 2000 ? 2 : 3)) + } + menu: ContextMenu { + MenuItem { text: qsTr("1.0 second"); onClicked: appSettings.vadSilenceMs = 1000 } + MenuItem { text: qsTr("1.5 seconds"); onClicked: appSettings.vadSilenceMs = 1500 } + MenuItem { text: qsTr("2.0 seconds"); onClicked: appSettings.vadSilenceMs = 2000 } + MenuItem { text: qsTr("3.0 seconds"); onClicked: appSettings.vadSilenceMs = 3000 } + } + } + + ComboBox { + label: qsTr("Mic sensitivity") + currentIndex: appSettings.vadThreshold <= 500 ? 0 + : (appSettings.vadThreshold <= 1000 ? 1 : 2) + menu: ContextMenu { + MenuItem { text: qsTr("High (even soft voices)"); onClicked: appSettings.vadThreshold = 400 } + MenuItem { text: qsTr("Medium"); onClicked: appSettings.vadThreshold = 800 } + MenuItem { text: qsTr("Low (noisy rooms)"); onClicked: appSettings.vadThreshold = 1600 } + } + } + SectionHeader { text: qsTr("Account") } Button { diff --git a/rpm/harbour-hermes.spec b/rpm/harbour-hermes.spec index 6e5ad08..0b5e545 100644 --- a/rpm/harbour-hermes.spec +++ b/rpm/harbour-hermes.spec @@ -1,6 +1,6 @@ Name: harbour-hermes Summary: Client for the Hermes Web UI with voice -Version: 0.1.2 +Version: 0.2.0 Release: 1 Group: Qt/Qt License: MIT @@ -24,6 +24,11 @@ Sessions list, chat with live streaming replies, voice dictation (server-side TTS), plus a hands-free conversation mode. %changelog +* Sat Sep 12 2026 Carlo Baratto - 0.2.0-1 +- Modalita' dialogo (voce continua): ascolto con stop automatico sul + silenzio (QAudioInput + VAD), trascrizione, invio automatico, lettura + della risposta e riascolto in ciclo; pausa e sensibilita' microfono + configurabili; la dettatura manuale puo' inviare da sola * Sat Sep 12 2026 Carlo Baratto - 0.1.2-1 - Fix lista sessioni vuota sul device: le QVariantList di mappe non espongono i campi nei delegate su Qt 5.6 -> nuovo SessionsModel diff --git a/src/main.cpp b/src/main.cpp index 8ab8347..6f121b1 100644 --- a/src/main.cpp +++ b/src/main.cpp @@ -9,6 +9,7 @@ #include "player.h" #include "recorder.h" #include "settings.h" +#include "voicerecorder.h" Q_DECL_EXPORT int main(int argc, char *argv[]) { @@ -18,15 +19,16 @@ Q_DECL_EXPORT int main(int argc, char *argv[]) // AppConfigLocation = ~/.config/harbour/hermes (zona persistente della sandbox). app->setOrganizationName(QStringLiteral("harbour")); app->setApplicationName(QStringLiteral("hermes")); - app->setApplicationVersion(QStringLiteral("0.1.0")); + app->setApplicationVersion(QStringLiteral("0.2.0")); - qDebug() << "harbour-hermes v0.1.0 build" << __DATE__ << __TIME__; + qDebug() << "harbour-hermes v0.2.0 build" << __DATE__ << __TIME__; Settings settings; ApiClient api(&settings); ChatModel chat(&api); Recorder recorder; Player player; + VoiceRecorder voiceRec(&settings); // Flusso voce: registrazione -> trascrizione; TTS pronto -> riproduzione. QObject::connect(&recorder, &Recorder::finished, &api, &ApiClient::transcribeFile); @@ -38,6 +40,7 @@ Q_DECL_EXPORT int main(int argc, char *argv[]) view->rootContext()->setContextProperty(QStringLiteral("chat"), &chat); view->rootContext()->setContextProperty(QStringLiteral("recorder"), &recorder); view->rootContext()->setContextProperty(QStringLiteral("player"), &player); + view->rootContext()->setContextProperty(QStringLiteral("voiceRec"), &voiceRec); view->rootContext()->setContextProperty(QStringLiteral("appVersion"), QCoreApplication::applicationVersion()); diff --git a/src/settings.cpp b/src/settings.cpp index dc594d6..bfdd94d 100644 --- a/src/settings.cpp +++ b/src/settings.cpp @@ -70,6 +70,45 @@ void Settings::setTtsAutoRead(bool on) emit ttsAutoReadChanged(); } +bool Settings::dictationAutoSend() const +{ + return m_settings.value(QStringLiteral("dictationAutoSend"), true).toBool(); +} + +void Settings::setDictationAutoSend(bool on) +{ + if (on == dictationAutoSend()) + return; + m_settings.setValue(QStringLiteral("dictationAutoSend"), on); + emit dictationAutoSendChanged(); +} + +int Settings::vadSilenceMs() const +{ + return m_settings.value(QStringLiteral("vadSilenceMs"), 1500).toInt(); +} + +void Settings::setVadSilenceMs(int ms) +{ + if (ms == vadSilenceMs()) + return; + m_settings.setValue(QStringLiteral("vadSilenceMs"), ms); + emit vadSilenceMsChanged(); +} + +int Settings::vadThreshold() const +{ + return m_settings.value(QStringLiteral("vadThreshold"), 800).toInt(); +} + +void Settings::setVadThreshold(int threshold) +{ + if (threshold == vadThreshold()) + return; + m_settings.setValue(QStringLiteral("vadThreshold"), threshold); + emit vadThresholdChanged(); +} + QString Settings::defaultWorkspace() const { return m_settings.value(QStringLiteral("defaultWorkspace"), QString()).toString(); diff --git a/src/settings.h b/src/settings.h index 3d8ea18..74a05bd 100644 --- a/src/settings.h +++ b/src/settings.h @@ -19,6 +19,12 @@ class Settings : public QObject Q_PROPERTY(QString ttsVoice READ ttsVoice WRITE setTtsVoice NOTIFY ttsVoiceChanged) Q_PROPERTY(QString ttsEngine READ ttsEngine WRITE setTtsEngine NOTIFY ttsEngineChanged) Q_PROPERTY(bool ttsAutoRead READ ttsAutoRead WRITE setTtsAutoRead NOTIFY ttsAutoReadChanged) + Q_PROPERTY(bool dictationAutoSend READ dictationAutoSend WRITE setDictationAutoSend + NOTIFY dictationAutoSendChanged) + Q_PROPERTY(int vadSilenceMs READ vadSilenceMs WRITE setVadSilenceMs + NOTIFY vadSilenceMsChanged) + Q_PROPERTY(int vadThreshold READ vadThreshold WRITE setVadThreshold + NOTIFY vadThresholdChanged) Q_PROPERTY(QString defaultWorkspace READ defaultWorkspace WRITE setDefaultWorkspace NOTIFY defaultWorkspaceChanged) @@ -37,6 +43,15 @@ public: bool ttsAutoRead() const; void setTtsAutoRead(bool on); + bool dictationAutoSend() const; + void setDictationAutoSend(bool on); + + int vadSilenceMs() const; + void setVadSilenceMs(int ms); + + int vadThreshold() const; + void setVadThreshold(int threshold); + QString defaultWorkspace() const; void setDefaultWorkspace(const QString &ws); @@ -45,6 +60,9 @@ signals: void ttsVoiceChanged(); void ttsEngineChanged(); void ttsAutoReadChanged(); + void dictationAutoSendChanged(); + void vadSilenceMsChanged(); + void vadThresholdChanged(); void defaultWorkspaceChanged(); private: diff --git a/src/voicerecorder.cpp b/src/voicerecorder.cpp new file mode 100644 index 0000000..1a47b5d --- /dev/null +++ b/src/voicerecorder.cpp @@ -0,0 +1,264 @@ +#include "voicerecorder.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "settings.h" + +namespace { + +void append16(QByteArray &b, quint16 v) +{ + b.append(char(v & 0xff)); + b.append(char((v >> 8) & 0xff)); +} + +void append32(QByteArray &b, quint32 v) +{ + b.append(char(v & 0xff)); + b.append(char((v >> 8) & 0xff)); + b.append(char((v >> 16) & 0xff)); + b.append(char((v >> 24) & 0xff)); +} + +// Picco (valore assoluto massimo) dei campioni 16 bit little-endian del blocco. +int peakOf(const QByteArray &chunk) +{ + int peak = 0; + const char *d = chunk.constData(); + const int n = chunk.size() - (chunk.size() % 2); + for (int i = 0; i < n; i += 2) { + const qint16 s = qint16(quint8(d[i]) | (quint8(d[i + 1]) << 8)); + const int v = qAbs(int(s)); + if (v > peak) + peak = v; + } + return peak; +} + +} // namespace + +VoiceRecorder::VoiceRecorder(Settings *settings, QObject *parent) + : QObject(parent) + , m_settings(settings) + , m_input(nullptr) + , m_device(nullptr) + , m_silenceTimer(new QTimer(this)) + , m_lastVoiceMs(0) + , m_lastLevelEmitMs(0) + , m_hasVoice(false) + , m_active(false) + , m_level(0.0) +{ + m_silenceTimer->setInterval(150); + connect(m_silenceTimer, &QTimer::timeout, this, &VoiceRecorder::onSilenceCheck); +} + +void VoiceRecorder::start() +{ + if (m_active) + return; + + const QAudioDeviceInfo info = QAudioDeviceInfo::defaultInputDevice(); + if (info.isNull()) { + emit failed(tr("No audio input device")); + return; + } + + // PCM 16 bit mono: 16 kHz di solito basta per la voce; altrimenti si + // ripiega su frequenze piu' comuni mantenendo il 16 bit mono. + QAudioFormat format; + format.setChannelCount(1); + format.setSampleSize(16); + format.setCodec(QStringLiteral("audio/pcm")); + format.setByteOrder(QAudioFormat::LittleEndian); + format.setSampleType(QAudioFormat::SignedInt); + + const int rates[] = { 16000, 48000, 44100, 32000, 8000 }; + bool found = false; + for (unsigned i = 0; i < sizeof(rates) / sizeof(rates[0]); ++i) { + format.setSampleRate(rates[i]); + if (info.isFormatSupported(format)) { + found = true; + break; + } + } + if (!found) { + emit failed(tr("Audio input format not supported")); + return; + } + + m_format = format; + m_pcm.clear(); + m_hasVoice = false; + m_level = 0.0; + m_lastVoiceMs = 0; + m_lastLevelEmitMs = 0; + + m_input = new QAudioInput(info, m_format, this); + m_device = m_input->start(); + if (!m_device) { + delete m_input; + m_input = nullptr; + emit failed(tr("Could not start audio capture")); + return; + } + connect(m_device, &QIODevice::readyRead, this, &VoiceRecorder::onReadyRead); + + m_elapsed.start(); + m_active = true; + emit activeChanged(); + emit levelChanged(); + emit listening(); + qDebug() << "[voice] ascolto avviato:" << m_format.sampleRate() << "Hz mono 16 bit"; + m_silenceTimer->start(); +} + +void VoiceRecorder::stop() +{ + if (!m_active) + return; + finalize(); +} + +void VoiceRecorder::onReadyRead() +{ + if (!m_device) + return; + const QByteArray chunk = m_device->readAll(); + if (chunk.isEmpty()) + return; + m_pcm.append(chunk); + + const int peak = peakOf(chunk); + const qint64 elapsed = m_elapsed.elapsed(); + if (peak > m_settings->vadThreshold()) { + m_lastVoiceMs = elapsed; + m_hasVoice = true; + } + + if (elapsed - m_lastLevelEmitMs >= 150) { + m_lastLevelEmitMs = elapsed; + m_level = qreal(peak) / 32768.0; + emit levelChanged(); + } +} + +void VoiceRecorder::onSilenceCheck() +{ + if (!m_active) + return; + const qint64 elapsed = m_elapsed.elapsed(); + + // Nessuna voce dopo 12 s: la cattura viene scartata (failed "no-speech"). + if (!m_hasVoice && elapsed >= 12000) { + qDebug() << "[voice] nessuna voce rilevata, stop"; + finalize(); + return; + } + + // Stop automatico: almeno un frammento di voce udito e pausa oltre soglia. + if (m_hasVoice + && elapsed - m_lastVoiceMs >= qint64(m_settings->vadSilenceMs()) + && elapsed >= 800) { + qDebug() << "[voice] stop automatico: pausa di" << (elapsed - m_lastVoiceMs) + << "ms dopo voce a" << m_lastVoiceMs << "ms"; + finalize(); + return; + } + + // Tetto di sicurezza anti-registrazione infinita. + if (elapsed >= 60000) { + qDebug() << "[voice] stop automatico: durata massima raggiunta"; + finalize(); + } +} + +void VoiceRecorder::finalize() +{ + m_silenceTimer->stop(); + + if (m_device) { + m_pcm.append(m_device->readAll()); + m_device = nullptr; + } + if (m_input) { + m_input->stop(); + m_input->deleteLater(); + m_input = nullptr; + } + + m_active = false; + m_level = 0.0; + emit activeChanged(); + emit levelChanged(); + + if (!m_hasVoice) { + m_pcm.clear(); + emit failed(QStringLiteral("no-speech")); + return; + } + + const QString path = makeWavPath(); + const int bytes = m_pcm.size(); + if (!writeWav(path)) { + m_pcm.clear(); + emit failed(tr("Could not write the audio file")); + return; + } + m_pcm.clear(); + qDebug() << "[voice] WAV pronto:" << path << bytes << "byte"; + emit finished(path); +} + +QString VoiceRecorder::makeWavPath() const +{ + const QString dir = QStandardPaths::writableLocation(QStandardPaths::AppDataLocation) + + QStringLiteral("/voice"); + QDir().mkpath(dir); + return dir + QStringLiteral("/v_") + + QString::number(QDateTime::currentMSecsSinceEpoch()) + + QStringLiteral(".wav"); +} + +bool VoiceRecorder::writeWav(const QString &path) +{ + QFile f(path); + if (!f.open(QIODevice::WriteOnly)) + return false; + + const quint16 channels = quint16(m_format.channelCount()); + const quint32 sampleRate = quint32(m_format.sampleRate()); + const quint16 bits = quint16(m_format.sampleSize()); + const quint32 dataSize = quint32(m_pcm.size()); + const quint32 byteRate = sampleRate * channels * bits / 8; + const quint16 blockAlign = quint16(channels * bits / 8); + + QByteArray header; + header.reserve(44); + header.append("RIFF"); + append32(header, 36 + dataSize); + header.append("WAVE"); + header.append("fmt "); + append32(header, 16); + append16(header, 1); // PCM + append16(header, channels); + append32(header, sampleRate); + append32(header, byteRate); + append16(header, blockAlign); + append16(header, bits); + header.append("data"); + append32(header, dataSize); + + f.write(header); + f.write(m_pcm); + f.close(); + return true; +} diff --git a/src/voicerecorder.h b/src/voicerecorder.h new file mode 100644 index 0000000..0f192c6 --- /dev/null +++ b/src/voicerecorder.h @@ -0,0 +1,73 @@ +#ifndef VOICERECORDER_H +#define VOICERECORDER_H + +#include +#include +#include +#include +#include + +class QAudioInput; +class QIODevice; +class QTimer; +class Settings; + +// Registratore per la modalità dialogo (voce continua). +// +// Differenze rispetto a Recorder (dettatura manuale, QAudioRecorder/OGG): +// - usa QAudioInput con PCM 16 bit mono per analizzare il livello del +// segnale in tempo reale; +// - ferma la registrazione DA SOLO dopo un periodo di silenzio (pausa e +// soglia configurabili nelle impostazioni) oppure al tetto massimo; +// - produce un WAV che il server trascrive come gli altri formati. +// +// finished(path) arriva per stop automatico o manuale; se non è stata +// rilevata voce emette failed("no-speech") (il WAV viene scartato). +// Il segnale listening() conferma l'avvio della cattura. +class VoiceRecorder : public QObject +{ + Q_OBJECT + Q_PROPERTY(bool active READ active NOTIFY activeChanged) + Q_PROPERTY(qreal level READ level NOTIFY levelChanged) + +public: + explicit VoiceRecorder(Settings *settings, QObject *parent = nullptr); + + bool active() const { return m_active; } + qreal level() const { return m_level; } + + Q_INVOKABLE void start(); + Q_INVOKABLE void stop(); + +signals: + void activeChanged(); + void levelChanged(); + void listening(); + void finished(const QString &wavPath); + void failed(const QString &error); + +private slots: + void onReadyRead(); + void onSilenceCheck(); + +private: + void finalize(); + QString makeWavPath() const; + bool writeWav(const QString &path); + + Settings *m_settings; + QAudioInput *m_input; + QIODevice *m_device; + QTimer *m_silenceTimer; + QAudioFormat m_format; + + QByteArray m_pcm; + QElapsedTimer m_elapsed; + qint64 m_lastVoiceMs; + qint64 m_lastLevelEmitMs; + bool m_hasVoice; + bool m_active; + qreal m_level; +}; + +#endif // VOICERECORDER_H diff --git a/tests/core_test.pro b/tests/core_test.pro index 1d978ce..6a6a4c2 100644 --- a/tests/core_test.pro +++ b/tests/core_test.pro @@ -21,6 +21,7 @@ SOURCES += \ $$SRC/chatmodel.cpp \ $$SRC/sessionsmodel.cpp \ $$SRC/recorder.cpp \ + $$SRC/voicerecorder.cpp \ $$SRC/player.cpp \ core_test.cpp @@ -32,4 +33,5 @@ HEADERS += \ $$SRC/chatmodel.h \ $$SRC/sessionsmodel.h \ $$SRC/recorder.h \ + $$SRC/voicerecorder.h \ $$SRC/player.h