feat(voice): modalità dialogo — voce continua hands-free (v0.2.0)
- Nuovo VoiceRecorder (QAudioInput, PCM 16 bit mono): analisi del livello
in tempo reale e stop automatico dopo il silenzio (pausa e soglia
configurabili dalle impostazioni); produce WAV per /api/transcribe;
emette failed("no-speech") se non rileva voce (cattura scartata)
- ChatPage: modalità dialogo (menu "Voice mode"): ascolto -> trascrizione
-> invio -> lettura della risposta -> riascolto in ciclo; tocco sul
testo di stato per uscire; limite di 3 errori consecutivi poi esce
- Dettatura manuale: invio automatico dopo la trascrizione (opzione,
default attivo, disattivabile)
- Settings: sezione "Voice dialog" (auto-invio, pausa auto-stop,
sensibilità microfono)
- Il percorso di dettatura manuale (Recorder OGG) resta invariato
This commit is contained in:
@@ -29,6 +29,6 @@ speaks (validated live against a real instance).
|
|||||||
|
|
||||||
## Status
|
## Status
|
||||||
|
|
||||||
Version 0.1.2 — protocol and C++ core validated against a real `hermes-webui`
|
Version 0.2.0 — protocol and C++ core validated against a real `hermes-webui`
|
||||||
(login, sessions, streaming chat, TTS, STT); on-device testing (audio paths,
|
(login, sessions, streaming chat, TTS, STT); on-device testing (audio paths,
|
||||||
Sailjail prompts, keyboard handling) still pending.
|
Sailjail prompts, keyboard handling) still pending.
|
||||||
|
|||||||
+3
-3
@@ -76,12 +76,12 @@ richiede il probe WebUI su :8899). Esito 12/09/2026: unit 9/9, live completo OK.
|
|||||||
|
|
||||||
```bash
|
```bash
|
||||||
cd /home/kaneda/workspace
|
cd /home/kaneda/workspace
|
||||||
tar czf harbour-hermes-0.1.2.tar.gz \
|
tar czf harbour-hermes-0.2.0.tar.gz \
|
||||||
--transform 's,^harbour-hermes,harbour-hermes-0.1.2,' \
|
--transform 's,^harbour-hermes,harbour-hermes-0.2.0,' \
|
||||||
--exclude='.git' harbour-hermes
|
--exclude='.git' harbour-hermes
|
||||||
```
|
```
|
||||||
|
|
||||||
Lo spec si aspetta la directory `harbour-hermes-0.1.2/` (pattern degli altri
|
Lo spec si aspetta la directory `harbour-hermes-0.2.0/` (pattern degli altri
|
||||||
progetti: cercato in modo robusto anche per i sorgenti live di sfdk).
|
progetti: cercato in modo robusto anche per i sorgenti live di sfdk).
|
||||||
|
|
||||||
## 5. Dipendenze lato server (WebUI)
|
## 5. Dipendenze lato server (WebUI)
|
||||||
|
|||||||
@@ -16,6 +16,7 @@ SOURCES += \
|
|||||||
src/chatmodel.cpp \
|
src/chatmodel.cpp \
|
||||||
src/sessionsmodel.cpp \
|
src/sessionsmodel.cpp \
|
||||||
src/recorder.cpp \
|
src/recorder.cpp \
|
||||||
|
src/voicerecorder.cpp \
|
||||||
src/player.cpp
|
src/player.cpp
|
||||||
|
|
||||||
HEADERS += \
|
HEADERS += \
|
||||||
@@ -26,6 +27,7 @@ HEADERS += \
|
|||||||
src/chatmodel.h \
|
src/chatmodel.h \
|
||||||
src/sessionsmodel.h \
|
src/sessionsmodel.h \
|
||||||
src/recorder.h \
|
src/recorder.h \
|
||||||
|
src/voicerecorder.h \
|
||||||
src/player.h
|
src/player.h
|
||||||
|
|
||||||
OTHER_FILES += \
|
OTHER_FILES += \
|
||||||
|
|||||||
+172
-10
@@ -7,6 +7,11 @@ Page {
|
|||||||
|
|
||||||
property string sessionId
|
property string sessionId
|
||||||
property string noticeText: ""
|
property string noticeText: ""
|
||||||
|
// Modalità dialogo (voce continua): ascolto → trascrizione → invio →
|
||||||
|
// lettura della risposta → riascolto (vedi toggleVoiceMode()).
|
||||||
|
property bool voiceMode: false
|
||||||
|
property string voiceState: "idle" // listening|transcribing|sending|streaming|speaking
|
||||||
|
property int voiceErrors: 0
|
||||||
// Altezza tastiera: alza il composer quando la tastiera e' aperta.
|
// Altezza tastiera: alza il composer quando la tastiera e' aperta.
|
||||||
property real kbHeight: Qt.inputMethod.visible
|
property real kbHeight: Qt.inputMethod.visible
|
||||||
? Math.max(0, page.height - Qt.inputMethod.keyboardRectangle.y)
|
? Math.max(0, page.height - Qt.inputMethod.keyboardRectangle.y)
|
||||||
@@ -49,6 +54,10 @@ Page {
|
|||||||
}
|
}
|
||||||
|
|
||||||
PullDownMenu {
|
PullDownMenu {
|
||||||
|
MenuItem {
|
||||||
|
text: page.voiceMode ? qsTr("Voice mode: on") : qsTr("Voice mode: off")
|
||||||
|
onClicked: page.toggleVoiceMode()
|
||||||
|
}
|
||||||
MenuItem {
|
MenuItem {
|
||||||
text: qsTr("Read last reply aloud")
|
text: qsTr("Read last reply aloud")
|
||||||
onClicked: page.readLast()
|
onClicked: page.readLast()
|
||||||
@@ -119,18 +128,28 @@ Page {
|
|||||||
spacing: Theme.paddingSmall / 2
|
spacing: Theme.paddingSmall / 2
|
||||||
|
|
||||||
Label {
|
Label {
|
||||||
|
id: statusLbl
|
||||||
x: Theme.horizontalPageMargin
|
x: Theme.horizontalPageMargin
|
||||||
width: parent.width - 2 * Theme.horizontalPageMargin
|
width: parent.width - 2 * Theme.horizontalPageMargin
|
||||||
font.pixelSize: Theme.fontSizeExtraSmall
|
font.pixelSize: Theme.fontSizeExtraSmall
|
||||||
color: recorder.recording ? Theme.highlightColor
|
color: page.voiceMode ? Theme.highlightColor
|
||||||
: Theme.secondaryHighlightColor
|
: (recorder.recording ? Theme.highlightColor
|
||||||
visible: recorder.recording || api.transcribing || page.noticeText.length > 0
|
: Theme.secondaryHighlightColor)
|
||||||
|
visible: page.voiceMode || recorder.recording || api.transcribing || page.noticeText.length > 0
|
||||||
truncationMode: TruncationMode.Fade
|
truncationMode: TruncationMode.Fade
|
||||||
text: recorder.recording
|
text: page.voiceMode
|
||||||
|
? page.voiceStatusText()
|
||||||
|
: (recorder.recording
|
||||||
? qsTr("Recording… %1 — tap Stop to transcribe").arg(page.formatMs(recorder.durationMs))
|
? qsTr("Recording… %1 — tap Stop to transcribe").arg(page.formatMs(recorder.durationMs))
|
||||||
: api.transcribing
|
: (api.transcribing ? qsTr("Transcribing…") : page.noticeText))
|
||||||
? qsTr("Transcribing…")
|
|
||||||
: page.noticeText
|
// In modalità dialogo: un tocco sul testo esce (MouseArea
|
||||||
|
// interna al Label: nessuna interferenza con la Column).
|
||||||
|
MouseArea {
|
||||||
|
anchors.fill: parent
|
||||||
|
enabled: page.voiceMode
|
||||||
|
onClicked: page.exitVoiceMode()
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
Row {
|
Row {
|
||||||
@@ -142,7 +161,7 @@ Page {
|
|||||||
id: micBtn
|
id: micBtn
|
||||||
text: recorder.recording ? qsTr("Stop") : qsTr("Mic")
|
text: recorder.recording ? qsTr("Stop") : qsTr("Mic")
|
||||||
// Button non ha "highlighted": lo stato è nel testo (Mic/Stop)
|
// Button non ha "highlighted": lo stato è nel testo (Mic/Stop)
|
||||||
enabled: !api.transcribing && !chat.busy || recorder.recording
|
enabled: !page.voiceMode && (!api.transcribing && !chat.busy || recorder.recording)
|
||||||
onClicked: {
|
onClicked: {
|
||||||
if (recorder.recording)
|
if (recorder.recording)
|
||||||
recorder.stop()
|
recorder.stop()
|
||||||
@@ -184,11 +203,30 @@ Page {
|
|||||||
onTriggered: page.noticeText = ""
|
onTriggered: page.noticeText = ""
|
||||||
}
|
}
|
||||||
|
|
||||||
|
Timer {
|
||||||
|
id: voiceRetry
|
||||||
|
interval: 1200
|
||||||
|
onTriggered: {
|
||||||
|
if (page.voiceMode)
|
||||||
|
page.startListening()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
Connections {
|
Connections {
|
||||||
target: chat
|
target: chat
|
||||||
onRowCountChanged: scrollDebounce.restart()
|
onRowCountChanged: scrollDebounce.restart()
|
||||||
onStreamTick: scrollDebounce.restart()
|
onStreamTick: scrollDebounce.restart()
|
||||||
onAssistantFinished: {
|
onAssistantFinished: {
|
||||||
|
if (page.voiceMode) {
|
||||||
|
var t = chat.lastAssistantText()
|
||||||
|
if (t.length > 0) {
|
||||||
|
page.voiceState = "speaking"
|
||||||
|
api.speak(t)
|
||||||
|
} else {
|
||||||
|
voiceRetry.restart()
|
||||||
|
}
|
||||||
|
return
|
||||||
|
}
|
||||||
if (appSettings.ttsAutoRead)
|
if (appSettings.ttsAutoRead)
|
||||||
page.readLast()
|
page.readLast()
|
||||||
}
|
}
|
||||||
@@ -198,11 +236,39 @@ Page {
|
|||||||
Connections {
|
Connections {
|
||||||
target: api
|
target: api
|
||||||
onTranscriptionReady: {
|
onTranscriptionReady: {
|
||||||
|
page.voiceErrors = 0
|
||||||
|
if (page.voiceMode) {
|
||||||
|
if (text.trim().length === 0) {
|
||||||
|
voiceRetry.restart()
|
||||||
|
return
|
||||||
|
}
|
||||||
|
input.text = text
|
||||||
|
page.voiceState = "sending"
|
||||||
|
page.doSend()
|
||||||
|
return
|
||||||
|
}
|
||||||
input.text = text
|
input.text = text
|
||||||
input.focus = true
|
input.focus = true
|
||||||
|
if (appSettings.dictationAutoSend)
|
||||||
|
page.doSend()
|
||||||
|
}
|
||||||
|
onTranscriptionFailed: {
|
||||||
|
page.showNotice(error)
|
||||||
|
if (page.voiceMode) {
|
||||||
|
page.voiceErrors += 1
|
||||||
|
if (page.voiceErrors >= 3) {
|
||||||
|
page.exitVoiceMode()
|
||||||
|
} else {
|
||||||
|
voiceRetry.restart()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
onTtsFailed: {
|
||||||
|
if (page.voiceMode)
|
||||||
|
voiceRetry.restart()
|
||||||
|
else
|
||||||
|
page.showNotice(error)
|
||||||
}
|
}
|
||||||
onTranscriptionFailed: page.showNotice(error)
|
|
||||||
onTtsFailed: page.showNotice(error)
|
|
||||||
}
|
}
|
||||||
|
|
||||||
Component.onCompleted: {
|
Component.onCompleted: {
|
||||||
@@ -210,6 +276,102 @@ Page {
|
|||||||
chat.openSession(sessionId)
|
chat.openSession(sessionId)
|
||||||
}
|
}
|
||||||
|
|
||||||
|
Component.onDestruction: {
|
||||||
|
if (page.voiceMode) {
|
||||||
|
page.voiceMode = false
|
||||||
|
voiceRec.stop()
|
||||||
|
player.stop()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Connections {
|
||||||
|
target: player
|
||||||
|
onFinished: {
|
||||||
|
if (page.voiceMode)
|
||||||
|
page.startListening()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
Connections {
|
||||||
|
target: voiceRec
|
||||||
|
onListening: {
|
||||||
|
if (page.voiceMode)
|
||||||
|
page.voiceState = "listening"
|
||||||
|
}
|
||||||
|
onFinished: {
|
||||||
|
if (!page.voiceMode)
|
||||||
|
return
|
||||||
|
if (wavPath.length === 0)
|
||||||
|
return
|
||||||
|
page.voiceState = "transcribing"
|
||||||
|
api.transcribeFile(wavPath)
|
||||||
|
}
|
||||||
|
onFailed: {
|
||||||
|
if (error === "no-speech") {
|
||||||
|
if (page.voiceMode)
|
||||||
|
voiceRetry.restart()
|
||||||
|
return
|
||||||
|
}
|
||||||
|
page.showNotice(error)
|
||||||
|
if (page.voiceMode) {
|
||||||
|
page.voiceErrors += 1
|
||||||
|
if (page.voiceErrors >= 3) {
|
||||||
|
page.exitVoiceMode()
|
||||||
|
page.showNotice(qsTr("Voice mode off after repeated errors"))
|
||||||
|
} else {
|
||||||
|
voiceRetry.restart()
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
function toggleVoiceMode() {
|
||||||
|
if (page.voiceMode)
|
||||||
|
page.exitVoiceMode()
|
||||||
|
else
|
||||||
|
page.enterVoiceMode()
|
||||||
|
}
|
||||||
|
|
||||||
|
function enterVoiceMode() {
|
||||||
|
if (chat.busy || recorder.recording) {
|
||||||
|
page.showNotice(qsTr("Wait for the current reply first"))
|
||||||
|
return
|
||||||
|
}
|
||||||
|
player.stop()
|
||||||
|
page.voiceErrors = 0
|
||||||
|
page.voiceMode = true
|
||||||
|
page.voiceState = "listening"
|
||||||
|
voiceRec.start()
|
||||||
|
}
|
||||||
|
|
||||||
|
function exitVoiceMode() {
|
||||||
|
page.voiceMode = false
|
||||||
|
page.voiceState = "idle"
|
||||||
|
voiceRec.stop()
|
||||||
|
player.stop()
|
||||||
|
}
|
||||||
|
|
||||||
|
function startListening() {
|
||||||
|
if (!page.voiceMode)
|
||||||
|
return
|
||||||
|
page.voiceState = "listening"
|
||||||
|
voiceRec.start()
|
||||||
|
}
|
||||||
|
|
||||||
|
function voiceStatusText() {
|
||||||
|
if (page.voiceState === "listening")
|
||||||
|
return qsTr("● Listening — speak, I stop at the pause (tap to exit)")
|
||||||
|
if (page.voiceState === "transcribing")
|
||||||
|
return qsTr("Transcribing…")
|
||||||
|
if (page.voiceState === "sending")
|
||||||
|
return qsTr("Sending…")
|
||||||
|
if (page.voiceState === "streaming")
|
||||||
|
return qsTr("Hermes is writing…")
|
||||||
|
if (page.voiceState === "speaking")
|
||||||
|
return qsTr("Speaking…")
|
||||||
|
return qsTr("Voice mode on")
|
||||||
|
}
|
||||||
|
|
||||||
function doSend() {
|
function doSend() {
|
||||||
var t = input.text
|
var t = input.text
|
||||||
if (t.trim().length === 0)
|
if (t.trim().length === 0)
|
||||||
|
|||||||
@@ -94,6 +94,40 @@ Page {
|
|||||||
onCheckedChanged: appSettings.ttsAutoRead = checked
|
onCheckedChanged: appSettings.ttsAutoRead = checked
|
||||||
}
|
}
|
||||||
|
|
||||||
|
SectionHeader { text: qsTr("Voice dialog") }
|
||||||
|
|
||||||
|
TextSwitch {
|
||||||
|
text: qsTr("Send right after dictation")
|
||||||
|
description: qsTr("Send the transcribed text automatically, without tapping Send")
|
||||||
|
checked: appSettings.dictationAutoSend
|
||||||
|
onCheckedChanged: appSettings.dictationAutoSend = checked
|
||||||
|
}
|
||||||
|
|
||||||
|
ComboBox {
|
||||||
|
label: qsTr("Auto-stop pause")
|
||||||
|
currentIndex: {
|
||||||
|
var v = appSettings.vadSilenceMs
|
||||||
|
return v <= 1000 ? 0 : (v <= 1500 ? 1 : (v <= 2000 ? 2 : 3))
|
||||||
|
}
|
||||||
|
menu: ContextMenu {
|
||||||
|
MenuItem { text: qsTr("1.0 second"); onClicked: appSettings.vadSilenceMs = 1000 }
|
||||||
|
MenuItem { text: qsTr("1.5 seconds"); onClicked: appSettings.vadSilenceMs = 1500 }
|
||||||
|
MenuItem { text: qsTr("2.0 seconds"); onClicked: appSettings.vadSilenceMs = 2000 }
|
||||||
|
MenuItem { text: qsTr("3.0 seconds"); onClicked: appSettings.vadSilenceMs = 3000 }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
ComboBox {
|
||||||
|
label: qsTr("Mic sensitivity")
|
||||||
|
currentIndex: appSettings.vadThreshold <= 500 ? 0
|
||||||
|
: (appSettings.vadThreshold <= 1000 ? 1 : 2)
|
||||||
|
menu: ContextMenu {
|
||||||
|
MenuItem { text: qsTr("High (even soft voices)"); onClicked: appSettings.vadThreshold = 400 }
|
||||||
|
MenuItem { text: qsTr("Medium"); onClicked: appSettings.vadThreshold = 800 }
|
||||||
|
MenuItem { text: qsTr("Low (noisy rooms)"); onClicked: appSettings.vadThreshold = 1600 }
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
SectionHeader { text: qsTr("Account") }
|
SectionHeader { text: qsTr("Account") }
|
||||||
|
|
||||||
Button {
|
Button {
|
||||||
|
|||||||
@@ -1,6 +1,6 @@
|
|||||||
Name: harbour-hermes
|
Name: harbour-hermes
|
||||||
Summary: Client for the Hermes Web UI with voice
|
Summary: Client for the Hermes Web UI with voice
|
||||||
Version: 0.1.2
|
Version: 0.2.0
|
||||||
Release: 1
|
Release: 1
|
||||||
Group: Qt/Qt
|
Group: Qt/Qt
|
||||||
License: MIT
|
License: MIT
|
||||||
@@ -24,6 +24,11 @@ Sessions list, chat with live streaming replies, voice dictation
|
|||||||
(server-side TTS), plus a hands-free conversation mode.
|
(server-side TTS), plus a hands-free conversation mode.
|
||||||
|
|
||||||
%changelog
|
%changelog
|
||||||
|
* Sat Sep 12 2026 Carlo Baratto <carlo@carlobaratto.it> - 0.2.0-1
|
||||||
|
- Modalita' dialogo (voce continua): ascolto con stop automatico sul
|
||||||
|
silenzio (QAudioInput + VAD), trascrizione, invio automatico, lettura
|
||||||
|
della risposta e riascolto in ciclo; pausa e sensibilita' microfono
|
||||||
|
configurabili; la dettatura manuale puo' inviare da sola
|
||||||
* Sat Sep 12 2026 Carlo Baratto <carlo@carlobaratto.it> - 0.1.2-1
|
* Sat Sep 12 2026 Carlo Baratto <carlo@carlobaratto.it> - 0.1.2-1
|
||||||
- Fix lista sessioni vuota sul device: le QVariantList di mappe non
|
- Fix lista sessioni vuota sul device: le QVariantList di mappe non
|
||||||
espongono i campi nei delegate su Qt 5.6 -> nuovo SessionsModel
|
espongono i campi nei delegate su Qt 5.6 -> nuovo SessionsModel
|
||||||
|
|||||||
+5
-2
@@ -9,6 +9,7 @@
|
|||||||
#include "player.h"
|
#include "player.h"
|
||||||
#include "recorder.h"
|
#include "recorder.h"
|
||||||
#include "settings.h"
|
#include "settings.h"
|
||||||
|
#include "voicerecorder.h"
|
||||||
|
|
||||||
Q_DECL_EXPORT int main(int argc, char *argv[])
|
Q_DECL_EXPORT int main(int argc, char *argv[])
|
||||||
{
|
{
|
||||||
@@ -18,15 +19,16 @@ Q_DECL_EXPORT int main(int argc, char *argv[])
|
|||||||
// AppConfigLocation = ~/.config/harbour/hermes (zona persistente della sandbox).
|
// AppConfigLocation = ~/.config/harbour/hermes (zona persistente della sandbox).
|
||||||
app->setOrganizationName(QStringLiteral("harbour"));
|
app->setOrganizationName(QStringLiteral("harbour"));
|
||||||
app->setApplicationName(QStringLiteral("hermes"));
|
app->setApplicationName(QStringLiteral("hermes"));
|
||||||
app->setApplicationVersion(QStringLiteral("0.1.0"));
|
app->setApplicationVersion(QStringLiteral("0.2.0"));
|
||||||
|
|
||||||
qDebug() << "harbour-hermes v0.1.0 build" << __DATE__ << __TIME__;
|
qDebug() << "harbour-hermes v0.2.0 build" << __DATE__ << __TIME__;
|
||||||
|
|
||||||
Settings settings;
|
Settings settings;
|
||||||
ApiClient api(&settings);
|
ApiClient api(&settings);
|
||||||
ChatModel chat(&api);
|
ChatModel chat(&api);
|
||||||
Recorder recorder;
|
Recorder recorder;
|
||||||
Player player;
|
Player player;
|
||||||
|
VoiceRecorder voiceRec(&settings);
|
||||||
|
|
||||||
// Flusso voce: registrazione -> trascrizione; TTS pronto -> riproduzione.
|
// Flusso voce: registrazione -> trascrizione; TTS pronto -> riproduzione.
|
||||||
QObject::connect(&recorder, &Recorder::finished, &api, &ApiClient::transcribeFile);
|
QObject::connect(&recorder, &Recorder::finished, &api, &ApiClient::transcribeFile);
|
||||||
@@ -38,6 +40,7 @@ Q_DECL_EXPORT int main(int argc, char *argv[])
|
|||||||
view->rootContext()->setContextProperty(QStringLiteral("chat"), &chat);
|
view->rootContext()->setContextProperty(QStringLiteral("chat"), &chat);
|
||||||
view->rootContext()->setContextProperty(QStringLiteral("recorder"), &recorder);
|
view->rootContext()->setContextProperty(QStringLiteral("recorder"), &recorder);
|
||||||
view->rootContext()->setContextProperty(QStringLiteral("player"), &player);
|
view->rootContext()->setContextProperty(QStringLiteral("player"), &player);
|
||||||
|
view->rootContext()->setContextProperty(QStringLiteral("voiceRec"), &voiceRec);
|
||||||
view->rootContext()->setContextProperty(QStringLiteral("appVersion"),
|
view->rootContext()->setContextProperty(QStringLiteral("appVersion"),
|
||||||
QCoreApplication::applicationVersion());
|
QCoreApplication::applicationVersion());
|
||||||
|
|
||||||
|
|||||||
@@ -70,6 +70,45 @@ void Settings::setTtsAutoRead(bool on)
|
|||||||
emit ttsAutoReadChanged();
|
emit ttsAutoReadChanged();
|
||||||
}
|
}
|
||||||
|
|
||||||
|
bool Settings::dictationAutoSend() const
|
||||||
|
{
|
||||||
|
return m_settings.value(QStringLiteral("dictationAutoSend"), true).toBool();
|
||||||
|
}
|
||||||
|
|
||||||
|
void Settings::setDictationAutoSend(bool on)
|
||||||
|
{
|
||||||
|
if (on == dictationAutoSend())
|
||||||
|
return;
|
||||||
|
m_settings.setValue(QStringLiteral("dictationAutoSend"), on);
|
||||||
|
emit dictationAutoSendChanged();
|
||||||
|
}
|
||||||
|
|
||||||
|
int Settings::vadSilenceMs() const
|
||||||
|
{
|
||||||
|
return m_settings.value(QStringLiteral("vadSilenceMs"), 1500).toInt();
|
||||||
|
}
|
||||||
|
|
||||||
|
void Settings::setVadSilenceMs(int ms)
|
||||||
|
{
|
||||||
|
if (ms == vadSilenceMs())
|
||||||
|
return;
|
||||||
|
m_settings.setValue(QStringLiteral("vadSilenceMs"), ms);
|
||||||
|
emit vadSilenceMsChanged();
|
||||||
|
}
|
||||||
|
|
||||||
|
int Settings::vadThreshold() const
|
||||||
|
{
|
||||||
|
return m_settings.value(QStringLiteral("vadThreshold"), 800).toInt();
|
||||||
|
}
|
||||||
|
|
||||||
|
void Settings::setVadThreshold(int threshold)
|
||||||
|
{
|
||||||
|
if (threshold == vadThreshold())
|
||||||
|
return;
|
||||||
|
m_settings.setValue(QStringLiteral("vadThreshold"), threshold);
|
||||||
|
emit vadThresholdChanged();
|
||||||
|
}
|
||||||
|
|
||||||
QString Settings::defaultWorkspace() const
|
QString Settings::defaultWorkspace() const
|
||||||
{
|
{
|
||||||
return m_settings.value(QStringLiteral("defaultWorkspace"), QString()).toString();
|
return m_settings.value(QStringLiteral("defaultWorkspace"), QString()).toString();
|
||||||
|
|||||||
@@ -19,6 +19,12 @@ class Settings : public QObject
|
|||||||
Q_PROPERTY(QString ttsVoice READ ttsVoice WRITE setTtsVoice NOTIFY ttsVoiceChanged)
|
Q_PROPERTY(QString ttsVoice READ ttsVoice WRITE setTtsVoice NOTIFY ttsVoiceChanged)
|
||||||
Q_PROPERTY(QString ttsEngine READ ttsEngine WRITE setTtsEngine NOTIFY ttsEngineChanged)
|
Q_PROPERTY(QString ttsEngine READ ttsEngine WRITE setTtsEngine NOTIFY ttsEngineChanged)
|
||||||
Q_PROPERTY(bool ttsAutoRead READ ttsAutoRead WRITE setTtsAutoRead NOTIFY ttsAutoReadChanged)
|
Q_PROPERTY(bool ttsAutoRead READ ttsAutoRead WRITE setTtsAutoRead NOTIFY ttsAutoReadChanged)
|
||||||
|
Q_PROPERTY(bool dictationAutoSend READ dictationAutoSend WRITE setDictationAutoSend
|
||||||
|
NOTIFY dictationAutoSendChanged)
|
||||||
|
Q_PROPERTY(int vadSilenceMs READ vadSilenceMs WRITE setVadSilenceMs
|
||||||
|
NOTIFY vadSilenceMsChanged)
|
||||||
|
Q_PROPERTY(int vadThreshold READ vadThreshold WRITE setVadThreshold
|
||||||
|
NOTIFY vadThresholdChanged)
|
||||||
Q_PROPERTY(QString defaultWorkspace READ defaultWorkspace WRITE setDefaultWorkspace
|
Q_PROPERTY(QString defaultWorkspace READ defaultWorkspace WRITE setDefaultWorkspace
|
||||||
NOTIFY defaultWorkspaceChanged)
|
NOTIFY defaultWorkspaceChanged)
|
||||||
|
|
||||||
@@ -37,6 +43,15 @@ public:
|
|||||||
bool ttsAutoRead() const;
|
bool ttsAutoRead() const;
|
||||||
void setTtsAutoRead(bool on);
|
void setTtsAutoRead(bool on);
|
||||||
|
|
||||||
|
bool dictationAutoSend() const;
|
||||||
|
void setDictationAutoSend(bool on);
|
||||||
|
|
||||||
|
int vadSilenceMs() const;
|
||||||
|
void setVadSilenceMs(int ms);
|
||||||
|
|
||||||
|
int vadThreshold() const;
|
||||||
|
void setVadThreshold(int threshold);
|
||||||
|
|
||||||
QString defaultWorkspace() const;
|
QString defaultWorkspace() const;
|
||||||
void setDefaultWorkspace(const QString &ws);
|
void setDefaultWorkspace(const QString &ws);
|
||||||
|
|
||||||
@@ -45,6 +60,9 @@ signals:
|
|||||||
void ttsVoiceChanged();
|
void ttsVoiceChanged();
|
||||||
void ttsEngineChanged();
|
void ttsEngineChanged();
|
||||||
void ttsAutoReadChanged();
|
void ttsAutoReadChanged();
|
||||||
|
void dictationAutoSendChanged();
|
||||||
|
void vadSilenceMsChanged();
|
||||||
|
void vadThresholdChanged();
|
||||||
void defaultWorkspaceChanged();
|
void defaultWorkspaceChanged();
|
||||||
|
|
||||||
private:
|
private:
|
||||||
|
|||||||
@@ -0,0 +1,264 @@
|
|||||||
|
#include "voicerecorder.h"
|
||||||
|
|
||||||
|
#include <QAudioDeviceInfo>
|
||||||
|
#include <QAudioInput>
|
||||||
|
#include <QDateTime>
|
||||||
|
#include <QDebug>
|
||||||
|
#include <QDir>
|
||||||
|
#include <QFile>
|
||||||
|
#include <QIODevice>
|
||||||
|
#include <QStandardPaths>
|
||||||
|
#include <QTimer>
|
||||||
|
|
||||||
|
#include "settings.h"
|
||||||
|
|
||||||
|
namespace {
|
||||||
|
|
||||||
|
void append16(QByteArray &b, quint16 v)
|
||||||
|
{
|
||||||
|
b.append(char(v & 0xff));
|
||||||
|
b.append(char((v >> 8) & 0xff));
|
||||||
|
}
|
||||||
|
|
||||||
|
void append32(QByteArray &b, quint32 v)
|
||||||
|
{
|
||||||
|
b.append(char(v & 0xff));
|
||||||
|
b.append(char((v >> 8) & 0xff));
|
||||||
|
b.append(char((v >> 16) & 0xff));
|
||||||
|
b.append(char((v >> 24) & 0xff));
|
||||||
|
}
|
||||||
|
|
||||||
|
// Picco (valore assoluto massimo) dei campioni 16 bit little-endian del blocco.
|
||||||
|
int peakOf(const QByteArray &chunk)
|
||||||
|
{
|
||||||
|
int peak = 0;
|
||||||
|
const char *d = chunk.constData();
|
||||||
|
const int n = chunk.size() - (chunk.size() % 2);
|
||||||
|
for (int i = 0; i < n; i += 2) {
|
||||||
|
const qint16 s = qint16(quint8(d[i]) | (quint8(d[i + 1]) << 8));
|
||||||
|
const int v = qAbs(int(s));
|
||||||
|
if (v > peak)
|
||||||
|
peak = v;
|
||||||
|
}
|
||||||
|
return peak;
|
||||||
|
}
|
||||||
|
|
||||||
|
} // namespace
|
||||||
|
|
||||||
|
VoiceRecorder::VoiceRecorder(Settings *settings, QObject *parent)
|
||||||
|
: QObject(parent)
|
||||||
|
, m_settings(settings)
|
||||||
|
, m_input(nullptr)
|
||||||
|
, m_device(nullptr)
|
||||||
|
, m_silenceTimer(new QTimer(this))
|
||||||
|
, m_lastVoiceMs(0)
|
||||||
|
, m_lastLevelEmitMs(0)
|
||||||
|
, m_hasVoice(false)
|
||||||
|
, m_active(false)
|
||||||
|
, m_level(0.0)
|
||||||
|
{
|
||||||
|
m_silenceTimer->setInterval(150);
|
||||||
|
connect(m_silenceTimer, &QTimer::timeout, this, &VoiceRecorder::onSilenceCheck);
|
||||||
|
}
|
||||||
|
|
||||||
|
void VoiceRecorder::start()
|
||||||
|
{
|
||||||
|
if (m_active)
|
||||||
|
return;
|
||||||
|
|
||||||
|
const QAudioDeviceInfo info = QAudioDeviceInfo::defaultInputDevice();
|
||||||
|
if (info.isNull()) {
|
||||||
|
emit failed(tr("No audio input device"));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// PCM 16 bit mono: 16 kHz di solito basta per la voce; altrimenti si
|
||||||
|
// ripiega su frequenze piu' comuni mantenendo il 16 bit mono.
|
||||||
|
QAudioFormat format;
|
||||||
|
format.setChannelCount(1);
|
||||||
|
format.setSampleSize(16);
|
||||||
|
format.setCodec(QStringLiteral("audio/pcm"));
|
||||||
|
format.setByteOrder(QAudioFormat::LittleEndian);
|
||||||
|
format.setSampleType(QAudioFormat::SignedInt);
|
||||||
|
|
||||||
|
const int rates[] = { 16000, 48000, 44100, 32000, 8000 };
|
||||||
|
bool found = false;
|
||||||
|
for (unsigned i = 0; i < sizeof(rates) / sizeof(rates[0]); ++i) {
|
||||||
|
format.setSampleRate(rates[i]);
|
||||||
|
if (info.isFormatSupported(format)) {
|
||||||
|
found = true;
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (!found) {
|
||||||
|
emit failed(tr("Audio input format not supported"));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
m_format = format;
|
||||||
|
m_pcm.clear();
|
||||||
|
m_hasVoice = false;
|
||||||
|
m_level = 0.0;
|
||||||
|
m_lastVoiceMs = 0;
|
||||||
|
m_lastLevelEmitMs = 0;
|
||||||
|
|
||||||
|
m_input = new QAudioInput(info, m_format, this);
|
||||||
|
m_device = m_input->start();
|
||||||
|
if (!m_device) {
|
||||||
|
delete m_input;
|
||||||
|
m_input = nullptr;
|
||||||
|
emit failed(tr("Could not start audio capture"));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
connect(m_device, &QIODevice::readyRead, this, &VoiceRecorder::onReadyRead);
|
||||||
|
|
||||||
|
m_elapsed.start();
|
||||||
|
m_active = true;
|
||||||
|
emit activeChanged();
|
||||||
|
emit levelChanged();
|
||||||
|
emit listening();
|
||||||
|
qDebug() << "[voice] ascolto avviato:" << m_format.sampleRate() << "Hz mono 16 bit";
|
||||||
|
m_silenceTimer->start();
|
||||||
|
}
|
||||||
|
|
||||||
|
void VoiceRecorder::stop()
|
||||||
|
{
|
||||||
|
if (!m_active)
|
||||||
|
return;
|
||||||
|
finalize();
|
||||||
|
}
|
||||||
|
|
||||||
|
void VoiceRecorder::onReadyRead()
|
||||||
|
{
|
||||||
|
if (!m_device)
|
||||||
|
return;
|
||||||
|
const QByteArray chunk = m_device->readAll();
|
||||||
|
if (chunk.isEmpty())
|
||||||
|
return;
|
||||||
|
m_pcm.append(chunk);
|
||||||
|
|
||||||
|
const int peak = peakOf(chunk);
|
||||||
|
const qint64 elapsed = m_elapsed.elapsed();
|
||||||
|
if (peak > m_settings->vadThreshold()) {
|
||||||
|
m_lastVoiceMs = elapsed;
|
||||||
|
m_hasVoice = true;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (elapsed - m_lastLevelEmitMs >= 150) {
|
||||||
|
m_lastLevelEmitMs = elapsed;
|
||||||
|
m_level = qreal(peak) / 32768.0;
|
||||||
|
emit levelChanged();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void VoiceRecorder::onSilenceCheck()
|
||||||
|
{
|
||||||
|
if (!m_active)
|
||||||
|
return;
|
||||||
|
const qint64 elapsed = m_elapsed.elapsed();
|
||||||
|
|
||||||
|
// Nessuna voce dopo 12 s: la cattura viene scartata (failed "no-speech").
|
||||||
|
if (!m_hasVoice && elapsed >= 12000) {
|
||||||
|
qDebug() << "[voice] nessuna voce rilevata, stop";
|
||||||
|
finalize();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Stop automatico: almeno un frammento di voce udito e pausa oltre soglia.
|
||||||
|
if (m_hasVoice
|
||||||
|
&& elapsed - m_lastVoiceMs >= qint64(m_settings->vadSilenceMs())
|
||||||
|
&& elapsed >= 800) {
|
||||||
|
qDebug() << "[voice] stop automatico: pausa di" << (elapsed - m_lastVoiceMs)
|
||||||
|
<< "ms dopo voce a" << m_lastVoiceMs << "ms";
|
||||||
|
finalize();
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
// Tetto di sicurezza anti-registrazione infinita.
|
||||||
|
if (elapsed >= 60000) {
|
||||||
|
qDebug() << "[voice] stop automatico: durata massima raggiunta";
|
||||||
|
finalize();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
void VoiceRecorder::finalize()
|
||||||
|
{
|
||||||
|
m_silenceTimer->stop();
|
||||||
|
|
||||||
|
if (m_device) {
|
||||||
|
m_pcm.append(m_device->readAll());
|
||||||
|
m_device = nullptr;
|
||||||
|
}
|
||||||
|
if (m_input) {
|
||||||
|
m_input->stop();
|
||||||
|
m_input->deleteLater();
|
||||||
|
m_input = nullptr;
|
||||||
|
}
|
||||||
|
|
||||||
|
m_active = false;
|
||||||
|
m_level = 0.0;
|
||||||
|
emit activeChanged();
|
||||||
|
emit levelChanged();
|
||||||
|
|
||||||
|
if (!m_hasVoice) {
|
||||||
|
m_pcm.clear();
|
||||||
|
emit failed(QStringLiteral("no-speech"));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
|
||||||
|
const QString path = makeWavPath();
|
||||||
|
const int bytes = m_pcm.size();
|
||||||
|
if (!writeWav(path)) {
|
||||||
|
m_pcm.clear();
|
||||||
|
emit failed(tr("Could not write the audio file"));
|
||||||
|
return;
|
||||||
|
}
|
||||||
|
m_pcm.clear();
|
||||||
|
qDebug() << "[voice] WAV pronto:" << path << bytes << "byte";
|
||||||
|
emit finished(path);
|
||||||
|
}
|
||||||
|
|
||||||
|
QString VoiceRecorder::makeWavPath() const
|
||||||
|
{
|
||||||
|
const QString dir = QStandardPaths::writableLocation(QStandardPaths::AppDataLocation)
|
||||||
|
+ QStringLiteral("/voice");
|
||||||
|
QDir().mkpath(dir);
|
||||||
|
return dir + QStringLiteral("/v_")
|
||||||
|
+ QString::number(QDateTime::currentMSecsSinceEpoch())
|
||||||
|
+ QStringLiteral(".wav");
|
||||||
|
}
|
||||||
|
|
||||||
|
bool VoiceRecorder::writeWav(const QString &path)
|
||||||
|
{
|
||||||
|
QFile f(path);
|
||||||
|
if (!f.open(QIODevice::WriteOnly))
|
||||||
|
return false;
|
||||||
|
|
||||||
|
const quint16 channels = quint16(m_format.channelCount());
|
||||||
|
const quint32 sampleRate = quint32(m_format.sampleRate());
|
||||||
|
const quint16 bits = quint16(m_format.sampleSize());
|
||||||
|
const quint32 dataSize = quint32(m_pcm.size());
|
||||||
|
const quint32 byteRate = sampleRate * channels * bits / 8;
|
||||||
|
const quint16 blockAlign = quint16(channels * bits / 8);
|
||||||
|
|
||||||
|
QByteArray header;
|
||||||
|
header.reserve(44);
|
||||||
|
header.append("RIFF");
|
||||||
|
append32(header, 36 + dataSize);
|
||||||
|
header.append("WAVE");
|
||||||
|
header.append("fmt ");
|
||||||
|
append32(header, 16);
|
||||||
|
append16(header, 1); // PCM
|
||||||
|
append16(header, channels);
|
||||||
|
append32(header, sampleRate);
|
||||||
|
append32(header, byteRate);
|
||||||
|
append16(header, blockAlign);
|
||||||
|
append16(header, bits);
|
||||||
|
header.append("data");
|
||||||
|
append32(header, dataSize);
|
||||||
|
|
||||||
|
f.write(header);
|
||||||
|
f.write(m_pcm);
|
||||||
|
f.close();
|
||||||
|
return true;
|
||||||
|
}
|
||||||
@@ -0,0 +1,73 @@
|
|||||||
|
#ifndef VOICERECORDER_H
|
||||||
|
#define VOICERECORDER_H
|
||||||
|
|
||||||
|
#include <QAudioFormat>
|
||||||
|
#include <QByteArray>
|
||||||
|
#include <QElapsedTimer>
|
||||||
|
#include <QObject>
|
||||||
|
#include <QString>
|
||||||
|
|
||||||
|
class QAudioInput;
|
||||||
|
class QIODevice;
|
||||||
|
class QTimer;
|
||||||
|
class Settings;
|
||||||
|
|
||||||
|
// Registratore per la modalità dialogo (voce continua).
|
||||||
|
//
|
||||||
|
// Differenze rispetto a Recorder (dettatura manuale, QAudioRecorder/OGG):
|
||||||
|
// - usa QAudioInput con PCM 16 bit mono per analizzare il livello del
|
||||||
|
// segnale in tempo reale;
|
||||||
|
// - ferma la registrazione DA SOLO dopo un periodo di silenzio (pausa e
|
||||||
|
// soglia configurabili nelle impostazioni) oppure al tetto massimo;
|
||||||
|
// - produce un WAV che il server trascrive come gli altri formati.
|
||||||
|
//
|
||||||
|
// finished(path) arriva per stop automatico o manuale; se non è stata
|
||||||
|
// rilevata voce emette failed("no-speech") (il WAV viene scartato).
|
||||||
|
// Il segnale listening() conferma l'avvio della cattura.
|
||||||
|
class VoiceRecorder : public QObject
|
||||||
|
{
|
||||||
|
Q_OBJECT
|
||||||
|
Q_PROPERTY(bool active READ active NOTIFY activeChanged)
|
||||||
|
Q_PROPERTY(qreal level READ level NOTIFY levelChanged)
|
||||||
|
|
||||||
|
public:
|
||||||
|
explicit VoiceRecorder(Settings *settings, QObject *parent = nullptr);
|
||||||
|
|
||||||
|
bool active() const { return m_active; }
|
||||||
|
qreal level() const { return m_level; }
|
||||||
|
|
||||||
|
Q_INVOKABLE void start();
|
||||||
|
Q_INVOKABLE void stop();
|
||||||
|
|
||||||
|
signals:
|
||||||
|
void activeChanged();
|
||||||
|
void levelChanged();
|
||||||
|
void listening();
|
||||||
|
void finished(const QString &wavPath);
|
||||||
|
void failed(const QString &error);
|
||||||
|
|
||||||
|
private slots:
|
||||||
|
void onReadyRead();
|
||||||
|
void onSilenceCheck();
|
||||||
|
|
||||||
|
private:
|
||||||
|
void finalize();
|
||||||
|
QString makeWavPath() const;
|
||||||
|
bool writeWav(const QString &path);
|
||||||
|
|
||||||
|
Settings *m_settings;
|
||||||
|
QAudioInput *m_input;
|
||||||
|
QIODevice *m_device;
|
||||||
|
QTimer *m_silenceTimer;
|
||||||
|
QAudioFormat m_format;
|
||||||
|
|
||||||
|
QByteArray m_pcm;
|
||||||
|
QElapsedTimer m_elapsed;
|
||||||
|
qint64 m_lastVoiceMs;
|
||||||
|
qint64 m_lastLevelEmitMs;
|
||||||
|
bool m_hasVoice;
|
||||||
|
bool m_active;
|
||||||
|
qreal m_level;
|
||||||
|
};
|
||||||
|
|
||||||
|
#endif // VOICERECORDER_H
|
||||||
@@ -21,6 +21,7 @@ SOURCES += \
|
|||||||
$$SRC/chatmodel.cpp \
|
$$SRC/chatmodel.cpp \
|
||||||
$$SRC/sessionsmodel.cpp \
|
$$SRC/sessionsmodel.cpp \
|
||||||
$$SRC/recorder.cpp \
|
$$SRC/recorder.cpp \
|
||||||
|
$$SRC/voicerecorder.cpp \
|
||||||
$$SRC/player.cpp \
|
$$SRC/player.cpp \
|
||||||
core_test.cpp
|
core_test.cpp
|
||||||
|
|
||||||
@@ -32,4 +33,5 @@ HEADERS += \
|
|||||||
$$SRC/chatmodel.h \
|
$$SRC/chatmodel.h \
|
||||||
$$SRC/sessionsmodel.h \
|
$$SRC/sessionsmodel.h \
|
||||||
$$SRC/recorder.h \
|
$$SRC/recorder.h \
|
||||||
|
$$SRC/voicerecorder.h \
|
||||||
$$SRC/player.h
|
$$SRC/player.h
|
||||||
|
|||||||
Reference in New Issue
Block a user