feat(voice): modalità dialogo — voce continua hands-free (v0.2.0)

- Nuovo VoiceRecorder (QAudioInput, PCM 16 bit mono): analisi del livello
  in tempo reale e stop automatico dopo il silenzio (pausa e soglia
  configurabili dalle impostazioni); produce WAV per /api/transcribe;
  emette failed("no-speech") se non rileva voce (cattura scartata)
- ChatPage: modalità dialogo (menu "Voice mode"): ascolto -> trascrizione
  -> invio -> lettura della risposta -> riascolto in ciclo; tocco sul
  testo di stato per uscire; limite di 3 errori consecutivi poi esce
- Dettatura manuale: invio automatico dopo la trascrizione (opzione,
  default attivo, disattivabile)
- Settings: sezione "Voice dialog" (auto-invio, pausa auto-stop,
  sensibilità microfono)
- Il percorso di dettatura manuale (Recorder OGG) resta invariato
This commit is contained in:
2026-09-12 09:14:25 +02:00
parent 479edb0f99
commit 9deeedf5b3
12 changed files with 620 additions and 18 deletions
+173 -11
View File
@@ -7,6 +7,11 @@ Page {
property string sessionId
property string noticeText: ""
// Modalità dialogo (voce continua): ascolto → trascrizione → invio →
// lettura della risposta → riascolto (vedi toggleVoiceMode()).
property bool voiceMode: false
property string voiceState: "idle" // listening|transcribing|sending|streaming|speaking
property int voiceErrors: 0
// Altezza tastiera: alza il composer quando la tastiera e' aperta.
property real kbHeight: Qt.inputMethod.visible
? Math.max(0, page.height - Qt.inputMethod.keyboardRectangle.y)
@@ -49,6 +54,10 @@ Page {
}
PullDownMenu {
MenuItem {
text: page.voiceMode ? qsTr("Voice mode: on") : qsTr("Voice mode: off")
onClicked: page.toggleVoiceMode()
}
MenuItem {
text: qsTr("Read last reply aloud")
onClicked: page.readLast()
@@ -119,18 +128,28 @@ Page {
spacing: Theme.paddingSmall / 2
Label {
id: statusLbl
x: Theme.horizontalPageMargin
width: parent.width - 2 * Theme.horizontalPageMargin
font.pixelSize: Theme.fontSizeExtraSmall
color: recorder.recording ? Theme.highlightColor
: Theme.secondaryHighlightColor
visible: recorder.recording || api.transcribing || page.noticeText.length > 0
color: page.voiceMode ? Theme.highlightColor
: (recorder.recording ? Theme.highlightColor
: Theme.secondaryHighlightColor)
visible: page.voiceMode || recorder.recording || api.transcribing || page.noticeText.length > 0
truncationMode: TruncationMode.Fade
text: recorder.recording
? qsTr("Recording… %1 — tap Stop to transcribe").arg(page.formatMs(recorder.durationMs))
: api.transcribing
? qsTr("Transcribing…")
: page.noticeText
text: page.voiceMode
? page.voiceStatusText()
: (recorder.recording
? qsTr("Recording… %1 — tap Stop to transcribe").arg(page.formatMs(recorder.durationMs))
: (api.transcribing ? qsTr("Transcribing…") : page.noticeText))
// In modalità dialogo: un tocco sul testo esce (MouseArea
// interna al Label: nessuna interferenza con la Column).
MouseArea {
anchors.fill: parent
enabled: page.voiceMode
onClicked: page.exitVoiceMode()
}
}
Row {
@@ -142,7 +161,7 @@ Page {
id: micBtn
text: recorder.recording ? qsTr("Stop") : qsTr("Mic")
// Button non ha "highlighted": lo stato è nel testo (Mic/Stop)
enabled: !api.transcribing && !chat.busy || recorder.recording
enabled: !page.voiceMode && (!api.transcribing && !chat.busy || recorder.recording)
onClicked: {
if (recorder.recording)
recorder.stop()
@@ -184,11 +203,30 @@ Page {
onTriggered: page.noticeText = ""
}
Timer {
id: voiceRetry
interval: 1200
onTriggered: {
if (page.voiceMode)
page.startListening()
}
}
Connections {
target: chat
onRowCountChanged: scrollDebounce.restart()
onStreamTick: scrollDebounce.restart()
onAssistantFinished: {
if (page.voiceMode) {
var t = chat.lastAssistantText()
if (t.length > 0) {
page.voiceState = "speaking"
api.speak(t)
} else {
voiceRetry.restart()
}
return
}
if (appSettings.ttsAutoRead)
page.readLast()
}
@@ -198,11 +236,39 @@ Page {
Connections {
target: api
onTranscriptionReady: {
page.voiceErrors = 0
if (page.voiceMode) {
if (text.trim().length === 0) {
voiceRetry.restart()
return
}
input.text = text
page.voiceState = "sending"
page.doSend()
return
}
input.text = text
input.focus = true
if (appSettings.dictationAutoSend)
page.doSend()
}
onTranscriptionFailed: {
page.showNotice(error)
if (page.voiceMode) {
page.voiceErrors += 1
if (page.voiceErrors >= 3) {
page.exitVoiceMode()
} else {
voiceRetry.restart()
}
}
}
onTtsFailed: {
if (page.voiceMode)
voiceRetry.restart()
else
page.showNotice(error)
}
onTranscriptionFailed: page.showNotice(error)
onTtsFailed: page.showNotice(error)
}
Component.onCompleted: {
@@ -210,6 +276,102 @@ Page {
chat.openSession(sessionId)
}
Component.onDestruction: {
if (page.voiceMode) {
page.voiceMode = false
voiceRec.stop()
player.stop()
}
}
Connections {
target: player
onFinished: {
if (page.voiceMode)
page.startListening()
}
}
Connections {
target: voiceRec
onListening: {
if (page.voiceMode)
page.voiceState = "listening"
}
onFinished: {
if (!page.voiceMode)
return
if (wavPath.length === 0)
return
page.voiceState = "transcribing"
api.transcribeFile(wavPath)
}
onFailed: {
if (error === "no-speech") {
if (page.voiceMode)
voiceRetry.restart()
return
}
page.showNotice(error)
if (page.voiceMode) {
page.voiceErrors += 1
if (page.voiceErrors >= 3) {
page.exitVoiceMode()
page.showNotice(qsTr("Voice mode off after repeated errors"))
} else {
voiceRetry.restart()
}
}
}
}
function toggleVoiceMode() {
if (page.voiceMode)
page.exitVoiceMode()
else
page.enterVoiceMode()
}
function enterVoiceMode() {
if (chat.busy || recorder.recording) {
page.showNotice(qsTr("Wait for the current reply first"))
return
}
player.stop()
page.voiceErrors = 0
page.voiceMode = true
page.voiceState = "listening"
voiceRec.start()
}
function exitVoiceMode() {
page.voiceMode = false
page.voiceState = "idle"
voiceRec.stop()
player.stop()
}
function startListening() {
if (!page.voiceMode)
return
page.voiceState = "listening"
voiceRec.start()
}
function voiceStatusText() {
if (page.voiceState === "listening")
return qsTr("● Listening — speak, I stop at the pause (tap to exit)")
if (page.voiceState === "transcribing")
return qsTr("Transcribing…")
if (page.voiceState === "sending")
return qsTr("Sending…")
if (page.voiceState === "streaming")
return qsTr("Hermes is writing…")
if (page.voiceState === "speaking")
return qsTr("Speaking…")
return qsTr("Voice mode on")
}
function doSend() {
var t = input.text
if (t.trim().length === 0)
+34
View File
@@ -94,6 +94,40 @@ Page {
onCheckedChanged: appSettings.ttsAutoRead = checked
}
SectionHeader { text: qsTr("Voice dialog") }
TextSwitch {
text: qsTr("Send right after dictation")
description: qsTr("Send the transcribed text automatically, without tapping Send")
checked: appSettings.dictationAutoSend
onCheckedChanged: appSettings.dictationAutoSend = checked
}
ComboBox {
label: qsTr("Auto-stop pause")
currentIndex: {
var v = appSettings.vadSilenceMs
return v <= 1000 ? 0 : (v <= 1500 ? 1 : (v <= 2000 ? 2 : 3))
}
menu: ContextMenu {
MenuItem { text: qsTr("1.0 second"); onClicked: appSettings.vadSilenceMs = 1000 }
MenuItem { text: qsTr("1.5 seconds"); onClicked: appSettings.vadSilenceMs = 1500 }
MenuItem { text: qsTr("2.0 seconds"); onClicked: appSettings.vadSilenceMs = 2000 }
MenuItem { text: qsTr("3.0 seconds"); onClicked: appSettings.vadSilenceMs = 3000 }
}
}
ComboBox {
label: qsTr("Mic sensitivity")
currentIndex: appSettings.vadThreshold <= 500 ? 0
: (appSettings.vadThreshold <= 1000 ? 1 : 2)
menu: ContextMenu {
MenuItem { text: qsTr("High (even soft voices)"); onClicked: appSettings.vadThreshold = 400 }
MenuItem { text: qsTr("Medium"); onClicked: appSettings.vadThreshold = 800 }
MenuItem { text: qsTr("Low (noisy rooms)"); onClicked: appSettings.vadThreshold = 1600 }
}
}
SectionHeader { text: qsTr("Account") }
Button {