feat(voice): modalità dialogo — voce continua hands-free (v0.2.0)
- Nuovo VoiceRecorder (QAudioInput, PCM 16 bit mono): analisi del livello
in tempo reale e stop automatico dopo il silenzio (pausa e soglia
configurabili dalle impostazioni); produce WAV per /api/transcribe;
emette failed("no-speech") se non rileva voce (cattura scartata)
- ChatPage: modalità dialogo (menu "Voice mode"): ascolto -> trascrizione
-> invio -> lettura della risposta -> riascolto in ciclo; tocco sul
testo di stato per uscire; limite di 3 errori consecutivi poi esce
- Dettatura manuale: invio automatico dopo la trascrizione (opzione,
default attivo, disattivabile)
- Settings: sezione "Voice dialog" (auto-invio, pausa auto-stop,
sensibilità microfono)
- Il percorso di dettatura manuale (Recorder OGG) resta invariato
This commit is contained in:
+173
-11
@@ -7,6 +7,11 @@ Page {
|
||||
|
||||
property string sessionId
|
||||
property string noticeText: ""
|
||||
// Modalità dialogo (voce continua): ascolto → trascrizione → invio →
|
||||
// lettura della risposta → riascolto (vedi toggleVoiceMode()).
|
||||
property bool voiceMode: false
|
||||
property string voiceState: "idle" // listening|transcribing|sending|streaming|speaking
|
||||
property int voiceErrors: 0
|
||||
// Altezza tastiera: alza il composer quando la tastiera e' aperta.
|
||||
property real kbHeight: Qt.inputMethod.visible
|
||||
? Math.max(0, page.height - Qt.inputMethod.keyboardRectangle.y)
|
||||
@@ -49,6 +54,10 @@ Page {
|
||||
}
|
||||
|
||||
PullDownMenu {
|
||||
MenuItem {
|
||||
text: page.voiceMode ? qsTr("Voice mode: on") : qsTr("Voice mode: off")
|
||||
onClicked: page.toggleVoiceMode()
|
||||
}
|
||||
MenuItem {
|
||||
text: qsTr("Read last reply aloud")
|
||||
onClicked: page.readLast()
|
||||
@@ -119,18 +128,28 @@ Page {
|
||||
spacing: Theme.paddingSmall / 2
|
||||
|
||||
Label {
|
||||
id: statusLbl
|
||||
x: Theme.horizontalPageMargin
|
||||
width: parent.width - 2 * Theme.horizontalPageMargin
|
||||
font.pixelSize: Theme.fontSizeExtraSmall
|
||||
color: recorder.recording ? Theme.highlightColor
|
||||
: Theme.secondaryHighlightColor
|
||||
visible: recorder.recording || api.transcribing || page.noticeText.length > 0
|
||||
color: page.voiceMode ? Theme.highlightColor
|
||||
: (recorder.recording ? Theme.highlightColor
|
||||
: Theme.secondaryHighlightColor)
|
||||
visible: page.voiceMode || recorder.recording || api.transcribing || page.noticeText.length > 0
|
||||
truncationMode: TruncationMode.Fade
|
||||
text: recorder.recording
|
||||
? qsTr("Recording… %1 — tap Stop to transcribe").arg(page.formatMs(recorder.durationMs))
|
||||
: api.transcribing
|
||||
? qsTr("Transcribing…")
|
||||
: page.noticeText
|
||||
text: page.voiceMode
|
||||
? page.voiceStatusText()
|
||||
: (recorder.recording
|
||||
? qsTr("Recording… %1 — tap Stop to transcribe").arg(page.formatMs(recorder.durationMs))
|
||||
: (api.transcribing ? qsTr("Transcribing…") : page.noticeText))
|
||||
|
||||
// In modalità dialogo: un tocco sul testo esce (MouseArea
|
||||
// interna al Label: nessuna interferenza con la Column).
|
||||
MouseArea {
|
||||
anchors.fill: parent
|
||||
enabled: page.voiceMode
|
||||
onClicked: page.exitVoiceMode()
|
||||
}
|
||||
}
|
||||
|
||||
Row {
|
||||
@@ -142,7 +161,7 @@ Page {
|
||||
id: micBtn
|
||||
text: recorder.recording ? qsTr("Stop") : qsTr("Mic")
|
||||
// Button non ha "highlighted": lo stato è nel testo (Mic/Stop)
|
||||
enabled: !api.transcribing && !chat.busy || recorder.recording
|
||||
enabled: !page.voiceMode && (!api.transcribing && !chat.busy || recorder.recording)
|
||||
onClicked: {
|
||||
if (recorder.recording)
|
||||
recorder.stop()
|
||||
@@ -184,11 +203,30 @@ Page {
|
||||
onTriggered: page.noticeText = ""
|
||||
}
|
||||
|
||||
Timer {
|
||||
id: voiceRetry
|
||||
interval: 1200
|
||||
onTriggered: {
|
||||
if (page.voiceMode)
|
||||
page.startListening()
|
||||
}
|
||||
}
|
||||
|
||||
Connections {
|
||||
target: chat
|
||||
onRowCountChanged: scrollDebounce.restart()
|
||||
onStreamTick: scrollDebounce.restart()
|
||||
onAssistantFinished: {
|
||||
if (page.voiceMode) {
|
||||
var t = chat.lastAssistantText()
|
||||
if (t.length > 0) {
|
||||
page.voiceState = "speaking"
|
||||
api.speak(t)
|
||||
} else {
|
||||
voiceRetry.restart()
|
||||
}
|
||||
return
|
||||
}
|
||||
if (appSettings.ttsAutoRead)
|
||||
page.readLast()
|
||||
}
|
||||
@@ -198,11 +236,39 @@ Page {
|
||||
Connections {
|
||||
target: api
|
||||
onTranscriptionReady: {
|
||||
page.voiceErrors = 0
|
||||
if (page.voiceMode) {
|
||||
if (text.trim().length === 0) {
|
||||
voiceRetry.restart()
|
||||
return
|
||||
}
|
||||
input.text = text
|
||||
page.voiceState = "sending"
|
||||
page.doSend()
|
||||
return
|
||||
}
|
||||
input.text = text
|
||||
input.focus = true
|
||||
if (appSettings.dictationAutoSend)
|
||||
page.doSend()
|
||||
}
|
||||
onTranscriptionFailed: {
|
||||
page.showNotice(error)
|
||||
if (page.voiceMode) {
|
||||
page.voiceErrors += 1
|
||||
if (page.voiceErrors >= 3) {
|
||||
page.exitVoiceMode()
|
||||
} else {
|
||||
voiceRetry.restart()
|
||||
}
|
||||
}
|
||||
}
|
||||
onTtsFailed: {
|
||||
if (page.voiceMode)
|
||||
voiceRetry.restart()
|
||||
else
|
||||
page.showNotice(error)
|
||||
}
|
||||
onTranscriptionFailed: page.showNotice(error)
|
||||
onTtsFailed: page.showNotice(error)
|
||||
}
|
||||
|
||||
Component.onCompleted: {
|
||||
@@ -210,6 +276,102 @@ Page {
|
||||
chat.openSession(sessionId)
|
||||
}
|
||||
|
||||
Component.onDestruction: {
|
||||
if (page.voiceMode) {
|
||||
page.voiceMode = false
|
||||
voiceRec.stop()
|
||||
player.stop()
|
||||
}
|
||||
}
|
||||
|
||||
Connections {
|
||||
target: player
|
||||
onFinished: {
|
||||
if (page.voiceMode)
|
||||
page.startListening()
|
||||
}
|
||||
}
|
||||
|
||||
Connections {
|
||||
target: voiceRec
|
||||
onListening: {
|
||||
if (page.voiceMode)
|
||||
page.voiceState = "listening"
|
||||
}
|
||||
onFinished: {
|
||||
if (!page.voiceMode)
|
||||
return
|
||||
if (wavPath.length === 0)
|
||||
return
|
||||
page.voiceState = "transcribing"
|
||||
api.transcribeFile(wavPath)
|
||||
}
|
||||
onFailed: {
|
||||
if (error === "no-speech") {
|
||||
if (page.voiceMode)
|
||||
voiceRetry.restart()
|
||||
return
|
||||
}
|
||||
page.showNotice(error)
|
||||
if (page.voiceMode) {
|
||||
page.voiceErrors += 1
|
||||
if (page.voiceErrors >= 3) {
|
||||
page.exitVoiceMode()
|
||||
page.showNotice(qsTr("Voice mode off after repeated errors"))
|
||||
} else {
|
||||
voiceRetry.restart()
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
function toggleVoiceMode() {
|
||||
if (page.voiceMode)
|
||||
page.exitVoiceMode()
|
||||
else
|
||||
page.enterVoiceMode()
|
||||
}
|
||||
|
||||
function enterVoiceMode() {
|
||||
if (chat.busy || recorder.recording) {
|
||||
page.showNotice(qsTr("Wait for the current reply first"))
|
||||
return
|
||||
}
|
||||
player.stop()
|
||||
page.voiceErrors = 0
|
||||
page.voiceMode = true
|
||||
page.voiceState = "listening"
|
||||
voiceRec.start()
|
||||
}
|
||||
|
||||
function exitVoiceMode() {
|
||||
page.voiceMode = false
|
||||
page.voiceState = "idle"
|
||||
voiceRec.stop()
|
||||
player.stop()
|
||||
}
|
||||
|
||||
function startListening() {
|
||||
if (!page.voiceMode)
|
||||
return
|
||||
page.voiceState = "listening"
|
||||
voiceRec.start()
|
||||
}
|
||||
|
||||
function voiceStatusText() {
|
||||
if (page.voiceState === "listening")
|
||||
return qsTr("● Listening — speak, I stop at the pause (tap to exit)")
|
||||
if (page.voiceState === "transcribing")
|
||||
return qsTr("Transcribing…")
|
||||
if (page.voiceState === "sending")
|
||||
return qsTr("Sending…")
|
||||
if (page.voiceState === "streaming")
|
||||
return qsTr("Hermes is writing…")
|
||||
if (page.voiceState === "speaking")
|
||||
return qsTr("Speaking…")
|
||||
return qsTr("Voice mode on")
|
||||
}
|
||||
|
||||
function doSend() {
|
||||
var t = input.text
|
||||
if (t.trim().length === 0)
|
||||
|
||||
Reference in New Issue
Block a user