diff --git a/patches/@capacitor-community+speech-recognition+7.0.1.patch b/patches/@capacitor-community+speech-recognition+7.0.1.patch
index 851fc64..abf9ad3 100644
--- a/patches/@capacitor-community+speech-recognition+7.0.1.patch
+++ b/patches/@capacitor-community+speech-recognition+7.0.1.patch
@@ -1,23 +1,481 @@
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/results.bin b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/results.bin
+new file mode 100644
+index 0000000..7ed749e
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/results.bin
+@@ -0,0 +1 @@
++o/bundleLibRuntimeToDirDebug
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/bundleLibRuntimeToDirDebug_dex/com/getcapacitor/community/speechrecognition/Constants.dex b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/bundleLibRuntimeToDirDebug_dex/com/getcapacitor/community/speechrecognition/Constants.dex
+new file mode 100644
+index 0000000..720cd1e
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/bundleLibRuntimeToDirDebug_dex/com/getcapacitor/community/speechrecognition/Constants.dex differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/bundleLibRuntimeToDirDebug_dex/com/getcapacitor/community/speechrecognition/Receiver.dex b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/bundleLibRuntimeToDirDebug_dex/com/getcapacitor/community/speechrecognition/Receiver.dex
+new file mode 100644
+index 0000000..5d21223
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/bundleLibRuntimeToDirDebug_dex/com/getcapacitor/community/speechrecognition/Receiver.dex differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/bundleLibRuntimeToDirDebug_dex/com/getcapacitor/community/speechrecognition/SpeechRecognition$SpeechRecognitionListener.dex b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/bundleLibRuntimeToDirDebug_dex/com/getcapacitor/community/speechrecognition/SpeechRecognition$SpeechRecognitionListener.dex
+new file mode 100644
+index 0000000..668e631
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/bundleLibRuntimeToDirDebug_dex/com/getcapacitor/community/speechrecognition/SpeechRecognition$SpeechRecognitionListener.dex differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/bundleLibRuntimeToDirDebug_dex/com/getcapacitor/community/speechrecognition/SpeechRecognition.dex b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/bundleLibRuntimeToDirDebug_dex/com/getcapacitor/community/speechrecognition/SpeechRecognition.dex
+new file mode 100644
+index 0000000..59b8c5b
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/bundleLibRuntimeToDirDebug_dex/com/getcapacitor/community/speechrecognition/SpeechRecognition.dex differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/desugar_graph.bin b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/desugar_graph.bin
+new file mode 100644
+index 0000000..d60cb42
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/3b668565d422d3defdbffac23d2ae7a9/transformed/bundleLibRuntimeToDirDebug/desugar_graph.bin differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/a9f31cd11f0b4f2759dee65fdc12b529/results.bin b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/a9f31cd11f0b4f2759dee65fdc12b529/results.bin
+new file mode 100644
+index 0000000..0d259dd
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/a9f31cd11f0b4f2759dee65fdc12b529/results.bin
+@@ -0,0 +1 @@
++o/classes
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/a9f31cd11f0b4f2759dee65fdc12b529/transformed/classes/classes_dex/classes.dex b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/a9f31cd11f0b4f2759dee65fdc12b529/transformed/classes/classes_dex/classes.dex
+new file mode 100644
+index 0000000..d4d71e5
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/.transforms/a9f31cd11f0b4f2759dee65fdc12b529/transformed/classes/classes_dex/classes.dex differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/aapt_friendly_merged_manifests/debug/processDebugManifest/aapt/AndroidManifest.xml b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/aapt_friendly_merged_manifests/debug/processDebugManifest/aapt/AndroidManifest.xml
+new file mode 100644
+index 0000000..b81ecea
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/aapt_friendly_merged_manifests/debug/processDebugManifest/aapt/AndroidManifest.xml
+@@ -0,0 +1,15 @@
++
++
++
++
++
++
++
++
++
++
++
++
++
++
+\ No newline at end of file
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/aapt_friendly_merged_manifests/debug/processDebugManifest/aapt/output-metadata.json b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/aapt_friendly_merged_manifests/debug/processDebugManifest/aapt/output-metadata.json
+new file mode 100644
+index 0000000..b3c0af9
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/aapt_friendly_merged_manifests/debug/processDebugManifest/aapt/output-metadata.json
+@@ -0,0 +1,18 @@
++{
++ "version": 3,
++ "artifactType": {
++ "type": "AAPT_FRIENDLY_MERGED_MANIFESTS",
++ "kind": "Directory"
++ },
++ "applicationId": "com.getcapacitor.community.speechrecognition.speechrecognition",
++ "variantName": "debug",
++ "elements": [
++ {
++ "type": "SINGLE",
++ "filters": [],
++ "attributes": [],
++ "outputFile": "AndroidManifest.xml"
++ }
++ ],
++ "elementType": "File"
++}
+\ No newline at end of file
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/aar_metadata/debug/writeDebugAarMetadata/aar-metadata.properties b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/aar_metadata/debug/writeDebugAarMetadata/aar-metadata.properties
+new file mode 100644
+index 0000000..1211b1e
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/aar_metadata/debug/writeDebugAarMetadata/aar-metadata.properties
+@@ -0,0 +1,6 @@
++aarFormatVersion=1.0
++aarMetadataVersion=1.0
++minCompileSdk=1
++minCompileSdkExtension=0
++minAndroidGradlePluginVersion=1.0.0
++coreLibraryDesugaringEnabled=false
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/annotation_processor_list/debug/javaPreCompileDebug/annotationProcessors.json b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/annotation_processor_list/debug/javaPreCompileDebug/annotationProcessors.json
+new file mode 100644
+index 0000000..9e26dfe
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/annotation_processor_list/debug/javaPreCompileDebug/annotationProcessors.json
+@@ -0,0 +1 @@
++{}
+\ No newline at end of file
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/compile_library_classes_jar/debug/bundleLibCompileToJarDebug/classes.jar b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/compile_library_classes_jar/debug/bundleLibCompileToJarDebug/classes.jar
+new file mode 100644
+index 0000000..ff6ad96
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/compile_library_classes_jar/debug/bundleLibCompileToJarDebug/classes.jar differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/compile_r_class_jar/debug/generateDebugRFile/R.jar b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/compile_r_class_jar/debug/generateDebugRFile/R.jar
+new file mode 100644
+index 0000000..eb6f866
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/compile_r_class_jar/debug/generateDebugRFile/R.jar differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/compile_symbol_list/debug/generateDebugRFile/R.txt b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/compile_symbol_list/debug/generateDebugRFile/R.txt
+new file mode 100644
+index 0000000..f7ac549
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/compile_symbol_list/debug/generateDebugRFile/R.txt
+@@ -0,0 +1,3 @@
++int id webview 0x0
++int layout bridge_layout_main 0x0
++int string my_string 0x0
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/compiled_local_resources/debug/compileDebugLibraryResources/out/layout_bridge_layout_main.xml.flat b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/compiled_local_resources/debug/compileDebugLibraryResources/out/layout_bridge_layout_main.xml.flat
+new file mode 100644
+index 0000000..183eadd
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/compiled_local_resources/debug/compileDebugLibraryResources/out/layout_bridge_layout_main.xml.flat differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/debug/packageDebugResources/compile-file-map.properties b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/debug/packageDebugResources/compile-file-map.properties
+new file mode 100644
+index 0000000..4e5a864
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/debug/packageDebugResources/compile-file-map.properties
+@@ -0,0 +1,2 @@
++#Mon Jul 20 15:42:26 EDT 2026
++com.getcapacitor.community.speechrecognition.speechrecognition.capacitor-community-speech-recognition-main-6\:/layout/bridge_layout_main.xml=/Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/packaged_res/debug/packageDebugResources/layout/bridge_layout_main.xml
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/debug/packageDebugResources/merged.dir/values/values.xml b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/debug/packageDebugResources/merged.dir/values/values.xml
+new file mode 100644
+index 0000000..e95bf66
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/debug/packageDebugResources/merged.dir/values/values.xml
+@@ -0,0 +1,4 @@
++
++
++ Just a simple string
++
+\ No newline at end of file
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/debug/packageDebugResources/merger.xml b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/debug/packageDebugResources/merger.xml
+new file mode 100644
+index 0000000..3058471
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/debug/packageDebugResources/merger.xml
+@@ -0,0 +1,2 @@
++
++Just a simple string
+\ No newline at end of file
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/mergeDebugAssets/merger.xml b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/mergeDebugAssets/merger.xml
+new file mode 100644
+index 0000000..8c51637
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/mergeDebugAssets/merger.xml
+@@ -0,0 +1,2 @@
++
++
+\ No newline at end of file
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/mergeDebugJniLibFolders/merger.xml b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/mergeDebugJniLibFolders/merger.xml
+new file mode 100644
+index 0000000..472210d
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/mergeDebugJniLibFolders/merger.xml
+@@ -0,0 +1,2 @@
++
++
+\ No newline at end of file
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/mergeDebugShaders/merger.xml b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/mergeDebugShaders/merger.xml
+new file mode 100644
+index 0000000..6bebd8c
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/incremental/mergeDebugShaders/merger.xml
+@@ -0,0 +1,2 @@
++
++
+\ No newline at end of file
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/javac/debug/compileDebugJavaWithJavac/classes/com/getcapacitor/community/speechrecognition/Constants.class b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/javac/debug/compileDebugJavaWithJavac/classes/com/getcapacitor/community/speechrecognition/Constants.class
+new file mode 100644
+index 0000000..3094bd9
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/javac/debug/compileDebugJavaWithJavac/classes/com/getcapacitor/community/speechrecognition/Constants.class differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/javac/debug/compileDebugJavaWithJavac/classes/com/getcapacitor/community/speechrecognition/Receiver.class b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/javac/debug/compileDebugJavaWithJavac/classes/com/getcapacitor/community/speechrecognition/Receiver.class
+new file mode 100644
+index 0000000..af9be6c
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/javac/debug/compileDebugJavaWithJavac/classes/com/getcapacitor/community/speechrecognition/Receiver.class differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/javac/debug/compileDebugJavaWithJavac/classes/com/getcapacitor/community/speechrecognition/SpeechRecognition$SpeechRecognitionListener.class b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/javac/debug/compileDebugJavaWithJavac/classes/com/getcapacitor/community/speechrecognition/SpeechRecognition$SpeechRecognitionListener.class
+new file mode 100644
+index 0000000..a056329
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/javac/debug/compileDebugJavaWithJavac/classes/com/getcapacitor/community/speechrecognition/SpeechRecognition$SpeechRecognitionListener.class differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/javac/debug/compileDebugJavaWithJavac/classes/com/getcapacitor/community/speechrecognition/SpeechRecognition.class b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/javac/debug/compileDebugJavaWithJavac/classes/com/getcapacitor/community/speechrecognition/SpeechRecognition.class
+new file mode 100644
+index 0000000..57a50fe
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/javac/debug/compileDebugJavaWithJavac/classes/com/getcapacitor/community/speechrecognition/SpeechRecognition.class differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/local_only_symbol_list/debug/parseDebugLocalResources/R-def.txt b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/local_only_symbol_list/debug/parseDebugLocalResources/R-def.txt
+new file mode 100644
+index 0000000..80ec5cb
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/local_only_symbol_list/debug/parseDebugLocalResources/R-def.txt
+@@ -0,0 +1,5 @@
++R_DEF: Internal format may change without notice
++local
++id webview
++layout bridge_layout_main
++string my_string
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/manifest_merge_blame_file/debug/processDebugManifest/manifest-merger-blame-debug-report.txt b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/manifest_merge_blame_file/debug/processDebugManifest/manifest-merger-blame-debug-report.txt
+new file mode 100644
+index 0000000..6010dd3
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/manifest_merge_blame_file/debug/processDebugManifest/manifest-merger-blame-debug-report.txt
+@@ -0,0 +1,21 @@
++1
++2
++4
++5
++6
++7
++7-->/Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:4:5-71
++7-->/Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:4:22-68
++8
++9
++9-->/Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:5:5-9:15
++10
++10-->/Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:6:7-8:16
++11
++11-->/Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:7:9-68
++11-->/Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:7:17-65
++12
++13
++14
++15
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/merged_manifest/debug/processDebugManifest/AndroidManifest.xml b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/merged_manifest/debug/processDebugManifest/AndroidManifest.xml
+new file mode 100644
+index 0000000..b81ecea
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/merged_manifest/debug/processDebugManifest/AndroidManifest.xml
+@@ -0,0 +1,15 @@
++
++
++
++
++
++
++
++
++
++
++
++
++
++
+\ No newline at end of file
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/navigation_json/debug/extractDeepLinksDebug/navigation.json b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/navigation_json/debug/extractDeepLinksDebug/navigation.json
+new file mode 100644
+index 0000000..0637a08
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/navigation_json/debug/extractDeepLinksDebug/navigation.json
+@@ -0,0 +1 @@
++[]
+\ No newline at end of file
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/nested_resources_validation_report/debug/generateDebugResources/nestedResourcesValidationReport.txt b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/nested_resources_validation_report/debug/generateDebugResources/nestedResourcesValidationReport.txt
+new file mode 100644
+index 0000000..08f4ebe
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/nested_resources_validation_report/debug/generateDebugResources/nestedResourcesValidationReport.txt
+@@ -0,0 +1 @@
++0 Warning/Error
+\ No newline at end of file
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/packaged_res/debug/packageDebugResources/layout/bridge_layout_main.xml b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/packaged_res/debug/packageDebugResources/layout/bridge_layout_main.xml
+new file mode 100644
+index 0000000..56fec15
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/packaged_res/debug/packageDebugResources/layout/bridge_layout_main.xml
+@@ -0,0 +1,15 @@
++
++
++
++
++
++
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/packaged_res/debug/packageDebugResources/values/values.xml b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/packaged_res/debug/packageDebugResources/values/values.xml
+new file mode 100644
+index 0000000..e95bf66
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/packaged_res/debug/packageDebugResources/values/values.xml
+@@ -0,0 +1,4 @@
++
++
++ Just a simple string
++
+\ No newline at end of file
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_dir/debug/bundleLibRuntimeToDirDebug/com/getcapacitor/community/speechrecognition/Constants.class b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_dir/debug/bundleLibRuntimeToDirDebug/com/getcapacitor/community/speechrecognition/Constants.class
+new file mode 100644
+index 0000000..3094bd9
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_dir/debug/bundleLibRuntimeToDirDebug/com/getcapacitor/community/speechrecognition/Constants.class differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_dir/debug/bundleLibRuntimeToDirDebug/com/getcapacitor/community/speechrecognition/Receiver.class b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_dir/debug/bundleLibRuntimeToDirDebug/com/getcapacitor/community/speechrecognition/Receiver.class
+new file mode 100644
+index 0000000..af9be6c
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_dir/debug/bundleLibRuntimeToDirDebug/com/getcapacitor/community/speechrecognition/Receiver.class differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_dir/debug/bundleLibRuntimeToDirDebug/com/getcapacitor/community/speechrecognition/SpeechRecognition$SpeechRecognitionListener.class b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_dir/debug/bundleLibRuntimeToDirDebug/com/getcapacitor/community/speechrecognition/SpeechRecognition$SpeechRecognitionListener.class
+new file mode 100644
+index 0000000..a056329
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_dir/debug/bundleLibRuntimeToDirDebug/com/getcapacitor/community/speechrecognition/SpeechRecognition$SpeechRecognitionListener.class differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_dir/debug/bundleLibRuntimeToDirDebug/com/getcapacitor/community/speechrecognition/SpeechRecognition.class b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_dir/debug/bundleLibRuntimeToDirDebug/com/getcapacitor/community/speechrecognition/SpeechRecognition.class
+new file mode 100644
+index 0000000..57a50fe
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_dir/debug/bundleLibRuntimeToDirDebug/com/getcapacitor/community/speechrecognition/SpeechRecognition.class differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_jar/debug/bundleLibRuntimeToJarDebug/classes.jar b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_jar/debug/bundleLibRuntimeToJarDebug/classes.jar
+new file mode 100644
+index 0000000..c701569
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/runtime_library_classes_jar/debug/bundleLibRuntimeToJarDebug/classes.jar differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/symbol_list_with_package_name/debug/generateDebugRFile/package-aware-r.txt b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/symbol_list_with_package_name/debug/generateDebugRFile/package-aware-r.txt
+new file mode 100644
+index 0000000..f0ee474
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/intermediates/symbol_list_with_package_name/debug/generateDebugRFile/package-aware-r.txt
+@@ -0,0 +1,4 @@
++com.getcapacitor.community.speechrecognition.speechrecognition
++id webview
++layout bridge_layout_main
++string my_string
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/outputs/logs/manifest-merger-debug-report.txt b/node_modules/@capacitor-community/speech-recognition/android/build/outputs/logs/manifest-merger-debug-report.txt
+new file mode 100644
+index 0000000..102bc47
+--- /dev/null
++++ b/node_modules/@capacitor-community/speech-recognition/android/build/outputs/logs/manifest-merger-debug-report.txt
+@@ -0,0 +1,28 @@
++-- Merging decision tree log ---
++manifest
++ADDED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:2:3-10:14
++INJECTED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:2:3-10:14
++ package
++ INJECTED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml
++ xmlns:android
++ ADDED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:2:13-71
++uses-permission#android.permission.RECORD_AUDIO
++ADDED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:4:5-71
++ android:name
++ ADDED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:4:22-68
++queries
++ADDED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:5:5-9:15
++intent#action:name:android.speech.RecognitionService
++ADDED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:6:7-8:16
++action#android.speech.RecognitionService
++ADDED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:7:9-68
++ android:name
++ ADDED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml:7:17-65
++uses-sdk
++INJECTED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml reason: use-sdk injection requested
++INJECTED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml
++INJECTED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml
++ android:targetSdkVersion
++ INJECTED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml
++ android:minSdkVersion
++ INJECTED from /Users/mohamad-macbook-air/workspace/temp/dt-frontend-worktrees/support-audio-input/node_modules/@capacitor-community/speech-recognition/android/src/main/AndroidManifest.xml
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/tmp/compileDebugJavaWithJavac/compileTransaction/stash-dir/SpeechRecognition$SpeechRecognitionListener.class.uniqueId1 b/node_modules/@capacitor-community/speech-recognition/android/build/tmp/compileDebugJavaWithJavac/compileTransaction/stash-dir/SpeechRecognition$SpeechRecognitionListener.class.uniqueId1
+new file mode 100644
+index 0000000..dac1e83
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/tmp/compileDebugJavaWithJavac/compileTransaction/stash-dir/SpeechRecognition$SpeechRecognitionListener.class.uniqueId1 differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/tmp/compileDebugJavaWithJavac/compileTransaction/stash-dir/SpeechRecognition.class.uniqueId0 b/node_modules/@capacitor-community/speech-recognition/android/build/tmp/compileDebugJavaWithJavac/compileTransaction/stash-dir/SpeechRecognition.class.uniqueId0
+new file mode 100644
+index 0000000..836f0ad
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/tmp/compileDebugJavaWithJavac/compileTransaction/stash-dir/SpeechRecognition.class.uniqueId0 differ
+diff --git a/node_modules/@capacitor-community/speech-recognition/android/build/tmp/compileDebugJavaWithJavac/previous-compilation-data.bin b/node_modules/@capacitor-community/speech-recognition/android/build/tmp/compileDebugJavaWithJavac/previous-compilation-data.bin
+new file mode 100644
+index 0000000..c57ddd3
+Binary files /dev/null and b/node_modules/@capacitor-community/speech-recognition/android/build/tmp/compileDebugJavaWithJavac/previous-compilation-data.bin differ
diff --git a/node_modules/@capacitor-community/speech-recognition/android/src/main/java/com/getcapacitor/community/speechrecognition/SpeechRecognition.java b/node_modules/@capacitor-community/speech-recognition/android/src/main/java/com/getcapacitor/community/speechrecognition/SpeechRecognition.java
-index a99ea69..2bea32e 100644
+index a99ea69..5cabbd7 100644
--- a/node_modules/@capacitor-community/speech-recognition/android/src/main/java/com/getcapacitor/community/speechrecognition/SpeechRecognition.java
+++ b/node_modules/@capacitor-community/speech-recognition/android/src/main/java/com/getcapacitor/community/speechrecognition/SpeechRecognition.java
-@@ -168,6 +168,7 @@ public class SpeechRecognition extends Plugin implements Constants {
+@@ -81,13 +81,18 @@ public class SpeechRecognition extends Plugin implements Constants {
+ String prompt = call.getString("prompt", null);
+ boolean partialResults = call.getBoolean("partialResults", false);
+ boolean popup = call.getBoolean("popup", false);
+- beginListening(language, maxResults, prompt, partialResults, popup, call);
++ JSArray contextualStrings = call.getArray("contextualStrings", new JSArray());
++ beginListening(language, maxResults, prompt, partialResults, popup, contextualStrings, call);
+ }
+
+ @PluginMethod
+ public void stop(final PluginCall call) {
+ try {
+ stopListening();
++ // This never resolved the call on success, only rejected on
++ // exception — any caller doing `await stop()` (e.g. a JS-side
++ // restart loop) would hang forever waiting on this promise.
++ call.resolve();
+ } catch (Exception ex) {
+ call.reject(ex.getLocalizedMessage());
+ }
+@@ -157,6 +162,7 @@ public class SpeechRecognition extends Plugin implements Constants {
+ String prompt,
+ final boolean partialResults,
+ boolean showPopup,
++ JSArray contextualStrings,
+ PluginCall call
+ ) {
+ Logger.info(getLogTag(), "Beginning to listen for audible speech");
+@@ -168,6 +174,19 @@ public class SpeechRecognition extends Plugin implements Constants {
intent.putExtra(RecognizerIntent.EXTRA_CALLING_PACKAGE, bridge.getActivity().getPackageName());
intent.putExtra(RecognizerIntent.EXTRA_PARTIAL_RESULTS, partialResults);
intent.putExtra("android.speech.extra.DICTATION_MODE", partialResults);
+ intent.putExtra(RecognizerIntent.EXTRA_PREFER_OFFLINE, true);
++ if (Build.VERSION.SDK_INT >= Build.VERSION_CODES.TIRAMISU && contextualStrings != null && contextualStrings.length() > 0) {
++ ArrayList biasingStrings = new ArrayList<>();
++ for (int i = 0; i < contextualStrings.length(); i++) {
++ String value = contextualStrings.optString(i, null);
++ if (value != null && !value.trim().isEmpty()) {
++ biasingStrings.add(value);
++ }
++ }
++ if (!biasingStrings.isEmpty()) {
++ intent.putStringArrayListExtra(RecognizerIntent.EXTRA_BIASING_STRINGS, biasingStrings);
++ }
++ }
if (prompt != null) {
intent.putExtra(RecognizerIntent.EXTRA_PROMPT, prompt);
+@@ -283,6 +302,25 @@ public class SpeechRecognition extends Plugin implements Constants {
+ SpeechRecognition.this.stopListening();
+ String errorMssg = getErrorText(error);
+
++ // Unlike onEndOfSpeech, this previously never told the JS side the
++ // session ended — silence commonly ends a session via ERROR_SPEECH_TIMEOUT
++ // / ERROR_NO_MATCH rather than onEndOfSpeech, so without this the caller's
++ // restart loop never fires and listening state gets stuck forever.
++ bridge
++ .getWebView()
++ .post(() -> {
++ try {
++ SpeechRecognition.this.lock.lock();
++ SpeechRecognition.this.listening(false);
++
++ JSObject ret = new JSObject();
++ ret.put("status", "stopped");
++ SpeechRecognition.this.notifyListeners(LISTENING_EVENT, ret);
++ } finally {
++ SpeechRecognition.this.lock.unlock();
++ }
++ });
++
+ if (this.call != null) {
+ call.reject(errorMssg);
+ }
diff --git a/node_modules/@capacitor-community/speech-recognition/ios/Plugin/Plugin.swift b/node_modules/@capacitor-community/speech-recognition/ios/Plugin/Plugin.swift
-index 5d1b35b..810277d 100644
+index 5d1b35b..899c7a6 100644
--- a/node_modules/@capacitor-community/speech-recognition/ios/Plugin/Plugin.swift
+++ b/node_modules/@capacitor-community/speech-recognition/ios/Plugin/Plugin.swift
-@@ -79,6 +79,9 @@ public class SpeechRecognition: CAPPlugin {
+@@ -54,6 +54,7 @@ public class SpeechRecognition: CAPPlugin {
+ let language: String = call.getString("language") ?? "en-US"
+ let maxResults: Int = call.getInt("maxResults") ?? self.defaultMatches
+ let partialResults: Bool = call.getBool("partialResults") ?? false
++ let contextualStrings: [String] = call.getArray("contextualStrings", String.self) ?? []
+
+ if self.recognitionTask != nil {
+ self.recognitionTask?.cancel()
+@@ -79,6 +80,12 @@ public class SpeechRecognition: CAPPlugin {
self.recognitionRequest = SFSpeechAudioBufferRecognitionRequest()
self.recognitionRequest?.shouldReportPartialResults = partialResults
++ if !contextualStrings.isEmpty {
++ self.recognitionRequest?.contextualStrings = contextualStrings
++ }
+ if #available(iOS 13, *), self.speechRecognizer?.supportsOnDeviceRecognition == true {
+ self.recognitionRequest?.requiresOnDeviceRecognition = true
+ }
diff --git a/src/service/VoiceInputService.js b/src/service/VoiceInputService.js
index f752b17..2383c2f 100644
--- a/src/service/VoiceInputService.js
+++ b/src/service/VoiceInputService.js
@@ -19,12 +19,63 @@ import { Capacitor } from '@capacitor/core'
const SILENCE_COMMIT_MS = 2200
const RESTART_DELAY_MS = 250
+// Defense-in-depth: some Android OEM recognizers can die (e.g. after a speech
+// timeout error) without emitting any event at all, which would otherwise
+// leave the mic looking "still listening" forever with nothing restarting it.
+// If no native event of any kind has arrived in this long, assume the session
+// is dead and force a restart even with no pending partial text.
+const HEARTBEAT_TIMEOUT_MS = 6000
+// Native recognizers accept a limited vocabulary hint list; keep it small so
+// the common names/labels actually get weighted rather than diluted.
+const MAX_CONTEXTUAL_STRINGS = 100
+// A known Android build of the plugin never resolved stop()'s promise on
+// success — any `await`ed native call here hanging silently would otherwise
+// wedge the whole restart loop (and the mic would look stuck "listening"
+// forever). Cap every native await so a broken plugin promise can't do that.
+const NATIVE_CALL_TIMEOUT_MS = 1500
+// Android forwards both the interim AND the true final transcript (from
+// onResults) through the same partialResults event, with the final one
+// typically landing a couple hundred ms after the session is reported
+// "stopped" — and no flag distinguishes them. The final result is usually
+// MORE accurate than the last interim (it benefits from the full-utterance
+// language model rather than a streaming guess), which matters most exactly
+// on names — the same uncertainty behind "Moutaz" being misheard as
+// "Models". So rather than committing immediately and discarding the late
+// final as noise, wait this long after a session ends for it to arrive and
+// supersede the interim before actually committing.
+const FINAL_RESULT_GRACE_MS = 450
+// Safety net for a final result arriving even later than the grace window
+// (or a duplicate slipping through some other path) — still not committed as
+// a second task if it looks like the same utterance.
+const DUPLICATE_GUARD_MS = 3000
+// Below this fraction of shared words, two transcripts are treated as
+// different utterances rather than a re-delivery of the same one.
+const DUPLICATE_WORD_OVERLAP = 0.6
-const START_OPTIONS = {
- language: 'en-US',
- maxResults: 1,
- partialResults: true,
- popup: false,
+const withTimeout = (promise, ms) =>
+ Promise.race([promise, new Promise(resolve => setTimeout(resolve, ms))])
+
+const normalizeForDupeCheck = text =>
+ text
+ .trim()
+ .toLowerCase()
+ .replace(/[.,!?]/g, '')
+
+// Word-overlap rather than exact/prefix match: names are exactly the words
+// ASR is least confident about (the same uncertainty behind "Moutaz" heard as
+// "Models"), so the final transcript commonly comes back with a different
+// word around a name than the interim partial that already got committed.
+// Requiring every character to match would miss that; requiring most of the
+// same words to match still catches it as the same utterance.
+const wordOverlapRatio = (a, b) => {
+ const wordsA = new Set(a.split(/\s+/).filter(Boolean))
+ const wordsB = new Set(b.split(/\s+/).filter(Boolean))
+ if (wordsA.size === 0 || wordsB.size === 0) return 0
+ let shared = 0
+ for (const word of wordsA) {
+ if (wordsB.has(word)) shared++
+ }
+ return shared / Math.max(wordsA.size, wordsB.size)
}
class VoiceInputService {
@@ -33,9 +84,26 @@ class VoiceInputService {
this._callbacks = null
this._partial = ''
this._lastSpeechAt = 0
+ this._lastNativeEventAt = 0
this._silenceTimer = null
this._restarting = false
+ this._restartPromise = null
this._webRecognition = null
+ this._contextualStrings = []
+ this._lastCommittedText = ''
+ this._lastCommittedAt = 0
+ this._awaitingFinal = false
+ this._resolveAwaitingFinal = null
+ }
+
+ _startOptions() {
+ return {
+ language: 'en-US',
+ maxResults: 1,
+ partialResults: true,
+ popup: false,
+ contextualStrings: this._contextualStrings,
+ }
}
get isNative() {
@@ -78,12 +146,21 @@ class VoiceInputService {
}
}
- async start(callbacks) {
+ // vocabulary: circle member names + label names, used to bias native
+ // recognition toward the words that matter most for task capture (iOS
+ // contextualStrings / Android 13+ EXTRA_BIASING_STRINGS). Without this, an
+ // unfamiliar name like "Moutaz" can get auto-corrected to a dictionary word.
+ async start(callbacks, vocabulary = []) {
if (this._active) return
this._callbacks = callbacks
this._active = true
this._partial = ''
this._lastSpeechAt = Date.now()
+ this._lastNativeEventAt = Date.now()
+ this._contextualStrings = [...new Set(vocabulary.filter(Boolean))].slice(
+ 0,
+ MAX_CONTEXTUAL_STRINGS,
+ )
if (this.isNative) {
await this._startNative()
@@ -103,16 +180,35 @@ class VoiceInputService {
clearInterval(this._silenceTimer)
this._silenceTimer = null
}
+ // Let any in-flight restart (triggered by a native "stopped" event or the
+ // heartbeat) finish tearing down first, so it doesn't resurrect a session
+ // right after the user asked to stop.
+ if (this._restartPromise) {
+ await this._restartPromise
+ }
if (this.isNative) {
+ let SpeechRecognition
try {
- const { SpeechRecognition } = await import(
+ ;({ SpeechRecognition } = await import(
'@capacitor-community/speech-recognition'
- )
- await SpeechRecognition.stop()
- await SpeechRecognition.removeAllListeners()
+ ))
+ await withTimeout(SpeechRecognition.stop(), NATIVE_CALL_TIMEOUT_MS)
} catch {
// recognizer may already be stopped
}
+ // Wait for a possible late-arriving final result while listeners are
+ // still attached — removing them first would mean it's never heard.
+ // Always runs, even if the native stop() call above failed, so we
+ // never skip committing whatever was captured.
+ await this._finalizeSegment()
+ try {
+ await withTimeout(
+ SpeechRecognition?.removeAllListeners(),
+ NATIVE_CALL_TIMEOUT_MS,
+ )
+ } catch {
+ // non-fatal
+ }
} else if (this._webRecognition) {
const rec = this._webRecognition
this._webRecognition = null
@@ -121,27 +217,67 @@ class VoiceInputService {
} catch {
// already stopped
}
+ this._commitPartial()
+ } else {
+ this._commitPartial()
+ }
+ this._callbacks?.onStateChange?.(false)
+ }
+
+ // Called when a session has ended (or is being torn down for restart) and
+ // whatever's in `_partial` is ready to become a task — except Android's
+ // true final transcript, if there is one, is usually still in flight and
+ // hasn't replaced it yet. Give it a brief window to land first.
+ async _finalizeSegment() {
+ if (this._partial.trim() && this.isNative) {
+ this._awaitingFinal = true
+ await new Promise(resolve => {
+ this._resolveAwaitingFinal = resolve
+ setTimeout(resolve, FINAL_RESULT_GRACE_MS)
+ })
+ this._awaitingFinal = false
+ this._resolveAwaitingFinal = null
}
this._commitPartial()
- this._callbacks?.onStateChange?.(false)
}
_commitPartial() {
const text = this._partial.trim()
this._partial = ''
this._callbacks?.onPartial?.('')
- if (text) this._callbacks?.onSegment?.(text)
+ if (text) {
+ this._lastCommittedText = normalizeForDupeCheck(text)
+ this._lastCommittedAt = Date.now()
+ this._callbacks?.onSegment?.(text)
+ }
+ }
+
+ // True if `text` looks like a re-delivery of what we just committed (exact
+ // match, or one is a prefix of the other — covers the final result being a
+ // trimmed/extended variant of the last partial we already committed on).
+ _isEchoOfLastCommit(text) {
+ if (!this._lastCommittedText) return false
+ if (Date.now() - this._lastCommittedAt > DUPLICATE_GUARD_MS) return false
+ const a = normalizeForDupeCheck(text)
+ const b = this._lastCommittedText
+ if (a === b || a.startsWith(b) || b.startsWith(a)) return true
+ return wordOverlapRatio(a, b) >= DUPLICATE_WORD_OVERLAP
}
_checkSilence() {
if (!this._active || this._restarting) return
- if (
- this._partial.trim() &&
- Date.now() - this._lastSpeechAt > SILENCE_COMMIT_MS
- ) {
+ const now = Date.now()
+ if (this._partial.trim() && now - this._lastSpeechAt > SILENCE_COMMIT_MS) {
// A pause means the utterance (= task) is complete: cycle the recognizer
// so the buffer commits and a fresh session begins.
this._restartNative()
+ return
+ }
+ if (now - this._lastNativeEventAt > HEARTBEAT_TIMEOUT_MS) {
+ // No native event of any kind for too long — the recognizer likely
+ // died silently (seen on some Android devices/OEMs). Force a restart
+ // so the mic doesn't sit "listening" forever with nothing happening.
+ this._restartNative()
}
}
@@ -152,14 +288,33 @@ class VoiceInputService {
await SpeechRecognition.removeAllListeners()
await SpeechRecognition.addListener('partialResults', ({ matches }) => {
+ this._lastNativeEventAt = Date.now()
const text = matches?.[0] || ''
if (!text) return
+
+ if (this._awaitingFinal) {
+ // This is the true final result we were waiting for — it's usually
+ // more accurate than the interim it's replacing, so use it and stop
+ // waiting out the rest of the grace window.
+ this._partial = text
+ this._callbacks?.onPartial?.(text)
+ this._resolveAwaitingFinal?.()
+ return
+ }
+
+ if (this._isEchoOfLastCommit(text)) {
+ // Arrived even later than the grace window (or some other stray
+ // delivery) — still don't let it look like a fresh spoken segment.
+ return
+ }
+
this._partial = text
this._lastSpeechAt = Date.now()
this._callbacks?.onPartial?.(text)
})
await SpeechRecognition.addListener('listeningState', ({ status }) => {
+ this._lastNativeEventAt = Date.now()
if (status === 'stopped' && this._active && !this._restarting) {
// OS ended the session on its own (silence on Android, session limit
// on iOS) — commit and start over.
@@ -169,34 +324,39 @@ class VoiceInputService {
// With partialResults the transcript arrives via listeners; the promise's
// resolution/rejection timing differs per platform, so don't rely on it.
- SpeechRecognition.start(START_OPTIONS).catch(() => {
+ SpeechRecognition.start(this._startOptions()).catch(() => {
if (this._active && !this._restarting) {
this._restartNative()
}
})
}
- async _restartNative() {
- if (this._restarting) return
+ _restartNative() {
+ if (this._restarting) return this._restartPromise
this._restarting = true
- try {
- const { SpeechRecognition } = await import(
- '@capacitor-community/speech-recognition'
- )
- try {
- await SpeechRecognition.stop()
- } catch {
- // already stopped
- }
- this._commitPartial()
- // Let the OS recognizer tear down before starting a new session
- await new Promise(r => setTimeout(r, RESTART_DELAY_MS))
- if (this._active) {
- SpeechRecognition.start(START_OPTIONS).catch(() => {})
- this._lastSpeechAt = Date.now()
- }
- } finally {
+ this._restartPromise = this._doRestartNative().finally(() => {
this._restarting = false
+ this._restartPromise = null
+ })
+ return this._restartPromise
+ }
+
+ async _doRestartNative() {
+ const { SpeechRecognition } = await import(
+ '@capacitor-community/speech-recognition'
+ )
+ try {
+ await withTimeout(SpeechRecognition.stop(), NATIVE_CALL_TIMEOUT_MS)
+ } catch {
+ // already stopped
+ }
+ await this._finalizeSegment()
+ // Let the OS recognizer tear down before starting a new session
+ await new Promise(r => setTimeout(r, RESTART_DELAY_MS))
+ if (this._active) {
+ SpeechRecognition.start(this._startOptions()).catch(() => {})
+ this._lastSpeechAt = Date.now()
+ this._lastNativeEventAt = Date.now()
}
}
diff --git a/src/views/components/AddTaskModal.jsx b/src/views/components/AddTaskModal.jsx
index 63ec27f..ef9a402 100644
--- a/src/views/components/AddTaskModal.jsx
+++ b/src/views/components/AddTaskModal.jsx
@@ -1,7 +1,6 @@
import { Add } from '@mui/icons-material'
import { Box, Button, Typography } from '@mui/joy'
import { useMediaQuery } from '@mui/material'
-import { useQueryClient } from '@tanstack/react-query'
import * as chrono from 'chrono-node'
import moment from 'moment'
import { useCallback, useEffect, useRef, useState } from 'react'
@@ -26,7 +25,6 @@ import KeyboardShortcutHint from '../../components/common/KeyboardShortcutHint'
import { useDocumentScanner } from '../../hooks/useDocumentScanner'
import { localAIService } from '../../service/LocalAIService'
import { voiceInputService } from '../../service/VoiceInputService'
-import { CreateLabel } from '../../utils/Fetcher'
import { TASK_COLOR } from '../../utils/Colors'
import AdvancedOptionsSection, {
AdvancedOptionsTrigger,
@@ -71,7 +69,6 @@ const TaskInput = ({ onChoreUpdate, isModalOpen, onClose }) => {
useCircleMembers()
const { isLoading: isProjectsLoading } = useProjects()
const createChoreMutation = useCreateChore()
- const queryClient = useQueryClient()
const { data: userProfile } = useUserProfile()
@@ -623,73 +620,26 @@ const TaskInput = ({ onChoreUpdate, isModalOpen, onClose }) => {
}
}
- // Creates labels that were spoken but don't exist yet. Returns a Map of
- // lowercase name → label id covering both created and already-existing ones.
- const createMissingLabels = async newLabels => {
- const resolved = new Map(
- (userLabels || []).map(l => [l.name.toLowerCase(), l.id]),
- )
- let createdAny = false
- for (const label of newLabels) {
- const key = label.name.toLowerCase()
- if (resolved.has(key)) continue
- try {
- const resp = await CreateLabel({
- name: label.name,
- color: label.color || '#3b82f6',
- })
- const data = await resp.json()
- const created = data?.res ?? data
- if (created?.id) {
- resolved.set(key, created.id)
- createdAny = true
- }
- } catch (error) {
- console.error('Error creating label:', error)
- }
- }
- if (createdAny) {
- queryClient.invalidateQueries({ queryKey: ['labels'] })
- }
- return resolved
- }
-
// Single voice-captured task: land it in the smart input so the user
- // reviews it with the normal pickers before creating. Setting taskText
- // (rather than calling processText directly) lets the reparse effect run
- // exactly once, consuming any picker overrides from the panel.
- const handleVoiceSingle = async (text, overrides = {}) => {
+ // reviews it with the normal pickers before creating.
+ const handleVoiceSingle = (text, overrides = {}) => {
setShowVoice(false)
if (Object.keys(overrides).length > 0) {
pendingVoiceOverridesRef.current = overrides
}
setTaskText(text)
- const labels = parseLabels(text, userLabels || [])
- if (labels.newLabels?.length) {
- // Once the labels query refetches, the reparse links them automatically
- await createMissingLabels(labels.newLabels)
- }
}
// Multiple voice-captured tasks: they were reviewed as cards in the panel,
// so create them all directly.
const handleVoiceCreateMany = async parsedTasks => {
const notificationTemplates = getDefaultNotification()
- const allNewLabels = parsedTasks.flatMap(t => t.newLabels || [])
- const labelIdsByName =
- allNewLabels.length > 0 ? await createMissingLabels(allNewLabels) : null
for (const parsed of parsedTasks) {
- const extraLabelIds = (parsed.newLabels || [])
- .map(nl => labelIdsByName?.get(nl.name.toLowerCase()))
- .filter(id => id != null && !parsed.labelIds.includes(id))
- const chore = buildChorePayload(
- { ...parsed, labelIds: [...parsed.labelIds, ...extraLabelIds] },
- {
- userProfile,
- projectId,
- notificationTemplates,
- },
- )
+ const chore = buildChorePayload(parsed, {
+ userProfile,
+ projectId,
+ notificationTemplates,
+ })
try {
const result = await createChoreMutation.mutateAsync(chore)
if (result?._pendingCreate) {
diff --git a/src/views/components/VoiceToTask/VoicePanel.jsx b/src/views/components/VoiceToTask/VoicePanel.jsx
index b33391a..97601d2 100644
--- a/src/views/components/VoiceToTask/VoicePanel.jsx
+++ b/src/views/components/VoiceToTask/VoicePanel.jsx
@@ -144,14 +144,6 @@ const buildChips = (effective, frequencyLabel, { members, currentUserId }) => {
label: name,
})
})
- effective.newLabels.forEach(label => {
- chips.push({
- key: `new-label-${label.name}`,
- color: 'warning',
- icon: ,
- label: `${label.name} · new`,
- })
- })
if (effective.isAnyone) {
chips.push({
key: 'assignee',
@@ -436,9 +428,10 @@ const VoicePanel = ({
patchSegment,
reset,
isNative,
- } = useVoiceToTask({ members })
+ } = useVoiceToTask({ members, userLabels })
const [creating, setCreating] = useState(false)
const autoStartedRef = useRef(false)
+ const segmentsScrollRef = useRef(null)
const parseCtx = useMemo(
() => ({ userLabels, members, currentUserId: userProfile?.id }),
@@ -459,6 +452,12 @@ const VoicePanel = ({
}
}, [open, startHandsFree])
+ // Keep the newest captured task visible as more are added
+ useEffect(() => {
+ const el = segmentsScrollRef.current
+ if (el) el.scrollTop = el.scrollHeight
+ }, [segments.length])
+
if (!open) return null
const isListening = phase === 'listening'
@@ -552,6 +551,7 @@ const VoicePanel = ({
{/* ── Captured task cards ── */}
{segments.length > 0 && (
label.name),
- newLabels: labels.newLabels || [],
assignees,
isAnyone,
frequency: repeat.result,
diff --git a/src/views/components/VoiceToTask/useVoiceToTask.js b/src/views/components/VoiceToTask/useVoiceToTask.js
index 6739a33..69352f6 100644
--- a/src/views/components/VoiceToTask/useVoiceToTask.js
+++ b/src/views/components/VoiceToTask/useVoiceToTask.js
@@ -32,8 +32,19 @@ const haptic = async kind => {
}
}
+// Vocabulary fed to the native recognizer as a biasing hint so unfamiliar
+// names/labels aren't auto-corrected to a dictionary word (e.g. "Moutaz" →
+// "Models"). Best-effort only — unsupported on iOS <13-without-on-device and
+// Android <13, which is why the normalizer also does fuzzy post-matching.
+const buildVocabulary = (members, userLabels) => [
+ ...members.flatMap(m =>
+ [m.displayName, m.displayName?.split(/\s+/)[0], m.username].filter(Boolean),
+ ),
+ ...userLabels.map(l => l.name).filter(Boolean),
+]
+
// phases: idle | listening | review | denied
-export function useVoiceToTask({ members = [] } = {}) {
+export function useVoiceToTask({ members = [], userLabels = [] } = {}) {
const [phase, setPhase] = useState('idle')
const [isLocked, setIsLocked] = useState(false)
const [partialText, setPartialText] = useState('')
@@ -44,6 +55,10 @@ export function useVoiceToTask({ members = [] } = {}) {
const segmentsRef = useRef([])
const membersRef = useRef(members)
membersRef.current = members
+ const userLabelsRef = useRef(userLabels)
+ userLabelsRef.current = userLabels
+ const vocabularyRef = useRef(buildVocabulary(members, userLabels))
+ vocabularyRef.current = buildVocabulary(members, userLabels)
const phaseRef = useRef(phase)
phaseRef.current = phase
@@ -54,6 +69,12 @@ export function useVoiceToTask({ members = [] } = {}) {
const pressStartedListeningRef = useRef(false)
const lastActivityRef = useRef(0)
const watchdogRef = useRef(null)
+ // While the mic is held (not locked), a mid-hold restart (Android session
+ // limits, forced silence boundary) shouldn't split into a new task — the
+ // user is still holding the button, so it's still one entry. This tracks
+ // which segment is the "active" one for the current hold to merge onto;
+ // reset to null on release so the *next* hold starts a fresh entry.
+ const activeHoldSegmentIdRef = useRef(null)
const applySegments = useCallback(next => {
segmentsRef.current = next
@@ -64,6 +85,7 @@ export function useVoiceToTask({ members = [] } = {}) {
rawText => {
const normalized = normalizeSpokenText(rawText, {
members: membersRef.current,
+ userLabels: userLabelsRef.current,
})
const { text, dropPrevious } = applyScratchThat(normalized)
const pieces = splitSpokenSegments(text)
@@ -71,14 +93,52 @@ export function useVoiceToTask({ members = [] } = {}) {
let base = segmentsRef.current
if (dropPrevious && base.length > 0) {
+ const dropped = base[base.length - 1]
base = base.slice(0, -1)
+ if (activeHoldSegmentIdRef.current === dropped.id) {
+ activeHoldSegmentIdRef.current = null
+ }
haptic('medium')
}
- if (pieces.length > 0) haptic('light')
- applySegments([
- ...base,
- ...pieces.map(piece => ({ id: generateUUID(), text: piece })),
- ])
+ if (pieces.length === 0) {
+ applySegments(base)
+ return
+ }
+ haptic('light')
+
+ if (!lockedRef.current) {
+ // Hold-to-talk: the first piece continues the entry already active
+ // for this hold (if any); only a spoken separator within the same
+ // commit starts additional new entries.
+ const activeIndex = base.findIndex(
+ s => s.id === activeHoldSegmentIdRef.current,
+ )
+ if (activeIndex !== -1) {
+ const merged = [...base]
+ merged[activeIndex] = {
+ ...merged[activeIndex],
+ text: `${merged[activeIndex].text} ${pieces[0]}`.trim(),
+ }
+ const rest = pieces.slice(1).map(piece => ({
+ id: generateUUID(),
+ text: piece,
+ }))
+ if (rest.length > 0) {
+ activeHoldSegmentIdRef.current = rest[rest.length - 1].id
+ }
+ applySegments([...merged, ...rest])
+ return
+ }
+ }
+
+ const newPieces = pieces.map(piece => ({
+ id: generateUUID(),
+ text: piece,
+ }))
+ if (!lockedRef.current) {
+ activeHoldSegmentIdRef.current = newPieces[newPieces.length - 1].id
+ }
+ applySegments([...base, ...newPieces])
},
[applySegments],
)
@@ -91,6 +151,9 @@ export function useVoiceToTask({ members = [] } = {}) {
await voiceInputService.stop()
setPartialText('')
setIsLocked(false)
+ // Release ends the current hold — the next hold-press starts a fresh
+ // entry rather than continuing to merge onto this one
+ activeHoldSegmentIdRef.current = null
// stop() commits any buffered partial synchronously through onSegment,
// so the ref is up to date by the time we read it
setPhase(segmentsRef.current.length > 0 ? 'review' : 'idle')
@@ -104,19 +167,25 @@ export function useVoiceToTask({ members = [] } = {}) {
return false
}
lastActivityRef.current = Date.now()
- await voiceInputService.start({
- onPartial: text => {
- lastActivityRef.current = Date.now()
- setPartialText(
- normalizeSpokenText(text, { members: membersRef.current }),
- )
+ await voiceInputService.start(
+ {
+ onPartial: text => {
+ lastActivityRef.current = Date.now()
+ setPartialText(
+ normalizeSpokenText(text, {
+ members: membersRef.current,
+ userLabels: userLabelsRef.current,
+ }),
+ )
+ },
+ onSegment: commitSegment,
+ onError: () => {
+ setPhase('denied')
+ },
+ onStateChange: () => {},
},
- onSegment: commitSegment,
- onError: () => {
- setPhase('denied')
- },
- onStateChange: () => {},
- })
+ vocabularyRef.current,
+ )
setPhase('listening')
haptic('medium')
@@ -156,7 +225,9 @@ export function useVoiceToTask({ members = [] } = {}) {
const held = Date.now() - pressStartedAtRef.current
if (pressStartedListeningRef.current) {
if (held < TAP_THRESHOLD_MS) {
- // Quick tap → hands-free lock
+ // Quick tap → hands-free lock; from here on, silence boundaries
+ // should start new entries again, not merge onto the last one
+ activeHoldSegmentIdRef.current = null
setIsLocked(true)
} else {
// Hold-to-talk → release ends the capture
@@ -211,6 +282,7 @@ export function useVoiceToTask({ members = [] } = {}) {
applySegments([])
setPartialText('')
setIsLocked(false)
+ activeHoldSegmentIdRef.current = null
setPhase('idle')
}, [applySegments])
diff --git a/src/views/components/VoiceToTask/voiceNormalizer.js b/src/views/components/VoiceToTask/voiceNormalizer.js
index a7cef88..17c3a44 100644
--- a/src/views/components/VoiceToTask/voiceNormalizer.js
+++ b/src/views/components/VoiceToTask/voiceNormalizer.js
@@ -1,7 +1,7 @@
// Deterministic transforms that turn spoken language into the typed syntax
// CustomParsers understands. No LLM — instant, predictable, fully offline.
//
-// "label groceries" → "#groceries"
+// "label groceries" → "#groceries" (only when it matches an existing label)
// "assign to Sarah" → "@Sarah" (only when Sarah is a circle member)
// "worth five points" → "*5"
// "p one" / "top priority" → "priority 1" (parsePriority already handles that)
@@ -55,22 +55,23 @@ const normalizePoints = text =>
(_, n) => `*${NUMBER_WORDS[n.toLowerCase()] || n} points`,
)
-const normalizeLabels = text =>
- text.replace(
- /\b(?:with\s+)?(?:hash\s?tag|labell?ed(?:\s+as)?|label|tagged(?:\s+as)?|tag)\s+([\p{L}\p{N}_]+)/giu,
- '#$1',
- )
-
// "assign to Sarah" / "assigned to Sarah" / "assign Sarah" / "for Sarah".
-// Speech engines spell names their own way ("Sara" for Sarah) and add
-// punctuation, so exact display-name matching alone misses real speech —
-// an edit-distance-1 fuzzy pass catches those, but only after an explicit
-// assign verb so ordinary words never convert.
+// Speech engines spell names their own way — and worse, can auto-correct an
+// unfamiliar name to an unrelated dictionary word entirely ("Moutaz" heard as
+// "Models"), which plain edit-distance can't recover (too many edits apart).
+// But right after an assign verb, the next word has essentially no other
+// legitimate reading — it IS a name — so we take the *relative best* match
+// among circle members rather than requiring it to be objectively close.
+// A same-first-letter guard keeps this from firing on totally unrelated
+// words. Contextual-string biasing in VoiceInputService is the primary
+// defense (it can make the recognizer hear "Moutaz" correctly in the first
+// place); this is the fallback for when biasing isn't supported or still
+// mishears.
const ASSIGN_VERB = '(?:assign(?:ed|ee)?(?:\\s+(?:this|it))?(?:\\s+to)?|for)'
const STRICT_ASSIGN_VERB = '(?:assign(?:ed|ee)?(?:\\s+(?:this|it))?(?:\\s+to)?)'
+const MIN_MATCH_SCORE = 0.2
const levenshtein = (a, b) => {
- if (Math.abs(a.length - b.length) > 1) return 2
const prev = Array.from({ length: b.length + 1 }, (_, i) => i)
for (let i = 1; i <= a.length; i++) {
let diag = prev[0]
@@ -88,6 +89,9 @@ const levenshtein = (a, b) => {
return prev[b.length]
}
+const similarity = (a, b) =>
+ 1 - levenshtein(a, b) / Math.max(a.length, b.length)
+
const memberNameVariants = member =>
[
member.displayName,
@@ -95,19 +99,75 @@ const memberNameVariants = member =>
member.username,
].filter(n => n && n.length > 1)
-const findMemberFuzzy = (candidate, members) => {
+// Best-scoring item for `candidate` among `items`, requiring only that it
+// beats all others and shares a first letter — not an absolute closeness
+// threshold. `getVariants` returns the name strings to compare a given item
+// against (e.g. a member's display name/first name/username, or a label's
+// name). Shared by assignee and label matching since both face the same
+// problem: ASR is least confident on exactly the words that matter here.
+const findBestFuzzyMatch = (candidate, items, getVariants) => {
const c = candidate.toLowerCase()
- let close = null
- for (const member of members) {
- for (const name of memberNameVariants(member)) {
+ if (c.length < 3) return null
+ let best = null
+ let bestScore = MIN_MATCH_SCORE
+ for (const item of items) {
+ for (const name of getVariants(item)) {
const n = name.toLowerCase()
- if (n === c) return member
- if (!close && n.length >= 4 && c.length >= 4 && levenshtein(n, c) <= 1) {
- close = member
+ if (n === c) return item
+ if (n.length < 3 || n[0] !== c[0]) continue
+ const score = similarity(n, c)
+ if (score > bestScore) {
+ bestScore = score
+ best = item
}
}
}
- return close
+ return best
+}
+
+const findMemberFuzzy = (candidate, members) =>
+ findBestFuzzyMatch(candidate, members, memberNameVariants)
+
+// "label groceries" / "tag groceries" / "labeled as groceries" — only ever
+// converts to a label that already exists (matched exactly or as the closest
+// existing one), never invents a new one. Restricted to single-word label
+// names: CustomParsers' hashtag pattern (#([\p{L}\p{N}_]+)) can't span a
+// space, so a multi-word label like "Home Maintenance" could never be
+// represented as "#Home Maintenance" anyway — same limitation typing it by
+// hand would hit.
+const LABEL_VERB =
+ '(?:with\\s+)?(?:hash\\s?tag|labell?ed(?:\\s+as)?|label|tagged(?:\\s+as)?|tag)'
+
+const normalizeLabels = (text, userLabels = []) => {
+ const singleWordLabels = userLabels.filter(l => l.name && !/\s/.test(l.name))
+ let out = text
+
+ // Exact pass first so a clean spoken match always wins over the fuzzy pass
+ const byLengthDesc = [...singleWordLabels].sort(
+ (a, b) => b.name.length - a.name.length,
+ )
+ for (const label of byLengthDesc) {
+ out = out.replace(
+ new RegExp(
+ `\\b${LABEL_VERB}\\s+${escapeRegex(label.name)}\\b[,.]?`,
+ 'gi',
+ ),
+ `#${label.name}`,
+ )
+ }
+
+ // Fuzzy pass — the spoken word after the verb, matched against the closest
+ // existing single-word label
+ out = out.replace(
+ new RegExp(`\\b${LABEL_VERB}\\s+([\\p{L}][\\p{L}'-]*)[,.]?`, 'giu'),
+ (match, candidate) => {
+ const label = findBestFuzzyMatch(candidate, singleWordLabels, l => [
+ l.name,
+ ])
+ return label ? `#${label.name}` : match
+ },
+ )
+ return out
}
const normalizeAssignees = (text, members = []) => {
@@ -146,11 +206,14 @@ const normalizeAssignees = (text, members = []) => {
return out
}
-export const normalizeSpokenText = (text, { members = [] } = {}) => {
+export const normalizeSpokenText = (
+ text,
+ { members = [], userLabels = [] } = {},
+) => {
let out = stripFillers(text)
out = normalizePriority(out)
out = normalizePoints(out)
- out = normalizeLabels(out)
+ out = normalizeLabels(out, userLabels)
out = normalizeAssignees(out, members)
return out.replace(/\s+/g, ' ').trim()
}