diff --git a/.cspell.yaml b/.cspell.yaml index a843652..3ac12e8 100644 --- a/.cspell.yaml +++ b/.cspell.yaml @@ -53,6 +53,7 @@ words: - unsupplied - denoise - Denoise + - backpressure - Downmix - downmix - Downmixes @@ -153,6 +154,12 @@ words: - undiscarded - unreset - undecoded + - overwritable + - provisionals + - precheck + - unrequested + - unawaited + - unconfigured # Exclude common build artifacts, dependencies, and vendored third-party code ignorePaths: diff --git a/.reviewmark.yaml b/.reviewmark.yaml index 6269874..8a58d52 100644 --- a/.reviewmark.yaml +++ b/.reviewmark.yaml @@ -735,19 +735,38 @@ reviews: - "docs/design/speech/recognition-subsystem.md" # subsystem design + inline fallback units - "docs/verification/speech/recognition-subsystem.md" # subsystem verification + inline units - # Speech-RecognitionSubsystem-ISpeechRecognizer Review (one per unit) - - id: Speech-RecognitionSubsystem-ISpeechRecognizer - title: Review that Speech RecognitionSubsystem ISpeechRecognizer Implementation is Correct + # Speech-RecognitionSubsystem-ISpeechRecognizerEngine Review (one per unit) + - id: Speech-RecognitionSubsystem-ISpeechRecognizerEngine + title: Review that Speech RecognitionSubsystem ISpeechRecognizerEngine Implementation is Correct context: - "docs/design/speech.md" - "docs/design/speech/recognition-subsystem.md" - "docs/reqstream/speech.yaml" - "docs/reqstream/speech/recognition-subsystem.yaml" paths: - - "docs/reqstream/speech/recognition-subsystem/i-speech-recognizer.yaml" - - "docs/design/speech/recognition-subsystem/i-speech-recognizer.md" - - "docs/verification/speech/recognition-subsystem/i-speech-recognizer.md" - - "src/DemaConsulting.Speech/RecognitionSubsystem/ISpeechRecognizer.cs" + - "docs/reqstream/speech/recognition-subsystem/i-speech-recognizer-engine.yaml" + - "docs/reqstream/speech/recognition-subsystem/recognition-session-lease.yaml" + - "docs/design/speech/recognition-subsystem/i-speech-recognizer-engine.md" + - "docs/verification/speech/recognition-subsystem/i-speech-recognizer-engine.md" + - "src/DemaConsulting.Speech/RecognitionSubsystem/ISpeechRecognizerEngine.cs" + - "src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionEngineBusyException.cs" + + # Speech-RecognitionSubsystem-IRecognitionSession Review (one per unit) + - id: Speech-RecognitionSubsystem-IRecognitionSession + title: Review that Speech RecognitionSubsystem IRecognitionSession Implementation is Correct + context: + - "docs/design/speech.md" + - "docs/design/speech/recognition-subsystem.md" + - "docs/reqstream/speech.yaml" + - "docs/reqstream/speech/recognition-subsystem.yaml" + paths: + - "docs/reqstream/speech/recognition-subsystem/i-recognition-session.yaml" + - "docs/design/speech/recognition-subsystem/i-recognition-session.md" + - "docs/verification/speech/recognition-subsystem/i-recognition-session.md" + - "src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionSession.cs" + - "src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionSessionState.cs" + - "src/DemaConsulting.Speech/RecognitionSubsystem/SessionStateChangedEventArgs.cs" + - "src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionSessionFaultedException.cs" - "src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognitionResult.cs" - "src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognitionEvent.cs" @@ -766,48 +785,69 @@ reviews: - "src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognizerFactory.cs" - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SpeechRecognizerFactoryTests.cs" - # Speech-RecognitionSubsystem-SherpaOnnxSpeechRecognizer Review (one per unit) - - id: Speech-RecognitionSubsystem-SherpaOnnxSpeechRecognizer - title: Review that Speech RecognitionSubsystem SherpaOnnxSpeechRecognizer Implementation is Correct + # Speech-RecognitionSubsystem-SherpaOnnxRecognitionSession Review (one per unit) + - id: Speech-RecognitionSubsystem-SherpaOnnxRecognitionSession + title: Review that Speech RecognitionSubsystem SherpaOnnxRecognitionSession Implementation is Correct context: - "docs/design/speech.md" - "docs/design/speech/recognition-subsystem.md" - "docs/reqstream/speech.yaml" - "docs/reqstream/speech/recognition-subsystem.yaml" paths: - - "docs/reqstream/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.yaml" - - "docs/design/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md" - - "docs/verification/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md" - - "src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxSpeechRecognizer.cs" - - "src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionEngine.cs" - - "src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionEngineFactory.cs" + - "docs/reqstream/speech/recognition-subsystem/sherpa-onnx-recognition-session.yaml" + - "docs/reqstream/speech/recognition-subsystem/dedicated-worker.yaml" + - "docs/design/speech/recognition-subsystem/sherpa-onnx-recognition-session.md" + - "docs/verification/speech/recognition-subsystem/sherpa-onnx-recognition-session.md" + - "src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionSession.cs" + - "src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxSpeechRecognizerEngine.cs" + - "src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionBackend.cs" + - "src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionBackendFactory.cs" - "src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionEngine.cs" - "src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionEngineFactory.cs" - "src/DemaConsulting.Speech/RecognitionSubsystem/AudioFrameResampler.cs" - - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerTests.cs" + - "src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionResultBuffer.cs" + - "src/DemaConsulting.Speech/RecognitionSubsystem/DedicatedWorker.cs" + - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionSessionTests.cs" + - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerEngineTests.cs" - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionEngineTests.cs" - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionEngineAccuracyTests.cs" + - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/DedicatedWorkerTests.cs" - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/WordErrorRateCalculator.cs" - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/WordErrorRateCalculatorTests.cs" - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/AudioFrameResamplerTests.cs" - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/Fakes/FakeRecognitionEngine.cs" - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/Fakes/FakeRecognitionEngineFactory.cs" - # Speech-RecognitionSubsystem-UnavailableSpeechRecognizer Review (one per unit) - - id: Speech-RecognitionSubsystem-UnavailableSpeechRecognizer - title: Review that Speech RecognitionSubsystem UnavailableSpeechRecognizer Implementation is Correct + # Speech-RecognitionSubsystem-UnavailableSpeechRecognizerEngine Review (one per unit) + - id: Speech-RecognitionSubsystem-UnavailableSpeechRecognizerEngine + title: Review that Speech RecognitionSubsystem UnavailableSpeechRecognizerEngine Implementation is Correct context: - "docs/design/speech.md" - "docs/design/speech/recognition-subsystem.md" - "docs/reqstream/speech.yaml" - "docs/reqstream/speech/recognition-subsystem.yaml" paths: - - "docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer.yaml" - - "docs/design/speech/recognition-subsystem/unavailable-speech-recognizer.md" - - "docs/verification/speech/recognition-subsystem/unavailable-speech-recognizer.md" - - "src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableSpeechRecognizer.cs" + - "docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer-engine.yaml" + - "docs/design/speech/recognition-subsystem/unavailable-speech-recognizer-engine.md" + - "docs/verification/speech/recognition-subsystem/unavailable-speech-recognizer-engine.md" + - "src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableSpeechRecognizerEngine.cs" - "src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognizerUnavailableException.cs" - - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableSpeechRecognizerTests.cs" + - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableSpeechRecognizerEngineTests.cs" + + # Speech-RecognitionSubsystem-UnavailableRecognitionSession Review (one per unit) + - id: Speech-RecognitionSubsystem-UnavailableRecognitionSession + title: Review that Speech RecognitionSubsystem UnavailableRecognitionSession Implementation is Correct + context: + - "docs/design/speech.md" + - "docs/design/speech/recognition-subsystem.md" + - "docs/reqstream/speech.yaml" + - "docs/reqstream/speech/recognition-subsystem.yaml" + paths: + - "docs/reqstream/speech/recognition-subsystem/unavailable-recognition-session.yaml" + - "docs/design/speech/recognition-subsystem/unavailable-recognition-session.md" + - "docs/verification/speech/recognition-subsystem/unavailable-recognition-session.md" + - "src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableRecognitionSession.cs" + - "test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableRecognitionSessionTests.cs" # Speech-SynthesisSubsystem Review (one per subsystem) - id: Speech-SynthesisSubsystem @@ -842,21 +882,37 @@ reviews: - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/AudioTagCatalogTests.cs" - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/AudioTagParserTests.cs" - # Speech-SynthesisSubsystem-ISpeechSynthesizer Review (one per unit) - - id: Speech-SynthesisSubsystem-ISpeechSynthesizer - title: Review that Speech SynthesisSubsystem ISpeechSynthesizer Implementation is Correct + # Speech-SynthesisSubsystem-ISpeechSynthesizerEngine Review (one per unit) + - id: Speech-SynthesisSubsystem-ISpeechSynthesizerEngine + title: Review that Speech SynthesisSubsystem ISpeechSynthesizerEngine Implementation is Correct context: - "docs/design/speech.md" - "docs/design/speech/synthesis-subsystem.md" - "docs/reqstream/speech.yaml" - "docs/reqstream/speech/synthesis-subsystem.yaml" paths: - - "docs/reqstream/speech/synthesis-subsystem/i-speech-synthesizer.yaml" - - "docs/design/speech/synthesis-subsystem/i-speech-synthesizer.md" - - "docs/verification/speech/synthesis-subsystem/i-speech-synthesizer.md" - - "src/DemaConsulting.Speech/SynthesisSubsystem/ISpeechSynthesizer.cs" + - "docs/reqstream/speech/synthesis-subsystem/i-speech-synthesizer-engine.yaml" + - "docs/design/speech/synthesis-subsystem/i-speech-synthesizer-engine.md" + - "docs/verification/speech/synthesis-subsystem/i-speech-synthesizer-engine.md" + - "src/DemaConsulting.Speech/SynthesisSubsystem/ISpeechSynthesizerEngine.cs" - "src/DemaConsulting.Speech/SynthesisSubsystem/SynthesizedSpeech.cs" + # Speech-SynthesisSubsystem-ISynthesisSession Review (one per unit) + - id: Speech-SynthesisSubsystem-ISynthesisSession + title: Review that Speech SynthesisSubsystem ISynthesisSession Implementation is Correct + context: + - "docs/design/speech.md" + - "docs/design/speech/synthesis-subsystem.md" + - "docs/reqstream/speech.yaml" + - "docs/reqstream/speech/synthesis-subsystem.yaml" + paths: + - "docs/reqstream/speech/synthesis-subsystem/i-synthesis-session.yaml" + - "docs/design/speech/synthesis-subsystem/i-synthesis-session.md" + - "docs/verification/speech/synthesis-subsystem/i-synthesis-session.md" + - "src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisSession.cs" + - "src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisSessionState.cs" + - "src/DemaConsulting.Speech/SynthesisSubsystem/SessionStateChangedEventArgs.cs" + # Speech-SynthesisSubsystem-SpeechSynthesizerFactory Review (one per unit) - id: Speech-SynthesisSubsystem-SpeechSynthesizerFactory title: Review that Speech SynthesisSubsystem SpeechSynthesizerFactory Implementation is Correct @@ -872,22 +928,39 @@ reviews: - "src/DemaConsulting.Speech/SynthesisSubsystem/SpeechSynthesizerFactory.cs" - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SpeechSynthesizerFactoryTests.cs" - # Speech-SynthesisSubsystem-SherpaOnnxSpeechSynthesizer Review (one per unit) - - id: Speech-SynthesisSubsystem-SherpaOnnxSpeechSynthesizer - title: Review that Speech SynthesisSubsystem SherpaOnnxSpeechSynthesizer Implementation is Correct + # Speech-SynthesisSubsystem-SherpaOnnxSpeechSynthesizerEngine Review (one per unit) + - id: Speech-SynthesisSubsystem-SherpaOnnxSpeechSynthesizerEngine + title: Review that Speech SynthesisSubsystem SherpaOnnxSpeechSynthesizerEngine Implementation is Correct context: - "docs/design/speech.md" - "docs/design/speech/synthesis-subsystem.md" - "docs/reqstream/speech.yaml" - "docs/reqstream/speech/synthesis-subsystem.yaml" paths: - - "docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.yaml" - - "docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md" - - "docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md" - - "src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSpeechSynthesizer.cs" - - "src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisEngine.cs" + - "docs/reqstream/speech/synthesis-subsystem/synthesis-session-lease.yaml" + - "docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md" + - "docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md" + - "src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerEngine.cs" + - "src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisEngineBusyException.cs" + - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerEngineTests.cs" + + # Speech-SynthesisSubsystem-SherpaOnnxSynthesisSession Review (one per unit) + - id: Speech-SynthesisSubsystem-SherpaOnnxSynthesisSession + title: Review that Speech SynthesisSubsystem SherpaOnnxSynthesisSession Implementation is Correct + context: + - "docs/design/speech.md" + - "docs/design/speech/synthesis-subsystem.md" + - "docs/reqstream/speech.yaml" + - "docs/reqstream/speech/synthesis-subsystem.yaml" + paths: + - "docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.yaml" + - "docs/design/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md" + - "docs/verification/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md" + - "src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisSession.cs" + - "src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisSessionFaultedException.cs" + - "src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisBackend.cs" - "src/DemaConsulting.Speech/SynthesisSubsystem/EngineAudio.cs" - - "src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisEngineFactory.cs" + - "src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisBackendFactory.cs" - "src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisEngine.cs" - "src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisEngineFactory.cs" - "src/DemaConsulting.Speech/SynthesisSubsystem/IModelCapabilityProfile.cs" @@ -897,28 +970,45 @@ reviews: - "src/DemaConsulting.Speech/SynthesisSubsystem/SpeechParameterConventions.cs" - "src/DemaConsulting.Speech/SynthesisSubsystem/SentenceChunker.cs" - "src/DemaConsulting.Speech/SynthesisSubsystem/PlaybackAudioResampler.cs" - - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerTests.cs" + - "src/DemaConsulting.Speech/SynthesisSubsystem/DedicatedWorker.cs" + - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSynthesisSessionTests.cs" - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/DefaultModelCapabilityProfileTests.cs" - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SentenceChunkerTests.cs" - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/PlaybackAudioResamplerTests.cs" + - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/DedicatedWorkerTests.cs" - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/Fakes/FakeSynthesisEngine.cs" - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/Fakes/FakeSynthesisEngineFactory.cs" - # Speech-SynthesisSubsystem-UnavailableSpeechSynthesizer Review (one per unit) - - id: Speech-SynthesisSubsystem-UnavailableSpeechSynthesizer - title: Review that Speech SynthesisSubsystem UnavailableSpeechSynthesizer Implementation is Correct + # Speech-SynthesisSubsystem-UnavailableSpeechSynthesizerEngine Review (one per unit) + - id: Speech-SynthesisSubsystem-UnavailableSpeechSynthesizerEngine + title: Review that Speech SynthesisSubsystem UnavailableSpeechSynthesizerEngine Implementation is Correct context: - "docs/design/speech.md" - "docs/design/speech/synthesis-subsystem.md" - "docs/reqstream/speech.yaml" - "docs/reqstream/speech/synthesis-subsystem.yaml" paths: - - "docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer.yaml" - - "docs/design/speech/synthesis-subsystem/unavailable-speech-synthesizer.md" - - "docs/verification/speech/synthesis-subsystem/unavailable-speech-synthesizer.md" - - "src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSpeechSynthesizer.cs" + - "docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.yaml" + - "docs/design/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.md" + - "docs/verification/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.md" + - "src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSpeechSynthesizerEngine.cs" - "src/DemaConsulting.Speech/SynthesisSubsystem/SpeechSynthesizerUnavailableException.cs" - - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSpeechSynthesizerTests.cs" + - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSpeechSynthesizerEngineTests.cs" + + # Speech-SynthesisSubsystem-UnavailableSynthesisSession Review (one per unit) + - id: Speech-SynthesisSubsystem-UnavailableSynthesisSession + title: Review that Speech SynthesisSubsystem UnavailableSynthesisSession Implementation is Correct + context: + - "docs/design/speech.md" + - "docs/design/speech/synthesis-subsystem.md" + - "docs/reqstream/speech.yaml" + - "docs/reqstream/speech/synthesis-subsystem.yaml" + paths: + - "docs/reqstream/speech/synthesis-subsystem/unavailable-synthesis-session.yaml" + - "docs/design/speech/synthesis-subsystem/unavailable-synthesis-session.md" + - "docs/verification/speech/synthesis-subsystem/unavailable-synthesis-session.md" + - "src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSynthesisSession.cs" + - "test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSynthesisSessionTests.cs" # SpeechDemo-Architecture Review (one per system) - id: SpeechDemo-Architecture diff --git a/README.md b/README.md index 5c953b7..4356f5e 100644 --- a/README.md +++ b/README.md @@ -122,18 +122,29 @@ var captureDevice = new AudioDeviceFactory().CreateCaptureDevice( AudioDeviceSelection.SystemDefault, model.AudioFormat); -// 5. Compose the recognizer and stream recognized text as it arrives. Create never throws for an -// ordinary machine state (model not installed, no microphone) - check IsAvailable instead. -using var recognizer = SpeechRecognizerFactory.Create(model, catalog, captureDevice); -if (recognizer.IsAvailable) +// 5. Load the engine once, create a session bound to the capture device, and stream +// recognized text as it arrives. LoadAsync never throws for an ordinary machine state +// (model not installed, no microphone) - check IsAvailable instead. +await using var engine = await SpeechRecognizerFactory.LoadAsync(model, catalog); +if (engine.IsAvailable) { - recognizer.ResultReceived += (_, args) => - Console.WriteLine($"{(args.Result.IsFinal ? "final" : "partial")}: {args.Result.Text}"); + await using var session = await engine.CreateSessionAsync(captureDevice); - recognizer.Start(); + await session.StartAsync(); Console.WriteLine("Listening - press any key to stop..."); + + var resultsTask = Task.Run(async () => + { + await foreach (var evt in session.GetResultsAsync()) + { + var status = evt.Result.IsFinal ? "final" : "partial"; + Console.WriteLine($"{status}: {evt.Result.Text}"); + } + }); + Console.ReadKey(intercept: true); - recognizer.Stop(); + await session.StopAsync(); + await resultsTask; } ``` @@ -164,16 +175,16 @@ var playbackDevice = new AudioDeviceFactory().CreatePlaybackDevice( AudioDeviceSelection.SystemDefault, model.PreferredAudioFormat); -// 5. Compose the synthesizer and speak. Create never throws for an ordinary machine state (model -// not installed, no speakers) - check IsAvailable instead. -using var synthesizer = SpeechSynthesizerFactory.Create(model, catalog, playbackDevice); -if (synthesizer.IsAvailable) +// 5. Load the engine and speak a one-shot phrase. LoadAsync never throws for an ordinary +// machine state (model not installed, no speakers) - check IsAvailable instead. +await using var engine = await SpeechSynthesizerFactory.LoadAsync(model, catalog); +if (engine.IsAvailable) { - await synthesizer.SpeakAsync("To be, or not to be. [short pause] That is the question."); + await engine.SpeakAsync(playbackDevice, "To be, or not to be. [short pause] That is the question."); } ``` -Both `Create(...)` factories never throw for an ordinary machine state: a model that isn't +Both `LoadAsync(...)` factories never throw for an ordinary machine state: a model that isn't installed, a machine with no microphone/speakers, and a missing speech-engine native runtime all return `IsAvailable == false` instead of an exception. `AudioDeviceFactory` also exposes `RefreshDevices()` to re-scan for hot-plugged hardware, surfacing `AudioDeviceInUseException` if a @@ -182,17 +193,16 @@ device from the factory is currently active. `SpeakAsync` recognizes Natural Language Audio Tags (such as `[whispers]`, `[short pause]`, or `[excited]`), renders each one per the model's own declared capability, chunks narration into sentence-sized pieces, and pipelines synthesis with playback - an earlier chunk plays while a -later chunk is still synthesizing. `Stop()` cancels an in-flight `SpeakAsync` call deterministically -and is a safe no-op when nothing is speaking. +later chunk is still synthesizing. Passing a cancelled `CancellationToken` to `SpeakAsync` cancels +an in-flight call deterministically. For a model that declares tunable parameters - such as Kokoro's `voice` choice or VITS/Piper's -numeric `speaker` id - pass a `parameterValues` bag keyed by each parameter's `Id`: +numeric `speaker` id - pass a `parameterValues` bag keyed by each parameter's `Id` to `LoadAsync(...)`: ```csharp -using var synthesizer = SpeechSynthesizerFactory.Create( +await using var engine = await SpeechSynthesizerFactory.LoadAsync( model, catalog, - playbackDevice, parameterValues: new Dictionary { ["voice"] = "af_bella" }); ``` @@ -203,8 +213,8 @@ different models without breaking composition. A supplied value for a parameter declare, but that fails that parameter's own validation - the wrong CLR type, a number outside its declared range, a fractional value for a whole-number-only parameter, or a string that matches none of a `ChoiceParameter`'s declared options - throws `ArgumentException` synchronously -from `Create()`, naming the parameter, the model, and the reason the value is invalid. This same -rule applies to `SpeechRecognizerFactory.Create`'s `parameterValues` argument. +from `LoadAsync(...)`, naming the parameter, the model, and the reason the value is invalid. This +same rule applies to `SpeechRecognizerFactory.LoadAsync`'s `parameterValues` argument. See the [user guide][link-user-guide] for the full API walkthrough, voice/speaker catalogs, and Natural Language Audio Tag vocabulary. diff --git a/docs/design/definition.yaml b/docs/design/definition.yaml index d63a2b3..f50fea3 100644 --- a/docs/design/definition.yaml +++ b/docs/design/definition.yaml @@ -41,10 +41,21 @@ input-files: - docs/design/speech/model-management-subsystem/speech-model-descriptor.md - docs/design/speech/model-management-subsystem/speech-model-catalog.md - docs/design/speech/recognition-subsystem.md - - docs/design/speech/recognition-subsystem/i-speech-recognizer.md + - docs/design/speech/recognition-subsystem/i-speech-recognizer-engine.md + - docs/design/speech/recognition-subsystem/i-recognition-session.md - docs/design/speech/recognition-subsystem/speech-recognizer-factory.md - - docs/design/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md - - docs/design/speech/recognition-subsystem/unavailable-speech-recognizer.md + - docs/design/speech/recognition-subsystem/sherpa-onnx-recognition-session.md + - docs/design/speech/recognition-subsystem/unavailable-speech-recognizer-engine.md + - docs/design/speech/recognition-subsystem/unavailable-recognition-session.md + - docs/design/speech/synthesis-subsystem.md + - docs/design/speech/synthesis-subsystem/i-speech-synthesizer-engine.md + - docs/design/speech/synthesis-subsystem/i-synthesis-session.md + - docs/design/speech/synthesis-subsystem/speech-synthesizer-factory.md + - docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md + - docs/design/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md + - docs/design/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.md + - docs/design/speech/synthesis-subsystem/unavailable-synthesis-session.md + - docs/design/speech/synthesis-subsystem/audio-tag-parser.md - docs/design/speech-demo.md - docs/design/speech-demo/shell-subsystem.md - docs/design/speech-demo/device-selection-subsystem.md diff --git a/docs/design/introduction.md b/docs/design/introduction.md index d3feb12..fb9c442 100644 --- a/docs/design/introduction.md +++ b/docs/design/introduction.md @@ -41,17 +41,18 @@ SpeechCli command-line tool system and its constituent software items, specifica catalog/contract seam: typed tunable-parameter descriptors, the common per-model contract and its role-marker interfaces, and catalog enumeration of the compiled-in known-model registry alongside install state -- **RecognitionSubsystem (Subsystem)** — Streaming speech-to-text: the public recognizer - contract and result types, the composition root with honest unavailable fallback, the - capture-to-engine pipeline with its audio-format converter, and the mockable - recognition-engine seam +- **RecognitionSubsystem (Subsystem)** — Streaming speech-to-text: the public async + Engine/Session recognizer contract and result types, the composition root (`LoadAsync`/ + `CreateSessionAsync`) with honest unavailable fallbacks and engine exclusivity lease, the + capture-to-engine pipeline with its audio-format converter, and the mockable internal + recognition-backend seam - **SynthesisSubsystem (Subsystem)** — Text-to-speech: the closed, fixed Natural Language Audio Tag vocabulary grouped into kinds and the model-independent Layer 1 parser that recognizes bracket syntax against it, the Layer 2 rendering strategy that turns a parsed span sequence into a model-appropriate `SpeechPlan`, sentence chunking for pipelined synthesis, the public - `ISpeechSynthesizer` streaming/playback contract with its composition root and honest - unavailable fallback, the mockable synthesis-engine seam, the real sherpa-onnx synthesis - engine, and the playback-format converter + async Engine/Session streaming/playback contract with its composition root and honest + unavailable fallbacks, the mockable internal synthesis-backend seam, the real sherpa-onnx + synthesis engine, and the playback-format converter The following software items of the SpeechDemo system are also covered: @@ -104,6 +105,9 @@ The following software items of the SpeechCli system are also covered: (`recognize`) plus the `SilenceTimeoutRecognizerSession` idle-timeout utility, extending `ModelCommandsSubsystem`'s `ICliModelCatalog` seam with recognition-side members symmetric to the synthesis pair +- **ConversationCommandSubsystem (Subsystem)** — The one voice-conversation subcommand (`ask`), + running a speak-then-listen turn as one invocation by reusing `SynthesisCommandSubsystem`'s + and `RecognitionCommandSubsystem`'s existing seam members rather than introducing a new one The following OTS items are also covered: @@ -156,19 +160,22 @@ four concrete models covering both roles - two recognition models two synthesis models (`SherpaOnnxVitsLibriTtsEnglishSynthesisModel`, `SherpaOnnxKokoroEnglishSynthesisModel`) - so a host may use the compiled-in registry directly or supply its own. A fourth subsystem, -`RecognitionSubsystem`, consumes both of the above: it defines the public streaming -speech-to-text contract, composes a recognizer for an installed recognition model and a capture -device without ever throwing, and converts captured audio into the format a model requires before -streaming it through an internal, mockable recognition-engine seam. A fifth subsystem, +`RecognitionSubsystem`, consumes both of the above: it defines the public async Engine/Session +streaming speech-to-text contract (`ISpeechRecognizerEngine`/`IRecognitionSession`), composing a +recognizer engine for an installed recognition model via `LoadAsync` and then a session bound to +a capture device via the engine's `CreateSessionAsync` without ever throwing for an ordinary +machine state, and converts captured audio into the format a model requires before streaming it +through an internal, mockable recognition-backend seam. A fifth subsystem, `SynthesisSubsystem`, provides the text-to-speech side: its closed, fixed Natural Language Audio Tag vocabulary and the model-independent Layer 1 parser (`AudioTagCatalog`/`AudioTagParser`) that recognizes bracket syntax against it, its Layer 2 rendering strategy (`IModelCapabilityProfile`/`DefaultModelCapabilityProfile`) that turns a parsed span sequence into a model-appropriate `SpeechPlan` of `SpeechSegment`s, its `SentenceChunker` for -pipeline-friendly chunk boundaries, and its public `ISpeechSynthesizer` streaming/playback -contract, which composes a synthesizer for an installed synthesis model and a playback device -without ever throwing and pipelines chunked synthesis with playback through an internal, mockable -synthesis-engine seam. +pipeline-friendly chunk boundaries, and its public async Engine/Session streaming/playback +contract (`ISpeechSynthesizerEngine`/`ISynthesisSession`), which composes a synthesizer engine +for an installed synthesis model via `LoadAsync` and then a session bound to a playback device +via the engine's `CreateSessionAsync` without ever throwing and pipelines chunked synthesis with +playback through an internal, mockable synthesis-backend seam. A second, sibling system, `SpeechDemo`, sits alongside `Speech` in the model. It is the Avalonia desktop application that consumes the library's public API and demonstrates it, and it is @@ -184,13 +191,16 @@ nor knows about `SpeechDemo`. A third, sibling system, `SpeechCli`, also sits alongside `Speech` in the model. It is the cross-platform .NET global tool (`speech-cli`) that exposes the library's capabilities from the -command line, and it is structured with four subsystems: `ModelCommandsSubsystem` (the five +command line, and it is structured with five subsystems: `ModelCommandsSubsystem` (the five model-management subcommands over a CLI-owned catalog seam), `DeviceCommandsSubsystem` (the three device-related subcommands consuming the library's probe/factory interfaces directly), `SynthesisCommandSubsystem` (the `speak` subcommand, extending the catalog seam with synthesis -members), and `RecognitionCommandSubsystem` (the `recognize` subcommand and its silence-timeout -utility, extending the same seam with recognition members). As with `SpeechDemo`, the dependency -runs one way only: `SpeechCli` references `Speech`, and `Speech` neither references nor knows +members), `RecognitionCommandSubsystem` (the `recognize` subcommand and its silence-timeout +utility, extending the same seam with recognition members), and `ConversationCommandSubsystem` +(the `ask` subcommand, a speak-then-listen conversation turn reusing the synthesis and +recognition subsystems' existing seam members rather than introducing a new one). As with +`SpeechDemo`, the dependency runs one way only: `SpeechCli` references `Speech`, and `Speech` +neither references nor knows about `SpeechCli`. ## Folder Layout @@ -261,18 +271,27 @@ src/DemaConsulting.Speech/ ```text src/DemaConsulting.Speech/ └── RecognitionSubsystem/ - ├── ISpeechRecognizer.cs — Streaming speech-to-text contract - ├── SpeechRecognitionResult.cs — Recognized text plus provisional/final flag - ├── SpeechRecognitionEvent.cs — Result-carrying recognition event payload - ├── SpeechRecognizerUnavailableException.cs — Thrown by unavailable recognizers when misused - ├── SpeechRecognizerFactory.cs — Composition entry point for recognizer creation - ├── UnavailableSpeechRecognizer.cs — Honest unavailable recognizer fallback - ├── IRecognitionEngine.cs — Mockable speech-inference seam - ├── IRecognitionEngineFactory.cs — Mockable engine-loading seam - ├── SherpaOnnxRecognitionEngine.cs — Real sherpa-onnx streaming engine adapter - ├── SherpaOnnxRecognitionEngineFactory.cs — Real model-driven engine loader - ├── AudioFrameResampler.cs — Downmix and rate conversion for captured audio - └── SherpaOnnxSpeechRecognizer.cs — Real streaming recognition pipeline + ├── ISpeechRecognizerEngine.cs — Loaded-model engine contract (Layer 3) + ├── IRecognitionSession.cs — Per-device session contract (Layer 5) + ├── RecognitionSessionState.cs — Session state machine enum + ├── SessionStateChangedEventArgs.cs — StateChanged event payload + ├── SpeechRecognitionResult.cs — Recognized text plus provisional/final flag + ├── SpeechRecognitionEvent.cs — Result-carrying recognition event payload + ├── SpeechRecognizerUnavailableException.cs — Thrown by unavailable engines/sessions when misused + ├── RecognitionEngineBusyException.cs — Thrown by a concurrent CreateSessionAsync while leased + ├── RecognitionSessionFaultedException.cs — Surfaced through GetResultsAsync on worker fault + ├── SpeechRecognizerFactory.cs — Composition entry point: LoadAsync returns an engine + ├── UnavailableSpeechRecognizerEngine.cs — Honest unavailable engine fallback + ├── UnavailableRecognitionSession.cs — Honest unavailable session fallback + ├── IRecognitionBackend.cs — Mockable speech-inference seam (internal) + ├── IRecognitionBackendFactory.cs — Mockable engine-loading seam (internal) + ├── SherpaOnnxRecognitionEngine.cs — Real sherpa-onnx streaming engine adapter + ├── SherpaOnnxRecognitionEngineFactory.cs — Real model-driven engine loader + ├── SherpaOnnxSpeechRecognizerEngine.cs — Real ISpeechRecognizerEngine implementation + ├── SherpaOnnxRecognitionSession.cs — Real IRecognitionSession streaming pipeline + ├── RecognitionResultBuffer.cs — Byte-capped backpressure buffer for GetResultsAsync + ├── DedicatedWorker.cs — Long-running worker thread with cooperative-cancel-then-abandon + └── AudioFrameResampler.cs — Downmix and rate conversion for captured audio ``` ```text @@ -292,17 +311,25 @@ src/DemaConsulting.Speech/ ├── DefaultModelCapabilityProfile.cs — Generically-correct default rendering strategy ├── SentenceChunker.cs — Sentence/clause-sized chunking for pipelining ├── EngineAudio.cs — Raw engine output: samples plus produced rate - ├── ISynthesisEngine.cs — Mockable speech-synthesis seam - ├── ISynthesisEngineFactory.cs — Mockable engine-loading seam + ├── ISpeechSynthesizerEngine.cs — Loaded-model engine contract (Layer 3) + ├── ISynthesisSession.cs — Per-device session contract (Layer 5) + ├── SynthesisSessionState.cs — Session state machine enum + ├── SessionStateChangedEventArgs.cs — StateChanged event payload + ├── SpeechSynthesizerUnavailableException.cs — Thrown by unavailable engines/sessions when misused + ├── SynthesisEngineBusyException.cs — Thrown by a concurrent CreateSessionAsync while leased + ├── SynthesisSessionFaultedException.cs — Surfaced on worker fault + ├── SpeechSynthesizerFactory.cs — Composition entry point: LoadAsync returns an engine + ├── UnavailableSpeechSynthesizerEngine.cs — Honest unavailable engine fallback + ├── UnavailableSynthesisSession.cs — Honest unavailable session fallback + ├── ISynthesisBackend.cs — Mockable speech-synthesis seam (internal) + ├── ISynthesisBackendFactory.cs — Mockable engine-loading seam (internal) ├── SherpaOnnxSynthesisEngine.cs — Real sherpa-onnx offline-TTS engine adapter ├── SherpaOnnxSynthesisEngineFactory.cs — Real model-driven engine loader + ├── SherpaOnnxSpeechSynthesizerEngine.cs — Real ISpeechSynthesizerEngine implementation + ├── SherpaOnnxSynthesisSession.cs — Real ISynthesisSession chunked/pipelined synthesis + ├── DedicatedWorker.cs — Long-running worker thread with cooperative-cancel-then-abandon ├── PlaybackAudioResampler.cs — Rate conversion and upmix for playback audio - ├── SynthesizedSpeech.cs — One synthesized segment's audio and silence - ├── ISpeechSynthesizer.cs — Chunked/streaming synthesis-and-playback contract - ├── SpeechSynthesizerUnavailableException.cs — Thrown by unavailable synthesizers when misused - ├── SpeechSynthesizerFactory.cs — Composition entry point for synthesizer creation - ├── UnavailableSpeechSynthesizer.cs — Honest unavailable synthesizer fallback - └── SherpaOnnxSpeechSynthesizer.cs — Real chunked/pipelined synthesis pipeline + └── SynthesizedSpeech.cs — One synthesized segment's audio and silence ``` The SpeechDemo application's folder structure likewise mirrors its software structure: @@ -381,9 +408,13 @@ src/DemaConsulting.Speech.Cli/ │ │ └── DoctorCommand.cs — `doctor` implementation │ ├── SynthesisCommandSubsystem/ │ │ └── SpeakCommand.cs — `speak` implementation -│ └── RecognitionCommandSubsystem/ -│ ├── RecognizeCommand.cs — `recognize` implementation -│ └── SilenceTimeoutRecognizerSession.cs — Mic idle-timeout utility for `recognize --mic` +│ ├── RecognitionCommandSubsystem/ +│ │ ├── RecognizeCommand.cs — `recognize` implementation +│ │ └── SilenceTimeoutRecognizerSession.cs — Mic idle-timeout utility for `recognize --mic` +│ └── ConversationCommandSubsystem/ +│ ├── AskCommand.cs — `ask` implementation +│ ├── ICliCaptureDeviceSource.cs — CLI-owned capture-device seam contract +│ └── AudioDeviceFactoryCaptureDeviceSource.cs — Real seam over AudioDeviceFactory ``` ## Document Conventions diff --git a/docs/design/ots/sherpa-onnx.md b/docs/design/ots/sherpa-onnx.md index 2541b90..bf2cf88 100644 --- a/docs/design/ots/sherpa-onnx.md +++ b/docs/design/ots/sherpa-onnx.md @@ -34,10 +34,10 @@ Sherpa-onnx types are confined to two places. Each model's backing class produce configuration through an internal member of `IRecognitionModel`, matching this library's "sherpa-onnx configuration for its own model architecture" responsibility. The RecognitionSubsystem then consumes that configuration behind its internal -`IRecognitionEngine`/`IRecognitionEngineFactory` seam, implemented for real by +`IRecognitionBackend`/`IRecognitionBackendFactory` seam, implemented for real by `SherpaOnnxRecognitionEngine`/`SherpaOnnxRecognitionEngineFactory`. No sherpa-onnx type appears -anywhere in the library's public API: public callers interact only through `ISpeechRecognizer`, -`SpeechRecognitionResult`, and `SpeechRecognizerFactory`, keeping this library's "engine +anywhere in the library's public API: public callers interact only through `ISpeechRecognizerEngine`, +`IRecognitionSession`, `SpeechRecognitionResult`, and `SpeechRecognizerFactory`, keeping this library's "engine backend stays swappable at the public API surface" promise intact. See _RecognitionSubsystem Design_ for the seam's structure. diff --git a/docs/design/speech-cli.md b/docs/design/speech-cli.md index 257540e..1a6caaf 100644 --- a/docs/design/speech-cli.md +++ b/docs/design/speech-cli.md @@ -106,8 +106,8 @@ consumes. | `SpeechModelCatalog` (model commands) | Outbound | Constructor/method call | Consumed via seam; no native runtime | | `IAudioCaptureDeviceProbe`/`IAudioPlaybackDeviceProbe` (device commands) | Outbound | Method call | Never throws | | Audio device hardware (`devices test`) | Bidirectional | PCM samples | Real I/O; fails cleanly if unavailable | -| `ISpeechSynthesizer` (`speak`) | Outbound | Method call | Real inference/audio I/O; graceful fallback | -| `ISpeechRecognizer` (`recognize`) | Outbound | Method call | Real inference/audio I/O; graceful fallback | +| `ISpeechSynthesizerEngine`/`ISynthesisSession` (`speak`) | Outbound | Method call | Real I/O; graceful fallback | +| `ISpeechRecognizerEngine`/`IRecognitionSession` (`recognize`) | Outbound | Method call | Real I/O; graceful fallback | ## Dependencies diff --git a/docs/design/speech-cli/conversation-command-subsystem.md b/docs/design/speech-cli/conversation-command-subsystem.md index f448d4b..9b94e17 100644 --- a/docs/design/speech-cli/conversation-command-subsystem.md +++ b/docs/design/speech-cli/conversation-command-subsystem.md @@ -6,9 +6,10 @@ The ConversationCommandSubsystem implements the one voice-conversation subcomman by `CommandDispatch`: `ask` - a new, eleventh subcommand added after all ten scaffolded subcommands were implemented. It contains one type, `AskCommand`, and introduces **no new `ICliModelCatalog` seam member**: `ask` combines `SynthesisCommandSubsystem`'s existing -`GetPreferredAudioFormat`/`CreateSynthesizer` members and `RecognitionCommandSubsystem`'s existing -`GetAudioFormat`/`CreateRecognizer` members - both already extending the original -`ModelCommandsSubsystem` seam - to run a speak-then-listen turn as one CLI invocation. +`GetPreferredAudioFormat`/`CreateSynthesizerEngineAsync` members and +`RecognitionCommandSubsystem`'s existing `GetAudioFormat`/`CreateRecognizerEngineAsync` members - +both already extending the original `ModelCommandsSubsystem` seam - to run a speak-then-listen +turn as one CLI invocation, built entirely on the async Engine/Session API. - **`AskCommand`**: implements `ask --tts-model --stt-model (--text | --file ) @@ -26,12 +27,14 @@ subcommands were implemented. It contains one type, `AskCommand`, and introduces ### Reuse of the Existing ModelCommandsSubsystem Seam (No New Member) -`ask` needs to construct both a real `ISpeechSynthesizer` and a real `ISpeechRecognizer` for its -two phases. Both capabilities already exist on `ICliModelCatalog`, added by the synthesis and -recognition passes respectively: `CreateSynthesizer(descriptor, playbackDevice, parameterValues)` -and `CreateRecognizer(descriptor, captureDevice, parameterValues)`, each already validating (via -`SpeechModelCatalogAdapter`'s own internal casts) that the resolved model actually has the -corresponding role. Because `ask` requires exactly one synthesis-role model and exactly one +`ask` needs to load both a real `ISpeechSynthesizerEngine` and a real `ISpeechRecognizerEngine` for +its two phases. Both capabilities already exist on `ICliModelCatalog`, added by the synthesis and +recognition passes respectively: `CreateSynthesizerEngineAsync(descriptor, parameterValues, +cancellationToken)` and `CreateRecognizerEngineAsync(descriptor, parameterValues, +cancellationToken)`, each already validating (via `SpeechModelCatalogAdapter`'s own internal +casts) that the resolved model actually has the corresponding role, and each returning an engine +with no device bound yet - the device is bound afterward via `engine.CreateSessionAsync(device, +cancellationToken)`. Because `ask` requires exactly one synthesis-role model and exactly one recognition-role model - never a single model serving both roles - it has no need for a third, combined seam member; it simply calls the two existing members once each, exactly as `speak` and `recognize` already do individually. This keeps `AskCommand`'s own logic fully unit-testable @@ -62,42 +65,47 @@ next step (`list-models --role tts`/`--role stt`, or `download `) is al parameters, reusing `ParameterBagParser.Resolve` unmodified from the synthesis pass, exactly as `recognize`'s own `--stt-param` handling already does. -**Two-phase execution.** Phase 1 (speak) resolves a real playback device from the injected -`ICliPlaybackDeviceSource` (honoring `--playback-device`, defaulting to the system default device -exactly as `speak` does), constructs the TTS synthesizer via `CreateSynthesizer`, and calls -`SpeakAsync` on the resolved text, waiting for it to complete before Phase 2's listen phase can -end - Phase 2's recognizer construction now begins concurrently with Phase 1's wait rather than -strictly once it finishes (see "Recognizer Pre-Warming" below); the two phases' *listening* -remains strictly sequential, matching a natural question-then-answer conversational turn - `ask` -never starts real microphone capture while the prompt is still being spoken. Phase 2 (listen) -resolves a real capture device from the injected `ICliCaptureDeviceSource` (honoring -`--capture-device`, defaulting to the system default device exactly as `recognize --mic` does), -constructs the STT recognizer via `CreateRecognizer`, and blocks on a `ManualResetEventSlim` until -one of: the first **final** recognition result arrives, a `SilenceTimeoutRecognizerSession` +**Two-phase execution.** `RunAsync` kicks off Phase 2's recognizer engine/session pre-warm on a +background `Task.Run` immediately (see _Recognizer Pre-Warming_ below), then runs Phase 1: resolves +a real playback device from the injected `ICliPlaybackDeviceSource` (honoring `--playback-device`, +defaulting to the system default device exactly as `speak` does), awaits +`catalog.CreateSynthesizerEngineAsync(...)` then `engine.CreateSessionAsync(playbackDevice, ...)`, +and calls `session.SpeakAsync(text, cancellationToken)`, waiting for it to complete before Phase +2's listen phase can begin - Phase 2's recognizer engine/session construction runs concurrently +with Phase 1's wait, but the two phases' _listening_ remains strictly sequential, matching a +natural question-then-answer conversational turn - `ask` never starts real microphone capture +while the prompt is still being spoken. Once Phase 1 completes, `RunAsync` adopts the pre-warmed +`PrewarmedRecognizer` (engine + session) and calls `Listen`, which drives the already-constructed +session via `StartAsync`/`GetResultsAsync`/`StopAsync`, wrapped in a `SilenceTimeoutRecognizerSession` (constructed and armed exactly as `recognize --mic`'s own, reused unmodified from -`RecognitionCommandSubsystem`) times out, or `Ctrl+C` is pressed. +`RecognitionCommandSubsystem`), blocking until one of: the first **final** recognition result +arrives, the silence/start timeout fires, or `Ctrl+C` is pressed (signaled externally via a shared +`ManualResetEventSlim`). -### Recognizer Pre-Warming (Concurrent Phase 1/Phase 2 Model Load) +### Recognizer Pre-Warming (Concurrent Phase 1/Phase 2 Engine Load) -Loading a recognition model into native memory - not `Start()`, which merely begins streaming -audio through an already-loaded model - is the expensive step in constructing an -`ISpeechRecognizer` (see `SpeechRecognizerFactory`'s own remarks). Deferring that load until +Loading a recognition model into native memory - not `StartAsync`, which merely begins streaming +audio through an already-loaded engine - is the expensive step in constructing an +`ISpeechRecognizerEngine` (see `SpeechRecognizerFactory`'s own remarks). Deferring that load until strictly after Phase 1's playback finishes therefore introduced an avoidable turnaround-gap -latency between finishing speaking and starting to listen. `AskCommand.RunAsync` now kicks off a -private `PrewarmRecognizer` method - which resolves Phase 2's capture device (via -`ICliCaptureDeviceSource`, honoring `--capture-device`) and then calls `CreateRecognizer`, exactly -what `Listen` used to do at its own start - on a background `Task.Run` immediately after both -phases' parameters are resolved, so this work overlaps with the `await SpeakPromptAsync(...)` call -that follows it. Capture-device resolution moved into this pre-warm step alongside recognizer -construction because `ICliModelCatalog.CreateRecognizer` requires an already-constructed -`IAudioCaptureDevice` as an argument - it cannot be deferred independently of the recognizer -itself. Critically, only the model *load* is pre-warmed: `ISpeechRecognizer.Start()` (which begins -real microphone capture) is still called only once Phase 2's `Listen` genuinely runs, so -pre-warming never risks capturing audio - including any acoustic bleed from the prompt still being -played - while Phase 1 is in progress. +latency between finishing speaking and starting to listen. `AskCommand.RunAsync` kicks off a +private `PrewarmRecognizerAsync` method - which resolves Phase 2's capture device (via +`ICliCaptureDeviceSource`, honoring `--capture-device`), calls +`catalog.CreateRecognizerEngineAsync(...)`, then `engine.CreateSessionAsync(captureDevice, ...)`, +bundling both into a `PrewarmedRecognizer(Engine, Session)` record struct - on a background +`Task.Run` immediately after both phases' parameters are resolved, so this work overlaps with the +`await SpeakPromptAsync(...)` call that follows it. Capture-device resolution moved into this +pre-warm step alongside engine/session construction because `ISpeechRecognizerEngine.CreateSessionAsync` +requires an already-constructed `IAudioCaptureDevice` as an argument - it cannot be deferred +independently of the engine itself. If session creation fails after the engine has already loaded, +`PrewarmRecognizerAsync` disposes the orphaned engine before rethrowing, so a session-creation +failure never leaks the engine. Critically, only the model _load_ and session creation are +pre-warmed: `session.StartAsync(...)` (which begins real microphone capture) is still called only +once Phase 2's `Listen` genuinely runs, so pre-warming never risks capturing audio - including any +acoustic bleed from the prompt still being played - while Phase 1 is in progress. On every exit path, the pre-warm task's result and any exception it raises are always eventually -observed and never left unobserved, but `RunAsync` only *synchronously* awaits it on the success +observed and never left unobserved, but `RunAsync` only _synchronously_ awaits it on the success path (Phase 2 genuinely starting); on the canceled/failed paths it hands the cleanup off fire-and-forget so a slow in-flight model load never delays `Ctrl+C`/fast-failure responsiveness: @@ -106,25 +114,27 @@ fire-and-forget so a slow in-flight model load never delays `Ctrl+C`/fast-failur `DisposePrewarmedRecognizerAsync` helper via `_ = DisposePrewarmedRecognizerAsync(prewarmTask);`, so `RunAsync` can return/rethrow immediately instead of blocking on the recognizer's model load finishing. `DisposePrewarmedRecognizerAsync` itself still awaits the task in the background and, - if it produced a recognizer, disposes it, swallowing any pre-warm exception - Phase 1's own - cancellation/failure has already been reported, and a concurrently failed pre-warm is not - separately actionable once the call is already ending that way. + if it produced a `PrewarmedRecognizer`, disposes its session and then its engine (via + `DisposeAsync`), swallowing any pre-warm exception - Phase 1's own cancellation/failure has + already been reported, and a concurrently failed pre-warm is not separately actionable once the + call is already ending that way. - **Phase 1 succeeds**: `RunAsync` awaits the pre-warm task directly, which re-throws (with its - original exception type, message, and stack trace) any failure `PrewarmRecognizer` raised - an - unknown/unavailable `--capture-device`, an invalid `--stt-param` value, etc. - at the start of - Phase 2, exactly as a synchronous call to the same logic would have. `Listen` itself is - simplified to accept the already-constructed recognizer directly, rather than constructing one - of its own; it still checks `stopSignal.IsSet` first (now also covering the case where `Ctrl+C` - landed during the concurrent pre-warm itself) before ever calling `Start()`, and still calls - `onRecognizerCreated`/disposes the recognizer in its own `finally` block exactly as before, so - the disposal-ordering/callback-timing contract several unit tests depend on is unchanged. + original exception type, message, and stack trace) any failure `PrewarmRecognizerAsync` raised - + an unknown/unavailable `--capture-device`, an invalid `--stt-param` value, etc. - at the start + of Phase 2, exactly as a synchronous call to the same logic would have. `RunAsync` adopts the + engine via `await using`, then calls `Listen` with the already-constructed session; `Listen` + itself still checks `stopSignal.IsSet` first (now also covering the case where `Ctrl+C` landed + during the concurrent pre-warm itself) before ever calling `StartAsync`, and still calls + `onRecognizerCreated`/disposes the session in its own `finally` block exactly as before, so the + disposal-ordering/callback-timing contract several unit tests depend on is unchanged. This is a design note for future `ICliModelCatalog` implementers: `SpeechModelCatalogAdapter` (the only production implementation) wraps a read-mostly `SpeechModelCatalog` with no documented -thread-safety concerns for concurrent `CreateSynthesizer`/`CreateRecognizer` calls on the same -instance, which is what makes running `CreateRecognizer` concurrently with Phase 1's -`CreateSynthesizer`-backed playback safe today; a future catalog implementation with non-thread-safe -side effects shared across both calls would need to account for this concurrency. +thread-safety concerns for concurrent `CreateSynthesizerEngineAsync`/`CreateRecognizerEngineAsync` +calls on the same instance, which is what makes running `CreateRecognizerEngineAsync` concurrently +with Phase 1's `CreateSynthesizerEngineAsync`-backed playback safe today; a future catalog +implementation with non-thread-safe side effects shared across both calls would need to account +for this concurrency. ### Capture-Device Resolution Through `ICliCaptureDeviceSource` (New Seam) @@ -133,7 +143,7 @@ Unlike `RecognizeCommand`, which resolves its capture device directly from an in small, CLI-owned seam introduced in this pass and mirroring `ICliPlaybackDeviceSource` exactly (same `*Probe` property/`Create*Device(selection)` method shape, same rationale). This exists because `AudioDeviceFactory.CreateCaptureDevice` checks real PortAudio initialization state -*before* even consulting an injected probe, so a test that only fakes the probe (as +_before_ even consulting an injected probe, so a test that only fakes the probe (as `RecognizeCommandTests` does for its own, narrower error-path-only coverage) still resolves to the honestly unavailable fallback on any machine - including every headless CI runner - where real PortAudio never initializes, regardless of the fake probe. `ICliCaptureDeviceSource` lets a test @@ -165,16 +175,18 @@ stdout by default, or written to the file named by `--output-text` instead - mir so no carriage-return-overwrite console logic from `recognize` is reused here. **Cancellation and disposal.** `Ctrl+C` is wired to cooperative cancellation via -`Console.CancelKeyPress`, exactly mirroring `SpeakCommand.Run`'s and `RecognizeCommand.Run`'s own -subscribe/unsubscribe-in-try/finally pattern, for the whole call rather than per-phase: a single -handler both cancels a shared `CancellationTokenSource` (aborting an in-flight `SpeakAsync` during -Phase 1) and, once a recognizer has been constructed, calls its own `Stop()` and signals the same -`ManualResetEventSlim` Phase 2 blocks on - so `Ctrl+C` cancels cleanly whether it lands during -playback or during listening, with no partial recognized-text or playback state ever leaked. A -canceled Phase 1 skips Phase 2 entirely (a canceled prompt is never followed by a listen attempt). -The synthesizer, playback device, recognizer, silence-timeout session (when present), and output -writer are each disposed exactly once via nested `finally` blocks in the same disposal order -`SpeakCommand`/`RecognizeCommand` each already established for their own resources. +`Console.CancelKeyPress` in `AskCommand.Run(Context)`, for the whole call rather than per-phase: a +single handler cancels a shared `CancellationTokenSource`, calls `StopAsync(CancellationToken.None)` +(fire-and-forget) on the live session once Phase 2 has created one (tracked under a lock since the +handler runs on a separate thread), and signals the same `ManualResetEventSlim` Phase 2 blocks on - +so `Ctrl+C` cancels cleanly whether it lands during playback or during listening, with no partial +recognized-text or playback state ever leaked. A canceled Phase 1 skips Phase 2 entirely (a +canceled prompt is never followed by a listen attempt), and the concurrently pre-warmed +engine/session is disposed via the fire-and-forget `DisposePrewarmedRecognizerAsync` path described +above. Phase 1's own session, engine, and playback-device lease are disposed via `await using`/ +`using` in `SpeakPromptAsync`; Phase 2's session is disposed inside `Listen`'s own `finally` block, +and its engine is disposed by `RunAsync` itself (via `await using var engineLease = prewarmed.Engine;`) +once `Listen` returns. **Distinguishing a genuine `Ctrl+C` from a legitimate empty Phase 2 result.** The same `ManualResetEventSlim` unblocks Phase 2's wait identically whether the final wake-up came from a @@ -201,7 +213,8 @@ reporting a false "Recognized text written" success. ### Interactions with Other Units `AskCommand` depends on the `ICliModelCatalog` seam (all members added by both the synthesis and -recognition passes; no new member of its own), `SynthesisCommandSubsystem`'s +recognition passes; no new member of its own), the library's public `ISpeechSynthesizerEngine`/ +`ISynthesisSession`/`ISpeechRecognizerEngine`/`IRecognitionSession` types, `SynthesisCommandSubsystem`'s `ICliPlaybackDeviceSource`/`ParameterBagParser`, `RecognitionCommandSubsystem`'s `SilenceTimeoutRecognizerSession`, and `DeviceCommandsSubsystem`'s `DevicesTestCommand.ResolveDeviceSelectionOrThrow` internal helper - all four reused entirely diff --git a/docs/design/speech-cli/model-commands-subsystem.md b/docs/design/speech-cli/model-commands-subsystem.md index 50153d6..835450d 100644 --- a/docs/design/speech-cli/model-commands-subsystem.md +++ b/docs/design/speech-cli/model-commands-subsystem.md @@ -112,26 +112,31 @@ progress-reporting behavior of all five commands to be verified deterministicall access and no real model download, using a fake `ICliModelCatalog` (see each command's own test file under `test/DemaConsulting.Speech.Cli.Tests/Commands/ModelCommandsSubsystem/`). -### Addendum (Pass 5): Seam Extension for SynthesisCommandSubsystem +### Addendum (Pass 5): Seam Extension for SynthesisCommandSubsystem and RecognitionCommandSubsystem -Pass 5's `speak` subcommand needs to construct a real `ISpeechSynthesizer`, which requires casting -a resolved model to the library's internal `ISynthesisModel` interface - a cast that is only -compilable inside this same assembly (`DemaConsulting.Speech.Cli`), not inside -`DemaConsulting.Speech.Cli.Tests`, which is not granted `InternalsVisibleTo` access to it. Rather -than introduce a second, competing seam interface for `SynthesisCommandSubsystem`, `ICliModelCatalog` -is extended with two further members that keep the same shape as the original five - a narrow, -purpose-specific operation, never throwing for a genuinely absent capability where the library -itself would return a graceful fallback: +`speak` needs to load a real `ISpeechSynthesizerEngine`, and `recognize` needs to load a real +`ISpeechRecognizerEngine`, each of which requires casting a resolved model to the library's +internal `ISynthesisModel`/`IRecognitionModel` interface - a cast that is only compilable inside +this same assembly (`DemaConsulting.Speech.Cli`), not inside `DemaConsulting.Speech.Cli.Tests`, +which is not granted `InternalsVisibleTo` access to either. Rather than introduce competing seam +interfaces for each subsystem, `ICliModelCatalog` is extended with four further members that keep +the same shape as the original five - narrow, purpose-specific operations, never throwing for a +genuinely absent capability where the library itself would return a graceful fallback: | Member | Returns | Behavior | | --- | --- | --- | | `GetPreferredAudioFormat(descriptor)` | `AudioFormat` | Throws for a non-synthesis model | -| `CreateSynthesizer(descriptor, device, values)` | `ISpeechSynthesizer` | Throws for a non-synthesis model | - -`SpeechModelCatalogAdapter` implements both through a private `RequireSynthesisModel` helper that -performs the `is ISynthesisModel` cast once, throwing a clean, model-id-naming `ArgumentException` -when it fails. `CreateSynthesizer`'s implementation forwards to the library's public -`SpeechSynthesizerFactory.Create(ISynthesisModel, SpeechModelCatalog, IAudioPlaybackDevice, -ISpeechDiagnostics?, IReadOnlyDictionary?)` overload, resolving the model's -installed directory internally via the adapter's own composed catalog. See _SpeechCli -SynthesisCommandSubsystem Design_ for how `SpeakCommand` consumes these two members. +| `CreateSynthesizerEngineAsync(...)` | `Task` | Non-synthesis model throws; no device bound | +| `GetAudioFormat(descriptor)` | `AudioFormat` | Throws for a non-recognition model | +| `CreateRecognizerEngineAsync(...)` | `Task` | Non-recognition model throws; no device bound | + +`SpeechModelCatalogAdapter` implements all four through private `RequireSynthesisModel`/ +`RequireRecognitionModel` helpers that perform the respective `is ISynthesisModel`/ +`is IRecognitionModel` cast once, throwing a clean, model-id-naming `ArgumentException` when it +fails. `CreateSynthesizerEngineAsync`/`CreateRecognizerEngineAsync` forward unchanged to the +library's `SpeechSynthesizerFactory.LoadAsync`/`SpeechRecognizerFactory.LoadAsync` static methods, +each returning an engine with no device bound yet; the caller (`SpeakCommand`/`RecognizeCommand`/ +`AskCommand`) binds the resolved playback/capture device afterward via +`engine.CreateSessionAsync(device, cancellationToken)`. See _SpeechCli SynthesisCommandSubsystem +Design_ and _SpeechCli RecognitionCommandSubsystem Design_ for how each command consumes its two +members. diff --git a/docs/design/speech-cli/recognition-command-subsystem.md b/docs/design/speech-cli/recognition-command-subsystem.md index 8b32f23..f7857bd 100644 --- a/docs/design/speech-cli/recognition-command-subsystem.md +++ b/docs/design/speech-cli/recognition-command-subsystem.md @@ -3,23 +3,25 @@ ### Overview The RecognitionCommandSubsystem implements the one speech-to-text subcommand dispatched to by -`CommandDispatch`: `recognize` - the last of all 10 subcommands to be implemented. It contains two -types and extends `ModelCommandsSubsystem`'s existing catalog seam with two further members: +`CommandDispatch`: `recognize`. It contains two types and extends `ModelCommandsSubsystem`'s +existing catalog seam with two further members: - **`RecognizeCommand`**: implements `recognize --stt-model (--input | --mic) [--capture-device ] [--silence-timeout ] [--start-timeout ] [--stt-param key=value ...] [--interim | --final-only] [--output-text ]` -- **`SilenceTimeoutRecognizerSession`**: a two-phase idle-timeout observer for mic-mode sessions, - arming its timer with a start-timeout grace period before the first recognition result and a - silence-timeout window (resetting on every result) thereafter, stopping the recognizer when the - currently-active window elapses with no reset -- **`ICliModelCatalog.GetAudioFormat`/`CreateRecognizer`**: the two new seam members (see - _Extending the ModelCommandsSubsystem Seam_ below), implemented by `SpeechModelCatalogAdapter` +- **`SilenceTimeoutRecognizerSession`**: a stateless `IAsyncEnumerable` + decorator around a mic-mode `IRecognitionSession`'s `GetResultsAsync`, racing a two-phase idle + timeout (a start-timeout grace period before the first recognition result, a silence-timeout + window resetting on every result thereafter) against each `MoveNextAsync`, stopping the wrapped + session when the currently-active window elapses with no result +- **`ICliModelCatalog.GetAudioFormat`/`CreateRecognizerEngineAsync`**: the two new seam members + (see _Extending the ModelCommandsSubsystem Seam_ below), implemented by + `SpeechModelCatalogAdapter` ### Extending the ModelCommandsSubsystem Seam -`recognize` needs to construct a real `ISpeechRecognizer`, which requires an `IRecognitionModel` - +`recognize` needs to load a real `ISpeechRecognizerEngine`, which requires an `IRecognitionModel` - the library's recognition-capable model interface, exposing `AudioFormat`. Exactly as `SpeakCommand`'s pass established for `ISynthesisModel`, that cast is only safe inside `SpeechModelCatalogAdapter` (same assembly as the library's internal model types), never inside a @@ -29,7 +31,7 @@ are added to `ICliModelCatalog`: | Member | Returns | Behavior | | --- | --- | --- | | `GetAudioFormat(descriptor)` | `AudioFormat` | Throws for a non-recognition model | -| `CreateRecognizer(descriptor, captureDevice, values)` | `ISpeechRecognizer` | Throws for a non-recognition model | +| `CreateRecognizerEngineAsync(...)` | `Task` | Non-recognition model throws; no device bound | `SpeechModelCatalogAdapter` implements both via a private `RequireRecognitionModel` helper, mirroring `RequireSynthesisModel` exactly: it performs the cast once and throws a clean @@ -37,9 +39,11 @@ mirroring `RequireSynthesisModel` exactly: it performs the cast once and throws unreachable through the CLI's own dispatch - `RecognizeCommand` already validates `descriptor.Role == SpeechModelRole.Recognition` itself before calling either new member - but is retained for the same defensive reason `SpeakCommand`'s pass documented. -`CreateRecognizer`'s production implementation forwards to the public -`SpeechRecognizerFactory.Create(IRecognitionModel, SpeechModelCatalog, IAudioCaptureDevice, -ISpeechDiagnostics?, IReadOnlyDictionary?)` overload. +`CreateRecognizerEngineAsync`'s production implementation forwards unchanged to +`SpeechRecognizerFactory.LoadAsync(IRecognitionModel, SpeechModelCatalog, ISpeechDiagnostics?, +IReadOnlyDictionary?, CancellationToken)`, returning an `ISpeechRecognizerEngine` +with no device bound; `RecognizeCommand` later binds the resolved capture device by calling +`engine.CreateSessionAsync(captureDevice, cancellationToken)`. This keeps `RecognizeCommand`'s own logic - argument parsing, input-source resolution, `--stt-param` validation, verbosity filtering, capture device dispatch, disposal ordering, cancellation - fully @@ -53,10 +57,6 @@ Parses its own flags (`--stt-model`, `--input`, `--mic`, `--capture-device`, `-- hand-rolled loop style as every other command in this tool, requiring `--stt-model` and rejecting an unsupported argument or a value-less flag with `ArgumentException`. `--silence-timeout` and `--start-timeout` each additionally require a positive number of seconds when given. -`--start-timeout` follows `--silence-timeout`'s own existing "inert without `--mic`" convention: -it parses and validates as its own positive-number flag but is only ever consulted inside the -`options.Mic && options.SilenceTimeoutSeconds is { } seconds` guard in `Run` - no new cross-flag -validator rejects `--start-timeout` given without `--silence-timeout` or `--mic`. **Validation ordering mirrors `SpeakCommand`'s own verified ordering exactly**: parse-time `--stt-param` token shape is validated inline as each token is parsed; input-source mutual exclusion @@ -73,37 +73,51 @@ role but not yet downloaded (suggesting `download ` by name). Once reso values are validated against the resolved model's own declared parameters, reusing `ParameterBagParser.Resolve` unmodified from the synthesis pass. -**File-input mode (`--input`) requires no explicit "wait until done" loop.** -`ISpeechRecognizer.Start()` calls the supplied capture device's own `Start()` synchronously (not -on a background thread); a `WavFileAudioCaptureDevice`'s own `Start()` is itself fully synchronous -and blocking, delivering every frame before returning. This command subscribes its own handler -directly to the device instance's own `EndOfFileReached` event (a member additional to -`IAudioCaptureDevice`, not the recognizer's internal subscription), and that handler calls -`recognizer.Stop()` - which runs **reentrantly, on the same thread, from inside -`ISpeechRecognizer.Start()`'s own call to the device's `Start()`** - immediately after the last -frame is delivered and immediately before the device's own `Start()` returns. -`ISpeechRecognizer.Stop()`'s drain is a genuine, synchronous block (confirmed by reading -`SherpaOnnxSpeechRecognizer.StopCore`/`WaitForConsumer`, which calls -`consumerTask.GetAwaiter().GetResult()`), so by the time `ISpeechRecognizer.Start()` returns to -this command, every result derived from the whole file has already been raised and the recognizer -has already fully stopped. No settle-wait or fixed sleep is added anywhere in this design; none is -needed. - -**Mic-input mode (`--mic`)** blocks the calling thread on a `ManualResetEventSlim` set either by a -`Ctrl+C` handler or, when `--silence-timeout` was given, by a `SilenceTimeoutRecognizerSession`'s -`TimedOut` event (the session itself already called `Stop()` before raising that event). The -session enforces two distinct idle windows: it arms its timer with `--start-timeout` (defaulting -to `--silence-timeout`'s value when `--start-timeout` is omitted) until the first recognition -result arrives, then re-arms with `--silence-timeout` for every result from the first onward - -giving the user a separate, typically longer grace period to start speaking without weakening the -brief end-of-utterance pause `--silence-timeout` alone controls. `Ctrl+C` -is wired to cooperative cancellation via `Console.CancelKeyPress`, exactly mirroring -`SpeakCommand.Run`'s own subscribe/unsubscribe-in-try/finally pattern. `--capture-device` resolves a real -capture device from the injected `AudioDeviceFactory`, reusing -`DevicesTestCommand.ResolveDeviceSelectionOrThrow` (internal, same assembly, different namespace) -to validate any requested `--capture-device` name before a device is actually created, exactly mirroring -`devices test`'s and `speak`'s own validate-before-create pattern; an unavailable resolved device -throws `InvalidOperationException` suggesting `--input` as an alternative. +**Async composition flow.** `Run` resolves the capture device, then awaits `catalog +.CreateRecognizerEngineAsync(descriptor, parameterValues, cancellationToken)` followed by +`engine.CreateSessionAsync(captureDevice, cancellationToken)`, both bound with `await using` so +they are disposed (session first, then engine, reflecting declaration order reversed) on every +exit path. The returned `IRecognitionSession` is driven uniformly via `StartAsync`/`StopAsync`/ +`GetResultsAsync` in both input modes, rather than the two modes using structurally different +APIs. + +**File-input mode (`--input`) requires no explicit "wait until done" loop, and no wrapper +session.** `IRecognitionSession.StartAsync` drives the supplied `WavFileAudioCaptureDevice`'s own +delivery of every frame synchronously before returning (a `WavFileAudioCaptureDevice`'s own +capture is fully blocking, not threaded), so by the time `StartAsync` returns in file mode, every +frame has already been delivered to the backend. This command follows that `await +session.StartAsync(cancellationToken)` with an explicit drain `await +session.StopAsync(cancellationToken)`, exactly mirroring `IRecognitionSession.StopAsync`'s own +documented drain contract, so any buffered-but-not-yet-decoded audio is flushed through and every +result derived from the whole file is guaranteed to have been produced by the time the pump loop +over `GetResultsAsync` settles. No `SilenceTimeoutRecognizerSession` wrapper, settle-wait, or fixed +sleep is added for file mode; none is needed, since `StartAsync` + the explicit `StopAsync` are +together already sufficient to reach a terminal state. + +**Mic-input mode (`--mic`)** always constructs a `SilenceTimeoutRecognizerSession` - even when +both `--silence-timeout` and `--start-timeout` are omitted (each then defaults to its own +documented default-seconds value) - so listening cannot block `recognize --mic` forever with only +`Ctrl+C` as an escape hatch. The command pumps +`silenceTimeout.GetResultsAsync(cancellationToken)` (rather than `session.GetResultsAsync` +directly) concurrently with `await session.StartAsync(cancellationToken)`, started first as a +genuinely async, non-blocking call so live interim results print as they arrive rather than only +once `StartAsync` returns. The session enforces two distinct idle windows: it races the wrapped +enumerator's `MoveNextAsync` against `--start-timeout` (defaulting to a fixed 8 seconds, +independent of `--silence-timeout`, when `--start-timeout` is omitted) until the first +recognition result arrives, then re-arms with +`--silence-timeout` for every result from the first onward - giving the user a separate, +typically longer grace period to start speaking without weakening the brief end-of-utterance +pause `--silence-timeout` alone controls. `Ctrl+C` is wired to cooperative cancellation via a +synchronous, fire-and-forget `Console.CancelKeyPress` handler that calls `session.StopAsync +(CancellationToken.None)` directly - stopping the session completes its result stream, ending the +pump naturally, with `cancellationToken` forwarded into `GetResultsAsync` as a belt-and-suspenders +safety net for the documented edge case where `StopAsync` called while the session never started +does not itself complete that stream. `--capture-device` resolves a real capture device from the +injected `AudioDeviceFactory`, reusing `DevicesTestCommand.ResolveDeviceSelectionOrThrow` +(internal, same assembly, different namespace) to validate any requested `--capture-device` name +before a device is actually created, exactly mirroring `devices test`'s and `speak`'s own +validate-before-create pattern; an unavailable resolved device throws `InvalidOperationException` +suggesting `--input` as an alternative. **`--interim`/`--final-only` console UX.** The default (neither flag) prints both: an interim (`IsFinal == false`) result is written with a carriage-return overwrite and no trailing newline, @@ -129,49 +143,60 @@ not an error, when a file destination is also given. **Disposal.** Neither `IAudioCaptureDevice`, `WavFileAudioCaptureDevice`, nor the real PortAudio-backed capture device implement `IDisposable` (confirmed directly from all three source files), so - unlike `speak`'s playback-device side - this command never needs a conditional -capture-device disposal cast. Only the recognizer and, in mic mode with a silence timeout, the -`SilenceTimeoutRecognizerSession` need disposal, both handled in nested `finally` blocks so every -exit path (EOF stop, silence-timeout stop, `Ctrl+C`, or an error) disposes them exactly once. +capture-device disposal cast. The engine and session are both `IAsyncDisposable` and bound with +`await using`, guaranteeing exactly-once disposal on every exit path (file-mode drain completion, +mic-mode silence/start timeout, `Ctrl+C`, or an error) with no nested `try`/`finally` needed for +them. ### SilenceTimeoutRecognizerSession -A pure event-driven observer composed alongside a recognizer, not a decorator around its -lifecycle API: it never intercepts `Start()`/`Stop()` calls made by its owner, and calls -`ISpeechRecognizer.Stop()` itself only proactively, on timeout. It enforces two distinct, -sequential idle windows using the constructor's `startTimeout` parameter (defaulting to -`idleTimeout` when omitted) only once, to arm the single-shot idle timer at construction, before -`ResultReceived` is even subscribed; the single stored `_idleTimeout` field (the -`--silence-timeout` value) then re-arms the timer (`Change(idleTimeout, -Timeout.InfiniteTimeSpan)`) on every subsequent `ResultReceived` event, partial or final, starting -with the very first. `startTimeout` itself is never stored as a field - it is only read once, -inline, at construction, since nothing after that point ever needs it again. When the timer fires -with no reset since it was last armed, it calls `Stop()`, then raises its own `TimedOut` event. No -additional "first result seen" boolean flag is needed to implement this phase transition: -construction and `OnResultReceived` are already distinct call sites, so arming with `startTimeout` -once at construction and unconditionally re-arming with `_idleTimeout` on every `OnResultReceived` -call naturally implements "start-timeout governs only the pre-first-result window; silence-timeout -governs every re-arm from the first result onward" with zero new mutable state and zero new -lock-guarded reads/writes - the existing `_gate`/`_idleCallbackDone` concurrency design is -unchanged. - -The idle timer is implemented with the injectable `System.TimeProvider` abstraction (available in -the BCL since .NET 8, requiring no new package reference) rather than a hard-coded -`Timer`/`Thread.Sleep`, so a unit test can exercise the full idle/reset/timeout sequence -deterministically with a fake `TimeProvider` and no real wall-clock delay. A `volatile bool -_isDisposed` flag guards against a timer callback racing a concurrent `Dispose()` call, since the -callback runs on a thread-pool thread independent of the constructing/disposing thread; `Dispose()` -itself is idempotent. +A stateless `IAsyncEnumerable` decorator wrapping an +`IRecognitionSession`'s `GetResultsAsync`, not a background event-subscriber: it never intercepts +`StartAsync`/`StopAsync` calls made by its owner on the underlying session directly, and calls +`IRecognitionSession.StopAsync` itself only proactively, on timeout, from inside its own iterator. +Its `GetResultsAsync(CancellationToken)` is an `async IAsyncEnumerable` +method that obtains `IAsyncEnumerator` from the wrapped session (disposed +via the compiler-generated `await using` on that enumerator, needing no `IDisposable` of its own) +and, in a loop, races `enumerator.MoveNextAsync().AsTask()` against `Task.Delay(timeout, +_timeProvider, cancellationToken)` via `Task.WhenAny`: + +- If the enumerator wins, the current result is yielded and the next iteration re-arms the + timeout with `_idleTimeout` (the `--silence-timeout` value). +- If the delay wins first, and the session has not already been stopped by a prior timeout on this + same enumeration, it awaits `_session.StopAsync(CancellationToken.None)` then raises `TimedOut` + once - then **continues looping** rather than returning immediately, so any already-buffered or + in-flight trailing result the stop drains through still surfaces to the caller before the + enumerable genuinely completes. + +It enforces two distinct, sequential idle windows using the constructor's `startTimeout` parameter +(defaulting to `idleTimeout` when omitted): the very first race uses `startTimeout`, before any +result has been yielded; every subsequent race, from the first yielded result onward, uses +`idleTimeout`. No additional "first result seen" boolean flag beyond a simple loop-local variable +is needed to implement this phase transition, since the loop already distinguishes "before the +first yield" from "after". + +The idle timeout is implemented against the injectable `System.TimeProvider` abstraction +(available in the BCL since .NET 8, requiring no new package reference) via `Task.Delay(TimeSpan, +TimeProvider, CancellationToken)` rather than a hard-coded `Timer`/`Thread.Sleep`, so a unit test +can exercise the full idle/reset/timeout sequence deterministically with a fake `TimeProvider` +(`FakeTimeProvider`, driving its own `FakeTimer`) and no real wall-clock delay. Because this type +is now a pure, stateless async-enumerable decorator with cleanup delegated entirely to the inner +enumerator's own `await using` disposal, it needs **no locks, no wait-handles, and no +`IDisposable`/`Dispose()` of its own** - the prior timer/event-based design's `_gate`/ +`_idleCallbackDone` concurrency guards no longer exist because there is no longer a +background-thread timer callback racing a concurrent caller-driven disposal. ### Interactions with Other Units `RecognizeCommand` depends on the `ICliModelCatalog` seam (both its original five members, the two added by the synthesis pass, and the two added by this pass), the library's public -`AudioFormat`/`ISpeechRecognizer`/`SpeechRecognitionEvent`/`SpeechRecognitionResult` types, -`AudioDeviceFactory` from the library's audio subsystem, `WavFileAudioCaptureDevice`, and reuses -`DeviceCommandsSubsystem`'s `DevicesTestCommand.ResolveDeviceSelectionOrThrow` internal helper and -`SynthesisCommandSubsystem`'s `ParameterBagParser` unmodified, rather than duplicating either. -`SilenceTimeoutRecognizerSession` depends only on the library's public `ISpeechRecognizer`/ -`SpeechRecognitionEvent` types and the BCL `System.TimeProvider` abstraction. None of this -subsystem touches `IRecognitionModel` or any other internal library type directly - that access is -confined entirely to `SpeechModelCatalogAdapter`, exactly as `ModelCommandsSubsystem`'s original -seam already established. +`AudioFormat`/`ISpeechRecognizerEngine`/`IRecognitionSession`/`SpeechRecognitionEvent`/ +`SpeechRecognitionResult` types, `AudioDeviceFactory` from the library's audio subsystem, +`WavFileAudioCaptureDevice`, and reuses `DeviceCommandsSubsystem`'s `DevicesTestCommand +.ResolveDeviceSelectionOrThrow` internal helper and `SynthesisCommandSubsystem`'s +`ParameterBagParser` unmodified, rather than duplicating either. `SilenceTimeoutRecognizerSession` +depends only on the library's public `IRecognitionSession`/`SpeechRecognitionEvent` types and the +BCL `System.TimeProvider` abstraction. None of this subsystem touches `IRecognitionModel` or any +other internal library type directly - that access is confined entirely to +`SpeechModelCatalogAdapter`, exactly as `ModelCommandsSubsystem`'s original seam already +established. diff --git a/docs/design/speech-cli/synthesis-command-subsystem.md b/docs/design/speech-cli/synthesis-command-subsystem.md index dd8e3c8..b80e890 100644 --- a/docs/design/speech-cli/synthesis-command-subsystem.md +++ b/docs/design/speech-cli/synthesis-command-subsystem.md @@ -12,8 +12,8 @@ small seam over real playback-device resolution: [--playback-device ] [--tts-param key=value ...] [--no-tags]` - **`ParameterBagParser`**: parses and validates repeatable `--tts-param key=value` tokens against a resolved model's declared parameters -- **`ICliModelCatalog.GetPreferredAudioFormat`/`CreateSynthesizer`**: the two new seam members - (see _Extending the ModelCommandsSubsystem Seam_ below), implemented by +- **`ICliModelCatalog.GetPreferredAudioFormat`/`CreateSynthesizerEngineAsync`**: the two new seam + members (see _Extending the ModelCommandsSubsystem Seam_ below), implemented by `SpeechModelCatalogAdapter` - **`ICliPlaybackDeviceSource`**/**`AudioDeviceFactoryPlaybackDeviceSource`**: a small CLI-owned seam over real-playback-device resolution (see _Deterministic Playback-Device Resolution_ @@ -22,10 +22,10 @@ small seam over real playback-device resolution: ### Extending the ModelCommandsSubsystem Seam -`speak` needs to construct a real `ISpeechSynthesizer`, which requires an `ISynthesisModel` - +`speak` needs to load a real `ISpeechSynthesizerEngine`, which requires an `ISynthesisModel` - the library's synthesis-capable model interface. `ISynthesisModel` itself, and the members -`speak` needs from it (`PreferredAudioFormat`, and the constructor path -`SpeechSynthesizerFactory.Create` requires), are only reachable once a `SpeechModelDescriptor.Model` +`speak` needs from it (`PreferredAudioFormat`, and the loader path +`SpeechSynthesizerFactory.LoadAsync` requires), are only reachable once a `SpeechModelDescriptor.Model` is known to implement that interface. That cast is safe to perform inside `SpeechModelCatalogAdapter` (which lives in the same assembly as the library's own internal model types, so its `is ISynthesisModel` check compiles), but is **not** safe to perform inside a @@ -41,7 +41,7 @@ are added to `ICliModelCatalog`: | Member | Returns | Behavior | | --- | --- | --- | | `GetPreferredAudioFormat(descriptor)` | `AudioFormat` | Throws for a non-synthesis model | -| `CreateSynthesizer(descriptor, device, values)` | `ISpeechSynthesizer` | Throws for a non-synthesis model | +| `CreateSynthesizerEngineAsync(...)` | `Task` | Non-synthesis model throws; no device bound | `SpeechModelCatalogAdapter` implements both via a private `RequireSynthesisModel` helper that performs the cast once and throws a clean `ArgumentException` naming the offending model id when @@ -49,13 +49,14 @@ it fails. In production this branch is unreachable through the CLI's own dispatc already validates `descriptor.Role == SpeechModelRole.Synthesis` itself before calling either new member - but is retained because `ICliModelCatalog` is an interface any future caller could misuse, and a defensive, well-worded exception is preferable to an unhandled `InvalidCastException` -surfacing as a stack trace. `CreateSynthesizer`'s production implementation forwards to the -public `SpeechSynthesizerFactory.Create(ISynthesisModel, SpeechModelCatalog, IAudioPlaybackDevice, -ISpeechDiagnostics?, IReadOnlyDictionary?)` overload, which resolves the model's -installed directory internally via the composed catalog's own store and never throws for "not -installed" or "no playback device" - it instead returns the library's own -`UnavailableSpeechSynthesizer.Instance` singleton, a graceful-degradation contract `SpeakCommand` -relies on rather than duplicates. +surfacing as a stack trace. `CreateSynthesizerEngineAsync`'s production implementation forwards +unchanged to `SpeechSynthesizerFactory.LoadAsync(ISynthesisModel, SpeechModelCatalog, +ISpeechDiagnostics?, IReadOnlyDictionary?, CancellationToken)`, returning an +`ISpeechSynthesizerEngine` with no device bound yet; `SpeakCommand` later binds the resolved +playback device by calling `engine.CreateSessionAsync(playbackDevice, cancellationToken)`. The +loader never throws for "not installed" or "no playback device" - it instead returns the +library's own honestly-unavailable engine per `SpeechSynthesizerFactory`'s own contract, a +graceful-degradation contract `SpeakCommand` relies on rather than duplicates. This keeps `SpeakCommand`'s own logic - argument parsing, text-source resolution, `--tts-param` validation, tag stripping, device dispatch, disposal ordering, cancellation - fully unit-testable @@ -123,7 +124,7 @@ _SpeechCli ModelCommandsSubsystem Design_). Resolution switches on the parameter - **`BooleanParameter`**: parsed via `bool.TryParse` (case-insensitive) - boxed as `bool` An unrecognized `--tts-param` key throws `ArgumentException` naming the key. This is a deliberate -divergence from `SpeechSynthesizerFactory.Create`'s own library-level contract, which silently +divergence from `SpeechSynthesizerFactory.LoadAsync`'s own library-level contract, which silently ignores (and Info-logs) an unrecognized parameter key rather than throwing: at the library level, silently ignoring an unrecognized key is the right graceful-degradation choice when parameter values are supplied programmatically and may target multiple engine versions, but at the CLI an @@ -177,16 +178,19 @@ devices before a device is actually created, exactly mirroring `devices test`'s validate-before-create pattern; an unavailable resolved device throws `InvalidOperationException` suggesting `--output-audio` as an alternative. -**Disposal ordering.** `ISpeechSynthesizer.Dispose()` does not dispose the playback device it was -constructed over (confirmed against the library's own `SherpaOnnxSpeechSynthesizer.Dispose()` -implementation, which disposes only its own inference engine) - so `SpeakCommand` disposes the -synthesizer first, in an inner `finally`, guaranteeing any in-flight playback/write has genuinely -quiesced, then disposes the playback device, in an outer `finally`, via `(playbackDevice as -IDisposable)?.Dispose()`. The conditional cast is necessary because `IAudioPlaybackDevice` itself -does not declare `IDisposable` - a real device manages its own native stream lifecycle entirely -through `Start()`/`Stop()` - but `WavFileAudioPlaybackDevice` (used for `--output-audio`) additionally -implements `IDisposable` to finalize its RIFF header, and the outer `finally` must dispose it -unconditionally when present while remaining a safe no-op for every other device kind. +**Async composition flow and disposal ordering.** `Run` resolves the playback device, then awaits +`catalog.CreateSynthesizerEngineAsync(descriptor, parameterValues, cancellationToken)` followed by +`engine.CreateSessionAsync(playbackDevice, cancellationToken)`, both bound with `await using` +immediately after the playback device's own conditional `using var playbackDeviceLease = +playbackDevice as IDisposable` lease - so disposal, in declaration order reversed, runs +**session first, then engine, then the playback device lease**, guaranteeing any in-flight +playback/write has genuinely quiesced (session disposal) before the engine is torn down, which in +turn happens before the device itself is disposed. The conditional cast for the playback-device +lease is necessary because `IAudioPlaybackDevice` itself does not declare `IDisposable` - a real +device manages its own native stream lifecycle entirely through its session's `SpeakAsync`-driven +playback - but `WavFileAudioPlaybackDevice` (used for `--output-audio`) additionally implements +`IDisposable` to finalize its RIFF header, and the lease must dispose it unconditionally when +present while remaining a safe no-op for every other device kind. **Cancellation.** `Ctrl+C` is wired to a `CancellationTokenSource` via `Console.CancelKeyPress`, mirroring `DownloadCommand.Run`'s exact subscribe/unsubscribe-in-try/finally pattern (see @@ -197,9 +201,9 @@ propagating as a stack trace. ### Interactions with Other Units `SpeakCommand` depends on the `ICliModelCatalog` seam (both its original five members and the two -new ones added by this pass), the library's public `AudioFormat`/`ISpeechSynthesizer`/ -`AudioTagParser`/`TaggedTextSpan` types, its own `ICliPlaybackDeviceSource` seam (composed over -`AudioDeviceFactory` from the library's audio subsystem in production, via +new ones added by this pass), the library's public `AudioFormat`/`ISpeechSynthesizerEngine`/ +`ISynthesisSession`/`AudioTagParser`/`TaggedTextSpan` types, its own `ICliPlaybackDeviceSource` +seam (composed over `AudioDeviceFactory` from the library's audio subsystem in production, via `AudioDeviceFactoryPlaybackDeviceSource`), `WavFileAudioPlaybackDevice`, and reuses `DeviceCommandsSubsystem`'s `DevicesTestCommand.ResolveDeviceSelectionOrThrow` internal helper rather than duplicating device resolution logic. `ParameterBagParser` depends only on the diff --git a/docs/design/speech-demo.md b/docs/design/speech-demo.md index 4629a4d..63a9ce6 100644 --- a/docs/design/speech-demo.md +++ b/docs/design/speech-demo.md @@ -63,10 +63,10 @@ external interfaces are its user interface and the interfaces it consumes. | `AudioDeviceSelection` | Outbound | Value construction | Name-only device identity | | `SpeechModelCatalog.Enumerate()` | Outbound | Method call/return | Consumed; never throws | | `SpeechModelCatalog.DownloadAsync(...)` | Outbound | Method call/return | Throws for unknown model id | -| `SpeechSynthesizerFactory.Create(...)` | Outbound | Method call/return | Consumed; never throws | -| `SpeechRecognizerFactory.Create(...)` | Outbound | Method call/return | Consumed; never throws | -| `ISpeechSynthesizer.SpeakAsync(...)` | Outbound | Method call/return | Consumed; cancellation stops playback | -| `ISpeechRecognizer.Start()` / `Stop()` / `ResultReceived` | Outbound | Method call/event | Consumed; UI-marshaled | +| `SpeechSynthesizerFactory.LoadAsync(...)` | Outbound | Method call/return | Consumed; never throws | +| `SpeechRecognizerFactory.LoadAsync(...)` | Outbound | Method call/return | Consumed; never throws | +| `ISynthesisSession.SpeakAsync(...)` | Outbound | Method call/return | Consumed; cancellation stops playback | +| `IRecognitionSession.Start/Stop/GetResultsAsync()` | Outbound | `IAsyncEnumerable` | UI-marshaled via `StateChanged` | ## Dependencies @@ -140,7 +140,7 @@ N/A - SpeechDemo provides no safety-critical functionality requiring risk contro 1. **Input**: The synthesis panel is constructed, or its refresh, Play, or Stop command is invoked 2. **Composition**: The demo's synthesizer session seam resolves the selected model's installed - directory and forwards to `SpeechSynthesizerFactory.Create(...)`, narrowing the model to the + directory and forwards to `SpeechSynthesizerFactory.LoadAsync(...)`, narrowing the model to the library's synthesis role 3. **Playback**: The composed synthesizer speaks the entered text (which may contain inline Natural Language Audio Tags) through the selected playback device, reporting each lifecycle @@ -153,7 +153,7 @@ N/A - SpeechDemo provides no safety-critical functionality requiring risk contro 1. **Input**: The recognition panel is constructed, or its refresh, Start, or Stop command is invoked 2. **Composition**: The demo's recognizer session seam resolves the selected model's installed - directory and forwards to `SpeechRecognizerFactory.Create(...)`, narrowing the model to the + directory and forwards to `SpeechRecognizerFactory.LoadAsync(...)`, narrowing the model to the library's recognition role 3. **Streaming**: The composed recognizer streams audio from the selected capture device, raising progressive partial and final results diff --git a/docs/design/speech-demo/model-settings-subsystem.md b/docs/design/speech-demo/model-settings-subsystem.md index 6597820..8ae1672 100644 --- a/docs/design/speech-demo/model-settings-subsystem.md +++ b/docs/design/speech-demo/model-settings-subsystem.md @@ -85,13 +85,13 @@ an expected state. a dictionary keyed by `Id`. The design describes this untyped key-value bag as the interface between a host's settings UI and per-model synthesis/recognition parameter handling. As of this pass, `SynthesisPanelViewModel.PlayAsync` forwards this bag as the `parameterValues` argument to -`ISynthesizerSessionFactory.Create`, which threads it through to the library's -`SpeechSynthesizerFactory.Create` and, from there, to `ISynthesisModel.ResolveSpeakerId` — so +`ISynthesizerSessionFactory.LoadAsync`, which threads it through to the library's +`SpeechSynthesizerFactory.LoadAsync` and, from there, to `ISynthesisModel.ResolveSpeakerId` — so selecting a voice in the TTS panel now genuinely changes synthesized output. The recognition -panel does not yet consume this bag: `ISpeechRecognizer`'s contract accepts no per-call parameter -bag on any member, so for recognition it remains built and fully exercised by unit tests, ready -for a future library revision exposing a per-call recognition parameter surface. This is stated -plainly here rather than silently implied. +panel does not yet consume this bag: `IRecognitionSession`'s contract accepts no per-call +parameter bag on any member, so for recognition it remains built and fully exercised by unit +tests, ready for a future library revision exposing a per-call recognition parameter surface. +This is stated plainly here rather than silently implied. ### Interactions with Other Units diff --git a/docs/design/speech-demo/recognition-panel-subsystem.md b/docs/design/speech-demo/recognition-panel-subsystem.md index 11f1f2f..7809619 100644 --- a/docs/design/speech-demo/recognition-panel-subsystem.md +++ b/docs/design/speech-demo/recognition-panel-subsystem.md @@ -7,11 +7,15 @@ The RecognitionPanelSubsystem provides the demo's speech-to-text panel. It contains the following units: -- **IRecognizerSessionFactory** / **RecognizerSessionFactory**: the demo-owned recognizer +- **IRecognizerSessionFactory** / **RecognizerSessionFactory**: the demo-owned recognizer-engine composition seam and its real implementation over the library's `SpeechModelStore` and `SpeechRecognizerFactory` - **RecognitionPanelViewModel**: the panel's presentation state — the installed recognition - models, the Start/Stop streaming lifecycle, and the progressive partial-then-final transcript + models, the async Start/Stop streaming lifecycle over an `ISpeechRecognizerEngine` and an + `IRecognitionSession`, and the progressive partial-then-final transcript +- **CaptureDebugRecorder** / **WavFileWriter**: internal, opt-in, TEMPORARY diagnostic + instrumentation (off by default) that records raw captured audio to a `.wav` file for offline + investigation of a reported live-microphone bug; see "Capture Debug Recording" below ### Interfaces @@ -21,37 +25,40 @@ The subsystem exposes one presentation surface, `RecognitionPanelViewModel`, con - `IModelCatalogService` and `IAudioDeviceService` — the shell-provided demo services for listing installed models and creating capture devices - the shared `DeviceSelectionViewModel` — the capture-device picker's presentation state -- `IRecognizerSessionFactory` (below) — the demo-owned recognizer composition seam +- `IRecognizerSessionFactory` (below) — the demo-owned recognizer-engine composition seam -The library composes a recognizer through the static -`SpeechRecognizerFactory.Create(IRecognitionModel, string, IAudioCaptureDevice, ISpeechDiagnostics?)` +The library composes a recognizer engine through the static +`SpeechRecognizerFactory.LoadAsync(IRecognitionModel, SpeechModelStore, ISpeechDiagnostics?, +IReadOnlyDictionary?, CancellationToken)` method, which requires an `IRecognitionModel` — an interface whose members are partly `internal` to the library, so only the library's own assemblies can implement it. A demo-owned seam therefore accepts the common `ISpeechModel` contract instead and performs the narrowing itself: | Member | Returns | Behavior | | --- | --- | --- | -| `Create(model, captureDevice)` | `ISpeechRecognizer` | Never throws | +| `LoadAsync(model, cancellationToken)` | `Task` | Never throws | This is what lets a demo test substitute a plain, publicly implementable fake model for every scenario — including "wrong role" — with no `InternalsVisibleTo` grant from the library, and works around the static factory method itself not being substitutable in a ViewModel unit test. +Only the engine is composed through this seam; a capture device is bound later, per run, through +`ISpeechRecognizerEngine.CreateSessionAsync` directly against the returned engine. ### Design `RecognizerSessionFactory` resolves the model's installed-files directory from the shared `SpeechModelStore`, narrows the model to `IRecognitionModel`, and forwards to -`SpeechRecognizerFactory.Create`, inheriting that factory's "nothing throws at composition" +`SpeechRecognizerFactory.LoadAsync`, inheriting that factory's "nothing throws at composition" contract. A model that declares a role other than recognition — and is therefore not an -`IRecognitionModel` — returns the library's own `UnavailableSpeechRecognizer.Instance`, exactly -like a model that is not installed, rather than throwing: a host that lets a user choose an -installed model with the wrong role must still get a working, if unavailable, recognizer back. +`IRecognitionModel` — returns the library's own `UnavailableSpeechRecognizerEngine.Instance`, +exactly like a model that is not installed, rather than throwing: a host that lets a user choose +an installed model with the wrong role must still get a working, if unavailable, engine back. #### RecognitionPanelViewModel | Member | Type | Purpose | | --- | --- | --- | -| `NoModelsMessage` etc. | `const string` | Explanation per honest outcome (no models, no selection, no device, error) | +| `NoModelsMessage` etc. | `const string` | Honest-outcome text (no models/selection/device, unavailable, faulted) | | `AvailableModels` | `ObservableCollection` | The installed recognition models | | `SelectedModel` | `ISpeechModel?` | The chosen model | | `State` | `RecognitionStreamingState` | The current lifecycle state: `Idle`, `Listening`, `Error` | @@ -61,14 +68,14 @@ installed model with the wrong role must still get a working, if unavailable, re | `HasModels` / `CanStart` / `CanStop` | `bool` | Derived enablement values | | `CanChangeModel` | `bool` | `true` only when `State != Listening` | | `RefreshCommand` | generated command | Re-reads the catalog | -| `StartCommand` | generated command | Begins a streaming session | -| `StopCommand` | generated command | Ends the in-flight session | +| `StartCommand` | generated async command | Begins a streaming session | +| `StopCommand` | generated async command (`AllowConcurrentExecutions`) | Ends the in-flight session | -Implements `IDisposable`: disposing releases a cached recognizer and unsubscribes its events, -and unsubscribes from `IModelCatalogService.ModelInstalled` and the shared -`DeviceSelectionViewModel`'s property-change notifications (both subscribed in the constructor - -see below), all idempotently, so a shell shutting down never leaks unmanaged inference resources, -a live capture device, or a stale event subscription. +Implements `IAsyncDisposable`: disposing stops and releases an active session and the cached +engine, and unsubscribes from `IModelCatalogService.ModelInstalled` and the shared +`DeviceSelectionViewModel`'s property-change notifications and pre-refresh hook (both registered +in the constructor - see below), all idempotently, so a shell shutting down never leaks unmanaged +inference resources, a live capture device, or a stale event subscription. **Model filtering and refresh.** `Refresh()` lists only descriptors whose role is `Recognition` and whose state is `Downloaded`, mirroring the synthesis panel's own filtering and refresh @@ -79,77 +86,94 @@ honestly transcribe audio. `IModelCatalogService.ModelInstalled`. The handler ignores any event whose `ModelInstalledEventArgs.Role` is not `SpeechModelRole.Recognition`, and otherwise calls `Refresh()` marshaled onto the UI thread through a `SynchronizationContext` captured at -construction (falling back to calling `Refresh()` synchronously when none was captured) - the -same pattern already used for `ResultReceived` below, since the event's raising thread is not -otherwise guaranteed. This is what lets a newly downloaded recognition model completed from the -Model Catalog panel appear in `AvailableModels` automatically, without a manual Refresh click or -app restart. `Dispose()` unsubscribes this handler. - -**Recognizer reuse.** Per `ISpeechRecognizer`'s own "hot reuse" guidance - composing a recognizer -is the expensive step (it loads the model into native memory), while `Start()`/`Stop()` are cheap -and may be called repeatedly on the same instance - this ViewModel caches at most one recognizer -at a time, bound to the model/capture-device pair it was composed for, and reuses it across many -Start/Stop clicks instead of composing (and reloading the model) on every click. The cache is -invalidated (the recognizer is disposed and the field cleared, so the next Start composes a fresh -one) in exactly three cases, each of which genuinely requires a different recognizer/device pair: -the selected model changes (`OnSelectedModelChanged`), the selected capture device changes -(`OnDeviceSelectionChanged`, subscribed to the shared `DeviceSelectionViewModel`'s -`PropertyChanged`), or a device refresh is pending (the pre-refresh hook below). A recognizer that -reports `SpeechRecognizerUnavailableException` from `Start()` is also invalidated immediately, -rather than retried, since that failure means the specific cached instance is now known-broken. - -Both `OnSelectedModelChanged` and `OnDeviceSelectionChanged` call `Stop()` first when `CanStop` is -`true`, before invalidating - defensively, not merely for tidiness: `SelectedModel` and the shared -`DeviceSelectionViewModel.SelectedCaptureDevice`/`CaptureSelection` both have public setters and -are only *disabled in the view* while listening (`CanChangeModel` for the model picker; the -capture picker is never disabled at all), so either change can genuinely arrive while a session is -active - not just programmatically, but from the capture picker, which has no `Listening` guard. -Disposing the recognizer without stopping it first would leave `State` stuck at `Listening` -forever, since a later `Stop()` would see no cached recognizer and no-op while the native -recognizer kept running, unreachable and unstoppable from the UI. +construction (falling back to calling `Refresh()` synchronously when none was captured). This is +what lets a newly downloaded recognition model completed from the Model Catalog panel appear in +`AvailableModels` automatically, without a manual Refresh click or app restart. `DisposeAsync()` +unsubscribes this handler. + +**Engine/session reuse.** Per `ISpeechRecognizerEngine`'s own reuse guidance, this ViewModel +loads an engine at most once per selected model and reuses it across many Start/Stop cycles, +instead of reloading its model on every click. Because an `IRecognitionSession` is single-use, a +fresh session is created from the cached engine for every Start and released (via +`InvalidateSessionAsync`) before the next one is created. The cached engine is invalidated (via +`InvalidateEngineAsync` - which first releases the session, then disposes and clears the engine) +only when the selected model changes; the cached session alone is invalidated (released, keeping +the engine) whenever the selected capture device changes or the shared device-selection panel +forces a device refresh - a session, not an engine, is bound to a capture device. + +Both `OnSelectedModelChanged` and `OnDeviceSelectionChanged` call `StopAsync()` first when +`CanStop` is `true`, before invalidating - defensively, not merely for tidiness: `SelectedModel` +and the shared `DeviceSelectionViewModel.SelectedCaptureDevice`/`CaptureSelection` both have +public setters and are only *disabled in the view* while listening (`CanChangeModel` for the +model picker; the capture picker is never disabled at all), so either change can genuinely arrive +while a session is active - not just programmatically, but from the capture picker, which has no +`Listening` guard. Disposing a session without stopping it first would leave `State` stuck at +`Listening` forever. **Stops deterministically before a shared device refresh.** The constructor also registers a -pre-refresh hook with the shared `DeviceSelectionViewModel` via `RegisterPreRefreshHook`, -mirroring the `ModelInstalled` subscription precedent above. The hook calls `Stop()` when -`CanStop` is `true`, then unconditionally invalidates the cached recognizer - its bound capture -device is about to become stale the instant the refresh completes, so it must not be reused - all -wrapped to satisfy the hook's `Func` contract by returning `Task.CompletedTask`: `Stop()` is -fully synchronous down to the capture device's own closure (it blocks on draining the recognizer -before returning), so no real awaiting is ever needed here. This is what lets the -`DeviceSelectionViewModel.Refresh()` clicked from the "Refresh devices" button stop an actively -listening session deterministically before the shared device table is re-scanned, rather than -relying on the `AudioDeviceInUseException` fallback. `Dispose()` unregisters this hook. - -**Start algorithm.** `Start()`: +pre-refresh hook (`StopBeforeDeviceRefreshAsync`) with the shared `DeviceSelectionViewModel` via +`RegisterPreRefreshHook`, mirroring the `ModelInstalled` subscription precedent above. The hook +delegates to the same stop-then-invalidate-session path used for a capture-device change, so an +actively listening session is stopped and released before the shared device table is re-scanned + +- its bound capture device would otherwise become stale the instant the refresh completes. This +is what lets the `DeviceSelectionViewModel.Refresh()` clicked from the "Refresh devices" button +stop an actively listening session deterministically, rather than relying on the +`AudioDeviceInUseException` fallback. `DisposeAsync()` unregisters this hook. + +**State derivation.** `RecognitionStreamingState` is never assigned ad hoc at each call site; +`MapSessionState` is the single, pure mapping from `RecognitionSessionState` (`Starting`, +`Running`, `Stopping` map to `Listening`; `Faulted` maps to `Error`; anything else maps to +`Idle`), applied every time `IRecognitionSession.StateChanged` raises. Because `StateChanged` and +the results pumped from `GetResultsAsync` are documented to raise/resume from the session's own +background thread, `StartAsync` captures the UI thread's `SynchronizationContext` before the +session starts, and `OnSessionStateChanged` posts every transition through it (falling back to +applying it synchronously when none was captured) before `ApplySessionStateChanged` touches +`State`; a session reporting `Faulted` additionally sets `StatusMessage` to +`SessionFaultedMessage`. + +**Start algorithm.** `StartAsync()`: 1. Reports `NoModelSelectedMessage` and enters `Error` when no model is selected -2. When no recognizer is cached: creates the capture device through `IAudioDeviceService`; - reports `NoCaptureDeviceMessage` and enters `Error` when it is unavailable; composes a - recognizer through `IRecognizerSessionFactory`; disposes it and reports - `RecognizerUnavailableMessage` and enters `Error` when it honestly reports itself unavailable; - otherwise subscribes to `ResultReceived` and caches both the recognizer and its capture device -3. Otherwise reuses the cached recognizer unchanged, skipping composition entirely -4. Clears `Finals`, `Partial`, and `StatusMessage`; captures the current `SynchronizationContext`; - and calls `Start()` on the (cached or newly composed) recognizer -5. On success, enters `Listening`; on `SpeechRecognizerUnavailableException` (the recognizer - reported itself available but its capture device failed to start), invalidates the cached - recognizer (see "Recognizer reuse" above), reports the exception's message, and enters `Error` - -**Transcript sequencing.** `ResultReceived` is documented to raise from the recognizer's own -background decoding thread, so `OnResultReceived` posts each event through the UI thread's -captured `SynchronizationContext` (falling back to applying it synchronously when none was -captured) before `ApplyResult` touches any observable property — mirroring the library's own -`ConfigureAwait(true)` marshaling used elsewhere in this demo. `ApplyResult` commits a final -result to `Finals` and clears `Partial`, or replaces `Partial` with a provisional one: exactly the -sequencing a live captioning display depends on, so the trailing guess is replaced rather than -accumulated as noise, and a finished utterance becomes a stable committed line the instant the -recognizer considers it final. `BuildTranscriptText()` renders every committed line followed by -any in-progress partial. - -**Stop algorithm.** `Stop()` is a safe no-op when nothing is cached. Otherwise it stops the -recognizer - without unsubscribing or disposing it, since it remains cached for reuse (see -"Recognizer reuse" above) - clears the captured context, enters `Idle`, and reports -`StoppedMessage`. +2. Creates the capture device through `IAudioDeviceService` before paying for the comparatively + expensive async engine load, so an honest "no device" outcome does not depend on what the + engine-loading seam happens to return for this combination; reports `NoCaptureDeviceMessage` + and enters `Error` when it is unavailable +3. When no engine is cached for the selected model: invalidates any stale engine, then loads one + through `IRecognizerSessionFactory.LoadAsync`; disposes it and reports + `RecognizerUnavailableMessage` and enters `Error` when it honestly reports itself unavailable +4. Otherwise reuses the cached engine unchanged, skipping the load entirely +5. Releases any previous (single-use) session via `InvalidateSessionAsync`, then creates a fresh + session through `engine.CreateSessionAsync(captureDevice, ...)`; on + `RecognitionEngineBusyException`, reports the exception's message and enters `Error` +6. Clears `Finals`, `Partial`, and `StatusMessage`; captures the current + `SynchronizationContext`; subscribes `StateChanged`; and calls `session.StartAsync()` +7. On `SpeechRecognizerUnavailableException` from `StartAsync()`, unsubscribes, disposes the + session, reports the exception's message, and enters `Error` +8. On success, starts a background pump task (`PumpResultsAsync`) over + `session.GetResultsAsync`, stored for later awaiting/canceling on Stop, invalidation, or + disposal + +**Transcript sequencing.** `PumpResultsAsync` runs an `await foreach` over +`session.GetResultsAsync(cancellationToken)`, relying on each continuation naturally resuming on +the UI thread context captured by `StartAsync` rather than explicit marshaling, and applies each +result via `ApplyResult`: a final result is committed to `Finals` and `Partial` is cleared; a +provisional result replaces `Partial`. This is exactly the sequencing a live captioning display +depends on, so the trailing guess is replaced rather than accumulated as noise, and a finished +utterance becomes a stable committed line the instant the recognizer considers it final. The +pump's own `cancellationToken` ends only the enumeration (not the session) when the session is +released while still producing results; `OperationCanceledException` from that is swallowed as +expected invalidation, while `RecognitionSessionFaultedException`/ +`SpeechRecognizerUnavailableException` are reported through `StatusMessage`. +`BuildTranscriptText()` renders every committed line followed by any in-progress partial. + +**Stop algorithm.** `StopAsync()` (`AllowConcurrentExecutions = true`, so a concurrent second +call completes without racing to double-dispose) is a safe no-op when nothing is cached. +Otherwise it calls `session.StopAsync()`, awaits the pump task to fully drain, finalizes any +diagnostic capture recording, clears the captured UI context, and - if not already in `Error` - +reports `StoppedMessage`; the session itself remains cached (released only by the next +invalidation), so a subsequent Start reuses it only implicitly through the still-cached engine +(a fresh session is always created on the next Start, since a session is single-use). **Model-switch guard.** `CanChangeModel` is `true` only when `State != RecognitionStreamingState.Listening`, computed via `[NotifyPropertyChangedFor(nameof(CanChangeModel))]` on the generated `State` @@ -158,17 +182,36 @@ state transition with no manual notification code. `RecognitionPanelView.axaml` model-selection `ComboBox`'s `IsEnabled` to it, so a user cannot switch the selected recognition model through the view while a streaming session is active. This is a view-level convenience, not the sole safeguard: `SelectedModel`'s setter remains public, so `OnSelectedModelChanged` (see -"Recognizer reuse" above) still stops an active session defensively before invalidating, rather -than assuming the view's guard makes a mid-session change unreachable. +"Engine/session reuse" above) still stops an active session defensively before invalidating, +rather than assuming the view's guard makes a mid-session change unreachable. + +#### Capture Debug Recording (diagnostic instrumentation) + +**`CaptureDebugRecorder`** and **`WavFileWriter`** are internal, opt-in, TEMPORARY diagnostic +instrumentation added to investigate a reported live-microphone bug (a long mid-sentence pause +sometimes dropping the first word(s) spoken right after the pause). They are not part of this +subsystem's public presentation surface and are off by default. When the +`DEMASPEECH_CAPTURE_DEBUG_DIR` environment variable names a writable directory, +`RecognitionPanelViewModel`'s Start algorithm calls `CaptureDebugRecorder.TryStart(captureDevice)` +to attach a second, entirely independent subscriber to the same `IAudioCaptureDevice.FrameCaptured` +event the real recognizer subscribes to, writing every captured frame - in the capture device's +own native sample rate and channel count, before any resampling the recognizer applies - to a +session-scoped `.wav` file via `WavFileWriter`, on a dedicated writer thread fed by a +producer/consumer queue so a slow or failing disk can never block or throw into the audio capture +callback. The Stop algorithm finalizes (disposes) any active recording before clearing the +captured UI context. A recorder that fails to start (for example, an inaccessible directory) +logs the failure to the console and live recognition proceeds completely unaffected; this +instrumentation is intended to be removed once the investigation concludes. #### Testability `RecognitionPanelViewModel` depends on `IModelCatalogService`, `IAudioDeviceService`, the shared `DeviceSelectionViewModel`, and `IRecognizerSessionFactory` — never on the library's recognition concretes directly. This is what allows the whole Start/Stop lifecycle, the partial-then-final -transcript sequencing, every unavailable-state path, and the auto-refresh-on-install behavior to -be verified with no downloaded model, no native runtime, and no real microphone. `RefreshCommand` -is not bound to a visible button in `RecognitionPanelView.axaml` - it exists solely for the -auto-refresh-on-install path above and is not a user-facing control, since a manually clickable -refresh beside Start/Stop/the model dropdown proved to be a redundant, confusingly placed control -once auto-refresh-on-install existed. +transcript sequencing, state derivation from `IRecognitionSession.StateChanged`, every +unavailable-state path, and the auto-refresh-on-install behavior to be verified with no +downloaded model, no native runtime, and no real microphone. `RefreshCommand` is not bound to a +visible button in `RecognitionPanelView.axaml` - it exists solely for the auto-refresh-on-install +path above and is not a user-facing control, since a manually clickable refresh beside +Start/Stop/the model dropdown proved to be a redundant, confusingly placed control once +auto-refresh-on-install existed. diff --git a/docs/design/speech-demo/shell-subsystem.md b/docs/design/speech-demo/shell-subsystem.md index 003c4ce..5993c73 100644 --- a/docs/design/speech-demo/shell-subsystem.md +++ b/docs/design/speech-demo/shell-subsystem.md @@ -29,7 +29,16 @@ both to sit at the application's root namespace. `RecognitionPanelViewModel` over those adapters 6. Construct `MainWindowViewModel` over the four panel view models and assign it as the main window's data context -7. Dispose the catalog when the desktop lifetime signals shutdown +7. On shutdown, asynchronously dispose the `SynthesisPanelViewModel` and + `RecognitionPanelViewModel` (each in turn awaits the library's own async engine/session + teardown), then dispose the catalog + +Disposing the two session-owning panels before the catalog matters because a live session +holds a lease against the engine the catalog's `SpeechModelCatalog`/`SpeechModelStore` loaded; +tearing the panels down first lets each session release its engine's lease cleanly before the +catalog that produced it goes away. `desktop.ShutdownRequested` has no async-aware overload, so +this handler is itself `async void`-shaped (fire-and-forget), the same accepted pattern used +throughout this demo's ViewModels for handlers whose signature cannot be `async Task`. No dependency-injection container is used. The demo's purpose is to show a reader exactly how a host application wires itself to the library, and a container would move that wiring into diff --git a/docs/design/speech-demo/synthesis-panel-subsystem.md b/docs/design/speech-demo/synthesis-panel-subsystem.md index e1d5e38..0841be3 100644 --- a/docs/design/speech-demo/synthesis-panel-subsystem.md +++ b/docs/design/speech-demo/synthesis-panel-subsystem.md @@ -8,11 +8,12 @@ The SynthesisPanelSubsystem provides the demo's text-to-speech panel. It contain units, each documented in its own file: - **SynthesizerSessionFactory** (`ISynthesizerSessionFactory` / `SynthesizerSessionFactory`): the - demo-owned synthesizer composition seam and its real implementation over the library's + demo-owned synthesizer-engine composition seam and its real implementation over the library's `SpeechModelStore` and `SpeechSynthesizerFactory` — see _SynthesizerSessionFactory Design_ - **SynthesisPanelViewModel**: the panel's presentation state — the installed synthesis models, - the embedded `ModelSettingsViewModel`, the text input, the audio-tag hints, and the Play/Stop - lifecycle — see _SynthesisPanelViewModel Design_ + the embedded `ModelSettingsViewModel`, the text input, the audio-tag hints, and the async + Play/Stop lifecycle over a cached `ISpeechSynthesizerEngine` and `ISynthesisSession` reused + across many Play calls — see _SynthesisPanelViewModel Design_ ### Interfaces @@ -29,9 +30,10 @@ consumes the library's `SpeechModelStore`, `SpeechSynthesizerFactory`, `ISpeechM `SynthesisPanelViewModel` depends only on the seam interface, the shared device/settings view models, and the library's public `ISpeechModel` contract — never on the library's synthesis -concretes directly. This is what allows the whole Play/Stop lifecycle, including every -unavailable-state path and the auto-refresh-on-install behavior, to be verified with no -downloaded model, no native runtime, and no real speakers. `SynthesizerSessionFactory` is the -sole unit that narrows a public model to `ISynthesisModel` and calls into the library's -synthesizer composition. See each unit's own design document for its data model, algorithms, and -error handling. +concretes directly. This is what allows the whole Play/Stop lifecycle, including state derivation +from `ISynthesisSession.StateChanged`, every unavailable-state path, the engine/session reuse +across Play calls, and the auto-refresh-on-install behavior, to be verified with no downloaded +model, no native runtime, and no real speakers. `SynthesizerSessionFactory` is the sole unit that +narrows a public model to `ISynthesisModel` and calls into the library's synthesizer-engine +composition. See each unit's own design document for its data model, algorithms, and error +handling. diff --git a/docs/design/speech-demo/synthesis-panel-subsystem/synthesis-panel-view-model.md b/docs/design/speech-demo/synthesis-panel-subsystem/synthesis-panel-view-model.md index 9653f75..7a11198 100644 --- a/docs/design/speech-demo/synthesis-panel-subsystem/synthesis-panel-view-model.md +++ b/docs/design/speech-demo/synthesis-panel-subsystem/synthesis-panel-view-model.md @@ -1,8 +1,9 @@ ### SynthesisPanelViewModel **Purpose**: Present the text-to-speech panel's installed synthesis models, the embedded -`ModelSettingsViewModel`, the text input, the audio-tag hints, and the Play/Stop lifecycle, -composing a synthesizer for each Play through `ISynthesizerSessionFactory`. +`ModelSettingsViewModel`, the text input, the audio-tag hints, and the async Play/Stop lifecycle, +caching and reusing an `ISpeechSynthesizerEngine` and an `ISynthesisSession` across many Play +calls instead of recomposing them on every call (the bug this redesign fixes). **Data Model**: @@ -20,7 +21,7 @@ composing a synthesizer for each Play through `ISynthesizerSessionFactory`. | `CanChangeModel` | `bool` | `true` only when `State` is `Idle` or `Error` | | `RefreshCommand` | generated command | Re-reads the catalog | | `PlayCommand` | generated async command | Synthesizes and speaks `Text` | -| `StopCommand` | generated command | Stops an in-flight Play | +| `StopCommand` | generated async command | Stops an in-flight Play | **Key Methods**: @@ -30,27 +31,36 @@ composing a synthesizer for each Play through `ISynthesizerSessionFactory`. model catalog panel's own refresh algorithm. - **PlayAsync(cancellationToken)**: 1. Reports `NoModelSelectedMessage` and enters `Error` when no model is selected - 2. Creates the playback device through `IAudioDeviceService`; reports - `NoPlaybackDeviceMessage` and enters `Error` when it is unavailable - 3. Composes a synthesizer through `ISynthesizerSessionFactory`, forwarding - `Settings.BuildValueBag()` (the embedded settings panel's currently selected parameter - values, for example a selected voice) as the `parameterValues` argument; disposes it and - reports `SynthesizerUnavailableMessage` and enters `Error` when it honestly reports itself + 2. Builds the current parameter value bag via `Settings.BuildValueBag()` and creates the + playback device through `IAudioDeviceService` before paying for the comparatively expensive + async engine load; reports `NoPlaybackDeviceMessage` and enters `Error` when it is unavailable - 4. Otherwise clears the status message, transitions through `Synthesizing` then `Playing` - (both reported before awaiting `SpeakAsync`, since the current library API has no separate - synthesizing-vs-playing callback granularity), and awaits `SpeakAsync` - 5. On success, settles on `Idle`; on `OperationCanceledException` (from `StopCommand`), settles - on `Idle` with `StoppedMessage`; on `SpeechSynthesizerUnavailableException`, enters `Error` - with the exception's message - 6. Always disposes the synthesizer in a `finally` block, whichever path was taken -- **Stop()**: Calls `Stop()` on the active synthesizer, if any, and cancels `PlayCommand`'s - token, which is what drives `PlayAsync`'s `OperationCanceledException` path. A safe no-op when + 3. When no engine is cached, or the selected model or built parameter values differ from the + ones the cached engine was loaded with: invalidates the stale engine (and its session), then + loads a fresh one through `ISynthesizerSessionFactory.LoadAsync`; disposes it and reports + `SynthesizerUnavailableMessage` and enters `Error` when it honestly reports itself + unavailable + 4. Otherwise reuses the cached engine unchanged, skipping the load entirely + 5. When no session is cached, or the selected playback device differs from the one the cached + session was created for: releases the stale session, then creates a fresh one through + `engine.CreateSessionAsync`; on `SynthesisEngineBusyException` or an unavailable session, + reports the appropriate message and enters `Error`; otherwise subscribes `StateChanged` + 6. Otherwise reuses the cached session unchanged — this is the central reuse path that fixes + the "recreate on every Play" bug + 7. Clears the status message and awaits `session.SpeakAsync(Text, cancellationToken)` + 8. On success, `StateChanged` has already driven `State` back to `Idle`; on + `OperationCanceledException` (from `StopCommand`), reports `StoppedMessage`; on + `SpeechSynthesizerUnavailableException`, enters `Error` with the exception's message; on + `SynthesisSessionFaultedException` (terminal for a session), releases the session so the + next Play creates a fresh one against the still-cached engine, and reports the exception's + message +- **StopAsync()**: Calls `StopAsync()` on the active session, if any, and cancels `PlayCommand`'s + token as defense-in-depth in case the session itself cannot be stopped. A safe no-op when nothing is playing. -- **Dispose()**: Unsubscribes from `IModelCatalogService.ModelInstalled` and, defensively, - disposes an active synthesizer if one is still held — ordinarily released by `PlayAsync`'s own - `finally` block, but no longer guaranteed once this ViewModel can be disposed independently of - any in-flight Play. Both actions are safe to repeat, so `Dispose()` is idempotent. +- **DisposeAsync()**: Unsubscribes from `IModelCatalogService.ModelInstalled` and the shared + `DeviceSelectionViewModel`'s pre-refresh hook; stops and awaits an in-flight Play, if any; then + releases the cached session and disposes the cached engine. All actions are safe to repeat, so + `DisposeAsync()` is idempotent. **Auto-refresh on install.** The constructor subscribes to `IModelCatalogService.ModelInstalled`. The handler ignores any event whose @@ -60,24 +70,49 @@ back to calling `Refresh()` synchronously when none was captured), mirroring `RecognitionPanelViewModel`'s identical pattern - since the event's raising thread is not otherwise guaranteed. +**Engine/session reuse (the bugfix).** Per `ISpeechSynthesizerEngine`'s and +`ISynthesisSession`'s own reuse guidance, this ViewModel loads an engine at most once per selected +model/parameter-value combination, and creates a session at most once per engine/playback-device +combination, reusing both across many Play calls instead of reloading the model and recreating +the session on every click - the defect a prior design had, where every `PlayAsync` call composed +a brand-new synthesizer (and therefore reloaded the model) even when nothing about the selection +had changed. `PlayAsync` detects a reason to reload lazily, on its next call, rather than +proactively on a property change: the cached engine is reloaded only when the selected model or +`Settings.BuildValueBag()`'s built parameter values (compared by key/value, since a freshly built +dictionary is never the same instance twice) differ from the ones it was last loaded with; the +cached session is independently recreated whenever the engine was just reloaded or the selected +playback device differs from the one it was last bound to. `ISynthesisSession.StateChanged` +drives `State` (via `MapSessionState`, below) rather than the explicit `State = ...` assignments +a prior design made around the `await SpeakAsync` call. + **Stops deterministically before a shared device refresh.** The constructor also registers a pre-refresh hook with the shared `DeviceSelectionViewModel` via `RegisterPreRefreshHook`, mirroring the `ModelInstalled` subscription precedent above. Unlike -`RecognitionPanelViewModel`'s equivalent hook, calling `Stop()` alone here does not guarantee the -playback device is actually closed by the time the hook returns: `Stop()` only cancels the -synthesizer's internal token and requests cancellation of `PlayCommand` — the real -`_playbackDevice.Stop()` call happens later, inside `PlayAsync`'s own `finally` block, as part of -the already-in-flight task. So the hook first checks `PlayCommand.IsRunning`; when `true`, it -calls `Stop()` and then awaits `PlayCommand.ExecutionTask` (from `IAsyncRelayCommand`), swallowing -the expected `OperationCanceledException` that `Stop()`'s cancellation causes, before returning. -This is what lets the `DeviceSelectionViewModel.Refresh()` clicked from the "Refresh devices" -button stop an in-flight Play and wait for the playback device to genuinely close before the -shared device table is re-scanned, rather than relying on the `AudioDeviceInUseException` -fallback. `Dispose()` unregisters this hook. +`RecognitionPanelViewModel`'s equivalent hook, calling `StopAsync()` alone here does not guarantee +the playback device is actually closed by the time the hook returns: the session's own +`StopAsync()` only requests cancellation of the in-flight `SpeakAsync` - the real playback +device's own stop happens as part of the already-in-flight `PlayAsync` task settling. So the hook +first checks `PlayCommand.IsRunning`; when `true`, it calls `StopAsync()` and then awaits +`PlayCommand.ExecutionTask` (from `IAsyncRelayCommand`), swallowing the expected +`OperationCanceledException` that cancellation causes, before releasing the cached session (never +the cached engine). This is what lets the `DeviceSelectionViewModel.Refresh()` clicked from the +"Refresh devices" button stop an in-flight Play and wait for the playback device to genuinely +close before the shared device table is re-scanned, rather than relying on the +`AudioDeviceInUseException` fallback. `DisposeAsync()` unregisters this hook. + +**State derivation.** `SynthesisPlaybackState` is never assigned ad hoc at each call site; +`MapSessionState` is the single, pure mapping from `SynthesisSessionState` (`Starting` maps to +`Synthesizing`; `Running`/`Stopping` map to `Playing`; `Faulted` maps to `Error`; anything else +maps to `Idle`), applied every time `ISynthesisSession.StateChanged` raises, marshaled onto the +UI thread captured when the session was created (falling back to applying it synchronously when +none was captured); a session reporting `Faulted` additionally sets `StatusMessage` to +`SessionFaultedMessage`. **Embedded settings.** Assigning `SelectedModel` also assigns the embedded `Settings.Model`, so the settings panel always presents the currently selected voice's declared parameters, reusing -the ModelSettingsSubsystem instead of duplicating its rendering logic. +the ModelSettingsSubsystem instead of duplicating its rendering logic. It does not proactively +invalidate the cached engine/session: `PlayAsync` detects the model change lazily on its next +call (see "Engine/session reuse" above). **Audio tag hints.** `ExampleTagHints` projects each entry in the library's `AudioTagCatalog.Tags` to its first alias, bracketed (for example, `[whispers]`). Drawing the hints from the library's @@ -99,14 +134,17 @@ so the wrapping `Border` correctly disables the settings view. This prevents a u switching the selected model or its voice/speaker parameters while synthesis is in flight or audio is playing. -**Error Handling**: Never lets a seam fault, a device fault, or an unavailable-synthesizer state -escape as an unhandled exception; every path resolves to `State`/`StatusMessage`. Cancellation is -distinguished from a genuine failure via `OperationCanceledException`. +**Error Handling**: Never lets a seam fault, a device fault, or an unavailable-engine/session +state escape as an unhandled exception; every path resolves to `State`/`StatusMessage`. +Cancellation is distinguished from a genuine failure via `OperationCanceledException`, and a +terminal session fault releases the session (keeping the engine) rather than leaving a known-dead +session cached for the next Play. **Dependencies**: `IModelCatalogService`, `IAudioDeviceService`, the shared `DeviceSelectionViewModel`, `ISynthesizerSessionFactory`, and the embedded `ModelSettingsViewModel` — never the library's synthesis concretes directly. This is what allows -the whole Play/Stop lifecycle, including every unavailable-state path and the +the whole Play/Stop lifecycle, including state derivation from `ISynthesisSession.StateChanged`, +every unavailable-state path, the engine/session reuse behavior, and the auto-refresh-on-install behavior, to be verified with no downloaded model, no native runtime, and no real speakers. diff --git a/docs/design/speech-demo/synthesis-panel-subsystem/synthesizer-session-factory.md b/docs/design/speech-demo/synthesis-panel-subsystem/synthesizer-session-factory.md index df5239f..7326c22 100644 --- a/docs/design/speech-demo/synthesis-panel-subsystem/synthesizer-session-factory.md +++ b/docs/design/speech-demo/synthesis-panel-subsystem/synthesizer-session-factory.md @@ -1,13 +1,13 @@ ### SynthesizerSessionFactory -**Purpose**: Provide a demo-owned synthesizer composition seam over the library's +**Purpose**: Provide a demo-owned synthesizer-engine composition seam over the library's `SpeechModelStore` and `SpeechSynthesizerFactory`, accepting the library's public `ISpeechModel` contract so the panel can be tested with a plain fake model and no `InternalsVisibleTo` grant from the library. -**Why a Demo-Owned Seam**: The library composes a synthesizer through the static -`SpeechSynthesizerFactory.Create(ISynthesisModel, SpeechModelStore, IAudioPlaybackDevice, -ISpeechDiagnostics?, IReadOnlyDictionary?)` method, which requires an +**Why a Demo-Owned Seam**: The library composes a synthesizer engine through the static +`SpeechSynthesizerFactory.LoadAsync(ISynthesisModel, SpeechModelStore, ISpeechDiagnostics?, +IReadOnlyDictionary?, CancellationToken)` method, which requires an `ISynthesisModel` — an interface whose members are partly `internal` to the library, so only the library's own assemblies can implement it. A demo-owned seam therefore accepts the common `ISpeechModel` contract instead and performs the @@ -15,35 +15,41 @@ narrowing itself. This is also what works around the static factory method itsel substitutable in a ViewModel unit test. `ISynthesizerSessionFactory` and `SynthesizerSessionFactory` are documented as one unit because the interface has no independently observable behavior of its own — every test exercises it through -`SynthesizerSessionFactory`, its sole implementation. +`SynthesizerSessionFactory`, its sole implementation. Only the engine is composed through this +seam; a playback device is bound later, per run, through +`ISpeechSynthesizerEngine.CreateSessionAsync` directly against the returned engine, so a host can +load one engine per model/parameter combination and reuse it across many sessions instead of +reloading the model on every Play. **Data Model**: | Member | Returns | Behavior | | --- | --- | --- | -| `Create(model, playbackDevice, parameterValues)` | `ISpeechSynthesizer` | Never throws | +| `LoadAsync(model, parameterValues, cancellationToken)` | `Task` | Never throws | **Key Methods**: -- **Create(model, playbackDevice, parameterValues?)**: Narrows the model to `ISynthesisModel` and - forwards it, the shared `SpeechModelStore`, the playback device, and the `parameterValues` bag - unchanged to `SpeechSynthesizerFactory.Create(ISynthesisModel, SpeechModelStore, - IAudioPlaybackDevice, ISpeechDiagnostics?, IReadOnlyDictionary?)`, which - resolves the model's installed-files directory internally, inheriting that factory's "nothing - throws at composition" contract. A model that declares a - role other than synthesis — and is therefore not an `ISynthesisModel` — returns the library's - own `UnavailableSpeechSynthesizer.Instance`, exactly like a model that is not installed, rather - than throwing: a host that lets a user choose an installed model with the wrong role must - still get a working, if unavailable, synthesizer back. The `parameterValues` argument is the - untyped key-value bag `Settings.BuildValueBag()` assembles from the embedded settings panel's - current parameter values (for example a selected voice); forwarding it unchanged is what lets - picking a voice in the demo's TTS panel genuinely change what is synthesized. - -**Error Handling**: Throws `ArgumentNullException` for a null model or playback device (an -explicit misuse of the seam's own contract); otherwise never throws, returning an honest -unavailable synthesizer for every unavailable composition state. +- **LoadAsync(model, parameterValues?, cancellationToken)**: Narrows the model to + `ISynthesisModel` and forwards it, the shared `SpeechModelStore`, and the `parameterValues` bag + unchanged to `SpeechSynthesizerFactory.LoadAsync(ISynthesisModel, SpeechModelStore, + ISpeechDiagnostics?, IReadOnlyDictionary?, CancellationToken)`, which resolves + the model's installed-files directory internally, inheriting that factory's "nothing throws at + composition" contract. A model that declares a role other than synthesis — and is therefore not + an `ISynthesisModel` — returns the library's own `UnavailableSpeechSynthesizerEngine.Instance`, + exactly like a model that is not installed, rather than throwing: a host that lets a user + choose an installed model with the wrong role must still get a working, if unavailable, engine + back - including when a non-null `parameterValues` bag is supplied, since the role check + happens before any parameter is consulted. The `parameterValues` argument is the untyped + key-value bag `Settings.BuildValueBag()` assembles from the embedded settings panel's current + parameter values (for example a selected voice); forwarding it unchanged is what lets picking a + voice in the demo's TTS panel genuinely change what is synthesized. + +**Error Handling**: Throws `ArgumentNullException` for a null model (an explicit misuse of the +seam's own contract); otherwise never throws, returning an honest unavailable engine for every +unavailable composition state. **Dependencies**: The library's `SpeechModelStore`, `SpeechSynthesizerFactory`, `ISpeechModel`, -`ISynthesisModel`, `IAudioPlaybackDevice`, `UnavailableSpeechSynthesizer`. +`ISynthesisModel`, `ISpeechSynthesizerEngine`, `UnavailableSpeechSynthesizerEngine`. -**Callers**: `SynthesisPanelViewModel` (composes a synthesizer through this seam on `Play`). +**Callers**: `SynthesisPanelViewModel` (loads a synthesizer engine through this seam, at most +once per selected model/parameter combination, reused across many Play calls). diff --git a/docs/design/speech.md b/docs/design/speech.md index e5af7b1..67ae0c9 100644 --- a/docs/design/speech.md +++ b/docs/design/speech.md @@ -24,20 +24,22 @@ consists of five subsystems: `SpeechModelCatalog`'s enumeration of the compiled-in known-model registry (four real, production models covering both roles) alongside install state — see _ModelManagementSubsystem Design_ -- **RecognitionSubsystem**: the public streaming speech-to-text contract - (`ISpeechRecognizer`/`SpeechRecognitionResult`), the `SpeechRecognizerFactory` composition root - that returns either a real recognizer or an honest unavailable fallback, the - `SherpaOnnxSpeechRecognizer` pipeline that converts captured audio to the model's required - format and streams it through a mockable recognition-engine seam, and - `UnavailableSpeechRecognizer` — see _RecognitionSubsystem Design_ +- **RecognitionSubsystem**: the public async Engine/Session streaming speech-to-text contract + (`ISpeechRecognizerEngine`/`IRecognitionSession`/`SpeechRecognitionResult`), the + `SpeechRecognizerFactory` composition root whose `LoadAsync` returns either a real engine or + an honest unavailable fallback, the `SherpaOnnxRecognitionSession` pipeline that converts + captured audio to the model's required format and streams it through a mockable internal + `IRecognitionBackend` seam, and `UnavailableSpeechRecognizerEngine`/ + `UnavailableRecognitionSession` — see _RecognitionSubsystem Design_ - **SynthesisSubsystem**: the closed, fixed Natural Language Audio Tag vocabulary and the model-independent Layer 1 parser (`AudioTagCatalog`/`AudioTagParser`) that recognizes bracket syntax against it, the Layer 2 rendering strategy (`IModelCapabilityProfile`/`DefaultModelCapabilityProfile`) that turns a parsed span sequence into a model-appropriate `SpeechPlan` of `SpeechSegment`s, the `SentenceChunker` used for - pipeline-friendly chunk boundaries, and the public `ISpeechSynthesizer` streaming/playback - contract, whose `SpeechSynthesizerFactory` composition root returns either the real - `SherpaOnnxSpeechSynthesizer` pipeline or the honest `UnavailableSpeechSynthesizer` fallback — + pipeline-friendly chunk boundaries, and the public async Engine/Session streaming/playback + contract (`ISpeechSynthesizerEngine`/`ISynthesisSession`), whose `SpeechSynthesizerFactory` + composition root's `LoadAsync` returns either the real `SherpaOnnxSynthesisSession` pipeline + or the honest `UnavailableSpeechSynthesizerEngine`/`UnavailableSynthesisSession` fallback — see _SynthesisSubsystem Design_ Within AudioSubsystem, a child **PortAudio** subsystem isolates PortAudioSharp2 and the @@ -82,23 +84,38 @@ The system exposes the following public API to external consumers: - **SpeechModelCatalog**: enumerates the compiled-in known-model registry (four real, production models covering both roles) alongside install state, and orchestrates downloading a known model by id -- **ISpeechRecognizer** / **SpeechRecognitionResult** / **SpeechRecognitionEvent**: the streaming - speech-to-text contract and the immutable result/event types it delivers -- **SpeechRecognizerFactory**: the composition entry point for obtaining a speech recognizer for - an installed recognition model and a capture device -- **UnavailableSpeechRecognizer** / **SpeechRecognizerUnavailableException**: the honest - unavailable recognizer fallback and the exception thrown when an operational member of an - unavailable recognizer is invoked +- **ISpeechRecognizerEngine** / **IRecognitionSession** / **SpeechRecognitionResult** / + **SpeechRecognitionEvent**: the loaded-model engine and per-device session halves of the + streaming speech-to-text contract, their state enums (`RecognitionSessionState`, + `SessionStateChangedEventArgs`), and the immutable result/event types a session's + `GetResultsAsync()` delivers +- **SpeechRecognizerFactory**: the composition entry point whose `LoadAsync` obtains a speech + recognizer engine for an installed recognition model, with the engine's own + `CreateSessionAsync` binding it to a capture device for a session's lifetime +- **UnavailableSpeechRecognizerEngine** / **UnavailableRecognitionSession** / + **SpeechRecognizerUnavailableException** / **RecognitionEngineBusyException** / + **RecognitionSessionFaultedException**: the honest unavailable engine/session fallbacks, the + exception thrown when an operational member of an unavailable engine/session is invoked, the + exception thrown by a concurrent `CreateSessionAsync` while the engine's one session lease is + already held, and the exception/fault surfaced through a session's result stream when its + dedicated worker faults - **NaturalLanguageAudioTag** / **NaturalLanguageAudioTagKind** / **AudioTagDescriptor** / **AudioTagCatalog** / **TaggedTextSpanKind** / **TaggedTextSpan** / **AudioTagParser**: the closed, fixed inline audio-tag vocabulary and the model-independent parser that recognizes it -- **ISpeechSynthesizer** / **SynthesizedSpeech**: the streaming synthesis-and-playback contract - and the immutable value type it yields -- **SpeechSynthesizerFactory**: the composition entry point for obtaining a speech synthesizer - for an installed synthesis model and a playback device -- **UnavailableSpeechSynthesizer** / **SpeechSynthesizerUnavailableException**: the honest - unavailable synthesizer fallback and the exception thrown when an operational member of an - unavailable synthesizer is invoked +- **ISpeechSynthesizerEngine** / **ISynthesisSession** / **SynthesizedSpeech**: the loaded-model + engine and per-device session halves of the streaming synthesis-and-playback contract, their + state enums (`SynthesisSessionState`, `SessionStateChangedEventArgs`), and the immutable value + type a session's `SpeakAsync`/`SynthesizeAsync` yields +- **SpeechSynthesizerFactory**: the composition entry point whose `LoadAsync` obtains a speech + synthesizer engine for an installed synthesis model, with the engine's own + `CreateSessionAsync` binding it to a playback device for a session's lifetime +- **UnavailableSpeechSynthesizerEngine** / **UnavailableSynthesisSession** / + **SpeechSynthesizerUnavailableException** / **SynthesisEngineBusyException** / + **SynthesisSessionFaultedException**: the honest unavailable engine/session fallbacks, the + exception thrown when an operational member of an unavailable engine/session is invoked, the + exception thrown by a concurrent `CreateSessionAsync` while the engine's one session lease is + already held, and the exception/fault surfaced through a session's result stream when its + dedicated worker faults | Interface | Direction | Format | Constraints | | --- | --- | --- | --- | @@ -118,14 +135,16 @@ The system exposes the following public API to external consumers: | `IModelDownloadClient.DownloadAsync(...)` | Inbound | Method call | Throws on any non-success/transport failure | | `SpeechModelCatalog.Enumerate()` | Outbound | Method call/return | Never throws | | `SpeechModelCatalog.DownloadAsync(...)` | Inbound/Outbound | Method call/return | Throws for unknown model id | -| `SpeechRecognizerFactory.Create(...)` | Inbound/Outbound | Method call/return | Never throws for unavailable states | -| `ISpeechRecognizer.Start()`/`.Stop()` | Inbound | Method call | Throws only on unavailable or first-use faults | -| `ISpeechRecognizer.ResultReceived` | Outbound | Event | Raised off the audio callback thread | -| `ISpeechRecognizer.Dispose()` | Inbound | Method call | Idempotent; implies `Stop()` | +| `SpeechRecognizerFactory.LoadAsync(...)` | Inbound/Outbound | Method call | Never throws for unavailable states | +| `ISpeechRecognizerEngine.CreateSessionAsync(...)` | Inbound/Outbound | Method call | Never throws; throws if leased | +| `IRecognitionSession.StartAsync()`/`.StopAsync()` | Inbound | Method call | Throws only unavailable/dispose/restart | +| `IRecognitionSession.GetResultsAsync(...)` | Outbound | `IAsyncEnumerable` | Single-consumer; flushes before stop | +| `IRecognitionSession.DisposeAsync()` | Inbound | Method call | Idempotent; implies `StopAsync()` | | `AudioTagParser.Parse(...)` | Inbound/Outbound | Method call/return | Never throws; folds unmatched brackets | -| `SpeechSynthesizerFactory.Create(...)` | Inbound/Outbound | Method call/return | Never throws for unavailable states | -| `ISpeechSynthesizer.SpeakAsync(...)` | Inbound/Outbound | Method call/return | Throws only on unavailable/first-use | -| `ISpeechSynthesizer.Dispose()` | Inbound | Method call | Idempotent | +| `SpeechSynthesizerFactory.LoadAsync(...)` | Inbound/Outbound | Method call | Never throws for unavailable states | +| `ISpeechSynthesizerEngine.CreateSessionAsync(...)` | Inbound/Outbound | Method call | Never throws; throws if leased | +| `ISynthesisSession.SpeakAsync`/`.SynthesizeAsync` | Inbound/Outbound | Method call | Unavailable-only; no overlap | +| `ISynthesisSession.DisposeAsync()` | Inbound | Method call | Idempotent | ## Dependencies @@ -232,37 +251,48 @@ direct safety impact. **Streaming recognition path:** -1. **Input**: A host passes an installed `IRecognitionModel`, that model's installed-files - directory, and an `IAudioCaptureDevice` to `SpeechRecognizerFactory.Create(...)` -2. **Composition**: The factory checks installation, model role, and device availability, then - loads the model's own engine configuration through the internal `IRecognitionEngineFactory` - seam; any failure returns `UnavailableSpeechRecognizer.Instance` with a structural diagnostic -3. **Capture**: `Start()` subscribes to `FrameCaptured` and starts the device; each captured +1. **Input**: A host passes an installed `IRecognitionModel` and that model's installed-files + directory to `SpeechRecognizerFactory.LoadAsync(...)`, then passes an `IAudioCaptureDevice` + to the returned engine's `CreateSessionAsync(...)` +2. **Composition**: `LoadAsync` checks installation and model role, then loads the model's own + engine configuration through the internal `IRecognitionBackendFactory` seam; any failure + returns `UnavailableSpeechRecognizerEngine.Instance` with a structural diagnostic. + `CreateSessionAsync` checks device availability and the engine's single-session lease, + returning `UnavailableRecognitionSession.Instance` or throwing + `RecognitionEngineBusyException` for a concurrent second lease attempt +3. **Capture**: `StartAsync()` subscribes to `FrameCaptured` and starts the device; each captured block is copied onto a bounded queue on the audio callback thread and nothing more -4. **Conversion and inference**: A single background consumer downmixes and resamples each block - from the device's reported `ChannelCount`/`SampleRate` to the model's declared `AudioFormat` - via `AudioFrameResampler`, feeds it to the `IRecognitionEngine`, and polls for results -5. **Output**: `ResultReceived` raises each provisional and final `SpeechRecognitionResult` off - the audio callback thread; `Stop()` drains the queue so no result derived from already-captured - audio is lost +4. **Conversion and inference**: A dedicated long-running worker thread downmixes and resamples + each block from the device's reported `ChannelCount`/`SampleRate` to the model's declared + `AudioFormat` via `AudioFrameResampler`, feeds it to the `IRecognitionBackend`, and polls for + results +5. **Output**: `GetResultsAsync()` yields each provisional and final `SpeechRecognitionResult` + through a byte-capped backpressure buffer that coalesces provisional results but never drops a + final one; `StopAsync()` drains the queue so no result derived from already-captured audio is + lost before the result stream completes **Streaming synthesis path:** -1. **Input**: A host passes an installed `ISynthesisModel`, that model's installed-files - directory, and an `IAudioPlaybackDevice` to `SpeechSynthesizerFactory.Create(...)` -2. **Composition**: The factory checks installation, model role, and device availability, then - loads the model's own engine configuration through the internal `ISynthesisEngineFactory` seam; - any failure returns `UnavailableSpeechSynthesizer.Instance` with a structural diagnostic +1. **Input**: A host passes an installed `ISynthesisModel` and that model's installed-files + directory to `SpeechSynthesizerFactory.LoadAsync(...)`, then passes an `IAudioPlaybackDevice` + to the returned engine's `CreateSessionAsync(...)` +2. **Composition**: `LoadAsync` checks installation and model role, then loads the model's own + engine configuration through the internal `ISynthesisBackendFactory` seam; any failure returns + `UnavailableSpeechSynthesizerEngine.Instance` with a structural diagnostic. + `CreateSessionAsync` checks device availability and the engine's single-session lease, + returning `UnavailableSynthesisSession.Instance` or throwing `SynthesisEngineBusyException` + for a concurrent second lease attempt 3. **Parsing and rendering**: `AudioTagParser` recognizes inline audio-tag bracket syntax against the fixed vocabulary, and the selected model's `IModelCapabilityProfile` renders the parsed span sequence into a `SpeechPlan` of `SpeechSegment`s, honoring the model's own declared audio-tag support 4. **Chunked synthesis and playback**: `SentenceChunker` splits synthesis-ready text into - pipeline-friendly chunks; each chunk streams through the `ISynthesisEngine` and the resulting - audio is resampled via `PlaybackAudioResampler` to the playback device's required format before - being written to it -5. **Output**: The playback device renders the synthesized audio as it streams; `Stop()` halts an - in-progress synthesis/playback cycle + pipeline-friendly chunks; each chunk streams through the `ISynthesisBackend` on a dedicated + long-running worker thread and the resulting audio is resampled via `PlaybackAudioResampler` + to the playback device's required format before being written to it +5. **Output**: The playback device renders the synthesized audio as it streams via + `SpeakAsync(...)` (or is returned as `SynthesizedSpeech` segments via `SynthesizeAsync(...)`); + `StopAsync()` halts an in-progress synthesis/playback cycle ## Design Constraints @@ -304,7 +334,7 @@ direct safety impact. hardware resolved, and a recognition model accepts exactly one mono rate, so the RecognitionSubsystem - not the AudioSubsystem and not the host - converts between them. The converter uses channel averaging and linear interpolation, a deliberate simplicity/quality - trade-off recorded in _SherpaOnnxSpeechRecognizer Design_ + trade-off recorded in _SherpaOnnxRecognitionSession Design_ - **Recognition and synthesis models both ship in the compiled-in catalog**: The recognition pipeline is proven end to end against the two real, production recognition models (`SherpaOnnxZipformerEnRecognitionModel`, `SherpaOnnxNemotronStreamingEnRecognitionModel`), and diff --git a/docs/design/speech/audio-subsystem/wav-file-audio-capture-device.md b/docs/design/speech/audio-subsystem/wav-file-audio-capture-device.md index 19cecb7..974eba6 100644 --- a/docs/design/speech/audio-subsystem/wav-file-audio-capture-device.md +++ b/docs/design/speech/audio-subsystem/wav-file-audio-capture-device.md @@ -1,8 +1,8 @@ ### WavFileAudioCaptureDevice **Purpose**: Read a mono, 16-bit PCM `.wav` file and deliver it as normalized capture frames -instead of capturing from real microphone hardware, so any host application - not only the future -`recognize --input` CLI command - can drive speech recognition from a pre-recorded file +instead of capturing from real microphone hardware, so any host application - including the +shipped `recognize --input` CLI command - can drive speech recognition from a pre-recorded file deterministically. **Data Model**: Holds the constructor-supplied file path and per-frame sample count. Resolved @@ -40,7 +40,7 @@ caller-configuration error to surface clearly, not a backend-availability failur **Dependencies**: The Base Class Library's `FileStream`/`BinaryReader` only; implements `IAudioCaptureDevice`. -**Callers**: Any host composing `IAudioCaptureDevice`-consuming code (for example, a future +**Callers**: Any host composing `IAudioCaptureDevice`-consuming code (for example, the shipped `recognize --input ` CLI command, which subscribes to `EndOfFileReached` to know when to call `Stop()` and stop waiting for further recognition results) that needs speech recognition driven from a pre-recorded file instead of real microphone hardware. diff --git a/docs/design/speech/audio-subsystem/wav-file-audio-playback-device.md b/docs/design/speech/audio-subsystem/wav-file-audio-playback-device.md index 85d7b54..01c37ea 100644 --- a/docs/design/speech/audio-subsystem/wav-file-audio-playback-device.md +++ b/docs/design/speech/audio-subsystem/wav-file-audio-playback-device.md @@ -1,8 +1,8 @@ ### WavFileAudioPlaybackDevice **Purpose**: Write synthesized speech to a `.wav` file as 16-bit PCM instead of rendering it to -real playback hardware, so any host application - not only the future `speak --output-audio` CLI -command - can capture synthesized audio deterministically. +real playback hardware, so any host application - including the shipped `speak --output-audio` +CLI command - can capture synthesized audio deterministically. **Data Model**: Holds the constructor-supplied `sampleRate`/`channelCount`, an open `FileStream` and `BinaryWriter` created immediately at construction, and a running total of sample data bytes @@ -35,6 +35,6 @@ never throw `AudioDeviceUnavailableException` the way a real device's operationa **Dependencies**: The Base Class Library's `FileStream`/`BinaryWriter` only; implements `IAudioPlaybackDevice` and `IDisposable`. -**Callers**: Any host composing `IAudioPlaybackDevice`-consuming code (for example, a future +**Callers**: Any host composing `IAudioPlaybackDevice`-consuming code (for example, the shipped `speak --output-audio ` CLI command) that needs synthesized speech captured to a file instead of played through real hardware. diff --git a/docs/design/speech/model-management-subsystem.md b/docs/design/speech/model-management-subsystem.md index 7590b09..14a6d59 100644 --- a/docs/design/speech/model-management-subsystem.md +++ b/docs/design/speech/model-management-subsystem.md @@ -50,6 +50,11 @@ model since neither declares a parameter yet. It contains the following units: are built around - **SpeechModelParameters** (`ISpeechModelParameter`, `NumericParameter`, `ChoiceParameter`, `BooleanParameter`): the typed, self-describing tunable-parameter descriptor hierarchy +- **SpeechModelParameterDiagnostics**: the shared internal helper, invoked once up front by both + composition factories' innermost `LoadAsync` overloads, that validates a supplied + `parameterValues` bag against a model's declared `Parameters`, throwing `ArgumentException` for + an invalid value and reporting (via an `Info` diagnostic) an unknown key rather than silently + dropping it - **SpeechModelContract** (`ISpeechModel`, `IRecognitionModel`, `ISynthesisModel`): the common per-model contract plus its two role-specific interfaces - `IRecognitionModel` exposing a public `AudioFormat` and internal engine-construction members, and `ISynthesisModel` exposing @@ -137,15 +142,17 @@ its declared `Parameters` (`ISpeechModelParameter` instances - `NumericParameter `ChoiceParameter`, or `BooleanParameter`, each self-validating an internally consistent range/ option-set/default at construction), its declared `AudioTagSupport` (a declaration only - the Layer 2 rendering logic is Phase 4), its `DownloadDescriptor`, its `InstallAsync` hook (a no-op -default, overridable to unpack an archive payload), and its `NormalizeText` hook (an identity -default, overridable for Phase 4 text normalization). `IRecognitionModel` now exposes a public -plain-data `AudioFormat` declaration, while keeping `CreateEngineConfig(installedModelDirectory)` +default, overridable to unpack an archive payload), its `NormalizeText` hook (an identity +default, overridable for Phase 4 text normalization), and its `LicenseName`/`LicenseUrl` +default-hook members (`"Unknown"`/`null` by default, overridable to declare a model's real +license name and an optional canonical URL to its full text). `IRecognitionModel` now exposes a +public plain-data `AudioFormat` declaration, while keeping `CreateEngineConfig(installedModelDirectory)` internal because it returns a sherpa-onnx type; this lets hosts compose capture devices around a model's required format without leaking native engine configuration into the public API. `ISynthesisModel` similarly exposes a public best-effort `PreferredAudioFormat` hint, while keeping `CreateEngineConfig`, `CapabilityProfile`, and `ResolveSpeakerId(parameterValues)` internal. The hint is intentionally non-authoritative: the real synthesis output rate is still -the loaded engine's `ISynthesisEngine.SampleRate`. +the loaded engine's `ISynthesisBackend.SampleRate`. `SpeechModelCatalog` composes a compiled-in `KnownModels` list - as of Phase 7a, this phase's two real recognition models - with a `SpeechModelStore` and a `SpeechModelDownloader`. `Enumerate()` diff --git a/docs/design/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.md b/docs/design/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.md index 438749d..47d0c0a 100644 --- a/docs/design/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.md +++ b/docs/design/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.md @@ -86,11 +86,11 @@ none of them is a distinct "happy"/"sad"/"excited" reading of the same voice - t deliberately avoids claiming an emotion-control capability this model does not have. **Voice Selection Resolved**: unlike the sibling VITS/Piper model, this model's voice selection -is genuinely wired end-to-end: `SherpaOnnxSpeechSynthesizer.GenerateSegment` now calls +is genuinely wired end-to-end: `SherpaOnnxSynthesisSession.GenerateSegmentAsync` now calls `ISynthesisModel.ResolveSpeakerId` (a new, non-breaking default-hook interface member) instead of -hard-coding `speakerId: 0`, and `SpeechSynthesizerFactory.Create` threads an optional +hard-coding `speakerId: 0`, and `SpeechSynthesizerFactory.LoadAsync` threads an optional `parameterValues` bag through to the synthesizer for this purpose. See -`sherpa-onnx-speech-synthesizer.md` for the full mechanism. +`sherpa-onnx-synthesis-session.md` for the full mechanism. **Dependencies**: `ISynthesisModel`, `SpeechModelDownloadDescriptor`, `SpeechModelDownloadFile`, `TarBz2ArchiveExtractor`, `ChoiceParameter`, `ChoiceParameterOption`, @@ -100,6 +100,6 @@ hard-coding `speakerId: 0`, and `SpeechSynthesizerFactory.Create` threads an opt **Callers**: `SpeechModelCatalog.KnownModels` (registers this instance); `SpeechModelDownloader` (invokes `InstallAsync` after checksum verification); `SherpaOnnxSynthesisEngine` (consumes the internal `CreateEngineConfig` member through the -SynthesisSubsystem); `SherpaOnnxSpeechSynthesizer.GenerateSegment` (consumes the internal +SynthesisSubsystem); `SherpaOnnxSynthesisSession.GenerateSegmentAsync` (consumes the internal `ResolveSpeakerId` member once per synthesized segment); and hosts or factory composition code that read `PreferredAudioFormat` before engine construction. diff --git a/docs/design/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.md b/docs/design/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.md index d37c16d..8e34242 100644 --- a/docs/design/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.md +++ b/docs/design/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.md @@ -86,5 +86,5 @@ name mapping exists for this model's speakers. **Callers**: `SpeechModelCatalog.KnownModels` (registers this instance); `SpeechModelDownloader` (invokes `InstallAsync` after checksum verification); `SherpaOnnxSynthesisEngine` (consumes the internal `CreateEngineConfig` member through the -SynthesisSubsystem); `SherpaOnnxSpeechSynthesizer.GenerateSegment` (consumes the internal +SynthesisSubsystem); `SherpaOnnxSynthesisSession.GenerateSegmentAsync` (consumes the internal `ResolveSpeakerId` member once per synthesized segment). diff --git a/docs/design/speech/model-management-subsystem/speech-model-catalog.md b/docs/design/speech/model-management-subsystem/speech-model-catalog.md index ff507a5..63ca5b2 100644 --- a/docs/design/speech/model-management-subsystem/speech-model-catalog.md +++ b/docs/design/speech/model-management-subsystem/speech-model-catalog.md @@ -52,4 +52,4 @@ a corruption signal). **Callers**: Hosts building a model-settings page (enumerate + download/delete actions per the demo-application scope). `SpeechRecognizerFactory`/`SpeechSynthesizerFactory` catalog-based -`Create` overloads (via `Store`). +`LoadAsync` overloads (via `Store`). diff --git a/docs/design/speech/model-management-subsystem/speech-model-contract.md b/docs/design/speech/model-management-subsystem/speech-model-contract.md index 33a187a..2a8d29d 100644 --- a/docs/design/speech/model-management-subsystem/speech-model-contract.md +++ b/docs/design/speech/model-management-subsystem/speech-model-contract.md @@ -26,7 +26,7 @@ documentation anticipated. `SpeechModelDownloader`'s `ISpeechModel`-aware `DownloadAsync` overload after checksum verification and before the atomic swap. - **ISpeechModel.NormalizeText(text)**: applies this model's own text normalization/correction - before inference. Defaults to the identity function. `SherpaOnnxSpeechSynthesizer` calls this + before inference. Defaults to the identity function. `SherpaOnnxSynthesisSession` calls this hook before Layer 1 tag parsing, so a synthesis model may correct punctuation or spelling without the SynthesisSubsystem needing to know how. - **ISpeechModel.LicenseName / LicenseUrl**: a model's declared license name/identifier and an @@ -94,7 +94,7 @@ documentation anticipated. synthesizer) rather than here. - **ISynthesisModel.PreferredAudioFormat** *(public)*: a best-effort mono playback-format hint a host may use before the native engine is loaded. This is deliberately not authoritative: the - true output rate remains the loaded engine's `ISynthesisEngine.SampleRate`, which may differ + true output rate remains the loaded engine's `ISynthesisBackend.SampleRate`, which may differ and therefore still drive playback resampling. - **ISynthesisModel.CapabilityProfile** *(internal)*: the `IModelCapabilityProfile` this model uses to render Natural Language Audio Tags into a `SpeechPlan`. Defaults to @@ -112,7 +112,8 @@ documentation anticipated. fault synthesis. Every `internal` member is deliberately not public. The design scopes the "must not leak -sherpa-onnx types" constraint to `ISpeechRecognizer`/`ISpeechSynthesizer`, and makes each model's +sherpa-onnx types" constraint to `ISpeechRecognizerEngine`/`IRecognitionSession`/ +`ISpeechSynthesizerEngine`/`ISynthesisSession`, and makes each model's backing class responsible for "sherpa-onnx configuration for its own model architecture", so returning a real recognizer/synthesizer configuration here is consistent with the approved design. Keeping those members internal keeps every sherpa-onnx type out of the library's public diff --git a/docs/design/speech/model-management-subsystem/speech-model-parameters.md b/docs/design/speech/model-management-subsystem/speech-model-parameters.md index a4fb19f..e786a53 100644 --- a/docs/design/speech/model-management-subsystem/speech-model-parameters.md +++ b/docs/design/speech/model-management-subsystem/speech-model-parameters.md @@ -38,7 +38,7 @@ rules for what counts as a valid supplied value. - **BooleanParameter(id, displayName, description, default)**: Validates only the common fields. - **SpeechModelParameterDiagnostics.ValidateAndReport(modelId, declaredParameters, parameterValues, diagnostics, category)**: Called once, up front, by both composition factories' - innermost `Create` overloads (before either factory does any other work). Iterates + innermost `LoadAsync` overloads (before either factory does any other work). Iterates `declaredParameters` in their declared order; for each one present as a key in `parameterValues`, validates the supplied value against that parameter's own rules - a `NumericParameter` value must be a `double`/`int`/`float` within `[Minimum, Maximum]`, and @@ -73,4 +73,4 @@ depend on none beyond each other (`ChoiceParameter` depends on `ChoiceParameterO **Callers**: `ISpeechModel.Parameters`; a model's own backing class declares instances of these types; a host UI renders controls from them and supplies values back in an untyped key-value bag keyed by `Id`. `SpeechModelParameterDiagnostics.ValidateAndReport` is called exclusively by -`SpeechRecognizerFactory.Create` and `SpeechSynthesizerFactory.Create`'s innermost overloads. +`SpeechRecognizerFactory.LoadAsync` and `SpeechSynthesizerFactory.LoadAsync`'s innermost overloads. diff --git a/docs/design/speech/recognition-subsystem.md b/docs/design/speech/recognition-subsystem.md index 2c1856e..f4b27a3 100644 --- a/docs/design/speech/recognition-subsystem.md +++ b/docs/design/speech/recognition-subsystem.md @@ -5,113 +5,236 @@ ### Overview The RecognitionSubsystem turns audio captured by the AudioSubsystem into text using a model -installed by the ModelManagementSubsystem. It owns the library's public streaming -speech-to-text contract, the composition root that decides whether recognition is possible on -the current machine, the running pipeline that converts and streams audio into an inference -engine, and the honest fallback used when recognition is not possible. It contains the -following direct units: - -- **ISpeechRecognizer**: the public streaming recognition contract, together with the - **SpeechRecognitionResult** and **SpeechRecognitionEvent** value types it delivers -- **SpeechRecognizerFactory**: composition root that returns either a real recognizer or the - honest unavailable fallback, and never throws for an ordinary machine state -- **SherpaOnnxSpeechRecognizer**: the real streaming implementation, together with the internal - **IRecognitionEngine**/**IRecognitionEngineFactory** seam, its real - **SherpaOnnxRecognitionEngine**/**SherpaOnnxRecognitionEngineFactory** implementations, and the - **AudioFrameResampler** that converts captured audio into the format a model requires -- **UnavailableSpeechRecognizer** and **SpeechRecognizerUnavailableException**: honest fallback - behavior when no model, no engine, or no capture device is available - -A later pass threads an optional, session-level `parameterValues` bag through -`SpeechRecognizerFactory.Create` into a new `IRecognitionModel.CreateEngineConfig(installedModelDirectory, -parameterValues)` default hook, giving the RecognitionSubsystem the same tunable-parameter seam -the SynthesisSubsystem already has via `ResolveSpeakerId`, with no shipped recognition model yet -declaring a parameter to interpret from it. +installed by the ModelManagementSubsystem. It exposes an async Engine/Session API across five +conceptual layers rather than one synchronous recognizer contract: a Layer 3 engine owns one +loaded model and leases exclusive use of it to a Layer 5 session bound to one capture device, so +a host can load a model once and run many sequential or future concurrent-by-device sessions +against it without reloading. It contains the following direct units: + +- **ISpeechRecognizerEngine**: the public Layer 3 contract for one loaded, reusable recognition + engine - `IsAvailable` plus `CreateSessionAsync`, which leases the engine to a new session +- **IRecognitionSession**: the public Layer 5 contract for one streaming recognition session + bound to one capture device - a forward-only `RecognitionSessionState` machine with + `StartAsync`/`StopAsync`/`GetResultsAsync`, together with the **RecognitionSessionState**, + **SessionStateChangedEventArgs**, **SpeechRecognitionResult**, and **SpeechRecognitionEvent** + value types it uses +- **SpeechRecognizerFactory**: composition root whose `LoadAsync` overloads return a real engine + or the honest unavailable fallback, and never throw for an ordinary machine state +- **SherpaOnnxSpeechRecognizerEngine** and **SherpaOnnxRecognitionSession**: the real Layer 3/5 + implementations, together with the internal + **IRecognitionBackend**/**IRecognitionBackendFactory** seam, its real + **SherpaOnnxRecognitionEngine**/**SherpaOnnxRecognitionEngineFactory** implementations, the + **AudioFrameResampler** that converts captured audio into the format a model requires, the + **RecognitionResultBuffer** backpressure buffer, and the **DedicatedWorker** pump-thread helper +- **UnavailableSpeechRecognizerEngine**, **UnavailableRecognitionSession**, and + **SpeechRecognizerUnavailableException**: honest fallback behavior when no model, no backend, + or no capture device is available +- **RecognitionEngineBusyException** and **RecognitionSessionFaultedException**: the two new + fault types introduced by the async redesign, signaling a concurrent lease attempt and a + mid-session fault surfaced through `GetResultsAsync`, respectively ### Interfaces -The subsystem exposes `ISpeechRecognizer`, `SpeechRecognitionResult`, `SpeechRecognitionEvent`, -`SpeechRecognizerFactory`, `UnavailableSpeechRecognizer`, and -`SpeechRecognizerUnavailableException` as its public API. It consumes `IAudioCaptureDevice` from +The subsystem exposes `ISpeechRecognizerEngine`, `IRecognitionSession`, `RecognitionSessionState`, +`SessionStateChangedEventArgs`, `SpeechRecognitionResult`, `SpeechRecognitionEvent`, +`SpeechRecognizerFactory`, `UnavailableSpeechRecognizerEngine`, `UnavailableRecognitionSession`, +`SpeechRecognizerUnavailableException`, `RecognitionEngineBusyException`, and +`RecognitionSessionFaultedException` as its public API. It consumes `IAudioCaptureDevice` from the AudioSubsystem for input audio, `IRecognitionModel` from the ModelManagementSubsystem for the engine configuration and required input format, and `ISpeechDiagnostics` from the Diagnostics subsystem to report structural composition, lifecycle, and fault facts without ever exposing recognized text. -Both cross-subsystem dependencies were extended in this phase, additively: -`IAudioCaptureDevice` gained `ChannelCount`/`SampleRate` so the resolved capture format can be -discovered (see _IAudioCaptureDevice Design_), and `IRecognitionModel` now exposes a public -`AudioFormat` plus internal `CreateEngineConfig` so each model owns both its input-format -declaration and its engine configuration (see _SpeechModelContract Design_). -`IRecognitionModel.CreateEngineConfig` further gained a two-argument, parameter-value-aware -default-hook overload, forwarded from `SpeechRecognizerFactory.Create`'s own new optional -`parameterValues` argument. - No member of the subsystem's public API names a sherpa-onnx type, per this library's "engine backend stays swappable at the public API surface" decision. The sherpa-onnx configuration type appears only on `IRecognitionModel`'s internal members and inside the -subsystem's internal engine seam. +subsystem's internal backend seam. ### Design -`SpeechRecognizerFactory` is the subsystem's composition root. It checks, in order, whether the -requested model's files exist on disk, whether the model declares the recognition role, and -whether the supplied capture device is available; any failure returns -`UnavailableSpeechRecognizer.Instance` with a structural diagnostic explaining which condition -failed. Only then does it ask an `IRecognitionEngineFactory` to load the model, and a failure -there - the missing-native-runtime case - degrades exactly the -same honest way rather than throwing. Nothing about "is recognition possible?" is left for the -host to work out from separate signals. When a caller already knows the chosen model, the -recommended composition pattern is to construct the capture device first with -`AudioDeviceFactory.CreateCaptureDevice(selection, model.AudioFormat)` and then pass that device -into `SpeechRecognizerFactory.Create(...)`; when the backend honors the hint, -`AudioFrameResampler` stays on its existing equal-rate no-op fast path. - -The running pipeline in `SherpaOnnxSpeechRecognizer` spans two threads by design. The capture -device raises frames on a high-priority audio callback thread, so the recognizer's frame handler -does nothing but copy the block into a bounded queue and return; all conversion, inference, and -event raising happens on a single background consumer task. Stopping completes the queue and -joins that task, so every result derived from audio captured before the stop request has been -delivered by the time the call returns. +#### The five-layer vocabulary + +The redesign separates "is a model loaded and usable?" from "is one device currently streaming +through it?": + +- **Layer 3 - `ISpeechRecognizerEngine`**: owns one loaded `IRecognitionBackend` for one + recognition model. `IsAvailable` reports whether the machine can recognize at all. + `CreateSessionAsync(IAudioCaptureDevice, CancellationToken)` leases exclusive use of the engine + to exactly one live session at a time and returns a new `IRecognitionSession` bound to the + supplied device. +- **Layer 5 - `IRecognitionSession`**: owns the per-device streaming pipeline - the capture + subscription, the resampler, the pump thread, and the result buffer - for exactly one + `StartAsync`/`StopAsync` cycle. A session is single-use: it cannot be restarted after + `Stopped`. + +This mirrors the library's existing "composition is expensive, per-use is cheap" pattern (loading +a model via `SpeechRecognizerFactory.LoadAsync` is the expensive step; creating and running a +session is comparatively cheap), but now makes the two steps independently awaitable, async, and +separately testable, and makes the engine's exclusivity an explicit, observable contract +(`RecognitionEngineBusyException`) rather than an implicit assumption a host had to honor by +convention. + +#### Session state machine + +`RecognitionSessionState` is a forward-only state machine; no transition other than those shown +below is valid, and `StateChanged` is raised for every transition: + +```mermaid +stateDiagram-v2 + [*] --> Created + Created --> Starting + Starting --> Running + Starting --> Faulted + Running --> Stopping + Running --> Faulted + Stopping --> Stopped + Stopping --> Faulted + Stopped --> Disposing + Faulted --> Disposing + Disposing --> Disposed + Disposed --> [*] +``` + +`Faulted` is reachable from `Starting`, `Running`, or `Stopping` - any point where a capture +device can fail or go unavailable mid-session - and is terminal for the ordinary start/stop +cycle: a faulted session is still disposed through the same `Disposing`/`Disposed` path, but +never returns to `Starting` or `Running`. A fault surfaces to a consumer of `GetResultsAsync` as a +`RecognitionSessionFaultedException` thrown from the active enumeration, never silently. + +#### Engine exclusivity and lease behavior + +`ISpeechRecognizerEngine.CreateSessionAsync` leases the engine's single `IRecognitionBackend` to +exactly one live session (Decision #2 of the async redesign). The real engine implements this +with a `SemaphoreSlim(1, 1)` acquired with a zero timeout: a second `CreateSessionAsync` call +while a session is still live fails fast with `RecognitionEngineBusyException` rather than +queueing or blocking the caller, since a recognition backend genuinely cannot usefully decode two +concurrent streams and silently queueing would hide that fact behind an unbounded wait. The lease +is released when the session created from it is disposed, so a host that always disposes its +sessions (directly or via `await using`) can safely call `CreateSessionAsync` again as soon as the +previous session's `DisposeAsync` completes. + +#### Cancellation and abandon policy + +Both the pump thread inside `SherpaOnnxRecognitionSession` and the model-load step inside +`SpeechRecognizerFactory` run on a `DedicatedWorker`: a dedicated, long-running +(`TaskCreationOptions.LongRunning`) thread rather than a pooled thread, because both can block +inside native interop for an unbounded time. `DedicatedWorker` applies a **cooperative-cancel- +then-abandon** policy (Decision #4): on cancellation it signals the delegate's own +`CancellationToken` first, then waits up to a bounded timeout (`DefaultAbandonTimeout`, 2 seconds) +for the delegate to observe it and return; if the delegate has not returned by then, the worker +*abandons* it - the returned `Task` completes (faulted with `OperationCanceledException`) without +waiting for the native call to return, and the abandonment itself is reported through +`ISpeechDiagnostics` at `Warning` level so a host can see that a native call did not cooperate. +This bounds how long `StopAsync`/`DisposeAsync` can ever block a caller, at the cost of leaving an +abandoned native thread to finish on its own; it never blocks indefinitely on a stuck backend. + +#### Backpressure policy + +`SherpaOnnxRecognitionSession` isolates a possibly-slow `GetResultsAsync` consumer from the pump +thread with two independent, bounded buffers (Decision #5): + +- Captured-but-not-yet-converted audio queues in a bounded, drop-oldest `Channel` + (`PendingFrameCapacity`, 64 blocks): recognition that has fallen behind live audio cannot be + caught up by queueing more of it, so the oldest unconverted block is dropped rather than + growing the backlog without bound. +- Converted results queue in a `RecognitionResultBuffer`: one overwritable "latest provisional" + slot (an unconsumed provisional is superseded by the next one, never queued) plus a byte-capped + (`MaxFinalResultBytes`, 64 KiB) FIFO of final results. A final result is only ever evicted when + the FIFO is full and no new final can otherwise be delivered, and that eviction is itself + reported through `ISpeechDiagnostics` at `Warning` level, since silently dropping a final result + without a trace would hide genuine data loss from a slow consumer. + +Both policies independently bound the subsystem's steady-state memory and latency instead of +letting either grow without limit when a consumer or the native backend cannot keep up. + +#### Composition + +`SpeechRecognizerFactory` is the subsystem's composition root. Its `LoadAsync` overloads check, +in order, whether the requested model's files exist on disk and whether the model declares the +recognition role; any failure returns `UnavailableSpeechRecognizerEngine.Instance` with a +structural diagnostic explaining which condition failed. Only then does it ask an +`IRecognitionBackendFactory` to load the model, and a failure there - the missing-native-runtime +case - degrades exactly the same honest way rather than throwing. Loading runs on a +`DedicatedWorker` thread and is awaited by the returned `Task`, so the expensive native-backend +construction never blocks the calling thread. A capture device is bound later, per session, via +`ISpeechRecognizerEngine.CreateSessionAsync` - not here - so one loaded engine can be reused +across many devices or many sequential sessions over its life. + +The running pipeline in `SherpaOnnxRecognitionSession` still spans two threads by design. The +capture device raises frames on a high-priority audio callback thread, so the session's frame +handler does nothing but copy the block into the bounded, drop-oldest queue and return; all +conversion, inference, and result buffering happens on the `DedicatedWorker` pump thread. +`StopAsync` completes the queue and awaits that pump task (subject to the abandon policy above), +so every result derived from audio captured before the stop request is either delivered or +accounted for as an evicted-with-diagnostic final by the time the call returns. `AudioFrameResampler` performs the format conversion, downmixing to mono and using simple linear interpolation for rate conversion except on the downsampling path, where a small windowed-sinc -FIR lowpass filter now runs immediately before decimation to attenuate above-target-Nyquist -energy. The internal `IRecognitionEngine`/`IRecognitionEngineFactory` seam confines every -sherpa-onnx call to `SherpaOnnxRecognitionEngine`/`SherpaOnnxRecognitionEngineFactory`. That seam -is the reason the whole pipeline is verifiable in CI: the recognizer's threading, conversion, -fault containment, and result ordering are all exercised through pure managed fakes with no model -file and no native inference binary present. It mirrors the `IPortAudioApi` seam used for audio +FIR lowpass filter runs immediately before decimation to attenuate above-target-Nyquist energy. +The internal `IRecognitionBackend`/`IRecognitionBackendFactory` seam confines every sherpa-onnx +call to `SherpaOnnxRecognitionEngine`/`SherpaOnnxRecognitionEngineFactory`. That seam is the +reason the whole pipeline is verifiable in CI: the session's threading, conversion, fault +containment, and result ordering are all exercised through pure managed fakes with no model file +and no native inference binary present. It mirrors the `IPortAudioApi` seam used for audio interop and the `IModelDownloadClient` seam used for downloads. -#### UnavailableSpeechRecognizer +#### UnavailableSpeechRecognizerEngine -**Purpose**: Provide a safe, always-obtainable `ISpeechRecognizer` fallback for use when +**Purpose**: Provide a safe, always-obtainable `ISpeechRecognizerEngine` fallback for use when recognition is not possible on the current machine. +**Data Model**: No instance fields. Exposes a single static `Instance` singleton; the constructor +is private. `IsAvailable` always returns `false`. + +**Key Methods**: + +- **CreateSessionAsync(IAudioCaptureDevice, CancellationToken)**: Never throws for the ordinary + unavailable machine state; returns a completed task holding + `UnavailableRecognitionSession.Instance` regardless of the supplied device's own availability. + Still throws `ArgumentNullException` synchronously for a null device, and + `OperationCanceledException` synchronously for an already-cancelled token, since those are + caller errors rather than machine states. + +**Error Handling**: Never throws for an ordinary unavailable machine state; unavailability is +represented entirely by the returned session's own behavior. A null device or an +already-cancelled token is still a caller error and throws synchronously, same as the real engine. + +**Dependencies**: `UnavailableRecognitionSession`; implements `ISpeechRecognizerEngine`. + +**Callers**: `SpeechRecognizerFactory.LoadAsync(...)` for every unavailable state. + +#### UnavailableRecognitionSession + +**Purpose**: Provide a safe, always-obtainable `IRecognitionSession` fallback returned by +`UnavailableSpeechRecognizerEngine.CreateSessionAsync`. + **Data Model**: No instance fields other than the never-invoked backing field for -`ResultReceived`. Exposes a single static `Instance` singleton; the constructor is private. -`IsAvailable` always returns `false`. +`StateChanged`. Exposes a single static `Instance` singleton; the constructor is private. +`IsAvailable` always returns `false`; `State` always reports `Created`, since this session never +runs and so never reaches any other state. **Key Methods**: -- **Start()** / **Stop()**: Always throw `SpeechRecognizerUnavailableException`. -- **Dispose()**: A no-op. Disposal must never throw or invalidate the shared instance, because a - host that wraps its recognizer in a disposal scope receives this instance and may dispose it - repeatedly. -- **ResultReceived**: Never raised; subscribing and unsubscribing are safe no-ops. +- **StartAsync()**: Always throws `SpeechRecognizerUnavailableException`. +- **StopAsync()** / **DisposeAsync()**: Safe no-ops that complete synchronously; this session + was never running and owns no engine, thread, or native resource, so there is nothing to stop + or release. Disposal must never throw or invalidate the shared instance, because a host that + wraps its session in a disposal scope receives this instance and may dispose it repeatedly. +- **GetResultsAsync()**: Throws `SpeechRecognizerUnavailableException` synchronously at the + point of invocation (not deferred to the first `MoveNextAsync`). +- **StateChanged**: Never raised; subscribing and unsubscribing are safe no-ops. -**Error Handling**: Signals misuse of a known-unavailable recognizer with +**Error Handling**: Signals misuse of a known-unavailable session with `SpeechRecognizerUnavailableException`. -**Dependencies**: `SpeechRecognizerUnavailableException`; implements `ISpeechRecognizer`. +**Dependencies**: `SpeechRecognizerUnavailableException`; implements `IRecognitionSession`. -**Callers**: `SpeechRecognizerFactory.Create(...)` for every unavailable state. +**Callers**: `UnavailableSpeechRecognizerEngine.CreateSessionAsync(...)` for every call. #### SpeechRecognizerUnavailableException -**Purpose**: Signal that an operational member of an unavailable recognizer was invoked, or that -a recognizer that claimed to be available failed on first use. +**Purpose**: Signal that an operational member of an unavailable recognizer engine or session was +invoked, or that a session that claimed to be available failed on first use. **Data Model**: No additional fields beyond the standard `Exception` base members. @@ -121,5 +244,41 @@ a recognizer that claimed to be available failed on first use. **Dependencies**: `Exception`. -**Callers**: `UnavailableSpeechRecognizer` for both operational members, and -`SherpaOnnxSpeechRecognizer.Start()` when its capture device fails to start. +**Callers**: `UnavailableRecognitionSession` for its operational members, and +`SherpaOnnxRecognitionSession.StartAsync()`/its frame handler when the capture device fails to +start or goes unavailable mid-session. + +#### RecognitionEngineBusyException + +**Purpose**: Signal that `ISpeechRecognizerEngine.CreateSessionAsync` was called while the +engine's single lease is already held by another live session (see "Engine exclusivity and lease +behavior" above). + +**Data Model**: No additional fields beyond the standard `Exception` base members. + +**Key Methods**: Standard three-constructor exception pattern. + +**Error Handling**: This type is itself the error-handling mechanism; the engine fails fast +(no queueing) rather than blocking the caller. + +**Dependencies**: `Exception`. + +**Callers**: `SherpaOnnxSpeechRecognizerEngine.CreateSessionAsync(...)`. + +#### RecognitionSessionFaultedException + +**Purpose**: Signal, through an active `GetResultsAsync` enumeration, that its session has +transitioned to `RecognitionSessionState.Faulted`. + +**Data Model**: No additional fields beyond the standard `Exception` base members. + +**Key Methods**: Standard three-constructor exception pattern. + +**Error Handling**: This type is itself the error-handling mechanism; it wraps the underlying +cause (for example a `SpeechRecognizerUnavailableException` from a capture device going +unavailable mid-session) as `InnerException`. + +**Dependencies**: `Exception`. + +**Callers**: `RecognitionResultBuffer.ReadAllAsync(...)` (consumed by +`SherpaOnnxRecognitionSession.GetResultsAsync`) once the buffer has been faulted. diff --git a/docs/design/speech/recognition-subsystem/i-recognition-session.md b/docs/design/speech/recognition-subsystem/i-recognition-session.md new file mode 100644 index 0000000..41c5634 --- /dev/null +++ b/docs/design/speech/recognition-subsystem/i-recognition-session.md @@ -0,0 +1,57 @@ +### IRecognitionSession + +**Purpose**: Define the Layer 5 contract for one streaming recognition session bound to one +capture device, so hosts and tests can depend on session lifecycle and result delivery without +depending on a specific inference backend. + +**Data Model**: `IsAvailable` indicates whether this session is backed by a real, leased backend +rather than an unavailable fallback. `State` reports the session's current +`RecognitionSessionState` - the forward-only `Created -> Starting -> Running -> Stopping -> +Stopped -> Disposing -> Disposed` machine (with `Faulted` reachable from `Starting`, `Running`, +or `Stopping`); see the subsystem-level design doc's state diagram. `StateChanged` is raised, as +a `SessionStateChangedEventArgs` carrying `Previous` and `Current`, for every transition. The +adjacent `SpeechRecognitionResult` record carries the full recognized `Text` of the current +utterance plus an `IsFinal` flag distinguishing a provisional hypothesis from a committed one; +`Text` is never a delta, so a host can render it directly. The `SpeechRecognitionEvent` record +wraps one result as the value yielded by `GetResultsAsync`, mirroring the AudioSubsystem's +`AudioCaptureFrameEventArgs` pattern. + +**Key Methods**: + +- **StartAsync(CancellationToken cancellationToken = default)**: Transitions `Created -> + Starting -> Running` and begins streaming recognition. Throws `InvalidOperationException` when + called more than once on the same session, since a session is single-use by design - a host + doing repeated, low-latency recognition should call `ISpeechRecognizerEngine.CreateSessionAsync` + again for the next turn rather than attempting to restart a stopped session. +- **StopAsync(CancellationToken cancellationToken = default)**: Transitions `Running -> + Stopping -> Stopped` and drains already-captured audio, so every result derived from audio + accepted before the call is either delivered or accounted for (see the subsystem-level + backpressure policy) before it returns - including the tail of an utterance that a streaming + backend could not otherwise decode without audio it will now never receive: an implementation + is expected to finalize and recover that trailing audio as one last final result rather than + silently losing it. This guarantee is best-effort in the same way backend faults elsewhere are: + if the backend itself faults while finalizing or resetting, the fault is reported rather than + thrown and `StopAsync` still completes, but the trailing audio and/or the backend's clean state + can no longer be guaranteed for that one call. Idempotent once stopped or faulted. +- **GetResultsAsync(CancellationToken cancellationToken = default)**: Returns an + `IAsyncEnumerable` of ordered provisional and final results. + Single-consumer: throws `InvalidOperationException` if called again while a previous + enumeration of the same session is still active. A session fault during enumeration surfaces as + `RecognitionSessionFaultedException` thrown from the enumerator, never silently. +- **DisposeAsync()**: Runs the full `Stopping -> Stopped` teardown first (if not already stopped + or faulted), then transitions `Disposing -> Disposed` and releases the engine's lease so the + owning `ISpeechRecognizerEngine` can create its next session. Idempotent. + +**Error Handling**: Real implementations and the unavailable fallback both use +`SpeechRecognizerUnavailableException` when an operational call is invalid because no usable +session is available or the capture device fails at first use or mid-session; such a fault also +transitions `State` to `Faulted` and is wrapped in `RecognitionSessionFaultedException` for any +active `GetResultsAsync` consumer. Starting more than once, or consuming `GetResultsAsync` from +more than one concurrent caller, throws `InvalidOperationException`. + +**Dependencies**: `RecognitionSessionState`, `SessionStateChangedEventArgs`, +`SpeechRecognitionResult`, `SpeechRecognitionEvent`, `SpeechRecognizerUnavailableException`, +`RecognitionSessionFaultedException`. + +**Callers**: `ISpeechRecognizerEngine.CreateSessionAsync(...)` constructs implementations; hosts +that render a live transcript call `StartAsync`, consume `GetResultsAsync`, and call `StopAsync`. diff --git a/docs/design/speech/recognition-subsystem/i-speech-recognizer-engine.md b/docs/design/speech/recognition-subsystem/i-speech-recognizer-engine.md new file mode 100644 index 0000000..95091b7 --- /dev/null +++ b/docs/design/speech/recognition-subsystem/i-speech-recognizer-engine.md @@ -0,0 +1,36 @@ +### ISpeechRecognizerEngine + +**Purpose**: Define the Layer 3 contract for one loaded, reusable speech-inference engine, so +hosts and tests can depend on "is recognition possible, and can I start a session?" without +depending on a specific inference backend. + +**Data Model**: `IsAvailable` indicates whether the engine is backed by a loaded backend and can +create a session at all. + +**Key Methods**: + +- **CreateSessionAsync(IAudioCaptureDevice device, CancellationToken + cancellationToken = default)**: Leases exclusive use of the loaded backend to a new + `IRecognitionSession` bound to `device`, and returns a task that completes with that + session. Never throws for an ordinary unavailable machine state - an unavailable engine returns + `UnavailableRecognitionSession.Instance` instead - but throws (faulting the returned task) + `RecognitionEngineBusyException` when another live session already holds the engine's lease, + since the backend cannot usefully decode two concurrent streams and this type never silently + queues a caller behind an unbounded wait. Precondition: `device` is non-null. + Postcondition: the returned session is never null, and either owns the leased backend or is the + shared unavailable instance. +- **DisposeAsync()**: Idempotent; disposes any still-active session this engine created before + releasing the engine's own native resources, so a host that disposes the engine directly (without + first disposing a session it is still holding) never leaks that session's native resources. + +**Error Handling**: `RecognitionEngineBusyException` is the one condition this contract +represents as an exception rather than a fallback value, because a concurrent lease attempt is a +caller-sequencing bug (create, use, and dispose one session before creating the next) rather than +an ordinary machine state. A null `device` throws (faulting the returned task) +`ArgumentNullException`. + +**Dependencies**: `IAudioCaptureDevice` from the AudioSubsystem; `IRecognitionSession`, +`RecognitionEngineBusyException`. + +**Callers**: `SpeechRecognizerFactory.LoadAsync(...)` constructs implementations; hosts call +`CreateSessionAsync` once per capture device they want to stream recognition from. diff --git a/docs/design/speech/recognition-subsystem/i-speech-recognizer.md b/docs/design/speech/recognition-subsystem/i-speech-recognizer.md deleted file mode 100644 index b2bbe46..0000000 --- a/docs/design/speech/recognition-subsystem/i-speech-recognizer.md +++ /dev/null @@ -1,45 +0,0 @@ -### ISpeechRecognizer - -**Purpose**: Define the contract for streaming speech-to-text, so hosts and tests can depend on -recognition behavior without depending on a specific inference engine. - -**Data Model**: `IsAvailable` indicates whether the recognizer is backed by a loaded engine and a -usable capture device. The adjacent `SpeechRecognitionResult` record carries the full recognized -`Text` of the current utterance plus an `IsFinal` flag distinguishing a provisional hypothesis -from a committed one; `Text` is never a delta, so a host can render it directly. The -`SpeechRecognitionEvent` record wraps one result as the event payload, mirroring the -AudioSubsystem's `AudioCaptureFrameEventArgs` pattern. - -**Key Methods**: - -- **Start()** / **Stop()**: Begin/end streaming recognition. Starting an already-running - recognizer and stopping one that is not running are both safe no-ops. `Stop()` drains - already-captured audio, so every result derived from audio accepted before the call is - delivered before it returns - including the tail of an utterance that a streaming engine could - not otherwise decode without audio it will now never receive (for example a push-to-talk - release with no trailing silence): an implementation is expected to finalize and recover that - trailing audio as one last final result rather than silently losing it, so no audio accepted - before `Stop()` carries over into (or is lost between) sessions. This guarantee is best-effort - in the same way engine faults elsewhere are: if the engine itself faults while finalizing or - resetting, the fault is reported rather than thrown and `Stop()` still completes, but the - trailing audio and/or the engine's clean state can no longer be guaranteed for that one call. A - recognizer supports many independent Start/Stop cycles on the same instance without - reconstruction - obtaining a recognizer from `SpeechRecognizerFactory` is the expensive step, so - a host doing repeated, low-latency recognition (for example, many turns of a voice - conversation) should construct one recognizer once and reuse it across cycles rather than - disposing and recreating it per turn. -- **Dispose()**: Releases the engine resources a running recognizer holds. Implies `Stop()` and - is idempotent. -- **ResultReceived**: Raised for each provisional or final result. Raised from the recognizer's - own background decoding thread, never from the audio callback thread, so a handler may do - moderate work. Handler exceptions are caught and reported, never propagated. - -**Error Handling**: Real implementations and the unavailable fallback both use -`SpeechRecognizerUnavailableException` when an operational call is invalid because no usable -recognizer is available or the capture device fails at first use. Starting a disposed recognizer -throws `ObjectDisposedException`. - -**Dependencies**: `SpeechRecognitionResult`, `SpeechRecognitionEvent`, -`SpeechRecognizerUnavailableException`. - -**Callers**: `SpeechRecognizerFactory.Create(...)` and hosts that render a live transcript. diff --git a/docs/design/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md b/docs/design/speech/recognition-subsystem/sherpa-onnx-recognition-session.md similarity index 55% rename from docs/design/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md rename to docs/design/speech/recognition-subsystem/sherpa-onnx-recognition-session.md index 68aaa55..1d262d3 100644 --- a/docs/design/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md +++ b/docs/design/speech/recognition-subsystem/sherpa-onnx-recognition-session.md @@ -1,96 +1,211 @@ -### SherpaOnnxSpeechRecognizer +### SherpaOnnxRecognitionSession -This chapter covers the streaming recognizer together with the two units it exists to -coordinate - the recognition-engine seam and the audio-format converter - because none of the -three can be reviewed meaningfully in isolation: the recognizer's whole job is to move audio from -the converter into the engine on the right thread. +This chapter covers the streaming session together with its companion Layer 3 engine and the +other units it exists to coordinate - the recognition-backend seam, the result buffer, the +dedicated pump-thread worker, and the audio-format converter - because none of them can be +reviewed meaningfully in isolation: the session's whole job is to move audio from the converter +into the backend on the right thread and buffer the results back out. -**Purpose**: Stream a capture device's audio through format conversion into a recognition engine +**Purpose**: Stream a capture device's audio through format conversion into a recognition backend and deliver the resulting provisional and final results, without ever performing recognition work on the audio callback thread. -**Data Model**: Holds the owned `IRecognitionEngine`, the `IAudioCaptureDevice` it streams from, -the owning `IRecognitionModel` (used only to normalize result text), an `AudioFrameResampler` -configured at construction from the device's reported format and the model's declared `AudioFormat.SampleRate`, and -a diagnostics sink. While running it also holds a bounded `Channel` of pending capture -blocks and the background consumer `Task` draining it; both are created on start and cleared on -stop, guarded by a lock together with the running and disposed flags. `IsAvailable` is always -`true`, because this type is only ever created after the engine loaded and the device reported -itself available - every unavailable case is represented by `UnavailableSpeechRecognizer` -instead. - -The pending-block queue holds 64 blocks and drops the oldest when full. Recognition that has -fallen behind live audio cannot be caught up by queueing more of it, so bounding the backlog -keeps both memory and latency flat instead of letting them grow without limit. +**Data Model**: Holds the leased `IRecognitionBackend` (owned by, and disposed by, the owning +`SherpaOnnxSpeechRecognizerEngine` - the session never disposes it, so the backend stays "hot" +across sessions), the `IAudioCaptureDevice` it streams from, the owning `IRecognitionModel` (used +only to normalize result text), an `AudioFrameResampler` configured at construction from the +device's reported format and the model's declared `AudioFormat.SampleRate`, a `DedicatedWorker` +that runs the pump loop, a `RecognitionResultBuffer`, a callback to release the engine's lease on +disposal, and a diagnostics sink. While running it also holds a bounded `Channel` of +pending capture blocks and the pump `Task` draining it; both are created on start and cleared on +stop, guarded by a lock together with `State`. `IsAvailable` is always `true`, because this type +is only ever created after the backend loaded and the device reported itself available - every +unavailable case is represented by `UnavailableRecognitionSession` instead. + +The pending-block queue holds `PendingFrameCapacity` (64) blocks and drops the oldest when full +(`BoundedChannelFullMode.DropOldest`). Recognition that has fallen behind live audio cannot be +caught up by queueing more of it, so bounding the backlog keeps both memory and latency flat +instead of letting them grow without limit. **Key Methods**: -- **Start()**: Creates the queue and consumer task, subscribes to `FrameCaptured`, then starts - the capture device. Subscribing before starting guarantees no captured block can be raised - before there is a handler to enqueue it. Starting an already-running recognizer is a no-op. - Precondition: not disposed. Postcondition: capture is running and results will be raised. -- **Stop()**: Unsubscribes, completes the queue, joins the consumer task, then stops the capture - device. Draining the consumer does two things in order: first it processes every already - queued block through the engine as normal, then - as its last action, still on the same - background decoding thread - it flushes the engine (`TryFlush`), finalizing and delivering any - trailing audio the engine had accepted but not yet decoded, for example the tail of an - utterance released with no trailing silence, which a streaming engine cannot normally decode - without more audio it will now never receive. Because of that flush, every result derived - from audio captured before the call - including that trailing fragment - has been delivered - when `Stop()` returns. The engine is then reset, discarding whatever the flush could not - recover and any stale hypothesis left over from the session just stopped, so a later `Start()` - on the same "hot" engine always begins decoding from a clean start-of-utterance state, - equivalent to a freshly constructed stream without the cost of reloading the model. Both the - flush and the reset are best-effort: each crosses the native decoder boundary, and a fault - there is reported through `ISpeechDiagnostics` rather than thrown, so `Stop()` still completes - and a later `Start()` is still permitted, but in that one failure case the trailing fragment - may go undelivered, the engine's state cannot be guaranteed clean, and in rare cases restart - may not fully recover the ability to decode. Stopping a recognizer that is not running is a - no-op. -- **Dispose()**: Performs the stop sequence (if running, which flushes and resets the engine and - reports the same way if either faults) and disposes the owned engine. Terminal regardless of - whether the flush or reset succeeded: `Dispose()` always disposes the engine and permanently - blocks a later `Start()`, so the restart guarantee above is specific to `Stop()`. Idempotent. -- **OnFrameCaptured(...)**: Runs on the audio callback thread. Copies the block and enqueues it; - nothing else. -- **ProcessFrame(...)**: Runs on the consumer thread. Converts the block through - `AudioFrameResampler`, feeds it to the engine, then polls the engine for results. Each result's - text is passed through the owning model's `IRecognitionModel.NormalizeText(text, isFinal)` - before `ResultReceived` raises it, so a model whose raw output needs casing/contraction/ - punctuation restoration (see `UppercaseTranscriptRestorer`) surfaces readable text to every - consumer instead of the engine's raw output; bounded at 32 results per block so a faulty engine - cannot livelock the consumer. -- **FlushFinal()**: Runs on the consumer thread, as the last action of the consumer loop once the - queue is drained and completed - so still before `Stop()`/`Dispose()` unblock their caller, and - still on the same background decoding thread as every other `ResultReceived` event. Calls the - engine's `TryFlush()` to finalize and decode any audio accepted but not yet decoded, and raises - the result the same way `ProcessFrame` does (same `NormalizeText` call) if there is one. - Contained the same way as `ProcessFrame`: a fault crossing the native decoder boundary, or from - a host's own handler, is reported rather than thrown. +- **StartAsync(CancellationToken)**: Transitions `Created -> Starting -> Running`. Creates the + queue, subscribes to `FrameCaptured`, then starts the capture device. Subscribing before + starting guarantees no captured block can be raised before there is a handler to enqueue it. + Throws `InvalidOperationException` when called more than once (sessions are single-use). + Precondition: not already started. Postcondition: capture is running and results will be + buffered. If the device fails to start, transitions to `Faulted`, faults the result buffer, and + throws `SpeechRecognizerUnavailableException` with the device's exception as `InnerException`. +- **StopAsync(CancellationToken)**: Transitions `Running -> Stopping -> Stopped` (idempotent once + `Stopped`/`Disposing`/`Disposed`/`Faulted`). Unsubscribes, completes the queue, cancels the pump + task's token (purely the `DedicatedWorker` abandon-timeout deadline - the pump loop itself never + observes this token while draining), awaits the pump task, completes the result buffer, resets + the backend, then stops the capture device. Draining the pump does two things in order: first + it processes every already-queued block through the backend as normal, then - as its last + action, still on the same pump thread - it flushes the backend (`TryFlush`), finalizing and + delivering any trailing audio the backend had accepted but not yet decoded, for example the tail + of an utterance released with no trailing silence, which a streaming backend cannot normally + decode without more audio it will now never receive. Because of that flush, every result derived + from audio captured before the call - including that trailing fragment - has been delivered (or + accounted for per the backpressure policy) when `StopAsync` returns. The backend is then reset, + discarding whatever the flush could not recover and any stale hypothesis left over from the + session just stopped, so the next session the owning engine creates on the same "hot" backend + always begins decoding from a clean start-of-utterance state, equivalent to a freshly constructed + stream without the cost of reloading the model. The flush, the reset, and an abandoned pump + thread are all best-effort: each is reported through `ISpeechDiagnostics` rather than thrown, so + `StopAsync` still completes, but in those cases the trailing fragment may go undelivered and/or + the backend's state cannot be guaranteed clean. +- **GetResultsAsync(CancellationToken)**: Single-consumer (`Interlocked.CompareExchange` guard, + throws `InvalidOperationException` on re-entry); awaits `RecognitionResultBuffer.ReadAllAsync` + and yields each result. +- **DisposeAsync()**: Runs the full `Stopping -> Stopped` teardown via `StopAsync` first (so a + session disposed directly from `Running` still visits that teardown rather than skipping it), + then transitions `Disposing -> Disposed` and invokes the release-lease callback so the owning + engine's next `CreateSessionAsync` can succeed. Idempotent. +- **OnFrameCaptured(...)**: Runs on the audio callback thread. Checks `_device.IsAvailable` + (faulting the session via `FaultSession` if it has gone unavailable mid-session), copies the + block, and enqueues it; nothing else. +- **PumpLoop(...)**: Runs on the `DedicatedWorker` pump thread. Converts each block through + `AudioFrameResampler`, feeds it to the backend, then polls the backend for results and writes + them into the `RecognitionResultBuffer`. Each result's text is passed through the owning model's + `IRecognitionModel.NormalizeText(text, isFinal)` before buffering, so a model whose raw output + needs casing/contraction/punctuation restoration (see `UppercaseTranscriptRestorer`) surfaces + readable text to every consumer instead of the backend's raw output; bounded at + `MaxResultsPerFrame` (32) results per block so a faulty backend cannot livelock the pump. As its + last action once the queue is drained and completed, flushes the backend (`TryFlush`) and + buffers the trailing final result the same way, before the queue-completion path returns. +- **FaultSession(Exception)**: Transitions to `Faulted` from `Starting`/`Running`/`Stopping`, + faults the result buffer with the cause (surfacing it to `GetResultsAsync` as + `RecognitionSessionFaultedException`), and cancels the pump token. **Error Handling**: Every stage is contained. A failure enqueueing a block, converting it, -running inference on it, or delivering a result to a host handler is caught and reported through +running inference on it, or delivering a result is caught and reported through `ISpeechDiagnostics`, never rethrown - an exception escaping the frame handler would propagate into the native PortAudio callback and tear down the audio stream, and a single bad block must -never end the session. A capture device that reported itself available but fails on `Start()` is -the one genuine failure the caller asked for, so it unwinds the partially started pipeline and is -surfaced as `SpeechRecognizerUnavailableException` with the device's exception as -`InnerException`. A device reporting a non-positive rate or channel count falls back to a -pass-through conversion with a warning rather than throwing at composition. Starting after -disposal throws `ObjectDisposedException`. +never end the session. A capture device that reported itself available but fails on `StartAsync` +or goes unavailable mid-session is surfaced as `SpeechRecognizerUnavailableException` (with the +device's exception as `InnerException` when applicable) and faults the session. A device +reporting a non-positive rate or channel count falls back to a pass-through conversion with a +warning rather than throwing at composition. Starting more than once, or consuming +`GetResultsAsync` from more than one concurrent caller, throws `InvalidOperationException`. + +**Dependencies**: `IRecognitionBackend`, `AudioFrameResampler`, `RecognitionResultBuffer`, +`DedicatedWorker`, `RecognitionSessionState`, `SessionStateChangedEventArgs`, +`SpeechRecognitionResult`, `SpeechRecognitionEvent`, `SpeechRecognizerUnavailableException`, +`RecognitionSessionFaultedException`, `IRecognitionModel` (from the ModelManagementSubsystem, for +`NormalizeText`), `IAudioCaptureDevice` and `AudioCaptureFrameEventArgs` from the AudioSubsystem, +and `ISpeechDiagnostics` from the Diagnostics subsystem. + +**Callers**: `SherpaOnnxSpeechRecognizerEngine.CreateSessionAsync(...)`. + +#### SherpaOnnxSpeechRecognizerEngine + +**Purpose**: Implement the Layer 3 `ISpeechRecognizerEngine` contract against a loaded +`IRecognitionBackend`, enforcing the subsystem's single-session-at-a-time exclusivity (see the +subsystem-level design doc's "Engine exclusivity and lease behavior"). + +**Data Model**: Holds the loaded `IRecognitionBackend`, the owning `IRecognitionModel`, a +`SemaphoreSlim(1, 1)` lease, and a diagnostics sink; constructs a new `DedicatedWorker` per +session it creates. `IsAvailable` is always `true`, because this type is only ever created after the +backend loaded successfully - every unavailable case is represented by +`UnavailableSpeechRecognizerEngine` instead. + +**Key Methods**: + +- **CreateSessionAsync(IAudioCaptureDevice, CancellationToken)**: When the supplied device + reports `IsAvailable` as `false` (no microphone or no working audio backend - an ordinary + machine state, exactly like a model not being installed), reports a Warning diagnostic and + returns `UnavailableRecognitionSession.Instance` without attempting the backend lease. Otherwise + attempts `_lease.Wait(0, ...)` - + a zero-timeout, fail-fast acquire with no queueing. On success, constructs and returns a new + `SherpaOnnxRecognitionSession` wrapping the shared backend, device, model, resampler + configuration, diagnostics sink, pump worker, and a release-lease callback that calls + `_lease.Release()`. On failure (the lease is already held), throws + `RecognitionEngineBusyException` without blocking. +- **Dispose()**: Disposes the owned backend and the lease semaphore. Only safe to call once no + session still holds the lease. + +**Error Handling**: `RecognitionEngineBusyException` is the only condition this type represents as +an exception; every other call either succeeds or is routed through the created session's own +error handling. + +**Dependencies**: `IRecognitionBackend`, `SherpaOnnxRecognitionSession`, `DedicatedWorker`, +`RecognitionEngineBusyException`; implements `ISpeechRecognizerEngine`. + +**Callers**: `SpeechRecognizerFactory.LoadAsync(...)` constructs it; hosts call +`CreateSessionAsync` once per capture device. + +#### RecognitionResultBuffer + +**Purpose**: Isolate a possibly-slow `GetResultsAsync` consumer from the pump thread with a +bounded, two-tier buffer (see the subsystem-level design doc's "Backpressure policy"). + +**Data Model**: One overwritable slot for the latest unconsumed provisional result, plus a +byte-capped (`MaxFinalResultBytes`, 64 KiB, measured via each final result's `Text` length) FIFO +queue of final results, backed by a `Channel`-based signal so `ReadAllAsync` can await new data +without polling. + +**Key Methods**: + +- **Write(SpeechRecognitionEvent)**: A provisional result overwrites the single pending + provisional slot (an unconsumed provisional is superseded by the next one, never queued). A + final result is enqueued into the FIFO; if enqueueing it would exceed `MaxFinalResultBytes`, the + oldest queued final is evicted first (reported through `ISpeechDiagnostics` at `Warning` level) + and only as a last resort. +- **ReadAllAsync(CancellationToken)**: An `IAsyncEnumerable` yielding the + latest provisional and every queued final in the order they became ready, until `Complete()` is + called, or throwing `RecognitionSessionFaultedException` once `Fault(...)` has been called. +- **Complete()**: Signals that no further results will be written; lets a draining + `ReadAllAsync` finish normally once its backlog is consumed. +- **Fault(Exception)**: Records the cause and signals every current and future `ReadAllAsync` + consumer to throw `RecognitionSessionFaultedException` wrapping it. + +**Error Handling**: Eviction of an unconsumed final result is the one data-loss case this type +can produce, and it is always reported through `ISpeechDiagnostics` at `Warning` level rather than +silently dropped. + +**Dependencies**: `SpeechRecognitionEvent`, `RecognitionSessionFaultedException`, +`ISpeechDiagnostics`. + +**Callers**: `SherpaOnnxRecognitionSession`'s pump loop writes to it; `GetResultsAsync` reads from +it. + +#### DedicatedWorker + +**Purpose**: Run one blocking delegate (a model load, or a session's pump loop) on its own +dedicated, long-running thread rather than a pooled thread, and bound how long a caller ever +waits for it to react to cancellation (see the subsystem-level design doc's "Cancellation and +abandon policy"). + +**Data Model**: An optional `abandonTimeout` (default `DefaultAbandonTimeout`, 2 seconds), an +`ISpeechDiagnostics` sink, and a diagnostics category string, all supplied at construction. + +**Key Methods**: + +- **RunAsync(Func<CancellationToken, Task> or Func<CancellationToken, T> work, + CancellationToken)**: Starts `work` on a dedicated thread created with + `TaskCreationOptions.LongRunning`, forwards `CancellationToken` to it, and returns a `Task`/ + `Task` representing the work. If the caller cancels, `work`'s own token is signaled first + (cooperative cancellation); if `work` has not completed within `abandonTimeout` of that signal, + the returned task completes faulted with `OperationCanceledException` without waiting further, + and the abandonment is reported through the diagnostics sink at `Warning` level, including which + diagnostics category the abandoned work belonged to. + +**Error Handling**: An abandoned delegate is never forcibly terminated (the CLR provides no safe +way to do so) - it is left to run to completion or forever on its own thread, and only the +*caller's view* of it (the returned `Task`) is abandoned. A delegate that throws is surfaced +through the returned task exactly as `Task.Run` would. -**Dependencies**: `IRecognitionEngine`, `AudioFrameResampler`, `SpeechRecognitionResult`, -`SpeechRecognitionEvent`, `SpeechRecognizerUnavailableException`, `IRecognitionModel` (from the -ModelManagementSubsystem, for `NormalizeText`), `IAudioCaptureDevice` and -`AudioCaptureFrameEventArgs` from the AudioSubsystem, and `ISpeechDiagnostics` from the -Diagnostics subsystem. +**Dependencies**: `ISpeechDiagnostics`. -**Callers**: `SpeechRecognizerFactory.Create(...)`. +**Callers**: `SpeechRecognizerFactory.LoadAsync(...)` for model loading; +`SherpaOnnxRecognitionSession` for its pump loop (one new worker constructed per session by +`SherpaOnnxSpeechRecognizerEngine.CreateSessionAsync`). -#### IRecognitionEngine and IRecognitionEngineFactory +#### IRecognitionBackend and IRecognitionBackendFactory **Purpose**: Confine every speech-inference interop call behind one mockable boundary, so the -recognizer's threading, conversion, and result-delivery logic is verifiable with pure managed +session's threading, conversion, and result-delivery logic is verifiable with pure managed fakes - no downloaded model and no platform-specific native binary are ever required in CI. This mirrors the `IPortAudioApi` seam used for audio interop and the `IModelDownloadClient` seam used for downloads. @@ -99,30 +214,30 @@ for downloads. **Key Methods**: -- **IRecognitionEngine.AcceptSamples(ReadOnlySpan<float>)**: Buffers one block of mono +- **IRecognitionBackend.AcceptSamples(ReadOnlySpan<float>)**: Buffers one block of mono samples, already at the model's required rate, into the current utterance. Never blocks on decoding; an empty block is a no-op. -- **IRecognitionEngine.TryDecode(out SpeechRecognitionResult?)**: Decodes as much buffered audio +- **IRecognitionBackend.TryDecode(out SpeechRecognitionResult?)**: Decodes as much buffered audio as possible and reports the next result, returning `false` when there is nothing new. Reporting "nothing new" instead of an empty result keeps the caller's event stream free of duplicates while silence is streaming. Callers may poll until it returns `false`. -- **IRecognitionEngine.TryFlush(out SpeechRecognitionResult?)**: Finalizes and decodes any +- **IRecognitionBackend.TryFlush(out SpeechRecognitionResult?)**: Finalizes and decodes any buffered-but-undecoded audio - the tail of an utterance released with no trailing silence, which `TryDecode` alone cannot decode without future context that will now never arrive - and reports it as one last final result if it produced non-empty text, or `false` if there was nothing to recover. Called once, at session end, immediately before `Reset()`. -- **IRecognitionEngine.Reset()**: Discards a partially decoded utterance. -- **IRecognitionEngineFactory.Create(IRecognitionModel, string installedModelDirectory)**: Loads - an engine from the model's own declared configuration and required input `AudioFormat`. +- **IRecognitionBackend.Reset()**: Discards a partially decoded utterance. +- **IRecognitionBackendFactory.Create(IRecognitionModel, string installedModelDirectory)**: Loads + a backend from the model's own declared configuration and required input `AudioFormat`. -**Error Handling**: Implementations are not thread-safe by contract; the recognizer calls them -from exactly one consumer thread. Load failures surface as exceptions from `Create(...)`, which +**Error Handling**: Implementations are not thread-safe by contract; the session calls them from +exactly one pump thread. Load failures surface as exceptions from `Create(...)`, which `SpeechRecognizerFactory` converts into the honest unavailable fallback. **Dependencies**: `SpeechRecognitionResult`; `IRecognitionModel` from the ModelManagementSubsystem. -**Callers**: `SherpaOnnxSpeechRecognizer` (engine) and `SpeechRecognizerFactory` (factory). +**Callers**: `SherpaOnnxRecognitionSession` (backend) and `SpeechRecognizerFactory` (factory). #### SherpaOnnxRecognitionEngine and SherpaOnnxRecognitionEngineFactory @@ -152,7 +267,7 @@ configuration, so adding a model never requires changing the factory. (see "Replay eligibility is gated on genuine recognized text since the last reset" below). Otherwise the text is emitted as provisional, suppressed when empty or unchanged since the previous call. -- **Reset()**: The session-end reset (called by `SherpaOnnxSpeechRecognizer.StopCore`). Creates a +- **Reset()**: The session-end reset (called by `SherpaOnnxRecognitionSession.StopAsync`). Creates a replacement stream and disposes the existing one, clears the remembered provisional text, and (if enabled) clears the warm-up buffer, the grace-period counter, and the `_hasRecognizedTextSinceReset` flag. Recreating the stream - not just calling @@ -254,7 +369,7 @@ members throw `ObjectDisposedException` after disposal. the ModelManagementSubsystem. **Callers**: `SpeechRecognizerFactory` constructs the factory; the factory constructs the engine; -`SherpaOnnxSpeechRecognizer` uses the engine. +`SherpaOnnxRecognitionSession` uses the engine. #### AudioFrameResampler @@ -307,7 +422,7 @@ throwing, since a capture device may legitimately deliver a very short block. lowpass stage, shared with `PlaybackAudioResampler`'s identical need in the synthesis subsystem; otherwise none beyond the Base Class Library. -**Callers**: `SherpaOnnxSpeechRecognizer`. +**Callers**: `SherpaOnnxRecognitionSession`. #### Design Constraints diff --git a/docs/design/speech/recognition-subsystem/speech-recognizer-factory.md b/docs/design/speech/recognition-subsystem/speech-recognizer-factory.md index 80ddfbd..471a6fc 100644 --- a/docs/design/speech/recognition-subsystem/speech-recognizer-factory.md +++ b/docs/design/speech/recognition-subsystem/speech-recognizer-factory.md @@ -1,93 +1,92 @@ ### SpeechRecognizerFactory -**Purpose**: Provide the single composition entry point for obtaining an `ISpeechRecognizer`, so -all "can this machine recognize speech right now?" logic lives in one reviewable place. +**Purpose**: Provide the single composition entry point for obtaining an +`ISpeechRecognizerEngine`, so all "can this machine recognize speech right now?" logic lives in +one reviewable place. -**Data Model**: A static class with no state. Three public `Create(...)` overloads exist: one -resolving an installed-model directory from a caller-supplied `string`, one resolving it from -a `SpeechModelStore` directly via `store.GetCurrentDirectory(model.Id)`, and one resolving it -from a `SpeechModelCatalog` directly via `catalog.Store.GetCurrentDirectory(model.Id)`. All call -through to the same internal composition logic, each with an internal counterpart that accepts -an injected `IRecognitionEngineFactory` so composition can be verified without model files or a +**Data Model**: A static class with no state. Three public `LoadAsync(...)` overloads exist: one +resolving an installed-model directory from a caller-supplied `string`, one resolving it from a +`SpeechModelStore` directly via `store.GetCurrentDirectory(model.Id)`, and one resolving it from a +`SpeechModelCatalog` directly via `catalog.Store.GetCurrentDirectory(model.Id)`. All call through +to the same internal composition logic, which has its own internal overload that accepts an +injected `IRecognitionBackendFactory` so composition can be verified without model files or a native runtime. **Key Methods**: -- **Create(IRecognitionModel model, string installedModelDirectory, IAudioCaptureDevice - captureDevice, ISpeechDiagnostics? diagnostics, IReadOnlyDictionary<string, object>? - parameterValues = null)**: Returns a real `SherpaOnnxSpeechRecognizer` - when the model's installed directory exists, the model declares `SpeechModelRole.Recognition`, - the capture device reports `IsAvailable`, and the engine loads. Otherwise returns - `UnavailableSpeechRecognizer.Instance`. Preconditions: `model` and `captureDevice` are - non-null. Postcondition: the returned recognizer is never null, and either owns a loaded engine - or is the shared unavailable instance. Loading the engine allocates native resources, so the - returned recognizer must be disposed. The recommended caller pattern is to compose - `captureDevice` first via `AudioDeviceFactory.CreateCaptureDevice(selection, model.AudioFormat)` - so the device opens already matching the model when the backend honors the hint. +- **LoadAsync(IRecognitionModel model, string installedModelDirectory, ISpeechDiagnostics? + diagnostics = null, IReadOnlyDictionary<string, object>? parameterValues = null, + CancellationToken cancellationToken = default)**: Returns a task that completes with a real + `SherpaOnnxSpeechRecognizerEngine` when the model's installed directory exists, the model + declares `SpeechModelRole.Recognition`, and the backend loads. Otherwise completes with + `UnavailableSpeechRecognizerEngine.Instance`. Preconditions: `model` is non-null. Postcondition: + the returned engine is never null, and either owns a loaded backend or is the shared unavailable + instance. Loading runs on a `DedicatedWorker` thread rather than blocking the calling thread + synchronously, since it allocates native resources and can take meaningful time. `parameterValues` is an optional session-level parameter value bag forwarded to the model's own - `CreateEngineConfig(installedModelDirectory, parameterValues)` overload when the engine is + `CreateEngineConfig(installedModelDirectory, parameterValues)` overload when the backend is constructed; `null` (or any bag, for either of today's two shipped models) resolves to today's - exact parameter-less behavior via that member's default hook. -- **Create(IRecognitionModel model, SpeechModelStore store, IAudioCaptureDevice captureDevice, - ISpeechDiagnostics? diagnostics, IReadOnlyDictionary<string, object>? parameterValues = - null)**: A convenience overload with byte-for-byte identical behavior - to the `string`-based overload above; it resolves `store.GetCurrentDirectory(model.Id)` for the + exact parameter-less behavior via that member's default hook. A capture device is *not* supplied + here - it is bound later, per session, via `ISpeechRecognizerEngine.CreateSessionAsync`. +- **LoadAsync(IRecognitionModel model, SpeechModelStore store, ISpeechDiagnostics? diagnostics = + null, IReadOnlyDictionary<string, object>? parameterValues = null, CancellationToken + cancellationToken = default)**: A convenience overload with byte-for-byte identical behavior to + the `string`-based overload above; it resolves `store.GetCurrentDirectory(model.Id)` for the caller and delegates to the same overload, so a host never needs to know - `SpeechModelStore`'s on-disk directory-naming scheme just to compose a recognizer. - Preconditions: `model`, `store`, and `captureDevice` are non-null. Preconditions unchanged. -- **Create(IRecognitionModel model, SpeechModelCatalog catalog, IAudioCaptureDevice - captureDevice, ISpeechDiagnostics? diagnostics, IReadOnlyDictionary<string, object>? - parameterValues = null)**: A convenience overload delegating through the + `SpeechModelStore`'s on-disk directory-naming scheme just to compose an engine. Preconditions: + `model` and `store` are non-null. +- **LoadAsync(IRecognitionModel model, SpeechModelCatalog catalog, ISpeechDiagnostics? + diagnostics = null, IReadOnlyDictionary<string, object>? parameterValues = null, + CancellationToken cancellationToken = default)**: A convenience overload delegating through the `SpeechModelStore`-based overload via the catalog's own `Store` property, so a host that already - owns a `SpeechModelCatalog` for enumeration and download can compose a recognizer through that - same catalog instance, without constructing a second, potentially divergent `SpeechModelStore`. - Preconditions: `model`, `catalog`, and `captureDevice` are non-null. + owns a `SpeechModelCatalog` for enumeration and download can compose an engine through that same + catalog instance, without constructing a second, potentially divergent `SpeechModelStore`. + Preconditions: `model` and `catalog` are non-null. The checks run in a deliberate order - parameter validation, then installed, then role, then -device, then engine load - so a caller-supplied parameter value invalid for a recognized -parameter is rejected synchronously and loudly before any of the ordinary, never-throw machine -state checks run, and so the cheapest and most common cause of unavailability (a model not -downloaded yet) is reported first among those and no native memory is allocated for a recognizer -that could never run. +backend load - so a caller-supplied parameter value invalid for a recognized parameter is +rejected synchronously and loudly before any of the ordinary, never-throw machine state checks +run, and so the cheapest and most common cause of unavailability (a model not downloaded yet) is +reported first among those and no native memory is allocated for an engine that could never run. -**Reuse and concurrent pre-warming**: Since a call to `Create` is the expensive step (it loads -the model into native memory) while `ISpeechRecognizer.Start`/`Stop` are cheap, a host doing -repeated, low-latency recognition should construct one recognizer once via `Create` and reuse it -across many `Start`/`Stop` cycles rather than calling `Create` again per turn - see -`ISpeechRecognizer`'s own design doc for that reuse contract. Because this factory itself holds -no state, a host may also call `Create` concurrently from a background task to overlap the -model-load step with other work (for example, pre-warming the next turn's recognizer while the -current turn's prompt is still speaking); this is safe with respect to the factory, but only -safe with respect to a caller-supplied `diagnostics` sink when that sink is itself safe for -concurrent use from multiple threads. +**Reuse and concurrent pre-warming**: Since a call to `LoadAsync` is the expensive step (it loads +the model into native memory) while `ISpeechRecognizerEngine.CreateSessionAsync` and the resulting +session's `StartAsync`/`StopAsync` cycle are cheap, a host doing repeated, low-latency recognition +should load one engine once via `LoadAsync` and reuse it across many sequential sessions rather +than calling `LoadAsync` again per turn - see `ISpeechRecognizerEngine`'s own design doc for that +reuse contract. Because this factory itself holds no state, a host may also call `LoadAsync` +concurrently from a background task to overlap the model-load step with other work (for example, +pre-warming the next turn's engine while the current turn's prompt is still speaking); this is +safe with respect to the factory, but only safe with respect to a caller-supplied `diagnostics` +sink when that sink is itself safe for concurrent use from multiple threads. -**Error Handling**: Every ordinary machine state is represented as the honest unavailable -recognizer plus a structural diagnostic, never as an exception, per this library's "nothing -throws at composition" decision. An engine load failure - the missing-native-runtime case for a -missing `org.k2fsa.sherpa.onnx.runtime.{RID}` binary or unusable model files - is caught and -degraded identically to a missing model. Only a null `model`, `store`, `catalog`, `captureDevice`, -or engine factory throws `ArgumentNullException`, since a null argument is a programming error -rather than a machine state. **Breaking change**: `parameterValues` is now validated against -`model.Parameters` before any other work runs. A supplied key that names a parameter *not* -declared by `model` is still silently ignored exactly as before (this deliberately preserves the -documented cross-model-compatibility contract - a host reusing one settings bag across different -models must not break just because model B doesn't declare a parameter model A had) but now also -reports an `Info` diagnostic. A supplied value for a parameter *that is declared* by `model` but -fails that parameter's own validation (wrong CLR type, a `NumericParameter` value outside -`[Minimum, Maximum]` or - when `IsInteger` is `true` - a non-integral value, an unrecognized -`ChoiceParameter` option, or a non-`bool` for a `BooleanParameter`) now throws `ArgumentException` -synchronously from `Create()` naming the parameter id, model id, and the reason the value is -invalid, via the shared `SpeechModelParameterDiagnostics.ValidateAndReport` helper. Previously -such a value was silently substituted with a default deeper in the composed recognizer; a caller -targeting a specific, declared parameter on this model with an invalid value is a caller bug that -should surface immediately rather than silently misbehave later. +**Error Handling**: Every ordinary machine state is represented as the honest unavailable engine +plus a structural diagnostic, never as an exception, per this library's "nothing throws at +composition" decision. A backend load failure - the missing-native-runtime case for a missing +`org.k2fsa.sherpa.onnx.runtime.{RID}` binary or unusable model files - is caught and degraded +identically to a missing model. A null `model`, `store`, `catalog`, or backend factory results in +an `ArgumentNullException`: for the `string`-based overload this faults the returned task (the +null check lives inside its `async` implementation), while for the `SpeechModelStore`-/ +`SpeechModelCatalog`-based overloads it is thrown synchronously, before any task is created, since +those overloads are ordinary synchronous methods that validate their own arguments and delegate +to the `string`-based overload. Either way the exception is a programming error, never a machine +state, and is indistinguishable to an `await`-based caller. A supplied key in `parameterValues` that names a parameter *not* +declared by `model` is silently ignored (this deliberately preserves the documented +cross-model-compatibility contract - a host reusing one settings bag across different models must +not break just because model B doesn't declare a parameter model A had) but reports an `Info` +diagnostic. A supplied value for a parameter *that is declared* by `model` but fails that +parameter's own validation (wrong CLR type, a `NumericParameter` value outside `[Minimum, +Maximum]` or - when `IsInteger` is `true` - a non-integral value, an unrecognized +`ChoiceParameter` option, or a non-`bool` for a `BooleanParameter`) faults the returned task with +`ArgumentException` naming the parameter id, model id, and the reason the value is invalid, via +the shared `SpeechModelParameterDiagnostics.ValidateAndReport` helper. A `cancellationToken` +cancelled before loading completes faults the returned task with `OperationCanceledException`. **Dependencies**: `IRecognitionModel`, `SpeechModelRole`, `SpeechModelStore`, `SpeechModelCatalog`, and `SpeechModelParameterDiagnostics` from the ModelManagementSubsystem, -`IAudioCaptureDevice` from the AudioSubsystem, `ISpeechDiagnostics`/`NullSpeechDiagnostics` from -the Diagnostics subsystem, and the subsystem's own `IRecognitionEngineFactory`, -`SherpaOnnxRecognitionEngineFactory`, `SherpaOnnxSpeechRecognizer`, and -`UnavailableSpeechRecognizer`. +`ISpeechDiagnostics`/`NullSpeechDiagnostics` from the Diagnostics subsystem, and the subsystem's +own `IRecognitionBackendFactory`, `SherpaOnnxRecognitionEngineFactory`, +`SherpaOnnxSpeechRecognizerEngine`, `UnavailableSpeechRecognizerEngine`, and `DedicatedWorker`. **Callers**: Host applications composing speech recognition at start-up, and the system-level integration tests. diff --git a/docs/design/speech/recognition-subsystem/unavailable-recognition-session.md b/docs/design/speech/recognition-subsystem/unavailable-recognition-session.md new file mode 100644 index 0000000..df2ccf4 --- /dev/null +++ b/docs/design/speech/recognition-subsystem/unavailable-recognition-session.md @@ -0,0 +1,30 @@ +### UnavailableRecognitionSession + +**Purpose**: Provide a safe, always-obtainable `IRecognitionSession` fallback, returned by +`UnavailableSpeechRecognizerEngine.CreateSessionAsync`, for use when recognition is not possible +on the current machine. + +**Data Model**: No instance state other than the never-invoked backing field for `StateChanged`. +Exposes a single static `Instance` singleton; the constructor is private. `IsAvailable` always +returns `false`; `State` always reports `RecognitionSessionState.Created`, since this session +never runs and so never reaches any other state. + +**Key Methods**: + +- **StartAsync()**: Always throws `SpeechRecognizerUnavailableException`. +- **StopAsync()** / **DisposeAsync()**: Safe no-ops that complete synchronously; this session + was never running and owns no engine, thread, or native resource, so there is nothing to stop + or release, and disposal never throws or invalidates `Instance` - so a host that wraps its + session in a disposal scope runs unchanged on a machine without recognition. +- **GetResultsAsync()**: Throws `SpeechRecognizerUnavailableException` synchronously, at the + point of invocation - not deferred to the first `MoveNextAsync` - since this session has no + real engine or capture device to stream results from. +- **StateChanged**: Never raised; subscription and unsubscription are safe no-ops. + +**Error Handling**: Obtaining and holding the instance never throws. Only the operational members +throw (or fault), and only when actually invoked - a caller that checks `IsAvailable` first never +triggers them. + +**Dependencies**: `SpeechRecognizerUnavailableException`; implements `IRecognitionSession`. + +**Callers**: `UnavailableSpeechRecognizerEngine.CreateSessionAsync(...)` for every call. diff --git a/docs/design/speech/recognition-subsystem/unavailable-speech-recognizer-engine.md b/docs/design/speech/recognition-subsystem/unavailable-speech-recognizer-engine.md new file mode 100644 index 0000000..e114a78 --- /dev/null +++ b/docs/design/speech/recognition-subsystem/unavailable-speech-recognizer-engine.md @@ -0,0 +1,27 @@ +### UnavailableSpeechRecognizerEngine + +**Purpose**: Provide a safe, always-obtainable `ISpeechRecognizerEngine` fallback for use when +recognition is not possible on the current machine. + +**Data Model**: No instance state. Exposes a single static `Instance` singleton; the constructor +is private, since the type carries no state and multiple instances would provide no value. +`IsAvailable` always returns `false`. + +**Key Methods**: + +- **CreateSessionAsync(IAudioCaptureDevice, CancellationToken)**: Throws `ArgumentNullException` + synchronously if `device` is `null`, and throws `OperationCanceledException` synchronously if + `cancellationToken` is already cancelled - both are programming/caller errors, not machine + states. Otherwise never throws; returns a completed task holding + `UnavailableRecognitionSession.Instance` regardless of the supplied device's own availability, + so a caller that composes without checking `IsAvailable` first still receives a usable, + honestly-unavailable session rather than an exception. + +**Error Handling**: Obtaining and holding the instance never throws. The returned session is the +one place operational misuse surfaces, and only when actually invoked. This mirrors +`UnavailableAudioCaptureDevice` exactly, so both subsystems degrade the same recognizable way. + +**Dependencies**: `UnavailableRecognitionSession`; implements `ISpeechRecognizerEngine`. + +**Callers**: `SpeechRecognizerFactory.LoadAsync(...)` when the model is not installed, the +model's role is not recognition, or the backend cannot be loaded. diff --git a/docs/design/speech/recognition-subsystem/unavailable-speech-recognizer.md b/docs/design/speech/recognition-subsystem/unavailable-speech-recognizer.md deleted file mode 100644 index 77c000a..0000000 --- a/docs/design/speech/recognition-subsystem/unavailable-speech-recognizer.md +++ /dev/null @@ -1,26 +0,0 @@ -### UnavailableSpeechRecognizer - -**Purpose**: Provide a safe, always-obtainable `ISpeechRecognizer` fallback for use when -recognition is not possible on the current machine. - -**Data Model**: No instance state other than the never-invoked backing field for -`ResultReceived`. Exposes a single static `Instance` singleton; the constructor is private, since -the type carries no state and multiple instances would provide no value. `IsAvailable` always -returns `false`. - -**Key Methods**: - -- **Start()** / **Stop()**: Always throw `SpeechRecognizerUnavailableException`. -- **Dispose()**: A no-op that never throws and never invalidates `Instance`, so a host that wraps - its recognizer in a disposal scope runs unchanged on a machine without recognition. -- **ResultReceived**: Never raised; subscription and unsubscription are safe no-ops. - -**Error Handling**: Obtaining and holding the instance never throws. Only the operational members -throw, and only when actually invoked - a caller that checks `IsAvailable` first never triggers -them. This mirrors `UnavailableAudioCaptureDevice` exactly, so both subsystems degrade the same -recognizable way. - -**Dependencies**: `SpeechRecognizerUnavailableException`; implements `ISpeechRecognizer`. - -**Callers**: `SpeechRecognizerFactory.Create(...)` when the model is not installed, the model's -role is not recognition, the capture device is unavailable, or the engine cannot be loaded. diff --git a/docs/design/speech/synthesis-subsystem.md b/docs/design/speech/synthesis-subsystem.md index dcab5b1..807d0bc 100644 --- a/docs/design/speech/synthesis-subsystem.md +++ b/docs/design/speech/synthesis-subsystem.md @@ -4,64 +4,75 @@ ### Overview -The SynthesisSubsystem provides text-to-speech capability. Sub-phase 4a shipped its Layer 1 -concern: the closed, fixed Natural Language Audio Tag vocabulary and the model-independent -parser that recognizes inline bracket syntax against it. Sub-phase 4b, described below, adds -Layer 2 per-model rendering of a parsed span sequence into a `SpeechPlan`, a chunked/streaming -pipeline that synthesizes and plays speech, a real sherpa-onnx synthesis engine, and the public -`ISpeechSynthesizer` contract with its honest unavailable fallback. Phase 10 threads an optional, -session-level `parameterValues` bag (for example a selected voice) through -`SpeechSynthesizerFactory.Create` into `SherpaOnnxSpeechSynthesizer`, which resolves it to a real -sherpa-onnx speaker id once per synthesized segment via the ModelManagementSubsystem's new -`ISynthesisModel.ResolveSpeakerId` hook, replacing a previously hard-coded `speakerId: 0` - -without disturbing the independent, per-segment `ParameterOverrides` speed/volume mechanism -already described below. It contains the following direct units: +The SynthesisSubsystem provides text-to-speech capability. It shipped its Layer 1 concern (the +closed, fixed Natural Language Audio Tag vocabulary and the model-independent parser) and its +Layer 2 concern (per-model rendering of a parsed span sequence into a `SpeechPlan`) unchanged by +this redesign. The public and internal text-to-speech API has since been redesigned from a +single, sync-ish `ISpeechSynthesizer` streaming contract to an explicit, 5-layer async +Engine/Session vocabulary that mirrors the RecognitionSubsystem's own Engine/Session redesign: +`ISpeechSynthesizerEngine` (Layer 3 - the public, loaded, expensive, native-backed model, obtained +once) creates cheap, reusable `ISynthesisSession` instances (Layer 5 - one per bound playback +device), each with its own explicit `SynthesisSessionState` lifecycle, so a host no longer has to +reconstruct a synthesizer per utterance to avoid the old type's single-session-per-instance +assumption. The internal `ISynthesisEngine`/`ISynthesisEngineFactory` seam was renamed to +`ISynthesisBackend`/`ISynthesisBackendFactory`, freeing the word "engine" for the public Layer 3 +contract. It contains the following direct units: - **AudioTagParser**: the pure, model-independent scanner, together with the **AudioTagCatalog** alias/kind lookup table it consumes and the **NaturalLanguageAudioTag** / **NaturalLanguageAudioTagKind** / **TaggedTextSpan** / **TaggedTextSpanKind** value types it - operates over (Sub-phase 4a, unchanged) + operates over (unchanged by this redesign) - **IModelCapabilityProfile** and **DefaultModelCapabilityProfile**: the Layer 2 rendering strategy that turns a parsed span sequence into an ordered **SpeechPlan** of **SpeechSegment**s, deciding per tag whether to pass it through, approximate it, or strip it, driven purely by the - selected model's own declared audio-tag support and parameters + selected model's own declared audio-tag support and parameters (unchanged by this redesign) - **SentenceChunker**: splits synthesis-ready plain text into sentence/clause-sized chunks so - chunked, pipelined synthesis has natural-sounding boundaries -- **ISpeechSynthesizer**: the public chunked/streaming synthesis-and-playback contract, together - with the **SynthesizedSpeech** value type it yields -- **SpeechSynthesizerFactory**: composition root that returns either a real synthesizer or the - honest unavailable fallback, and never throws for an ordinary machine state -- **SherpaOnnxSpeechSynthesizer**: the real chunked/pipelined implementation, together with the - internal **ISynthesisEngine**/**ISynthesisEngineFactory** seam, its real - **SherpaOnnxSynthesisEngine**/**SherpaOnnxSynthesisEngineFactory** implementations, and the + chunked synthesis has natural-sounding boundaries (unchanged by this redesign) +- **ISpeechSynthesizerEngine**: the public Layer 3 contract for a loaded synthesis model - + availability, `CreateSessionAsync`, and one-shot `SpeakAsync`/`SynthesizeAsync` convenience + overloads - together with the **SynthesizedSpeech** value type it and `ISynthesisSession` yield +- **ISynthesisSession**: the public Layer 5 contract for one session bound to one playback + device - availability, a `SynthesisSessionState` lifecycle with a `StateChanged` event, + non-overlapping `SpeakAsync`/`SynthesizeAsync` operations, and `StopAsync` +- **SpeechSynthesizerFactory**: composition root that asynchronously loads a real engine or + returns the honest unavailable fallback, and never throws for an ordinary machine state +- **SherpaOnnxSpeechSynthesizerEngine**: the real `ISpeechSynthesizerEngine` implementation, + owning one loaded **SynthesisBackend** (the renamed internal + **ISynthesisBackend**/**ISynthesisBackendFactory** seam and its real + **SherpaOnnxSynthesisEngine**/**SherpaOnnxSynthesisEngineFactory** implementations) and + enforcing single-session exclusivity via a fail-fast lease +- **SherpaOnnxSynthesisSession**: the real `ISynthesisSession` implementation, together with the **PlaybackAudioResampler** that converts synthesized audio into the format a playback device requires -- **UnavailableSpeechSynthesizer** and **SpeechSynthesizerUnavailableException**: honest fallback - behavior when no model, no engine, or no playback device is available +- **SynthesisSessionState** and **SessionStateChangedEventArgs**: the session lifecycle state + enum and its `StateChanged` event-argument type +- **SynthesisEngineBusyException** and **SynthesisSessionFaultedException**: exceptions signaling + a lease conflict and a terminal session fault respectively +- **DedicatedWorker**: internal utility applying a cooperative-cancel-then-abandon policy to a + non-cooperative native synthesis call, synthesis's own duplicated copy of + RecognitionSubsystem's identically-shaped internal utility +- **UnavailableSpeechSynthesizerEngine**, **UnavailableSynthesisSession**, and + **SpeechSynthesizerUnavailableException**: honest fallback behavior when no model, no engine + backend, or no playback device is available ### Interfaces The subsystem exposes `NaturalLanguageAudioTag`, `NaturalLanguageAudioTagKind`, `AudioTagDescriptor`, `AudioTagCatalog`, `TaggedTextSpanKind`, `TaggedTextSpan`, `AudioTagParser` -(Sub-phase 4a, unchanged), `ISpeechSynthesizer`, `SynthesizedSpeech`, `SpeechSynthesizerFactory`, -`UnavailableSpeechSynthesizer`, and `SpeechSynthesizerUnavailableException` as its public API. It -consumes `IAudioPlaybackDevice` from the AudioSubsystem for output audio, `ISynthesisModel` from -the ModelManagementSubsystem for the engine configuration, preferred playback-format hint, and -Layer 2 rendering strategy, and `ISpeechDiagnostics` from the Diagnostics subsystem to report -structural composition, lifecycle, and fault facts without ever exposing synthesized text. - -Both cross-subsystem dependencies were extended in this phase, additively, mirroring Phase 3's -identical recognition-direction additions: `IAudioPlaybackDevice` gained -`ChannelCount`/`SampleRate` so the resolved playback format can be discovered (see -_IAudioPlaybackDevice Design_), and `ISynthesisModel` gained a public best-effort -`PreferredAudioFormat` plus internal `CreateEngineConfig` and default-hook `CapabilityProfile` so -each model owns both its playback hint and its engine configuration/Layer 2 rendering strategy -(see _SpeechModelContract Design_). +(unchanged), `ISpeechSynthesizerEngine`, `ISynthesisSession`, `SynthesizedSpeech`, +`SynthesisSessionState`, `SessionStateChangedEventArgs`, `SpeechSynthesizerFactory`, +`UnavailableSpeechSynthesizerEngine`, `UnavailableSynthesisSession`, +`SynthesisEngineBusyException`, `SynthesisSessionFaultedException`, and +`SpeechSynthesizerUnavailableException` as its public API. It consumes `IAudioPlaybackDevice` from +the AudioSubsystem for output audio, `ISynthesisModel` from the ModelManagementSubsystem for the +engine configuration, preferred playback-format hint, and Layer 2 rendering strategy, and +`ISpeechDiagnostics` from the Diagnostics subsystem to report structural composition, lifecycle, +and fault facts without ever exposing synthesized text. No member of the subsystem's public API names a sherpa-onnx type, per this library's "engine backend stays swappable at the public API surface" decision. The sherpa-onnx configuration type appears only on `ISynthesisModel`'s internal members and inside the -subsystem's internal engine seam. +subsystem's internal `SynthesisBackend` seam. ### Design @@ -86,8 +97,7 @@ adjacent tags and text at either edge of the input never merge with neighboring The parser is pure and allocation-light: it holds no state beyond one `StringBuilder` for the in-progress plain-text run and returns a single ordered `IReadOnlyList`. This makes it trivially unit-testable with plain strings and no model, engine, or device double -required - the entire subsystem's public surface at this stage of the design has no threading, -disposal, or fault-containment concerns for the same reason. +required. `IModelCapabilityProfile.Render(IReadOnlyList spans, ISpeechModel model)` is the Layer 2 rendering strategy: it consumes the model-independent span sequence Layer 1 produced and @@ -95,226 +105,315 @@ turns it into an ordered `SpeechPlan` of `SpeechSegment`s, deciding per tag - ba `model.AudioTagSupport` and `model.Parameters` - whether to pass the tag through as its canonical bracket text (native support), approximate it as a timed pause or a numeric parameter override on the surrounding segment (parameter-mapped support), or strip it to plain narration (no support or -no matching convention). This signature deliberately differs from the parameter-dictionary shape -sketched in the Phase 4 plan: `DefaultModelCapabilityProfile.Instance` is a stateless singleton -shared by every model, and the public `ISpeechSynthesizer` contract has no caller-supplied -parameter bag for it to consume, so the strategy instead reads everything it needs directly from -the model it is rendering for. This keeps the contract trivially satisfiable by a model that -wants the default behavior (the `CapabilityProfile` hook on `ISynthesisModel` simply returns -`DefaultModelCapabilityProfile.Instance`) while still letting a model override it with bespoke, -non-generic rendering when its native tag support needs something the default cannot express. +no matching convention). `DefaultModelCapabilityProfile.Instance` is a stateless singleton shared +by every model; a model may override `ISynthesisModel.CapabilityProfile` with bespoke rendering +when its native tag support needs something the default cannot express. `SpeechParameterConventions`, a small internal helper, is where the one, deliberately narrow mapping from tag to numeric parameter lives: only `Fast`/`VeryFast`/`Slow`/`VerySlow` map to a speed override (a fraction of the parameter's declared range) and only `Loud`/`Soft`/`Whispers` map to a volume override; every other tag (every Emotion, every Non-verbal cue, `Breathy`, `Emphasis`) has no built-in numeric meaning and silently strips under parameter-mapped support. -This risk mitigation for unsupported tags is deliberately kept conservative: a wrong guess at -what "louder" or "more emphatic" numerically means for an arbitrary model would be worse than -narrating the plain text, so the default profile only maps the handful of tags with an -unambiguous numeric direction and lets a model override the profile entirely if it wants richer -behavior. Pause tags (`ShortPause`/`LongPause`) always render as real silence regardless of a -model's declared support, per this library's explicit "pauses require no model cooperation" -decision - no model text is spoken for a pause, so there is nothing for it to get wrong. Ordinary -narration text between tags is split into `SpeechSegment`s by `SentenceChunker` so the resulting -`SpeechPlan` already has chunk-sized boundaries lined up with natural speech units before the -pipeline ever synthesizes anything. A plain-text chunk whose text ends in a genuine ellipsis (per -`SentenceChunker`'s `ChunkWithMetadata`) additionally renders a longer, distinct -`EllipsisPauseMilliseconds` pause (~500ms, between the tag-triggered short/long pause durations), -modeling the natural, longer trailing-off pause a speaker takes at a genuine ellipsis; this is a -chunk-boundary-triggered mechanism, independent from the explicit `[pause]`/`[long-pause]` tag -path above, and every other chunk boundary still adds zero silence. +Pause tags (`ShortPause`/`LongPause`) always render as real silence regardless of a model's +declared support, per this library's explicit "pauses require no model cooperation" decision. +Ordinary narration text between tags is split into `SpeechSegment`s by `SentenceChunker` so the +resulting `SpeechPlan` already has chunk-sized boundaries lined up with natural speech units. A +plain-text chunk whose text ends in a genuine ellipsis additionally renders a longer, distinct +`EllipsisPauseMilliseconds` pause (~500ms, between the tag-triggered short/long pause durations). `SentenceChunker.Chunk(text, maxLength)` (and its metadata-carrying sibling -`ChunkWithMetadata(text, maxLength)`, see below) splits primary sentence-ending punctuation first, -then **unconditionally** splits every resulting sentence-level piece further on secondary clause -punctuation (commas, semicolons, colons) - not only when a piece is still over the length budget - -so every clause becomes its own chunk regardless of overall sentence length. This keeps the -audible gap at a clause boundary short and predictable even for a long, multi-clause sentence, -directly addressing reports of audible playback delay around sentence/clause boundaries during -long dictated text-to-speech playback. Only after that does a piece still over `maxLength` fall -back further to a plain whitespace budget split. A single word that alone exceeds `maxLength` is -still returned whole, never split mid-word, since a partial word cannot be synthesized -intelligibly. - -The primary sentence-ending pass treats a maximal run of consecutive sentence-ending characters -(any combination of `.`, `!`, `?` - e.g. an ellipsis `...`, or mixed terminators such as `?!` or -`!!`) as a single boundary, splitting only once after the last character of the run rather than -once per character. The whole run of punctuation stays attached to the sentence that precedes it, -matching how a human naturally pauses once at an ellipsis rather than stopping three separate -times. This merged-run handling applies only to _immediately adjacent_ boundary characters; a -whitespace-spaced punctuation run (e.g. `". . ."`, or a lone `,` surrounded by spaces) instead -produces one or more separate, word-less pieces after both punctuation passes. - -Rather than emit those word-less pieces as their own degenerate chunks - which produced audible -glitches/artifacts, since synthesizing a near-empty chunk of just punctuation is a poor unit of -speech - `SentenceChunker` runs a final merge pass: any piece whose trimmed text contains no -letter or digit character at all (checked with `char.IsLetterOrDigit`) has its text appended, -joined by a single space, onto the immediately preceding non-empty chunk, or is dropped entirely -if there is no preceding chunk (a degenerate run at the very start of the text). For example, -`"Sentence one . . . . . Sentence two"` produces exactly two chunks, -`"Sentence one . . . . ."` and `"Sentence two"`, not five near-empty punctuation chunks. - -`ChunkWithMetadata` additionally reports, per chunk, whether its final (merged) text ends in a -genuine ellipsis - three or more consecutive `.` characters, with or without interspersed -whitespace, found by scanning backward from the end of the chunk's text, skipping whitespace. This -metadata (`SentenceChunk.EndsWithEllipsis`) feeds `DefaultModelCapabilityProfile.Render`'s -ellipsis-triggered pause, described below; `Chunk` itself is a pure projection of -`ChunkWithMetadata`'s chunk text, so its signature and return shape are unchanged - though the -actual chunk boundaries it now produces differ from before this change, since clause punctuation -is now always split. - -`SherpaOnnxSpeechSynthesizer`'s chunked, pipelined design mirrors -`SherpaOnnxSpeechRecognizer`'s threading pattern but runs it in the synthesis direction. A -producer task walks the `SpeechPlan` chunk by chunk, calling the engine to synthesize each -`SpeechSegment` in turn and writing the resulting `SynthesizedSpeech` into a bounded -`Channel` of capacity 8 (raised from an original 5 once `SentenceChunker` -started splitting clause punctuation unconditionally, since a typical multi-clause sentence now -yields roughly 1.5-2x as many, smaller chunks, so the same segment count now covers less audio -duration than before); the caller's own task drains that channel and plays -each segment as it arrives, so synthesis of a later chunk runs concurrently with playback of an -earlier one. Unlike the recognition-direction channel, which uses `DropOldest` because live -capture audio can tolerate drops, this channel uses `BoundedChannelFullMode.Wait`: synthesized -audio has already cost real inference time, so it must never be silently discarded, and instead -the producer simply waits for the consumer to catch up. A pause segment (empty text) skips the -engine entirely and produces pure silence directly, since there is nothing for the engine to -synthesize. `PlaybackAudioResampler` performs the playback-direction format conversion - -resample from the engine's actual rate to the device's resolved rate, using the same small -windowed-sinc FIR anti-aliasing step before downsampling decimation that the recognition-side -resampler now uses, then upmix mono to the device's channel count - mirroring -`AudioFrameResampler`'s identical recognition-direction role. The internal -`ISynthesisEngine`/`ISynthesisEngineFactory` seam confines every sherpa-onnx call to -`SherpaOnnxSynthesisEngine`/`SherpaOnnxSynthesisEngineFactory`, mirroring the -`IRecognitionEngine`/`IRecognitionEngineFactory` seam so the whole pipeline - chunking, Layer 2 -rendering, pipelined synthesize-while-play ordering, cancellation, and fault containment - is +`ChunkWithMetadata(text, maxLength)`) splits primary sentence-ending punctuation first, then +unconditionally splits every resulting sentence-level piece further on secondary clause +punctuation (commas, semicolons, colons), falling back to a plain whitespace budget split only +when a piece is still over `maxLength`. A single word that alone exceeds `maxLength` is still +returned whole, never split mid-word. The primary sentence-ending pass treats a maximal run of +consecutive sentence-ending characters (e.g. an ellipsis `...`, or mixed terminators such as +`?!`) as a single boundary. A final merge pass folds any piece whose trimmed text contains no +letter or digit character onto the immediately preceding non-empty chunk, or drops it if none +precedes it. `ChunkWithMetadata.EndsWithEllipsis` reports whether a chunk's final text ends in a +genuine ellipsis, feeding `DefaultModelCapabilityProfile.Render`'s ellipsis-triggered pause. + +#### Layer 3/Layer 5 Vocabulary: Engine, Session, and the Lifecycle State Machine + +An `ISpeechSynthesizerEngine` represents one loaded synthesis model - expensive to obtain +(`SpeechSynthesizerFactory.LoadAsync` loads native model files), cheap to keep around for a +program's whole life. Calling `CreateSessionAsync(device, cancellationToken)` binds that engine to +one `IAudioPlaybackDevice` and returns an `ISynthesisSession` - cheap to obtain, bound to that one +device for its entire life, and safely reusable across many `SpeakAsync`/`SynthesizeAsync` calls +without reconstruction. This closes the Demo application's "reconstructs a synthesizer per Play" +bug class at the API level: a host now holds one session per device for as long as it needs it, +rather than one short-lived synthesizer per utterance. + +At most one session may be leased from a given engine at a time. `SherpaOnnxSpeechSynthesizerEngine` +enforces this with a fail-fast binary semaphore: `CreateSessionAsync` either acquires the lease +and returns a new session immediately, or - if the lease is already held by a still-undisposed +session - throws `SynthesisEngineBusyException` immediately, never queueing or waiting. The lease +is held for the leased session's entire life and is released exactly once, from that session's +`DisposeAsync`, so a new session can be created again only after the prior one is disposed. +Fail-fast was chosen over waiting because waiting would make `CreateSessionAsync`'s latency depend +on an unrelated session's teardown, with no caller-visible way to bound that wait; a caller that +legitimately wants more than one session simply obtains more than one engine. + +Each `ISynthesisSession` exposes an explicit `SynthesisSessionState` lifecycle: `Created`, +`Starting`, `Running`, `Stopping`, `Stopped`, `Disposing`, `Disposed`, `Faulted`. Unlike +recognition's continuous capture-window states, `Starting`/`Running`/`Stopping` denote one +discrete in-flight `SpeakAsync`/`SynthesizeAsync` operation, not a continuous stream: a session +returns to `Stopped` after each operation completes and is immediately ready to accept another +call. The `StateChanged` event reports every transition; a subscriber's handler exception is +caught and routed to diagnostics rather than propagated, so a misbehaving host handler can never +destabilize the session's own lifecycle. + +```mermaid +stateDiagram-v2 + [*] --> Created + Created --> Starting : SpeakAsync/SynthesizeAsync called + Starting --> Running + Running --> Stopping : operation completes, cancels, or faults + Stopping --> Stopped : success or cancellation + Stopping --> Faulted : non-cancellation failure + Stopped --> Starting : SpeakAsync/SynthesizeAsync called again + Stopped --> Disposing : DisposeAsync + Faulted --> Disposing : DisposeAsync + Disposing --> Disposed + Disposed --> [*] + + note right of Running + Starting/Running/Stopping denote one + discrete in-flight operation, not a + continuous stream. A session returns + to Stopped and is reusable after each + call. + end note +``` + +A second `SpeakAsync`/`SynthesizeAsync` call while the session is `Starting`, `Running`, or +`Stopping` throws `InvalidOperationException` immediately rather than queueing: the two methods +on a given session never overlap. This matches the lease's fail-fast philosophy and keeps the +session's own state machine simple - exactly one operation is ever in flight per session. A caller +that genuinely needs concurrent utterances obtains a second session (if the engine's lease +permits) or awaits the first operation's completion. + +Once a session faults - any non-cancellation failure during an operation, including a native call +the `DedicatedWorker` had to abandon rather than wait for indefinitely - it transitions to the +terminal `Faulted` state and every subsequent `SpeakAsync`/`SynthesizeAsync` call throws +`SynthesisSessionFaultedException`, wrapping the original fault as its inner exception. A faulted +session cannot resume; the caller must dispose it (releasing the engine's lease) and create a +replacement. `StopAsync` requests cooperative cancellation of whichever operation is currently in +flight (a no-op when none is) by cancelling that operation's own linked `CancellationTokenSource`; +the operation then unwinds through `Stopping` to `Stopped`, not `Faulted`, since a caller-requested +stop is an ordinary, successful outcome rather than a fault. + +`SherpaOnnxSynthesisSession`'s per-operation pipeline normalizes the input text, tag-parses it, +Layer 2 renders it into a `SpeechPlan`, then synthesizes each `SpeechSegment` in turn: unlike the +former streaming synthesizer, synthesis is no longer pipelined ahead of playback across an +unbounded/bounded channel - each segment is synthesized, then (for `SpeakAsync`) immediately +written to the bound playback device, before the next segment's synthesis begins. This keeps the +per-call state machine simple (one backend call in flight at a time) while still overlapping this +call's own synthesis-then-playback work normally, since playback of one segment and synthesis of +the next both still happen without the caller waiting for the whole plan up front. Each segment's +native `Generate` call runs through `DedicatedWorker.Run`, so a non-cooperative native call is +bounded by the cooperative-cancel-then-abandon policy (see below) rather than awaited +indefinitely. A pause segment (empty text) skips the backend entirely and produces pure silence +directly. For `SpeakAsync`, once every segment has been written, `WaitForPlaybackDrainAsync` polls +`IAudioPlaybackDevice.PendingSampleCount` until it reaches zero, plus a small fixed tail margin, +before stopping the device - genuinely waiting for the hardware to finish rendering rather than +treating "every segment enqueued" as "finished playing". A `finally` block stops the playback +device regardless of whether the loop, the drain wait, or neither completed, faulted, or was +cancelled, reporting (but not rethrowing) a failure to stop so the device is never left running. +`PlaybackAudioResampler` performs the playback-direction format conversion - resample from the +backend's actual rate to the device's resolved rate using a small windowed-sinc FIR +anti-aliasing step before downsampling decimation, then upmix mono to the device's channel +count - mirroring `AudioFrameResampler`'s identical recognition-direction role. The internal +`ISynthesisBackend`/`ISynthesisBackendFactory` seam (renamed from +`ISynthesisEngine`/`ISynthesisEngineFactory`, members unchanged) confines every sherpa-onnx call +to `SherpaOnnxSynthesisEngine`/`SherpaOnnxSynthesisEngineFactory`, so the whole pipeline is verifiable in CI with pure managed fakes. -`SpeakAsync` composes `SynthesizeStreamAsync` and `PlayStreamAsync` under one cancellable session: -calling `Stop()` cancels that session's linked `CancellationTokenSource` so an in-flight -utterance stops promptly and deterministically (for example in response to a user's barge-in -action), while a `finally` block around playback guarantees the playback device is stopped even -when the session is cancelled or a segment faults - a dropped playback device or a model failure -must fail the caller's awaited task honestly, never hang or crash the process. - -#### ISpeechSynthesizer - -**Purpose**: Define the contract for chunked, streaming text-to-speech synthesis and playback, so +`DedicatedWorker.Run` applies a cooperative-cancel-then-abandon policy to a delegate run on a +dedicated `TaskCreationOptions.LongRunning` task: on cancellation it waits a bounded, +injectable `DefaultAbandonTimeout` (2 seconds) for the delegate to stop cooperatively; if it has +not stopped by then, the worker reports a `Warning` diagnostic, detaches the still-running task +(observing, rather than propagating, whatever exception it eventually produces, so it can never +become an unobserved-exception crash), and throws `OperationCanceledException` to its own caller. +This is `SynthesisSubsystem`'s own duplicated copy of `RecognitionSubsystem`'s identically-shaped +internal utility: sharing it would require standing up a third, shared internal subsystem for one +~100-line utility, judged disproportionate to the duplication it would remove, and each +subsystem's copy is already fully covered by its own subsystem-scoped tests. + +`ISpeechSynthesizerEngine.SpeakAsync(device, text, cancellationToken)` and +`SynthesizeAsync(text, cancellationToken)` are one-shot convenience overloads: each internally +calls `CreateSessionAsync`, performs exactly one operation, and disposes the session before +returning (or faulting) - the common case of a single isolated utterance, without the caller +needing to manage a session's life at all. `SynthesizeAsync` (both on the engine and on a session) +needs no playback device at all: it returns the full, ordered `IReadOnlyList` +segment list produced by the plan, including pure-silence pause segments, so a caller that only +wants the synthesized audio (for example to save it, or to play it through a custom pipeline) can +get full fidelity without ever touching a playback device. + +#### ISpeechSynthesizerEngine + +**Purpose**: Define the public Layer 3 contract for a loaded, reusable text-to-speech model, so hosts and tests can depend on synthesis behavior without depending on a specific inference engine or native audio type. -**Data Model**: `IsAvailable` indicates whether the synthesizer is backed by a loaded engine and -a usable playback device. The adjacent `SynthesizedSpeech` record carries one segment's -normalized `Samples`, the `SampleRate` they were produced at, and the real `PreSilence`/ -`PostSilence` durations to play alongside them, so a caller playing segments as they arrive needs -nothing from any other segment to play this one correctly. +**Data Model**: `IsAvailable` indicates whether this engine is backed by a loaded native model. +The adjacent `SynthesizedSpeech` record carries one segment's normalized `Samples`, the +`SampleRate` they were produced at, and the real `PreSilence`/`PostSilence` durations to play +alongside them. **Key Methods**: -- **SynthesizeStreamAsync(text, cancellationToken)**: Normalizes, tag-parses, Layer 2 renders, - and chunks the input text, then synthesizes and yields one `SynthesizedSpeech` per chunk as it - becomes ready, so an early chunk is available before a later chunk finishes synthesizing. -- **PlayStreamAsync(stream, cancellationToken)**: Starts the playback device once, plays each - yielded segment's pre-silence, resampled/upmixed audio, and post-silence in order, and stops - the device once the stream ends or faults, guaranteeing the device is released either way. -- **SpeakAsync(text, cancellationToken)**: Composes the two above into one convenience call for - the common case of synthesizing and playing a whole utterance. -- **Stop()**: Cancels an in-flight `SpeakAsync` session deterministically. A safe no-op when no - session is in flight. -- **Dispose()**: Releases the engine resources a synthesizer holds. Implies `Stop()` and is - idempotent. +- **CreateSessionAsync(device, cancellationToken)**: Binds this engine to one playback device and + returns a reusable `ISynthesisSession`. Throws `SynthesisEngineBusyException` if another leased + session is still undisposed. +- **SpeakAsync(device, text, cancellationToken)**: One-shot convenience - creates a session, speaks + once, disposes the session. +- **SynthesizeAsync(text, cancellationToken)**: One-shot convenience - creates a session with no + device needed, returns the full-fidelity segment list, disposes the session. +- **DisposeAsync()**: Releases the engine's native resources. Disposes any still-active leased + session first. Idempotent. **Error Handling**: Real implementations and the unavailable fallback both use `SpeechSynthesizerUnavailableException` when an operational call is invalid because no usable -synthesizer is available or the playback device fails at first use. Calling an operational -member after disposal throws `ObjectDisposedException`. +engine is available. + +**Dependencies**: `SynthesizedSpeech`, `ISynthesisSession`, `SynthesisEngineBusyException`, +`SpeechSynthesizerUnavailableException`. + +**Callers**: `SpeechSynthesizerFactory.LoadAsync(...)` and hosts that speak synthesized text. + +#### ISynthesisSession + +**Purpose**: Define the public Layer 5 contract for one session bound to one playback device for +its entire life, cheap to obtain and safely reusable across many calls. + +**Data Model**: `IsAvailable` reflects disposal/fault status. `State` exposes the current +`SynthesisSessionState`; `StateChanged` reports every transition. + +**Key Methods**: -**Dependencies**: `SynthesizedSpeech`, `SpeechSynthesizerUnavailableException`. +- **SpeakAsync(text, cancellationToken)**: Synthesizes and plays text through the bound device. + Throws `InvalidOperationException` if called while another operation is already in flight. +- **SynthesizeAsync(text, cancellationToken)**: Synthesizes text and returns the full-fidelity + segment list without playing it. Same overlap rule as `SpeakAsync`. +- **StopAsync(cancellationToken)**: Requests cooperative cancellation of the in-flight operation, + if any. Safe no-op otherwise. +- **DisposeAsync()**: Cancels any in-flight operation, transitions to `Disposed`, and releases the + engine's exclusivity lease exactly once. Idempotent. -**Callers**: `SpeechSynthesizerFactory.Create(...)` and hosts that speak synthesized text. +**Error Handling**: `SpeechSynthesizerUnavailableException` when unavailable; +`SynthesisSessionFaultedException` once `Faulted`; `InvalidOperationException` on overlap; +`ObjectDisposedException` after disposal. + +**Dependencies**: `SynthesizedSpeech`, `SynthesisSessionState`, `SessionStateChangedEventArgs`, +`SynthesisSessionFaultedException`, `SpeechSynthesizerUnavailableException`. + +**Callers**: `ISpeechSynthesizerEngine.CreateSessionAsync(...)` callers and its own one-shot +convenience overloads. #### SpeechSynthesizerFactory -**Purpose**: Provide the single composition entry point for obtaining an `ISpeechSynthesizer`, so -all "can this machine speak right now?" logic lives in one reviewable place, mirroring -`SpeechRecognizerFactory` exactly. The recommended caller pattern is to compose -`playbackDevice` first via `AudioDeviceFactory.CreatePlaybackDevice(selection, -model.PreferredAudioFormat)` and then pass that device into `SpeechSynthesizerFactory.Create(...)`; -when the backend honors the hint and the loaded engine later reports the same rate, -`PlaybackAudioResampler` stays on its existing equal-rate no-op fast path. Because -`PreferredAudioFormat` is only a best-effort hint, the resampler remains the guaranteed fallback -when the loaded engine's real `SampleRate` differs. +**Purpose**: Provide the single composition entry point for obtaining an +`ISpeechSynthesizerEngine`, so all "can this machine speak right now?" logic lives in one +reviewable place, mirroring `SpeechRecognizerFactory`. Unlike the former synchronous factory, this +type no longer takes a playback device: an engine loaded here can create many sessions over its +life, each bound to its own device, via `ISpeechSynthesizerEngine.CreateSessionAsync`. The +blocking native model load itself runs on a `DedicatedWorker` rather than the calling thread, so +awaiting any `LoadAsync` overload never blocks a caller's synchronization context. -**Data Model**: A static class with no state. The public `Create(...)` overload composes against -the real sherpa-onnx engine factory; an internal overload accepts an injected -`ISynthesisEngineFactory` so composition can be verified without model files or a native runtime. +**Data Model**: A static class with no state. The public `LoadAsync(...)` overloads compose +against the real sherpa-onnx backend factory; internal overloads accept an injected +`ISynthesisBackendFactory` so composition can be verified without model files or a native runtime. **Key Methods**: -- **Create(ISynthesisModel model, string installedModelDirectory, IAudioPlaybackDevice - playbackDevice, ISpeechDiagnostics? diagnostics, IReadOnlyDictionary<string, object>? - parameterValues)**: Returns a real `SherpaOnnxSpeechSynthesizer` - when the model's installed directory exists, the model declares `SpeechModelRole.Synthesis`, - the playback device reports `IsAvailable`, and the engine loads. Otherwise returns - `UnavailableSpeechSynthesizer.Instance`. Preconditions: `model` and `playbackDevice` are - non-null. Postcondition: the returned synthesizer is never null, and either owns a loaded - engine or is the shared unavailable instance. Loading the engine allocates native resources, so - the returned synthesizer must be disposed. The new, optional `parameterValues` argument - (default `null`) is forwarded unchanged to the constructed synthesizer, which re-resolves it to - a speaker id via `ISynthesisModel.ResolveSpeakerId` once per synthesized segment; a `null` bag - (the pre-Phase-10 call shape) resolves to every model's own default voice, so this addition is - fully backward compatible. - -The checks run in the same deliberate order as the recognition-direction factory - installed, -then role, then device, then engine load - so the cheapest and most common cause of -unavailability (a model not downloaded yet) is reported first and no native memory is allocated -for a synthesizer that could never run. - -**Error Handling**: Every ordinary machine state is represented as the honest unavailable -synthesizer plus a structural diagnostic, never as an exception, per this library's "nothing -throws at composition" decision. An engine load failure is caught and degraded identically to a -missing model. Only a null `model`, `playbackDevice`, or engine factory throws -`ArgumentNullException`, since a null argument is a programming error rather than a machine -state. +- **LoadAsync(ISynthesisModel model, string installedModelDirectory, ISpeechDiagnostics? + diagnostics, IReadOnlyDictionary<string, object>? parameterValues, CancellationToken + cancellationToken)**: Returns a real `SherpaOnnxSpeechSynthesizerEngine` when the model's + installed directory exists, the model declares `SpeechModelRole.Synthesis`, and the backend + loads. Otherwise returns `UnavailableSpeechSynthesizerEngine.Instance`. Precondition: `model` is + non-null. Postcondition: the returned engine is never null, and either owns a loaded backend or + is the shared unavailable instance. The optional `parameterValues` bag (for example a selected + voice) is forwarded unchanged to the constructed engine and re-resolved via + `ISynthesisModel.ResolveSpeakerId` once per synthesized segment. +- **LoadAsync(ISynthesisModel model, SpeechModelStore store, ...)** / **LoadAsync(ISynthesisModel + model, SpeechModelCatalog catalog, ...)**: Convenience overloads that resolve the model's + installed-files directory from the supplied store/catalog before delegating to the overload + above. + +The checks run in the same deliberate order as the recognition-direction factory - parameter +validation, then installed, then role, then backend load - so the cheapest and most common cause +of unavailability (a model not downloaded yet) is reported first and no native memory is allocated +for an engine that could never run. + +**Error Handling**: Every ordinary machine state is represented as the honest unavailable engine +plus a structural diagnostic, never as an exception, per this library's "nothing throws at +composition" decision. A backend load failure is caught and degraded identically to a missing +model. Only a null `model`/`store`/`catalog`, an invalid `parameterValues` entry +(`ArgumentException`), or a cancelled `cancellationToken` (`OperationCanceledException`) throws, +since those are programming errors or an explicit caller request rather than a machine state. **Dependencies**: `ISynthesisModel` and `SpeechModelRole` from the ModelManagementSubsystem, -`IAudioPlaybackDevice` from the AudioSubsystem, `ISpeechDiagnostics`/`NullSpeechDiagnostics` from -the Diagnostics subsystem, and the subsystem's own `ISynthesisEngineFactory`, -`SherpaOnnxSynthesisEngineFactory`, `SherpaOnnxSpeechSynthesizer`, and -`UnavailableSpeechSynthesizer`. +`ISpeechDiagnostics`/`NullSpeechDiagnostics` from the Diagnostics subsystem, and the subsystem's +own `ISynthesisBackendFactory`, `SherpaOnnxSynthesisEngineFactory`, +`SherpaOnnxSpeechSynthesizerEngine`, and `UnavailableSpeechSynthesizerEngine`. **Callers**: Host applications composing speech synthesis at start-up, and the system-level integration tests. -#### UnavailableSpeechSynthesizer +#### UnavailableSpeechSynthesizerEngine -**Purpose**: Provide a safe, always-obtainable `ISpeechSynthesizer` fallback for use when +**Purpose**: Provide a safe, always-obtainable `ISpeechSynthesizerEngine` fallback for use when synthesis is not possible on the current machine. **Data Model**: No instance state. Exposes a single static `Instance` singleton; the constructor -is private, since the type carries no state and multiple instances would provide no value. -`IsAvailable` always returns `false`. +is private. `IsAvailable` always returns `false`. **Key Methods**: -- **SynthesizeStreamAsync(...)** / **PlayStreamAsync(...)** / **SpeakAsync(...)** / **Stop()**: - Always throw `SpeechSynthesizerUnavailableException`. -- **Dispose()**: A no-op that never throws and never invalidates `Instance`, so a host that wraps - its synthesizer in a disposal scope runs unchanged on a machine without synthesis. +- **CreateSessionAsync(device, cancellationToken)**: Always succeeds, returning the shared + `UnavailableSynthesisSession.Instance` - binding a device to an already unavailable engine is an + ordinary (if useless) composition, not an error. Throws `ArgumentNullException` for a null + `device`. +- **SpeakAsync(...)** / **SynthesizeAsync(...)**: Always throw + `SpeechSynthesizerUnavailableException`. +- **DisposeAsync()**: A no-op that never throws and never invalidates `Instance`. **Error Handling**: Obtaining and holding the instance never throws. Only the operational members -throw, and only when actually invoked - a caller that checks `IsAvailable` first never triggers -them. This mirrors `UnavailableSpeechRecognizer` exactly, so both subsystems degrade the same -recognizable way. +throw, and only when actually invoked. + +**Dependencies**: `SpeechSynthesizerUnavailableException`, `UnavailableSynthesisSession`; +implements `ISpeechSynthesizerEngine`. + +**Callers**: `SpeechSynthesizerFactory.LoadAsync(...)` when the model is not installed, the +model's role is not synthesis, or the backend cannot be loaded. + +#### UnavailableSynthesisSession + +**Purpose**: Provide the honest `ISynthesisSession` fallback returned by +`UnavailableSpeechSynthesizerEngine.CreateSessionAsync`. + +**Data Model**: No instance state. Exposes a single static `Instance` singleton. `IsAvailable` +always returns `false`; `State` always reports `SynthesisSessionState.Created` (it never +transitions). + +**Key Methods**: -**Dependencies**: `SpeechSynthesizerUnavailableException`; implements `ISpeechSynthesizer`. +- **SpeakAsync(...)** / **SynthesizeAsync(...)**: Always throw + `SpeechSynthesizerUnavailableException`. Throw `ArgumentNullException` for null `text`. +- **StopAsync(...)**: A safe no-op, since no operation is ever in flight. +- **DisposeAsync()**: A no-op that never throws and never invalidates `Instance`. -**Callers**: `SpeechSynthesizerFactory.Create(...)` when the model is not installed, the model's -role is not synthesis, the playback device is unavailable, or the engine cannot be loaded. +**Error Handling**: Mirrors `UnavailableSpeechSynthesizerEngine`. + +**Dependencies**: `SpeechSynthesizerUnavailableException`; implements `ISynthesisSession`. + +**Callers**: `UnavailableSpeechSynthesizerEngine.CreateSessionAsync(...)`. #### SpeechSynthesizerUnavailableException -**Purpose**: Signal that an operational member of an unavailable synthesizer was invoked, or that -a synthesizer that claimed to be available failed on first use. +**Purpose**: Signal that an operational member of an unavailable engine or session was invoked, or +that a session which claimed to be available faulted and was asked to operate again. **Data Model**: No additional fields beyond the standard `Exception` base members. @@ -324,48 +423,64 @@ a synthesizer that claimed to be available failed on first use. **Dependencies**: `Exception`. -**Callers**: `UnavailableSpeechSynthesizer` for every operational member, and -`SherpaOnnxSpeechSynthesizer` when its playback device fails to start. +**Callers**: `UnavailableSpeechSynthesizerEngine` and `UnavailableSynthesisSession` for every +operational member. + +#### SynthesisEngineBusyException + +**Purpose**: Signal that `ISpeechSynthesizerEngine.CreateSessionAsync` was called while this +engine's exclusivity lease is already held by another, still-undisposed session. + +**Data Model**: No additional fields beyond the standard `Exception` base members. + +**Key Methods**: Standard constructor pattern. + +**Dependencies**: `Exception`. + +**Callers**: `SherpaOnnxSpeechSynthesizerEngine.CreateSessionAsync`. + +#### SynthesisSessionFaultedException + +**Purpose**: Signal that a `SpeakAsync`/`SynthesizeAsync` call was made on a session that has +already transitioned to `SynthesisSessionState.Faulted`. + +**Data Model**: Carries the original fault as its `InnerException`. + +**Key Methods**: Standard constructor pattern taking a message and the original fault. + +**Dependencies**: `Exception`. + +**Callers**: `SherpaOnnxSynthesisSession`'s operation entry point, once faulted. #### Design Constraints **Volume is applied as post-hoc amplitude scaling, not a native engine parameter.** sherpa-onnx's offline TTS API has no volume/gain input, so a `Loud`/`Soft`/`Whispers` tag's numeric override is applied by scaling the generated segment's samples and clamping to `[-1.0, 1.0]`, rather than -being passed into `Generate(...)`. This is a pragmatic, model-independent way to honor a -parameter-mapped volume override without depending on any specific engine exposing gain control. +being passed into `Generate(...)`. **Speed is applied as a native engine parameter.** Unlike volume, sherpa-onnx's `Generate(text, speed, speakerId)` already accepts a speed multiplier, so a `Fast`/`VeryFast`/`Slow`/`VerySlow` -tag's override is passed straight through rather than post-processed, giving better quality than -resampling the output audio would. +tag's override is passed straight through rather than post-processed. **Voice/speaker selection is a session-level concern, not a per-segment override.** The `speakerId` argument to `Generate(...)` is resolved once per segment by calling `ISynthesisModel.ResolveSpeakerId(parameterValues)` against the constructor-supplied `parameterValues` bag - deliberately _not_ reused via the per-segment `ParameterOverrides` -mechanism above, which is Natural Language Audio Tag-scoped and transient (a `[fast]`/`[loud]` -tag applies only to the segment it annotates). A selected voice, by contrast, applies to the -whole session, so it is threaded through the constructor instead and re-resolved fresh per -segment (cheap and pure) rather than cached once for the whole session - keeping it correctly -independent of, and non-regressing for, the existing speed/volume tag mechanism. +mechanism above, which is Natural Language Audio Tag-scoped and transient. A selected voice, by +contrast, applies to the whole session, so it is threaded through the constructor instead and +re-resolved fresh per segment (cheap and pure) rather than cached once for the whole session. **Synthesized text is never reported through diagnostics.** Every diagnostic this subsystem emits -is a structural fact - composed, started, stopped, or a named fault - and never includes +is a structural fact - composed, started, stopped, faulted, or lease-related - and never includes synthesized text, honoring the same `ISpeechDiagnostics` contract the recognition direction honors for recognized text. -**Voice selection is now solved for every synthesis model this subsystem registers.** Phase 4 -shipped this subsystem against zero real synthesis models, proven only against a test model. -Phase 7b added the library's first real `ISynthesisModel` (VITS/Piper), which exposed only its -default speaker as a documented, accepted, out-of-scope limitation. Phase 10 closes that -limitation for models with real per-voice knowledge: the new `ISynthesisModel.ResolveSpeakerId` -hook lets a model such as `SherpaOnnxKokoroEnglishSynthesisModel` (11 genuinely distinct voices) -resolve a selected voice to sherpa-onnx's real speaker id, proven end-to-end in this project's -development sandbox by synthesizing the same sentence with two different voice selections and -observing genuinely different, non-silent output - not merely that the configuration field is -accepted. A later pass closes the VITS model's own remaining limitation too, by declaring a -plain numeric speaker parameter over its full 904-speaker range (LibriTTS-R's speaker embeddings -have no published name mapping, so a numeric index rather than a named choice is the honest -declaration) and overriding `ResolveSpeakerId` for it, proven the same way against two different -numeric speaker selections. +**Engine exclusivity fails fast rather than queueing.** A concurrent `CreateSessionAsync` call +while a lease is held throws `SynthesisEngineBusyException` immediately rather than waiting for +the held session to be disposed, so a caller's latency never depends on an unrelated session's +teardown with no caller-visible way to bound that wait. + +**`DedicatedWorker` is duplicated, not shared, between subsystems.** Standing up a third, shared +internal subsystem for one ~100-line utility was judged disproportionate to the duplication it +would remove; each subsystem's copy is independently and fully covered by its own tests. diff --git a/docs/design/speech/synthesis-subsystem/audio-tag-parser.md b/docs/design/speech/synthesis-subsystem/audio-tag-parser.md index 9fbf049..a78c0a1 100644 --- a/docs/design/speech/synthesis-subsystem/audio-tag-parser.md +++ b/docs/design/speech/synthesis-subsystem/audio-tag-parser.md @@ -47,7 +47,7 @@ content - however malformed - ever throws or is discarded; it always becomes lit **Dependencies**: None outside this unit's own types. The parser and catalog have no reference to any model, engine, or audio device. -**Callers**: `SherpaOnnxSpeechSynthesizer` calls `AudioTagParser.Parse` as its Layer 1 tag-parsing +**Callers**: `SherpaOnnxSynthesisSession` calls `AudioTagParser.Parse` as its Layer 1 tag-parsing step before Layer 2 rendering builds a `SpeechPlan` from the resulting spans, per this library's Layer 1 (model-independent parsing) / Layer 2 (model-specific rendering) split. diff --git a/docs/design/speech/synthesis-subsystem/i-speech-synthesizer-engine.md b/docs/design/speech/synthesis-subsystem/i-speech-synthesizer-engine.md new file mode 100644 index 0000000..8eeb990 --- /dev/null +++ b/docs/design/speech/synthesis-subsystem/i-speech-synthesizer-engine.md @@ -0,0 +1,40 @@ +### ISpeechSynthesizerEngine + +**Purpose**: Define the public Layer 3 contract for a loaded, reusable text-to-speech model, so +hosts and tests can depend on synthesis behavior without depending on a specific inference engine +or native audio type. + +**Data Model**: `IsAvailable` indicates whether this engine is backed by a loaded native model. +The adjacent `SynthesizedSpeech` record carries one segment's normalized `Samples`, the +`SampleRate` they were produced at, and the real `PreSilence`/`PostSilence` durations to play +alongside them, so a caller playing or saving segments needs nothing from any other segment to +use this one correctly. + +**Key Methods**: + +- **CreateSessionAsync(device, cancellationToken)**: Binds this engine to one playback device + and returns a reusable `ISynthesisSession`. Throws `SynthesisEngineBusyException` if another + leased session is still undisposed, and `ArgumentNullException` for a null `device`. +- **SpeakAsync(device, text, cancellationToken)**: One-shot convenience - internally creates a + session, speaks once, and disposes the session before returning, for the common case of a + single isolated utterance. +- **SynthesizeAsync(text, cancellationToken)**: One-shot convenience - internally creates a + session with no device needed, returns the full-fidelity `IReadOnlyList` + segment list (including pure-silence pause segments), and disposes the session. +- **DisposeAsync()**: Releases this engine's native resources. Disposes any still-active leased + session first (session disposal also releases the lease), then the owned backend. Idempotent. + +An engine supports many independent sessions over its life, one at a time - obtaining an engine +from `SpeechSynthesizerFactory` is the expensive step, so a host doing repeated, low-latency +synthesis (for example many turns of a voice conversation) should load one engine once and reuse +it, creating and disposing a session per device binding it needs, rather than reloading per turn. + +**Error Handling**: Real implementations and the unavailable fallback both use +`SpeechSynthesizerUnavailableException` when an operational call is invalid because no usable +engine is available. Calling an operational member after disposal throws +`ObjectDisposedException`. + +**Dependencies**: `SynthesizedSpeech`, `ISynthesisSession`, `SynthesisEngineBusyException`, +`SpeechSynthesizerUnavailableException`. + +**Callers**: `SpeechSynthesizerFactory.LoadAsync(...)` and hosts that speak synthesized text. diff --git a/docs/design/speech/synthesis-subsystem/i-speech-synthesizer.md b/docs/design/speech/synthesis-subsystem/i-speech-synthesizer.md deleted file mode 100644 index ae4e6f9..0000000 --- a/docs/design/speech/synthesis-subsystem/i-speech-synthesizer.md +++ /dev/null @@ -1,41 +0,0 @@ -### ISpeechSynthesizer - -**Purpose**: Define the contract for chunked, streaming text-to-speech synthesis and playback, so -hosts and tests can depend on synthesis behavior without depending on a specific inference engine -or native audio type. - -**Data Model**: `IsAvailable` indicates whether the synthesizer is backed by a loaded engine and -a usable playback device. The adjacent `SynthesizedSpeech` record carries one segment's -normalized `Samples`, the `SampleRate` they were produced at, and the real `PreSilence`/ -`PostSilence` durations to play alongside them, so a caller playing segments as they arrive needs -nothing from any other segment to play this one correctly. - -**Key Methods**: - -- **SynthesizeStreamAsync(text, cancellationToken)**: Normalizes, tag-parses, Layer 2 renders, - and chunks the input text, then synthesizes and yields one `SynthesizedSpeech` per chunk as it - becomes ready, so an early chunk is available before a later chunk finishes synthesizing. -- **PlayStreamAsync(stream, cancellationToken)**: Starts the playback device once, plays each - yielded segment's pre-silence, resampled/upmixed audio, and post-silence in order, and stops - the device once the stream ends or faults, guaranteeing the device is released either way. -- **SpeakAsync(text, cancellationToken)**: Composes the two above into one convenience call for - the common case of synthesizing and playing a whole utterance. -- **Stop()**: Cancels an in-flight `SpeakAsync` session deterministically. A safe no-op when no - session is in flight. -- **Dispose()**: Releases the engine resources a synthesizer holds. Implies `Stop()` and is - idempotent. - -A synthesizer supports many independent synthesize/play sessions on the same instance without -reconstruction - obtaining a synthesizer from `SpeechSynthesizerFactory` is the expensive step, so -a host doing repeated, low-latency synthesis (for example, many turns of a voice conversation) -should construct one synthesizer once and reuse it across sessions rather than disposing and -recreating it per turn. - -**Error Handling**: Real implementations and the unavailable fallback both use -`SpeechSynthesizerUnavailableException` when an operational call is invalid because no usable -synthesizer is available or the playback device fails at first use. Calling an operational -member after disposal throws `ObjectDisposedException`. - -**Dependencies**: `SynthesizedSpeech`, `SpeechSynthesizerUnavailableException`. - -**Callers**: `SpeechSynthesizerFactory.Create(...)` and hosts that speak synthesized text. diff --git a/docs/design/speech/synthesis-subsystem/i-synthesis-session.md b/docs/design/speech/synthesis-subsystem/i-synthesis-session.md new file mode 100644 index 0000000..bb0be42 --- /dev/null +++ b/docs/design/speech/synthesis-subsystem/i-synthesis-session.md @@ -0,0 +1,45 @@ +### ISynthesisSession + +**Purpose**: Define the public Layer 5 contract for one text-to-speech session, bound to exactly +one playback device for its entire life, cheap to obtain from +`ISpeechSynthesizerEngine.CreateSessionAsync` and safely reusable across many +`SpeakAsync`/`SynthesizeAsync` calls without reconstruction. + +**Data Model**: `IsAvailable` reflects disposal/fault status (`false` once disposed or `Faulted`). +`State` exposes the current `SynthesisSessionState` (`Created`, `Starting`, `Running`, +`Stopping`, `Stopped`, `Disposing`, `Disposed`, `Faulted`); `StateChanged` reports every +transition via a `SessionStateChangedEventArgs` carrying the previous and current state. Unlike +recognition's continuous capture-window states, `Starting`/`Running`/`Stopping` denote one +discrete in-flight operation, not a continuous stream - a session returns to `Stopped` after each +operation and is immediately ready for another call. See the `synthesis-subsystem.md` design +document for the full state diagram. + +**Key Methods**: + +- **SpeakAsync(text, cancellationToken)**: Synthesizes the text and plays it through the bound + playback device. Throws `InvalidOperationException` if called while another + `SpeakAsync`/`SynthesizeAsync` call on this same session is already in flight (the two methods + never overlap on one session), `SynthesisSessionFaultedException` once `Faulted`, + `SpeechSynthesizerUnavailableException` when unavailable, and `ArgumentNullException` for null + `text`. +- **SynthesizeAsync(text, cancellationToken)**: Synthesizes the text and returns the full-fidelity + `IReadOnlyList` segment list (including pure-silence pause segments) without + playing it - no playback device interaction at all. Same overlap/fault/availability error + behavior as `SpeakAsync`. +- **StopAsync(cancellationToken)**: Requests cooperative cancellation of whichever operation is + currently in flight, by cancelling that operation's own linked cancellation source. A safe + no-op when no operation is in flight. The cancelled operation unwinds to `Stopped`, not + `Faulted` - a caller-requested stop is an ordinary, successful outcome. +- **DisposeAsync()**: Cancels any in-flight operation, transitions through `Disposing` to + `Disposed`, and releases the owning engine's exclusivity lease exactly once. Idempotent. + +**Error Handling**: `SpeechSynthesizerUnavailableException` when unavailable; +`SynthesisSessionFaultedException` (wrapping the original fault) once `Faulted` - a faulted +session is terminal and must be disposed and replaced, never resumed; +`InvalidOperationException` on overlap; `ObjectDisposedException` after disposal. + +**Dependencies**: `SynthesizedSpeech`, `SynthesisSessionState`, `SessionStateChangedEventArgs`, +`SynthesisSessionFaultedException`, `SpeechSynthesizerUnavailableException`. + +**Callers**: `ISpeechSynthesizerEngine.CreateSessionAsync(...)` callers and +`ISpeechSynthesizerEngine`'s own one-shot `SpeakAsync`/`SynthesizeAsync` convenience overloads. diff --git a/docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md b/docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md new file mode 100644 index 0000000..c3ce233 --- /dev/null +++ b/docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md @@ -0,0 +1,39 @@ +### SherpaOnnxSpeechSynthesizerEngine + +**Purpose**: Implement the real `ISpeechSynthesizerEngine`, owning one loaded `SynthesisBackend` +and enforcing single-session exclusivity with a fail-fast lease. + +**Data Model**: Holds the owned `ISynthesisBackend`, the `ISynthesisModel` supplying +`NormalizeText`/`CapabilityProfile`/`ResolveSpeakerId`, an optional `parameterValues` bag (the +engine's default selected voice/tunable-parameter values, forwarded unchanged from +`SpeechSynthesizerFactory.LoadAsync` to every session it creates), a diagnostics sink, a binary +`SemaphoreSlim(1, 1)` exclusivity lease, and a disposed flag. `IsAvailable` is always `true`, +because this type is only ever created after the backend loaded successfully - every unavailable +case is represented by `UnavailableSpeechSynthesizerEngine` instead. + +**Key Methods**: + +- **CreateSessionAsync(device, cancellationToken)**: Attempts to acquire the lease with + `SemaphoreSlim.Wait(0, cancellationToken)` (zero timeout - fails immediately rather than + queueing). On success, constructs and returns a new `SherpaOnnxSynthesisSession` bound to + `device`, with a release callback that releases the lease exactly once when that session is + disposed. On failure (lease already held), throws `SynthesisEngineBusyException` immediately. + Throws `ArgumentNullException` for a null `device`. +- **DisposeAsync()**: If a session is currently leased, disposes it first (which releases the + lease as a side effect of that session's own `DisposeAsync`), then disposes the owned backend. + Idempotent: a second call disposes the backend exactly once. + +The fail-fast design (zero-timeout `Wait` rather than an unbounded or timed wait) means a +caller's `CreateSessionAsync` latency never depends on how long an unrelated, already-leased +session takes to be disposed; a caller that genuinely needs concurrent sessions must load a +second engine instance instead. + +**Error Handling**: `SynthesisEngineBusyException` when the lease is already held. +`ArgumentNullException` for a null `device`. Operational members throw +`ObjectDisposedException` after disposal. + +**Dependencies**: `ISynthesisBackend`, `SherpaOnnxSynthesisSession`, `SynthesisEngineBusyException`, +`IAudioPlaybackDevice` from the AudioSubsystem, `ISynthesisModel` from the +ModelManagementSubsystem, `ISpeechDiagnostics` from the Diagnostics subsystem. + +**Callers**: `SpeechSynthesizerFactory.LoadAsync(...)`. diff --git a/docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md b/docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md deleted file mode 100644 index b227d6a..0000000 --- a/docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md +++ /dev/null @@ -1,347 +0,0 @@ -### SherpaOnnxSpeechSynthesizer - -This chapter covers the streaming synthesizer together with the units it exists to coordinate - -the Layer 2 rendering strategy, the sentence chunker, the synthesis-engine seam, and the -playback-format converter - because none can be reviewed meaningfully in isolation: the -synthesizer's whole job is to move text through rendering and chunking, into the engine, and the -resulting audio through resampling onto the playback device, all while pipelining synthesis with -playback. - -**Purpose**: Chunk, render, and synthesize text into speech through a synthesis engine, and play -the result while continuing to synthesize later chunks, without ever performing synthesis or -inference work on the caller's UI thread. - -**Data Model**: Holds the owned `ISynthesisEngine`, the `IAudioPlaybackDevice` it plays through, -the `ISynthesisModel` supplying `NormalizeText`/`CapabilityProfile`/`ResolveSpeakerId`, an -optional `parameterValues` bag (the session's selected voice/tunable-parameter values, forwarded -unchanged from `SpeechSynthesizerFactory.Create`), and a diagnostics sink. While a session is in -flight it also holds a `CancellationTokenSource` linked to the caller's token (guarded by a lock, -so `Stop()` can cancel it safely from another thread) and, per `SynthesizeStreamAsync` call, a -bounded `Channel` of capacity 8 together with the background producer `Task` -filling it. `IsAvailable` is always `true`, because this type is only ever created after the -engine loaded and the device reported itself available - every unavailable case is represented by -`UnavailableSpeechSynthesizer` instead. - -The pending-segment channel holds at most 8 synthesized segments (raised from an original 2 to 5 -to smooth pacing over long multi-sentence text, since 2 could let playback catch up to and stall -on a still-synthesizing segment whenever one chunk took noticeably longer than its predecessor -took to play; raised again from 5 to 8 once `SentenceChunker` started splitting clause -punctuation unconditionally, since a typical multi-clause sentence now yields roughly 1.5-2x as -many, smaller chunks, eroding the anti-starvation margin the 2-to-5 increase provided) and uses -`BoundedChannelFullMode.Wait`: unlike the recognition-direction capture queue, -which can drop the oldest live audio, a segment here has already cost real inference time and -must never be silently discarded, so the producer simply waits for the consumer (playback) to -catch up. This is purely a buffer-size tuning change: it does not reduce the latency before the -very first word is spoken, which remains bounded by however long the first chunk alone takes to -synthesize. Bounded look-ahead *concurrency* (calling the engine for more than one segment at a -time) was investigated and deliberately rejected rather than implemented: `ISynthesisEngine`'s own -contract states implementations are not thread-safe with respect to concurrent calls, and a -regression test (`SherpaOnnxSpeechSynthesizerTests.SynthesizeStreamAsync_LongMultiSentenceInput_ProducesOrderedSegmentsSequentially`) -already asserts the engine is never called concurrently; the producer already starts the next -segment's synthesis immediately once the previous segment's channel write completes, so the -"pipeline synthesis with playback" goal this capacity exists for is already achieved without any -concurrent engine calls. - -`DrainPollInterval` (15ms) and `DrainTailMargin` (40ms) are fixed constants governing -`WaitForPlaybackDrainAsync`'s post-loop wait in `PlayStreamAsync` (see below): short enough that -`PlayStreamAsync` returns promptly once the hardware is genuinely done, long enough to avoid -busy-spinning the thread pool re-reading a value only a real-time audio callback thread can -change. - -**Key Methods**: - -- **SynthesizeStreamAsync(text, cancellationToken)**: Validates arguments eagerly (split into a - non-iterator validating wrapper plus a separate iterator method, `SynthesizeStreamCore`, so - argument errors surface on the calling `MoveNextAsync` rather than being deferred - a SonarQube - S4456 requirement for async-iterator methods with parameter validation). Normalizes the text via - `_model.NormalizeText`, parses it via `AudioTagParser.Parse`, renders it via - `_model.CapabilityProfile.Render(spans, _model)` into a `SpeechPlan`, spawns a producer task - synthesizing each `SpeechSegment` in turn onto the bounded channel, and yields from - `channel.Reader.ReadAllAsync(...)` inside a `try`/`finally` whose `finally` block unconditionally - awaits the producer task - on normal completion, on cancellation, and on any other exception - from the loop alike - so the producer (and whatever in-flight `_engine.Generate` call it may be - mid-way through) is guaranteed to have genuinely finished before this method ever returns - control to its caller. This closes a use-after-free window that previously existed only on the - cancellation path: `ReadAllAsync(cancellationToken)` observing cancellation threw - `OperationCanceledException` straight out of the `await foreach`, skipping a then-unconditional - post-loop await and orphaning the producer task; because the producer only checks its own - cancellation token between segments (never while inside `GenerateSegment`), an in-flight native - `_engine.Generate` call kept running, untracked, on a background thread even after this method - returned - and if the caller (for example `SpeakAsync`'s caller, on observing the same - cancellation) then disposed the synthesizer and its owned engine, that still-running native call - touched freed native memory, producing an `AccessViolationException` on the ThreadPool worker - thread running the producer. The `finally` block's await swallows a residual - `OperationCanceledException` from the producer task specifically (benign and expected on the - cancellation path, since `ProduceAsync` normally suppresses it internally and completes the - channel instead of faulting) without masking whatever exception, if any, is already propagating - out of the loop; a genuine (non-cancellation) producer fault on the normal-completion path still - propagates to the caller exactly as before. An unconditional await alone would not be safe, - though: the producer writes into the bounded channel via `writer.WriteAsync`, which only - unblocks a full write when either the reader keeps draining or the write's own token is - cancelled. Enumeration can also be abandoned for a reason that never touches the caller's - `cancellationToken` at all - most notably a consumer's own `await foreach` body throwing an - unrelated exception (for example `PlayStreamAsync` observing a playback device fault), which the - compiler's `await foreach` cleanup turns into a `DisposeAsync` on this iterator, resuming it - inside the same `finally` block. If the channel happened to be full at that moment, the reader - will never drain again and the caller's token was never cancelled, so the unconditional await - would hang forever. To close that hole, the producer task observes its own - `CancellationTokenSource` linked to (but distinct from) the caller's token, and the `finally` - block cancels it before awaiting the producer task on every exit path, guaranteeing - `writer.WriteAsync` always has a way to unblock regardless of why enumeration was abandoned. See - `SherpaOnnxSpeechSynthesizerTests.Stop_WhileSpeaking_CancelsInFlightSessionOnlyAfterInFlightGenerateReturns`, - `SherpaOnnxSpeechSynthesizerTests.SynthesizeStreamAsync_CancelledMidGenerate_AwaitsProducerBeforeEnumerationCompletesAndDisposalIsSafe`, - and - `SherpaOnnxSpeechSynthesizerTests.SynthesizeStreamAsync_EnumerationAbandonedWithoutCancellation_DisposesPromptlyInsteadOfHanging` - for regression coverage proving the producer task is never orphaned, disposal after cancellation - is safe, and abandoning enumeration for an unrelated reason cannot hang. -- **GenerateSegment(segment)**: An empty-text segment (a rendered pause) skips the engine - entirely and returns a pure-silence `SynthesizedSpeech` built directly from the segment's - declared silence durations. Otherwise resolves speed/volume overrides against the model's - declared numeric parameters via `SpeechParameterConventions`, resolves the speaker id for this - session's selected voice by calling `_model.ResolveSpeakerId(_parameterValues)` once per - segment (cheap and pure - re-evaluated fresh each call rather than cached for the whole - session, so it always reflects the constructor-supplied bag; this is deliberately independent - of the per-segment `ParameterOverrides` speed/volume mechanism above, which is Natural Language - Audio Tag-scoped and transient, not a fit for a session-level voice selection), calls - `_engine.Generate(text, speedRatio, speakerId)`, and applies any volume override as post-hoc - amplitude scaling (sherpa-onnx's offline TTS API has no native gain input), clamped to - `[-1.0, 1.0]`. -- **PlayStreamAsync(stream, cancellationToken)**: Calls `_playbackDevice.Start()`, builds one - `PlaybackAudioResampler` for the call from the engine's and device's reported formats, and for - each segment writes its pre-silence, resampled/upmixed audio, and post-silence in order. Once - every segment has been written, calls `WaitForPlaybackDrainAsync` before returning: `Write` is - fire-and-forget, so having enqueued every segment does not mean the hardware has rendered any - of it yet, and treating enqueue-complete as playback-complete was the root cause of a bug where - the TTS panel's status flashed back to idle and cut audio off almost instantly. A `finally` - block calls `_playbackDevice.Stop()` regardless of whether the loop, the drain wait, or neither - completed, faulted, or was cancelled - reporting (but not rethrowing) a failure to stop, so the - device is never left running. -- **WaitForPlaybackDrainAsync(cancellationToken)**: Polls `_playbackDevice.PendingSampleCount` - every `DrainPollInterval` (15ms) until it reaches zero - i.e. until the playback hardware has - genuinely rendered every sample this session wrote, not merely until every segment was handed - off - then applies one further `DrainTailMargin` (40ms) wait before returning, to absorb any - residual internal host buffering (e.g. PortAudio's own ring buffer) that has already left the - managed queue but has not yet actually reached the speakers. Honors `cancellationToken`: a - cancellation during either wait ends it immediately via `Task.Delay`'s own cancellation, letting - the caller's `finally` block still stop the device promptly instead of waiting out the full - drain. -- **SpeakAsync(text, cancellationToken)**: Creates the session's linked `CancellationTokenSource` - under the lock, composes `SynthesizeStreamAsync` with `PlayStreamAsync`, and clears the field - and disposes the token source in a `finally` block regardless of outcome. -- **Stop()**: Cancels `_sessionCancellation` if a session is in flight; a safe no-op otherwise. -- **Dispose()**: Calls `Stop()` then disposes the owned engine. Idempotent. - -**Error Handling**: A fault raised synthesizing a segment is caught by the producer task, -reported through `ISpeechDiagnostics`, and completes the channel with that exception -(`writer.TryComplete(ex)`) so the awaiting consumer observes it as a faulted enumeration rather -than hanging; a cancellation during production completes the channel normally instead. -`SynthesizeStreamCore`'s `finally` block guarantees the producer task is always awaited to -completion before the method returns on every exit path, so a caller can never observe control -returned while an in-flight `_engine.Generate` call is still running - see -`SynthesizeStreamAsync(text, cancellationToken)` above for the full rationale and regression -tests. A playback-device write failure propagates out of `PlayStreamAsync` after the device is still -stopped in `finally`. A cancellation while `WaitForPlaybackDrainAsync` is polling for drain (or -during its tail margin) propagates the same `Task.Delay`-raised `OperationCanceledException`, -ending the wait immediately rather than waiting out the full drain, while `finally` still stops -the device. An unavailable playback device throws promptly rather than hanging. -Calling an operational member after disposal throws `ObjectDisposedException`. - -**Dependencies**: `ISynthesisEngine`, `SpeechPlan`, `SpeechSegment`, `SpeechParameterConventions`, -`PlaybackAudioResampler`, `SynthesizedSpeech`, `SpeechSynthesizerUnavailableException`, -`AudioTagParser` from this subsystem's Sub-phase 4a units, `IAudioPlaybackDevice` from the -AudioSubsystem, and `ISpeechDiagnostics` from the Diagnostics subsystem. - -**Callers**: `SpeechSynthesizerFactory.Create(model, installedModelDirectory, playbackDevice, -diagnostics, parameterValues)`. - -#### ISynthesisEngine and ISynthesisEngineFactory - -**Purpose**: Confine every speech-synthesis interop call behind one mockable boundary, so the -synthesizer's chunking, Layer 2 rendering, pipelining, and fault-containment logic is verifiable -with pure managed fakes - no downloaded model and no platform-specific native binary are ever -required in CI. This mirrors the `IRecognitionEngine`/`IRecognitionEngineFactory` seam used for -recognition. - -**Data Model**: N/A (interfaces only). - -**Key Methods**: - -- **ISynthesisEngine.SampleRate**: The fixed rate this engine's `Generate(...)` output is - produced at. -- **ISynthesisEngine.Generate(text, speed, speakerId)**: Synthesizes one chunk of text at the - given speed multiplier and speaker id, returning an `EngineAudio` (samples plus the rate they - were produced at). -- **ISynthesisEngineFactory.Create(ISynthesisModel, string installedModelDirectory)**: Loads an - engine from the model's own declared configuration. - -**Error Handling**: Implementations are not thread-safe by contract; the synthesizer calls -`Generate` from exactly one producer thread per session. Load failures surface as exceptions from -`Create(...)`, which `SpeechSynthesizerFactory` converts into the honest unavailable fallback. - -**Dependencies**: `EngineAudio`; `ISynthesisModel` from the ModelManagementSubsystem. - -**Callers**: `SherpaOnnxSpeechSynthesizer` (engine) and `SpeechSynthesizerFactory` (factory). - -#### SherpaOnnxSynthesisEngine and SherpaOnnxSynthesisEngineFactory - -**Purpose**: Implement the engine seam against the real sherpa-onnx offline TTS API, reusing the -already-referenced `org.k2fsa.sherpa.onnx` package. These are the only types in this -subsystem that call speech-synthesis inference APIs. - -**Data Model**: The engine holds one loaded `OfflineTts`, its declared `SampleRate`, and a -disposed flag. The factory is stateless and holds no model-specific knowledge at all - -the design makes each model's backing class responsible for its own engine configuration, -so adding a synthesis model never requires changing the factory. - -**Key Methods**: - -- **Generate(...)**: Calls the underlying `OfflineTts.Generate(text, speed, speakerId)` and wraps - its output samples and rate in an `EngineAudio`. -- **Create(...)**: Asks the model for its `OfflineTtsConfig`, resolved against the - installed-files directory, and loads an engine from it. -- **Dispose()**: Releases the loaded `OfflineTts`. Safe to call more than once. - -**Error Handling**: Construction loads the model into native memory and therefore throws when the -native runtime for the current platform is absent or the model files are unusable; that exception -is what `SpeechSynthesizerFactory` converts into the honest unavailable fallback. Operational -members throw `ObjectDisposedException` after disposal. - -**Dependencies**: The sherpa-onnx managed API (see *SherpaOnnx Design*); `ISynthesisModel` from -the ModelManagementSubsystem. - -**Callers**: `SpeechSynthesizerFactory` constructs the factory; the factory constructs the -engine; `SherpaOnnxSpeechSynthesizer` uses the engine. - -#### IModelCapabilityProfile and DefaultModelCapabilityProfile - -**Purpose**: Decide, per Layer 1 span, how a Natural Language Audio Tag is realized for a -specific model - pass-through, approximation, or strip - turning a model-independent span -sequence into an ordered, model-appropriate `SpeechPlan`. - -**Data Model**: `IModelCapabilityProfile` declares one method, -`Render(IReadOnlyList spans, ISpeechModel model)`, returning a `SpeechPlan` (an -ordered `IReadOnlyList`). Each `SpeechSegment` carries the text to synthesize (or -empty, for a pure-silence pause segment), the pre/post silence durations to play alongside it, -and any numeric parameter overrides (speed/volume) to apply when synthesizing it. -`DefaultModelCapabilityProfile` is a stateless singleton (`Instance`) requiring no per-model -configuration, since every decision it makes is read from the `model` argument at call time. - -This signature intentionally differs from the parameter-dictionary shape sketched in the Phase 4 -plan (see the subsystem Design section above for the full rationale): a stateless singleton -shared by every model has no per-call parameter bag to consume, and reading everything from -`model` directly keeps the contract satisfiable with zero code by any model that wants the -default, generically-correct behavior. - -**Key Methods**: - -- **Render(spans, model)**: For each `TaggedTextSpanKind.Tag` span: if the tag is a pause, always - renders a silence-only `SpeechSegment` (300ms short / 900ms long) regardless of `model - .AudioTagSupport`. Otherwise, if `model.AudioTagSupport` is `Native`, passes the tag through as - its canonical bracket text so the model's own inference sees the control token. If - `ParameterMapped`, consults `SpeechParameterConventions` for a matching numeric parameter on - `model.Parameters` and attaches it as an override on the surrounding segment when one exists, - otherwise strips the tag. If `None`, strips every non-pause tag. For each - `TaggedTextSpanKind.PlainText` span, delegates to `SentenceChunker.ChunkWithMetadata(...)` and - emits one `SpeechSegment` per resulting chunk, setting that segment's `PostSilenceMs` to a new - `EllipsisPauseMilliseconds` constant (500ms) when the chunk's `EndsWithEllipsis` flag is set, - or `0` otherwise - a chunk-boundary-triggered pause distinct from, and never reusing, the - tag-triggered short/long pause durations above. - -**Error Handling**: Rejects a null `spans` or `model` with `ArgumentNullException`. Every other -input - any tag, any support level, any parameter set - is handled without throwing, per -this library's "never worse than plain narration" guarantee extended to Layer 2. - -**Dependencies**: `TaggedTextSpan`, `TaggedTextSpanKind`, `NaturalLanguageAudioTagKind` from this -subsystem's Sub-phase 4a units; `SentenceChunker`; `SpeechParameterConventions`; `SpeechSegment`, -`SpeechPlan`; `ISpeechModel`, `SpeechModelAudioTagSupport`, `NumericParameter` from the -ModelManagementSubsystem. - -**Callers**: `SherpaOnnxSpeechSynthesizer.SynthesizeStreamAsync(...)` via -`ISynthesisModel.CapabilityProfile`. - -#### SentenceChunker - -**Purpose**: Split synthesis-ready plain text into sentence/clause-sized chunks so chunked, -pipelined synthesis has natural-sounding boundaries and no chunk exceeds a length a synthesis -call can reasonably handle. - -**Data Model**: A stateless static class; `Chunk(text, maxLength)`/`ChunkWithMetadata(text, -maxLength)` take no configuration beyond their two arguments. `ChunkWithMetadata` returns a -`SentenceChunk` record struct per chunk (`Text`, `EndsWithEllipsis`); `Chunk` is a pure projection -of `ChunkWithMetadata`'s chunk text, so its signature and call pattern are unaffected - existing -callers of `Chunk` still receive plain chunk text with no code changes required, though the -chunk boundaries themselves now differ (see the `synthesis-subsystem.md` design document). - -**Key Methods**: - -- **Chunk(text, maxLength)** / **ChunkWithMetadata(text, maxLength)**: Splits on primary - sentence-ending punctuation (`.`, `!`, `?`) first, then **unconditionally** splits every - resulting piece further on secondary clause punctuation (`,`, `;`, `:`) - not only when the - piece is still over `maxLength` - so every clause becomes its own chunk. A punctuation - character embedded in a numeral is never treated as a boundary at either pass: a digit both - immediately before and after (e.g. the `:` in `"12:30"`, the `,` in `"1,000"`, or a decimal - point such as `"0.5"`) is always kept attached, and a decimal point specifically is also kept - attached when a digit follows immediately but none precedes, as long as the character before it - (if any) is not a letter - e.g. a leading fraction such as `".5"`, `"$.99"`, or `".5 units"` at - start-of-text/after whitespace/a sign/a currency symbol. This prevents such a decimal point from - being mistaken for a sentence-ending period and dropped by the degenerate-punctuation merge - below, which would otherwise silence the "point" when the number is later spoken (e.g. `".5"` - reading as "five" instead of "point five"). A genuine sentence-ending period directly followed - by a digit after a letter (e.g. `"Wait.5 more."`) still splits normally, since the numeral - exception only applies when no letter precedes the period. A piece still longer - than `maxLength` after both punctuation passes is split on a whitespace budget; a single word - that alone exceeds `maxLength` is returned whole, never split mid-word. Any resulting piece - whose trimmed text contains no letter or digit at all (just punctuation and/or whitespace, e.g. - a lone comma or a whitespace-spaced ellipsis like `". . ."`) is then merged onto the end of the - immediately preceding non-empty chunk (joined by a single space), or dropped if there is no - preceding chunk. `ChunkWithMetadata` additionally flags a chunk whose final text ends in three - or more consecutive `.` characters (with or without interspersed whitespace) as - `EndsWithEllipsis`. These numeral exceptions assume English/US-style numeral punctuation - (`,` as thousands separator, `.` as decimal point); locales that swap the two roles (e.g. - `"1.000,5"`) are not specially handled - every model in this library's current catalog is - English-only, so this is not currently a defect, but a future non-English model would need - these rules adjusted rather than assuming they generalize as-is. - -**Error Handling**: Rejects a non-positive `maxLength` with `ArgumentOutOfRangeException` and a -null `text` with `ArgumentNullException`, since neither describes a meaningful chunking request. -Empty and whitespace-only text return no chunks rather than throwing. - -**Dependencies**: None beyond the Base Class Library. - -**Callers**: `DefaultModelCapabilityProfile.Render(...)`. - -#### PlaybackAudioResampler - -**Purpose**: Convert one synthesized segment's mono audio, produced at a synthesis engine's fixed -rate, into the format a playback device requires - its resolved rate and channel count. Keeping -this conversion in a pure, dependency-free type mirrors `AudioFrameResampler`'s recognition -direction role and makes it exhaustively testable with plain float arrays. - -**Data Model**: Immutable: the engine sample rate, the target (device) sample rate, and the -target channel count, all validated as positive at construction. No per-call state, so one -instance serves a whole playback session. - -**Key Methods**: - -- **Resample(samples, sourceSampleRate, targetSampleRate)**: Produces - `floor(length * target / source)` samples via linear interpolation, clamped to the last input - sample at the boundary. Equal rates copy the input unchanged, so the identity case introduces no - error at all. When downsampling, a small Hamming-windowed sinc lowpass filter runs first to - attenuate above-target-Nyquist energy before decimation. -- **UpmixToChannels(samples, channelCount)**: Replicates each mono sample across every output - channel, interleaved. A single required channel copies the input unchanged. -- **Convert(samples)**: Composes `Resample` then `UpmixToChannels` into the single operation the - synthesis pipeline needs for every segment it plays. When the target channel count is 1, - `UpmixToChannels` is skipped entirely and the resampled result is returned directly, since - `UpmixToChannels`'s own single-channel case would only make a redundant copy of an array - `Convert` already owns exclusively. - -**Error Handling**: A non-positive sample rate or channel count throws -`ArgumentOutOfRangeException`. Empty input returns an empty result rather than throwing, since a -pure-silence segment legitimately has no samples to convert. - -**Dependencies**: `AudioSubsystem.WindowedSincLowpassFilter` for the downsampling anti-aliasing -lowpass stage, shared with `AudioFrameResampler`'s identical need in the recognition subsystem; -otherwise none beyond the Base Class Library. - -**Callers**: `SherpaOnnxSpeechSynthesizer.PlayStreamAsync(...)`. diff --git a/docs/design/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md b/docs/design/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md new file mode 100644 index 0000000..bc05b0a --- /dev/null +++ b/docs/design/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md @@ -0,0 +1,327 @@ +### SherpaOnnxSynthesisSession + +This chapter covers the real session implementation together with the units it exists to +coordinate - the Layer 2 rendering strategy, the sentence chunker, the synthesis-backend seam, +the dedicated-worker cancellation policy, and the playback-format converter - because none can be +reviewed meaningfully in isolation: the session's whole job is to move text through rendering and +chunking, into the backend (via a dedicated worker), and the resulting audio through resampling +onto its bound playback device. + +**Purpose**: Chunk, render, and synthesize text into speech through a synthesis backend, bound to +exactly one playback device for the session's entire life, without ever performing synthesis or +inference work on the caller's UI thread, and without pipelining synthesis ahead of playback +across an unbounded queue. + +**Data Model**: Holds the owned `ISynthesisBackend`, the `IAudioPlaybackDevice` it plays through, +the `ISynthesisModel` supplying `NormalizeText`/`CapabilityProfile`/`ResolveSpeakerId`, an +optional `parameterValues` bag (the session's selected voice/tunable-parameter values, forwarded +unchanged from `SherpaOnnxSpeechSynthesizerEngine`), a diagnostics sink, a release callback for +the engine's exclusivity lease, and a `_syncRoot` lock guarding its `SynthesisSessionState`, its +current operation's `CancellationTokenSource`, its fault (if any), and disposal/lease-released +flags. `IsAvailable` is `true` unless disposed or `Faulted`. + +Unlike the former streaming synthesizer's pending-segment channel, this session does not +pipeline synthesis ahead of playback: each segment is synthesized, then (for `SpeakAsync`) +immediately written to the playback device, before the next segment's synthesis begins. This +keeps the per-call state machine simple - exactly one backend call in flight at a time - while +still overlapping this call's own synthesis-then-playback work normally, since playback of one +segment proceeds without the caller waiting for the whole plan up front. + +`DrainPollInterval` (15ms) and `DrainTailMargin` (40ms) are fixed constants governing +`WaitForPlaybackDrainAsync`'s post-loop wait in `GenerateAndOptionallyPlayAsync` (see below): +short enough that a `SpeakAsync` call returns promptly once the hardware is genuinely done, long +enough to avoid busy-spinning the thread pool re-reading a value only a real-time audio callback +thread can change. + +**Key Methods**: + +- **SpeakAsync(text, cancellationToken)** / **SynthesizeAsync(text, cancellationToken)**: Both + validate `text` eagerly, then delegate to the shared `RunOperationAsync(text, + playAfterSynthesis, cancellationToken)`, which validates the overlap rule and current state + under the lock, transitions `Starting` → `Running`, runs + `GenerateAndOptionallyPlayAsync`, then transitions `Stopping` → `Stopped` on success or + cancellation, or to `Faulted` (reporting the fault via diagnostics) on any other exception. A + second call while an operation is already `Starting`/`Running`/`Stopping` throws + `InvalidOperationException`; a call while `Faulted` throws `SynthesisSessionFaultedException` + wrapping the original fault; a call after disposal throws `ObjectDisposedException`. +- **StopAsync(cancellationToken)**: Cancels the current operation's `CancellationTokenSource` + under the lock, if one exists, then awaits that same tracked operation task (abandoning the + wait, but not the operation itself, past the dedicated worker's abandon timeout) so the + returned task completes only once the in-flight operation has genuinely stopped; a safe no-op + otherwise. The cancelled operation still unwinds through the normal `Stopping` → `Stopped` + path, not `Faulted`. +- **GenerateAndOptionallyPlayAsync(text, playAfterSynthesis, cancellationToken)**: Normalizes the + text via `_model.NormalizeText`, parses it via `AudioTagParser.Parse`, renders it via + `_model.CapabilityProfile.Render(spans, _model)` into a `SpeechPlan`, then synthesizes each + `SpeechSegment` in turn via `GenerateSegmentAsync`. When `playAfterSynthesis` is `false` + (`SynthesizeAsync`), simply collects every segment's `SynthesizedSpeech` into the returned list. + When `true` (`SpeakAsync`), starts the playback device, builds one `PlaybackAudioResampler` for + the call, writes each segment's pre-silence/resampled audio/post-silence as it is produced, and + once every segment has been written, awaits `WaitForPlaybackDrainAsync` before a `finally` block + stops the device - reporting (but not rethrowing) a failure to stop, so the device is never left + running regardless of whether the loop, the drain wait, or neither completed, faulted, or was + cancelled. +- **GenerateSegmentAsync(segment, cancellationToken)**: An empty-text segment (a rendered pause) + skips the backend entirely and returns a pure-silence `SynthesizedSpeech` built directly from + the segment's declared silence durations. Otherwise resolves speed/volume overrides against the + model's declared numeric parameters via `SpeechParameterConventions`, resolves the speaker id + for this session's selected voice by calling `_model.ResolveSpeakerId(_parameterValues)` once + per segment (cheap and pure - re-evaluated fresh each call rather than cached for the whole + session; deliberately independent of the per-segment `ParameterOverrides` speed/volume + mechanism, which is Natural Language Audio Tag-scoped and transient, not a fit for a + session-level voice selection), and calls `_backend.Generate(text, speedRatio, speakerId)` + through `DedicatedWorker.Run` - so a non-cooperative native call is bounded by the + cooperative-cancel-then-abandon policy rather than awaited indefinitely - applying any volume + override as post-hoc amplitude scaling (sherpa-onnx's offline TTS API has no native gain input), + clamped to `[-1.0, 1.0]`. +- **WaitForPlaybackDrainAsync(cancellationToken)**: Polls `_device.PendingSampleCount` every + `DrainPollInterval` (15ms) until it reaches zero - i.e. until the playback hardware has + genuinely rendered every sample this operation wrote, not merely until every segment was handed + off - then applies one further `DrainTailMargin` (40ms) wait before returning, to absorb any + residual internal host buffering (e.g. PortAudio's own ring buffer) that has already left the + managed queue but has not yet actually reached the speakers. Honors `cancellationToken`: a + cancellation during either wait ends it immediately via `Task.Delay`'s own cancellation. +- **DisposeAsync()**: Cancels any in-flight operation's `CancellationTokenSource`, transitions + `Disposing` → `Disposed`, and releases the owning engine's exclusivity lease exactly once (via + the constructor-supplied release callback). Idempotent. + +**Error Handling**: A fault raised synthesizing or playing a segment transitions the session to +`Faulted` and is reported through `ISpeechDiagnostics` before being rethrown to the caller of the +operation that faulted; every subsequent `SpeakAsync`/`SynthesizeAsync` call then throws +`SynthesisSessionFaultedException` wrapping that original fault, until the session is disposed. A +playback-device write failure still stops the device via the `finally` block before propagating. +A cancellation while `WaitForPlaybackDrainAsync` is polling ends the wait immediately via +`Task.Delay`'s own cancellation, while `finally` still stops the device; the operation unwinds to +`Stopped`, not `Faulted`, since cancellation is an ordinary, successful outcome. An unavailable +playback device throws promptly rather than hanging. Calling an operational member after +disposal throws `ObjectDisposedException`; calling one while another is already in flight on the +same session throws `InvalidOperationException`. + +**Dependencies**: `ISynthesisBackend`, `SpeechPlan`, `SpeechSegment`, `SpeechParameterConventions`, +`PlaybackAudioResampler`, `SynthesizedSpeech`, `SynthesisSessionState`, +`SessionStateChangedEventArgs`, `SynthesisSessionFaultedException`, `DedicatedWorker`, +`SpeechSynthesizerUnavailableException`, `AudioTagParser` from this subsystem's unchanged units, +`IAudioPlaybackDevice` from the AudioSubsystem, and `ISpeechDiagnostics` from the Diagnostics +subsystem. + +**Callers**: `SherpaOnnxSpeechSynthesizerEngine.CreateSessionAsync(device, cancellationToken)`. + +#### ISynthesisBackend and ISynthesisBackendFactory + +**Purpose**: Confine every speech-synthesis interop call behind one mockable boundary, so the +session's chunking, Layer 2 rendering, per-segment synthesis, and fault-containment logic is +verifiable with pure managed fakes - no downloaded model and no platform-specific native binary +are ever required in CI. This mirrors the `IRecognitionBackend`/`IRecognitionBackendFactory` seam +used for recognition. Renamed from `ISynthesisEngine`/`ISynthesisEngineFactory` so the "engine" +vocabulary is reserved for the public Layer 3 `ISpeechSynthesizerEngine` contract; members are +unchanged by the rename. + +**Data Model**: N/A (interfaces only). + +**Key Methods**: + +- **ISynthesisBackend.SampleRate**: The fixed rate this backend's `Generate(...)` output is + produced at. +- **ISynthesisBackend.Generate(text, speed, speakerId)**: Synthesizes one chunk of text at the + given speed multiplier and speaker id, returning an `EngineAudio` (samples plus the rate they + were produced at). +- **ISynthesisBackendFactory.Create(ISynthesisModel, string installedModelDirectory)**: Loads a + backend from the model's own declared configuration. + +**Error Handling**: Implementations are not thread-safe by contract; a session calls `Generate` +from exactly one `DedicatedWorker` call at a time. Load failures surface as exceptions from +`Create(...)`, which `SpeechSynthesizerFactory` converts into the honest unavailable fallback. + +**Dependencies**: `EngineAudio`; `ISynthesisModel` from the ModelManagementSubsystem. + +**Callers**: `SherpaOnnxSynthesisSession` (backend) and `SpeechSynthesizerFactory`/ +`SherpaOnnxSpeechSynthesizerEngine` (factory). + +#### SherpaOnnxSynthesisEngine and SherpaOnnxSynthesisEngineFactory + +**Purpose**: Implement the backend seam against the real sherpa-onnx offline TTS API, reusing the +already-referenced `org.k2fsa.sherpa.onnx` package. These are the only types in this subsystem +that call speech-synthesis inference APIs. Class names are unchanged by the +`ISynthesisEngine`→`ISynthesisBackend` interface rename - only the interfaces they implement were +renamed. + +**Data Model**: The engine holds one loaded `OfflineTts`, its declared `SampleRate`, and a +disposed flag. The factory is stateless and holds no model-specific knowledge at all - the design +makes each model's backing class responsible for its own engine configuration, so adding a +synthesis model never requires changing the factory. + +**Key Methods**: + +- **Generate(...)**: Calls the underlying `OfflineTts.Generate(text, speed, speakerId)` and wraps + its output samples and rate in an `EngineAudio`. +- **Create(...)**: Asks the model for its `OfflineTtsConfig`, resolved against the + installed-files directory, and loads an engine from it. +- **Dispose()**: Releases the loaded `OfflineTts`. Safe to call more than once. + +**Error Handling**: Construction loads the model into native memory and therefore throws when the +native runtime for the current platform is absent or the model files are unusable; that exception +is what `SpeechSynthesizerFactory` converts into the honest unavailable fallback. Operational +members throw `ObjectDisposedException` after disposal. + +**Dependencies**: The sherpa-onnx managed API (see *SherpaOnnx Design*); `ISynthesisModel` from +the ModelManagementSubsystem. + +**Callers**: `SpeechSynthesizerFactory` constructs the factory; the factory constructs the +engine; `SherpaOnnxSpeechSynthesizerEngine` owns the engine and passes it to each +`SherpaOnnxSynthesisSession` it creates. + +#### IModelCapabilityProfile and DefaultModelCapabilityProfile + +**Purpose**: Decide, per Layer 1 span, how a Natural Language Audio Tag is realized for a +specific model - pass-through, approximation, or strip - turning a model-independent span +sequence into an ordered, model-appropriate `SpeechPlan`. + +**Data Model**: `IModelCapabilityProfile` declares one method, +`Render(IReadOnlyList spans, ISpeechModel model)`, returning a `SpeechPlan` (an +ordered `IReadOnlyList`). Each `SpeechSegment` carries the text to synthesize (or +empty, for a pure-silence pause segment), the pre/post silence durations to play alongside it, +and any numeric parameter overrides (speed/volume) to apply when synthesizing it. +`DefaultModelCapabilityProfile` is a stateless singleton (`Instance`) requiring no per-model +configuration, since every decision it makes is read from the `model` argument at call time. + +**Key Methods**: + +- **Render(spans, model)**: For each `TaggedTextSpanKind.Tag` span: if the tag is a pause, always + renders a silence-only `SpeechSegment` (300ms short / 900ms long) regardless of + `model.AudioTagSupport`. Otherwise, if `model.AudioTagSupport` is `Native`, passes the tag + through as its canonical bracket text so the model's own inference sees the control token. If + `ParameterMapped`, consults `SpeechParameterConventions` for a matching numeric parameter on + `model.Parameters` and attaches it as an override on the surrounding segment when one exists, + otherwise strips the tag. If `None`, strips every non-pause tag. For each + `TaggedTextSpanKind.PlainText` span, delegates to `SentenceChunker.ChunkWithMetadata(...)` and + emits one `SpeechSegment` per resulting chunk, setting that segment's `PostSilenceMs` to a new + `EllipsisPauseMilliseconds` constant (500ms) when the chunk's `EndsWithEllipsis` flag is set, or + `0` otherwise - a chunk-boundary-triggered pause distinct from, and never reusing, the + tag-triggered short/long pause durations above. + +**Error Handling**: Rejects a null `spans` or `model` with `ArgumentNullException`. Every other +input - any tag, any support level, any parameter set - is handled without throwing, per this +library's "never worse than plain narration" guarantee extended to Layer 2. + +**Dependencies**: `TaggedTextSpan`, `TaggedTextSpanKind`, `NaturalLanguageAudioTagKind` from this +subsystem's unchanged units; `SentenceChunker`; `SpeechParameterConventions`; `SpeechSegment`, +`SpeechPlan`; `ISpeechModel`, `SpeechModelAudioTagSupport`, `NumericParameter` from the +ModelManagementSubsystem. + +**Callers**: `SherpaOnnxSynthesisSession.GenerateAndOptionallyPlayAsync(...)` via +`ISynthesisModel.CapabilityProfile`. + +#### SentenceChunker + +**Purpose**: Split synthesis-ready plain text into sentence/clause-sized chunks so chunked +synthesis has natural-sounding boundaries and no chunk exceeds a length a synthesis call can +reasonably handle. + +**Data Model**: A stateless static class; `Chunk(text, maxLength)`/`ChunkWithMetadata(text, +maxLength)` take no configuration beyond their two arguments. `ChunkWithMetadata` returns a +`SentenceChunk` record struct per chunk (`Text`, `EndsWithEllipsis`); `Chunk` is a pure projection +of `ChunkWithMetadata`'s chunk text. + +**Key Methods**: + +- **Chunk(text, maxLength)** / **ChunkWithMetadata(text, maxLength)**: Splits on primary + sentence-ending punctuation (`.`, `!`, `?`) first, then **unconditionally** splits every + resulting piece further on secondary clause punctuation (`,`, `;`, `:`) - not only when the + piece is still over `maxLength` - so every clause becomes its own chunk. A punctuation + character embedded in a numeral is never treated as a boundary at either pass: a digit both + immediately before and after (e.g. the `:` in `"12:30"`, the `,` in `"1,000"`, or a decimal + point such as `"0.5"`) is always kept attached, and a decimal point specifically is also kept + attached when a digit follows immediately but none precedes, as long as the character before it + (if any) is not a letter - e.g. a leading fraction such as `".5"`, `"$.99"`, or `".5 units"` at + start-of-text/after whitespace/a sign/a currency symbol. A genuine sentence-ending period + directly followed by a digit after a letter (e.g. `"Wait.5 more."`) still splits normally, since + the numeral exception only applies when no letter precedes the period. A piece still longer + than `maxLength` after both punctuation passes is split on a whitespace budget; a single word + that alone exceeds `maxLength` is returned whole, never split mid-word. Any resulting piece + whose trimmed text contains no letter or digit at all (just punctuation and/or whitespace, e.g. + a lone comma or a whitespace-spaced ellipsis like `". . ."`) is then merged onto the end of the + immediately preceding non-empty chunk (joined by a single space), or dropped if there is no + preceding chunk. `ChunkWithMetadata` additionally flags a chunk whose final text ends in three + or more consecutive `.` characters (with or without interspersed whitespace) as + `EndsWithEllipsis`. These numeral exceptions assume English/US-style numeral punctuation (`,` + as thousands separator, `.` as decimal point); locales that swap the two roles (e.g. + `"1.000,5"`) are not specially handled - every model in this library's current catalog is + English-only, so this is not currently a defect, but a future non-English model would need + these rules adjusted rather than assuming they generalize as-is. + +**Error Handling**: Rejects a non-positive `maxLength` with `ArgumentOutOfRangeException` and a +null `text` with `ArgumentNullException`, since neither describes a meaningful chunking request. +Empty and whitespace-only text return no chunks rather than throwing. + +**Dependencies**: None beyond the Base Class Library. + +**Callers**: `DefaultModelCapabilityProfile.Render(...)`. + +#### PlaybackAudioResampler + +**Purpose**: Convert one synthesized segment's mono audio, produced at a synthesis backend's +fixed rate, into the format a playback device requires - its resolved rate and channel count. +Keeping this conversion in a pure, dependency-free type mirrors `AudioFrameResampler`'s +recognition direction role and makes it exhaustively testable with plain float arrays. + +**Data Model**: Immutable: the backend sample rate, the target (device) sample rate, and the +target channel count, all validated as positive at construction. No per-call state, so one +instance serves a whole operation. + +**Key Methods**: + +- **Resample(samples, sourceSampleRate, targetSampleRate)**: Produces + `floor(length * target / source)` samples via linear interpolation, clamped to the last input + sample at the boundary. Equal rates copy the input unchanged, so the identity case introduces no + error at all. When downsampling, a small Hamming-windowed sinc lowpass filter runs first to + attenuate above-target-Nyquist energy before decimation. +- **UpmixToChannels(samples, channelCount)**: Replicates each mono sample across every output + channel, interleaved. A single required channel copies the input unchanged. +- **Convert(samples)**: Composes `Resample` then `UpmixToChannels` into the single operation the + session needs for every segment it plays. When the target channel count is 1, + `UpmixToChannels` is skipped entirely and the resampled result is returned directly, since + `UpmixToChannels`'s own single-channel case would only make a redundant copy of an array + `Convert` already owns exclusively. + +**Error Handling**: A non-positive sample rate or channel count throws +`ArgumentOutOfRangeException`. Empty input returns an empty result rather than throwing, since a +pure-silence segment legitimately has no samples to convert. + +**Dependencies**: `AudioSubsystem.WindowedSincLowpassFilter` for the downsampling anti-aliasing +lowpass stage, shared with `AudioFrameResampler`'s identical need in the recognition subsystem; +otherwise none beyond the Base Class Library. + +**Callers**: `SherpaOnnxSynthesisSession.PlaySegment(...)` (via `GenerateAndOptionallyPlayAsync`). + +#### DedicatedWorker + +**Purpose**: Run one native synthesis call on a dedicated, long-running background thread, +applying a cooperative-cancel-then-abandon policy so a non-cooperative native call cannot hang a +caller's awaited task indefinitely. `SynthesisSubsystem`'s own duplicated copy of +`RecognitionSubsystem`'s identically-shaped internal utility. + +**Data Model**: A stateless static class. `Run(delegate, cancellationToken, diagnostics, +diagnosticsCategory, abandonTimeout = null)` takes no instance state; `abandonTimeout` defaults +to `DefaultAbandonTimeout` (2 seconds) when omitted/`null`, and is injectable for deterministic +test coverage of the abandon path. + +**Key Methods**: + +- **Run(delegate, cancellationToken, diagnostics, diagnosticsCategory, abandonTimeout)**: Starts + `delegate` on a `TaskCreationOptions.LongRunning` task. On normal completion, returns its + result. On cancellation, awaits the task for up to `abandonTimeout`; if it completes + cooperatively within that window, the cancellation propagates normally. If it does not, reports + a `Warning` diagnostic, detaches the still-running task (attaching a continuation that observes, + rather than propagates, whatever it eventually produces, so it can never become an + unobserved-exception crash), and throws `OperationCanceledException` to its own caller anyway - + the caller must not be held open indefinitely by a native call that refuses to stop. + +**Error Handling**: A genuine (non-cancellation) fault from `delegate` propagates normally from +the awaited task. The abandon path above is the sole case where this method returns/throws before +`delegate` has actually finished running. + +**Dependencies**: `ISpeechDiagnostics` from the Diagnostics subsystem. + +**Callers**: `SherpaOnnxSynthesisSession.GenerateSegmentAsync(...)` for every native +`ISynthesisBackend.Generate(...)` call, and `SpeechSynthesizerFactory.LoadAsync(...)` for the +blocking native model load. diff --git a/docs/design/speech/synthesis-subsystem/speech-synthesizer-factory.md b/docs/design/speech/synthesis-subsystem/speech-synthesizer-factory.md index b06bbe1..d72dfb3 100644 --- a/docs/design/speech/synthesis-subsystem/speech-synthesizer-factory.md +++ b/docs/design/speech/synthesis-subsystem/speech-synthesizer-factory.md @@ -1,77 +1,82 @@ ### SpeechSynthesizerFactory -**Purpose**: Provide the single composition entry point for obtaining an `ISpeechSynthesizer`, so -all "can this machine speak right now?" logic lives in one reviewable place, mirroring -`SpeechRecognizerFactory` exactly. +**Purpose**: Provide the single composition entry point for obtaining an +`ISpeechSynthesizerEngine`, so all "can this machine speak right now?" logic lives in one +reviewable place, mirroring `SpeechRecognizerFactory` exactly. Unlike the former synchronous +factory, this type no longer takes a playback device at all: an engine loaded here can create +many sessions over its life, each bound to its own device, via +`ISpeechSynthesizerEngine.CreateSessionAsync`. -**Data Model**: A static class with no state. Three public `Create(...)` overloads exist: one -resolving an installed-model directory from a caller-supplied `string`, one resolving it from -a `SpeechModelStore` directly via `store.GetCurrentDirectory(model.Id)`, and one resolving it -from a `SpeechModelCatalog` directly via `catalog.Store.GetCurrentDirectory(model.Id)`. All call +**Data Model**: A static class with no state. Three public `LoadAsync(...)` overloads exist: one +resolving an installed-model directory from a caller-supplied `string`, one resolving it from a +`SpeechModelStore` directly via `store.GetCurrentDirectory(model.Id)`, and one resolving it from +a `SpeechModelCatalog` directly via `catalog.Store.GetCurrentDirectory(model.Id)`. All call through to the same internal composition logic, each with an internal counterpart that accepts -an injected `ISynthesisEngineFactory` so composition can be verified without model files or a -native runtime. +an injected `ISynthesisBackendFactory` so composition can be verified without model files or a +native runtime. The blocking native model load itself runs on a `DedicatedWorker` rather than the +calling thread, so awaiting any `LoadAsync` overload never blocks a caller's synchronization +context. **Key Methods**: -- **Create(ISynthesisModel model, string installedModelDirectory, IAudioPlaybackDevice - playbackDevice, ISpeechDiagnostics? diagnostics, IReadOnlyDictionary<string, object>? - parameterValues = null)**: Returns a real `SherpaOnnxSpeechSynthesizer` - when the model's installed directory exists, the model declares `SpeechModelRole.Synthesis`, - the playback device reports `IsAvailable`, and the engine loads. Otherwise returns - `UnavailableSpeechSynthesizer.Instance`. Preconditions: `model` and `playbackDevice` are - non-null. Postcondition: the returned synthesizer is never null, and either owns a loaded - engine or is the shared unavailable instance. Loading the engine allocates native resources, so - the returned synthesizer must be disposed. `parameterValues` is an optional session-level - parameter value bag (for example a selected voice, built from the model's declared - `ISpeechModel.Parameters`), forwarded unchanged to the returned synthesizer, which re-resolves - it via `ISynthesisModel.ResolveSpeakerId` once per synthesized segment; `null` means every - model's own default voice/speaker. -- **Create(ISynthesisModel model, SpeechModelStore store, IAudioPlaybackDevice playbackDevice, - ISpeechDiagnostics? diagnostics, IReadOnlyDictionary<string, object>? parameterValues = - null)**: A convenience overload with byte-for-byte identical behavior to the `string`-based - overload above; it resolves `store.GetCurrentDirectory(model.Id)` for the caller and delegates - to the same overload, so a host never needs to know `SpeechModelStore`'s on-disk - directory-naming scheme just to compose a synthesizer. Preconditions: `model`, `store`, and - `playbackDevice` are non-null. -- **Create(ISynthesisModel model, SpeechModelCatalog catalog, IAudioPlaybackDevice - playbackDevice, ISpeechDiagnostics? diagnostics, IReadOnlyDictionary<string, object>? - parameterValues = null)**: A convenience overload delegating through the - `SpeechModelStore`-based overload via the catalog's own `Store` property, so a host that already - owns a `SpeechModelCatalog` for enumeration and download can compose a synthesizer through that - same catalog instance, without constructing a second, potentially divergent `SpeechModelStore`. - Preconditions: `model`, `catalog`, and `playbackDevice` are non-null. +- **LoadAsync(ISynthesisModel model, string installedModelDirectory, ISpeechDiagnostics? + diagnostics, IReadOnlyDictionary<string, object>? parameterValues = null, + CancellationToken cancellationToken = default)**: Returns a real + `SherpaOnnxSpeechSynthesizerEngine` when the model's installed directory exists, the model + declares `SpeechModelRole.Synthesis`, and the backend loads. Otherwise returns + `UnavailableSpeechSynthesizerEngine.Instance`. Precondition: `model` is non-null. Postcondition: + the returned engine is never null, and either owns a loaded backend or is the shared unavailable + instance. Loading the backend allocates native resources, so the returned engine must be + disposed. `parameterValues` is an optional engine-level parameter value bag (for example a + selected voice, built from the model's declared `ISpeechModel.Parameters`), forwarded unchanged + to the returned engine and onward to every session it creates, which re-resolves it via + `ISynthesisModel.ResolveSpeakerId` once per synthesized segment; `null` means every model's own + default voice/speaker. +- **LoadAsync(ISynthesisModel model, SpeechModelStore store, ISpeechDiagnostics? diagnostics, + IReadOnlyDictionary<string, object>? parameterValues = null, CancellationToken + cancellationToken = default)**: A convenience overload with byte-for-byte identical behavior to + the `string`-based overload above; it resolves `store.GetCurrentDirectory(model.Id)` for the + caller and delegates to the same overload, so a host never needs to know `SpeechModelStore`'s + on-disk directory-naming scheme just to compose an engine. Preconditions: `model` and `store` + are non-null. +- **LoadAsync(ISynthesisModel model, SpeechModelCatalog catalog, ISpeechDiagnostics? diagnostics, + IReadOnlyDictionary<string, object>? parameterValues = null, CancellationToken + cancellationToken = default)**: A convenience overload delegating through the + `SpeechModelStore`-based overload via the catalog's own `Store` property, so a host that + already owns a `SpeechModelCatalog` for enumeration and download can compose an engine through + that same catalog instance, without constructing a second, potentially divergent + `SpeechModelStore`. Preconditions: `model` and `catalog` are non-null. The checks run in the same deliberate order as the recognition-direction factory - parameter -validation, then installed, then role, then device, then engine load - so a caller-supplied -parameter value invalid for a recognized parameter is rejected synchronously and loudly before -any of the ordinary, never-throw machine state checks run, and so the cheapest and most common -cause of unavailability (a model not downloaded yet) is reported first among those and no native -memory is allocated for a synthesizer that could never run. +validation, then installed, then role, then backend load - so a caller-supplied parameter value +invalid for a recognized parameter is rejected synchronously and loudly before any of the +ordinary, never-throw machine state checks run, and so the cheapest and most common cause of +unavailability (a model not downloaded yet) is reported first among those and no native memory is +allocated for an engine that could never run. -**Error Handling**: Every ordinary machine state is represented as the honest unavailable -synthesizer plus a structural diagnostic, never as an exception, per this library's "nothing -throws at composition" decision. An engine load failure is caught and degraded identically to a -missing model. Only a null `model`, `store`, `catalog`, `playbackDevice`, or engine factory -throws `ArgumentNullException`, since a null argument is a programming error rather than a -machine state. **Breaking change**: `parameterValues` is now validated against `model.Parameters` +**Error Handling**: Every ordinary machine state is represented as the honest unavailable engine +plus a structural diagnostic, never as an exception, per this library's "nothing throws at +composition" decision. A backend load failure is caught and degraded identically to a missing +model. Only a null `model`/`store`/`catalog`/backend factory throws `ArgumentNullException`, an +invalid `parameterValues` entry throws `ArgumentException`, and a cancelled `cancellationToken` +throws `OperationCanceledException`, since those are programming errors or an explicit caller +request rather than a machine state. `parameterValues` is validated against `model.Parameters` before any other work runs, using the same shared `SpeechModelParameterDiagnostics.ValidateAndReport` helper as `SpeechRecognizerFactory`. A supplied key naming a parameter not declared by `model` is -still silently ignored exactly as before (preserving the documented cross-model-compatibility -contract) but now also reports an `Info` diagnostic. A supplied value for a parameter that *is* -declared by `model` but fails that parameter's own validation (wrong CLR type, out-of-range or -non-integral for a `NumericParameter`, unrecognized `ChoiceParameter` option, non-`bool` for a -`BooleanParameter`) now throws `ArgumentException` synchronously from `Create()` naming the -parameter id, model id, and the reason the value is invalid - previously such a value was -silently substituted with a default the first time `ResolveSpeakerId` ran per segment. This -validation happens once, up front, at `Create()`; it does not change `ResolveSpeakerId`'s or -`ResolveOverrideRatios`'s own existing never-throw, per-segment runtime contract. +silently ignored (preserving the documented cross-model-compatibility contract) but reports an +`Info` diagnostic. A supplied value for a parameter that *is* declared by `model` but fails that +parameter's own validation (wrong CLR type, out-of-range or non-integral for a `NumericParameter`, +unrecognized `ChoiceParameter` option, non-`bool` for a `BooleanParameter`) throws +`ArgumentException` synchronously from `LoadAsync()` naming the parameter id, model id, and the +reason the value is invalid. This validation happens once, up front, at `LoadAsync()`; it does +not change `ResolveSpeakerId`'s or `ResolveOverrideRatios`'s own existing never-throw, per-segment +runtime contract. **Dependencies**: `ISynthesisModel`, `SpeechModelRole`, `SpeechModelStore`, `SpeechModelCatalog`, -and `SpeechModelParameterDiagnostics` from the ModelManagementSubsystem, `IAudioPlaybackDevice` -from the AudioSubsystem, `ISpeechDiagnostics`/`NullSpeechDiagnostics` from the Diagnostics -subsystem, and the subsystem's own `ISynthesisEngineFactory`, `SherpaOnnxSynthesisEngineFactory`, -`SherpaOnnxSpeechSynthesizer`, and `UnavailableSpeechSynthesizer`. +and `SpeechModelParameterDiagnostics` from the ModelManagementSubsystem, +`ISpeechDiagnostics`/`NullSpeechDiagnostics` from the Diagnostics subsystem, and the subsystem's +own `ISynthesisBackendFactory`, `SherpaOnnxSynthesisEngineFactory`, +`SherpaOnnxSpeechSynthesizerEngine`, `UnavailableSpeechSynthesizerEngine`, and `DedicatedWorker`. **Callers**: Host applications composing speech synthesis at start-up, and the system-level integration tests. diff --git a/docs/design/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.md b/docs/design/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.md new file mode 100644 index 0000000..ec85c74 --- /dev/null +++ b/docs/design/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.md @@ -0,0 +1,30 @@ +### UnavailableSpeechSynthesizerEngine + +**Purpose**: Provide a safe, always-obtainable `ISpeechSynthesizerEngine` fallback for use when +synthesis is not possible on the current machine. + +**Data Model**: No instance state. Exposes a single static `Instance` singleton; the constructor +is private, since the type carries no state and multiple instances would provide no value. +`IsAvailable` always returns `false`. + +**Key Methods**: + +- **CreateSessionAsync(device, cancellationToken)**: Always succeeds, returning the shared + `UnavailableSynthesisSession.Instance` - binding a device to an already unavailable engine is + itself an ordinary (if useless) composition, not an error, so only the returned session's + operational members throw. Throws `ArgumentNullException` for a null `device`. +- **SpeakAsync(...)** / **SynthesizeAsync(...)**: Always throw + `SpeechSynthesizerUnavailableException`. +- **DisposeAsync()**: A no-op that never throws and never invalidates `Instance`, so a host that + wraps its engine in a disposal scope runs unchanged on a machine without synthesis. + +**Error Handling**: Obtaining and holding the instance never throws. Only the operational members +throw, and only when actually invoked - a caller that checks `IsAvailable` first never triggers +them. This mirrors `UnavailableSpeechRecognizerEngine` exactly, so both subsystems degrade the +same recognizable way. + +**Dependencies**: `SpeechSynthesizerUnavailableException`, `UnavailableSynthesisSession`; +implements `ISpeechSynthesizerEngine`. + +**Callers**: `SpeechSynthesizerFactory.LoadAsync(...)` when the model is not installed, the +model's role is not synthesis, or the backend cannot be loaded. diff --git a/docs/design/speech/synthesis-subsystem/unavailable-speech-synthesizer.md b/docs/design/speech/synthesis-subsystem/unavailable-speech-synthesizer.md deleted file mode 100644 index b397e51..0000000 --- a/docs/design/speech/synthesis-subsystem/unavailable-speech-synthesizer.md +++ /dev/null @@ -1,41 +0,0 @@ -### UnavailableSpeechSynthesizer - -**Purpose**: Provide a safe, always-obtainable `ISpeechSynthesizer` fallback for use when -synthesis is not possible on the current machine. - -**Data Model**: No instance state. Exposes a single static `Instance` singleton; the constructor -is private, since the type carries no state and multiple instances would provide no value. -`IsAvailable` always returns `false`. - -**Key Methods**: - -- **SynthesizeStreamAsync(...)** / **PlayStreamAsync(...)** / **SpeakAsync(...)** / **Stop()**: - Always throw `SpeechSynthesizerUnavailableException`. -- **Dispose()**: A no-op that never throws and never invalidates `Instance`, so a host that wraps - its synthesizer in a disposal scope runs unchanged on a machine without synthesis. - -**Error Handling**: Obtaining and holding the instance never throws. Only the operational members -throw, and only when actually invoked - a caller that checks `IsAvailable` first never triggers -them. This mirrors `UnavailableSpeechRecognizer` exactly, so both subsystems degrade the same -recognizable way. - -**Dependencies**: `SpeechSynthesizerUnavailableException`; implements `ISpeechSynthesizer`. - -**Callers**: `SpeechSynthesizerFactory.Create(...)` when the model is not installed, the model's -role is not synthesis, the playback device is unavailable, or the engine cannot be loaded. - -### SpeechSynthesizerUnavailableException - -**Purpose**: Signal that an operational member of an unavailable synthesizer was invoked, or that -a synthesizer that claimed to be available failed on first use. - -**Data Model**: No additional fields beyond the standard `Exception` base members. - -**Key Methods**: Standard three-constructor exception pattern. - -**Error Handling**: This type is itself the error-handling mechanism. - -**Dependencies**: `Exception`. - -**Callers**: `UnavailableSpeechSynthesizer` for every operational member, and -`SherpaOnnxSpeechSynthesizer` when its playback device fails to start. diff --git a/docs/design/speech/synthesis-subsystem/unavailable-synthesis-session.md b/docs/design/speech/synthesis-subsystem/unavailable-synthesis-session.md new file mode 100644 index 0000000..1540bae --- /dev/null +++ b/docs/design/speech/synthesis-subsystem/unavailable-synthesis-session.md @@ -0,0 +1,26 @@ +### UnavailableSynthesisSession + +**Purpose**: Provide the honest `ISynthesisSession` fallback returned by +`UnavailableSpeechSynthesizerEngine.CreateSessionAsync`, so a host that composes a session from an +unavailable engine still gets a safe, well-behaved object rather than a null or an exception from +composition itself. + +**Data Model**: No instance state. Exposes a single static `Instance` singleton; the constructor +is private. `IsAvailable` always returns `false`. `State` always reports +`SynthesisSessionState.Created`, since this session never transitions - there is no operation it +can ever genuinely perform. + +**Key Methods**: + +- **SpeakAsync(...)** / **SynthesizeAsync(...)**: Always throw + `SpeechSynthesizerUnavailableException`. Throw `ArgumentNullException` for a null `text`. +- **StopAsync(...)**: A safe no-op, since no operation is ever in flight. +- **DisposeAsync()**: A no-op that never throws and never invalidates `Instance`. + +**Error Handling**: Obtaining and holding the instance never throws. Only the operational members +throw, and only when actually invoked. Mirrors `UnavailableSpeechSynthesizerEngine`. + +**Dependencies**: `SpeechSynthesizerUnavailableException`, `SynthesisSessionState`; implements +`ISynthesisSession`. + +**Callers**: `UnavailableSpeechSynthesizerEngine.CreateSessionAsync(...)`. diff --git a/docs/reqstream/ots/sherpa-onnx.yaml b/docs/reqstream/ots/sherpa-onnx.yaml index b8d25d9..2132a81 100644 --- a/docs/reqstream/ots/sherpa-onnx.yaml +++ b/docs/reqstream/ots/sherpa-onnx.yaml @@ -32,5 +32,5 @@ sections: recognition contract so the engine backend stays swappable. tags: [ots] tests: - - SpeechRecognizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealRecognizer - - SpeechRecognizerFactory_Create_EngineLoadFails_ReturnsUnavailableRecognizerAndDoesNotThrow + - SpeechRecognizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine + - SpeechRecognizerFactory_LoadAsync_EngineLoadFails_ReturnsUnavailableEngineAndDoesNotFaultTask diff --git a/docs/reqstream/speech-cli.yaml b/docs/reqstream/speech-cli.yaml index 0649afe..5ee4d79 100644 --- a/docs/reqstream/speech-cli.yaml +++ b/docs/reqstream/speech-cli.yaml @@ -7,10 +7,10 @@ # ships as a separately packaged .NET global tool (`speech-cli`). # # This pass covers only the scaffold-level behavior actually implemented: subcommand dispatch, -# global options, version/help display, self-validation, and global-tool packaging. The ten +# global options, version/help display, self-validation, and global-tool packaging. The eleven # recognized subcommands (list-models, model-info, download, uninstall, clean, list-devices, -# devices, doctor, speak, recognize) are dispatched to by name and listed in --help, and all ten -# are now implemented. Requirements for the five model-management subcommands' functional +# devices, doctor, speak, recognize, ask) are dispatched to by name and listed in --help, and all +# eleven are now implemented. Requirements for the five model-management subcommands' functional # behavior are decomposed into docs/reqstream/speech-cli/model-commands-subsystem.yaml; # requirements for the three device-related subcommands' functional behavior are decomposed into # docs/reqstream/speech-cli/device-commands-subsystem.yaml; requirements for the speak @@ -129,6 +129,7 @@ sections: - SpeechCli-ModelCommands-Clean - SpeechCli-ModelCommands-LibraryDelegation - SpeechCli-ModelCommands-SynthesisSeamExtension + - SpeechCli-ModelCommands-RecognitionSeamExtension tests: - SpeechCli_ListModelsCommand_Invoked_ListsKnownModelsAsTable - DownloadCommand_RunAsync_MultipleModels_DownloadsEachSequentially @@ -195,8 +196,8 @@ sections: - SpeechCli-RecognitionCommands-CatalogSeamExtension - SpeechCli-RecognitionCommands-NullGuards tests: - - RecognizeCommand_Run_ValidParam_ForwardsToCreateRecognizer - - RecognizeCommand_Run_Success_DisposesRecognizerOnce + - RecognizeCommand_Run_ValidParam_ForwardsToCreateRecognizerEngine + - RecognizeCommand_Run_Success_DisposesEngineAndSessionOnce - id: SpeechCli-VoiceConversation title: >- diff --git a/docs/reqstream/speech-cli/conversation-command-subsystem.yaml b/docs/reqstream/speech-cli/conversation-command-subsystem.yaml index c20ba6c..c44038b 100644 --- a/docs/reqstream/speech-cli/conversation-command-subsystem.yaml +++ b/docs/reqstream/speech-cli/conversation-command-subsystem.yaml @@ -6,10 +6,10 @@ # docs/reqstream/speech-cli.yaml. # # This subsystem introduces no new ICliModelCatalog seam member: it reuses -# SynthesisCommandSubsystem's CreateSynthesizer and RecognitionCommandSubsystem's -# CreateRecognizer members exactly as speak and recognize already do individually, since ask -# requires exactly one synthesis-role model and exactly one recognition-role model, never a -# single model serving both roles. +# SynthesisCommandSubsystem's CreateSynthesizerEngineAsync and RecognitionCommandSubsystem's +# CreateRecognizerEngineAsync members exactly as speak and recognize already do individually, +# since ask requires exactly one synthesis-role model and exactly one recognition-role model, +# never a single model serving both roles. sections: - title: SpeechCli Requirements @@ -56,6 +56,8 @@ sections: - AskCommand_Run_NotDownloadedTtsModel_ThrowsArgumentExceptionWithDownloadHint - AskCommand_Run_NotDownloadedSttModel_ThrowsArgumentExceptionWithDownloadHint - AskCommand_Run_Success_SpeaksThenListensAndPrintsFinalResult + - Program_Run_WithAskCommand_DoesNotThrowNotImplemented + - SpeechCli_HelpFlag_Provided_ListsAllSubcommands - id: SpeechCli-ConversationCommands-ParamValidation title: >- @@ -63,7 +65,7 @@ sections: flags and validate each value against the respective resolved model's own declared parameter set, reusing the shared `ParameterBagParser` from the SynthesisCommandSubsystem, forwarding each fully-resolved parameter bag to its own - `CreateSynthesizer`/`CreateRecognizer` call. + `CreateSynthesizerEngineAsync`/`CreateRecognizerEngineAsync` call. tags: [subsystem] justification: | Both synthesis and recognition models declare tunable parameters; reusing the @@ -93,6 +95,7 @@ sections: - AskCommand_ParseArguments_TimeoutFlags_ParseAsPositiveDoubles - AskCommand_ParseArguments_NonPositiveSilenceTimeout_ThrowsArgumentException - AskCommand_Run_SilenceTimeoutWithNoFinalResult_EndsTurnWithEmptyText + - AskCommand_Run_NoTimeoutFlagsGiven_StillEndsTurnViaDefaultSilenceTimeout - id: SpeechCli-ConversationCommands-OutputDispatch title: >- @@ -161,7 +164,7 @@ sections: tags: [subsystem, performance] justification: | Loading a recognition model into native memory is the expensive step in - constructing an `ISpeechRecognizer`; deferring that load until after Phase 1's + constructing an `IRecognitionSession`; deferring that load until after Phase 1's playback finishes introduces an avoidable turnaround-gap latency between finishing speaking and starting to listen, which matters for a natural-feeling voice conversation. Overlapping the load with the time a person spends listening to the @@ -189,3 +192,9 @@ sections: - AskCommand_Run_NullDeviceSource_ThrowsArgumentNullException - AskCommand_Run_NullCaptureSource_ThrowsArgumentNullException - AskCommand_ParseArguments_NullArgs_ThrowsArgumentNullException + - AskCommand_RunAsync_NullContext_ThrowsArgumentNullException + - AskCommand_RunAsync_NullCatalog_ThrowsArgumentNullException + - AskCommand_RunAsync_NullDeviceSource_ThrowsArgumentNullException + - AskCommand_RunAsync_NullCaptureSource_ThrowsArgumentNullException + - AskCommand_RunAsync_NullStopSignal_ThrowsArgumentNullException + - AskCommand_RunAsync_NullOnRecognizerCreated_ThrowsArgumentNullException diff --git a/docs/reqstream/speech-cli/model-commands-subsystem.yaml b/docs/reqstream/speech-cli/model-commands-subsystem.yaml index 71f0c31..ac3b571 100644 --- a/docs/reqstream/speech-cli/model-commands-subsystem.yaml +++ b/docs/reqstream/speech-cli/model-commands-subsystem.yaml @@ -169,7 +169,7 @@ sections: - id: SpeechCli-ModelCommands-SynthesisSeamExtension title: >- The `ICliModelCatalog` seam shall additionally expose a resolved model's preferred - audio format and construct its `ISpeechSynthesizer`, for use by + audio format and construct its `ISpeechSynthesizerEngine`, for use by `SynthesisCommandSubsystem`'s `speak` subcommand, rejecting a descriptor whose model is not a synthesis model with a clean `ArgumentException` rather than an unhandled cast exception. @@ -184,6 +184,27 @@ sections: tests: - SpeechModelCatalogAdapter_GetPreferredAudioFormat_RecognitionRoleModel_ThrowsArgumentException - SpeechModelCatalogAdapter_GetPreferredAudioFormat_NullDescriptor_ThrowsArgumentNullException - - SpeechModelCatalogAdapter_CreateSynthesizer_RecognitionRoleModel_ThrowsArgumentException - - SpeechModelCatalogAdapter_CreateSynthesizer_NullDescriptor_ThrowsArgumentNullException - - SpeechModelCatalogAdapter_CreateSynthesizer_NullPlaybackDevice_ThrowsArgumentNullException + - SpeechModelCatalogAdapter_CreateSynthesizerEngineAsync_RecognitionRoleModel_ThrowsArgumentException + - SpeechModelCatalogAdapter_CreateSynthesizerEngineAsync_NullDescriptor_ThrowsArgumentNullException + + - id: SpeechCli-ModelCommands-RecognitionSeamExtension + title: >- + The `ICliModelCatalog` seam shall additionally expose a resolved model's required + audio format and construct its `ISpeechRecognizerEngine`, for use by + `RecognitionCommandSubsystem`'s `recognize` subcommand, rejecting a descriptor whose + model is not a recognition model with a clean `ArgumentException` rather than an + unhandled cast exception. + tags: [subsystem] + justification: | + Added in Pass 6 alongside `RecognitionCommandSubsystem`, symmetric to the synthesis + extension above: a resolved model must be cast to the library's internal + `IRecognitionModel` before either operation is possible, and that cast is only + compilable inside this same assembly, not inside the test project. Extending this + existing seam - rather than introducing a third, competing one - keeps a single, + narrow, already-established boundary between the CLI and the library's internal + model types. + tests: + - SpeechModelCatalogAdapter_GetAudioFormat_SynthesisRoleModel_ThrowsArgumentException + - SpeechModelCatalogAdapter_GetAudioFormat_NullDescriptor_ThrowsArgumentNullException + - SpeechModelCatalogAdapter_CreateRecognizerEngineAsync_SynthesisRoleModel_ThrowsArgumentException + - SpeechModelCatalogAdapter_CreateRecognizerEngineAsync_NullDescriptor_ThrowsArgumentNullException diff --git a/docs/reqstream/speech-cli/recognition-command-subsystem.yaml b/docs/reqstream/speech-cli/recognition-command-subsystem.yaml index db44384..0e1a8b2 100644 --- a/docs/reqstream/speech-cli/recognition-command-subsystem.yaml +++ b/docs/reqstream/speech-cli/recognition-command-subsystem.yaml @@ -5,10 +5,11 @@ # (recognize), decomposing the system-level dispatch requirements in docs/reqstream/speech-cli.yaml. # # This subsystem extends the ModelCommandsSubsystem seam (ICliModelCatalog) with two further -# members - GetAudioFormat and CreateRecognizer - rather than introduce a second, competing seam, -# since a resolved model must still be cast to the library's internal IRecognitionModel before -# either operation is possible, and that cast belongs in the same assembly-scoped adapter that -# already owns the original five members plus the two added by the SynthesisCommandSubsystem. +# members - GetAudioFormat and CreateRecognizerEngineAsync - rather than introduce a second, +# competing seam, since a resolved model must still be cast to the library's internal +# IRecognitionModel before either operation is possible, and that cast belongs in the same +# assembly-scoped adapter that already owns the original five members plus the two added by the +# SynthesisCommandSubsystem. sections: - title: SpeechCli Requirements @@ -42,7 +43,7 @@ sections: - RecognizeCommand_Run_UnknownModelId_ThrowsArgumentException - RecognizeCommand_Run_WrongRoleModel_ThrowsArgumentException - RecognizeCommand_Run_NotDownloadedModel_ThrowsArgumentExceptionWithDownloadHint - - RecognizeCommand_Run_Success_DisposesRecognizerOnce + - RecognizeCommand_Run_Success_DisposesEngineAndSessionOnce - SpeechCli_RecognizeCommandWithoutModel_Invoked_ReturnsCleanError - SpeechCli_RecognizeCommandWithUnknownModel_Invoked_ReturnsCleanError - SpeechCli_RecognizeCommandWithWrongRoleModel_Invoked_ReturnsCleanError @@ -56,7 +57,7 @@ sections: The tool shall parse repeatable `--stt-param key=value` flags and validate each value against the resolved recognition model's own declared parameter set, reusing the shared `ParameterBagParser` from the SynthesisCommandSubsystem, forwarding the - fully-resolved parameter bag to `CreateRecognizer`. + fully-resolved parameter bag to `CreateRecognizerEngineAsync`. tags: [subsystem] justification: | Recognition models declare tunable parameters exactly as synthesis models do (e.g. @@ -65,17 +66,17 @@ sections: `--stt-param` semantics across both `speak` and `recognize`. tests: - RecognizeCommand_ParseArguments_RepeatedParamFlags_AccumulatesInOrder - - RecognizeCommand_Run_ValidParam_ForwardsToCreateRecognizer + - RecognizeCommand_Run_ValidParam_ForwardsToCreateRecognizerEngine - RecognizeCommand_Run_InvalidParam_ThrowsArgumentException - id: SpeechCli-RecognitionCommands-FileInputEofDrivenStop title: >- The tool shall, in `--input` mode, drive the supplied `WavFileAudioCaptureDevice` to - completion via `ISpeechRecognizer.Start()`'s own synchronous, blocking call chain, - stopping the recognizer reentrantly from the device's own `EndOfFileReached` event - and returning control to the caller only once every result derived from the whole - file has already been raised and the recognizer has fully stopped, with no - additional wait or fixed sleep. + completion by explicitly draining the session after the capture device itself has + reported `EndOfFileReached`, awaiting `IRecognitionSession.StopAsync()` so that the + `GetResultsAsync()` stream completes and every result derived from the whole file + has already been consumed before returning control to the caller, with no additional + wait or fixed sleep. tags: [subsystem] justification: | A file-input recognition session is inherently finite and must terminate on its own @@ -83,16 +84,18 @@ sections: command needing an arbitrary settle delay that would be either too short (truncating results) or too long (wasting time) for an unknown file duration. tests: - - RecognizeCommand_Run_FileInput_StartsAndStopsRecognizerViaEndOfFile - - RecognizeCommand_Run_FileInput_PassesWavFileCaptureDeviceToCreateRecognizer + - RecognizeCommand_Run_FileInput_StartsAndStopsRecognizerViaExplicitDrain + - RecognizeCommand_Run_FileInput_PassesWavFileCaptureDeviceToCreateSession - id: SpeechCli-RecognitionCommands-SilenceTimeout title: >- The tool shall, in `--mic` mode with `--silence-timeout ` given, stop the - recognizer and end the session automatically once no recognition result (partial or - final) has arrived within the given idle window since the last result (or session - start), via `SilenceTimeoutRecognizerSession`, which shall re-arm its idle timer on - every result and never fire after `Dispose()`. + session and end its result stream automatically once no recognition result (partial + or final) has been yielded within the given idle window since the last result (or + session start), via `SilenceTimeoutRecognizerSession`, a stateless + `IAsyncEnumerable` decorator that re-arms a fresh idle timer + on every yielded result and raises `TimedOut` exactly once, after calling + `IRecognitionSession.StopAsync()`. tags: [subsystem] justification: | An unattended microphone session with no natural end-of-file signal needs an @@ -102,26 +105,22 @@ sections: - RecognizeCommand_ParseArguments_SilenceTimeoutFlag_ParsesSeconds - RecognizeCommand_ParseArguments_NonPositiveSilenceTimeout_ThrowsArgumentException - RecognizeCommand_ParseArguments_MalformedSilenceTimeout_ThrowsArgumentException - - SilenceTimeoutRecognizerSession_Construct_ArmsTimerWithGivenTimeout - - SilenceTimeoutRecognizerSession_PartialResultReceived_ResetsIdleTimer - - SilenceTimeoutRecognizerSession_FinalResultReceived_ResetsIdleTimer - - SilenceTimeoutRecognizerSession_IdleTimerFires_StopsRecognizerAndRaisesTimedOut - - SilenceTimeoutRecognizerSession_ResetThenFire_StopsOnlyOnActualFire - - SilenceTimeoutRecognizerSession_Dispose_UnsubscribesAndDisposesTimer - - SilenceTimeoutRecognizerSession_FireAfterDispose_DoesNotCallStopOrRaiseTimedOut - - SilenceTimeoutRecognizerSession_Construct_NullRecognizer_ThrowsArgumentNullException - - SilenceTimeoutRecognizerSession_Construct_NonPositiveTimeout_ThrowsArgumentOutOfRangeException - - SilenceTimeoutRecognizerSession_Construct_NullTimeProvider_UsesSystemTimeProvider + - SilenceTimeoutRecognizerSession_Construct_NullSession_ThrowsArgumentNullException + - SilenceTimeoutRecognizerSession_Construct_NonPositiveIdleTimeout_ThrowsArgumentOutOfRangeException + - SilenceTimeoutRecognizerSession_Construct_NullTimeProvider_DoesNotThrow + - SilenceTimeoutRecognizerSession_GetResultsAsync_TimeoutBeforeAnyResult_StopsSessionAndRaisesTimedOutOnce + - SilenceTimeoutRecognizerSession_GetResultsAsync_TimeoutAfterResult_StopsSessionAndRaisesTimedOutOnce + - SilenceTimeoutRecognizerSession_GetResultsAsync_CancellationRequested_PropagatesOperationCanceledException - id: SpeechCli-RecognitionCommands-StartTimeout title: >- - The tool shall, in `--mic` mode with `--silence-timeout` given, accept an optional - `--start-timeout ` flag configuring a separate idle window used only - before the first recognition result arrives (defaulting to `--silence-timeout`'s - value when omitted), via `SilenceTimeoutRecognizerSession`, which shall arm its idle - timer with the start-timeout value at construction and switch to the - silence-timeout value for every re-arm from the first result onward, rejecting a - non-positive or malformed `--start-timeout` with a clean error. + The tool shall, in `--mic` mode, accept an optional `--start-timeout ` + flag configuring a separate idle window used only before the first recognition + result is yielded, defaulting to a fixed 8-second value - independent of + `--silence-timeout` - when omitted, via `SilenceTimeoutRecognizerSession`, which + shall arm its idle timer with the start-timeout value for the first iteration and + switch to the idle-timeout value for every re-arm from the first yielded result + onward, rejecting a non-positive or malformed `--start-timeout` with a clean error. tags: [subsystem] justification: | A grace period for a user to begin speaking is typically much longer than the brief @@ -130,18 +129,22 @@ sections: up too quickly on a slow starter or one that lingers too long after the user has actually stopped talking. A separate, optional `--start-timeout` resolves this without changing `--silence-timeout`'s own existing behavior when the new flag is - not used. + not used, and a fixed default (rather than borrowing `--silence-timeout`'s value) + keeps the start-timeout's own meaning independent of whatever silence-timeout an + operator happens to choose. tests: - RecognizeCommand_ParseArguments_StartTimeoutFlag_ParsesSeconds - RecognizeCommand_ParseArguments_StartTimeoutOmitted_DefaultsToNull - RecognizeCommand_ParseArguments_NonPositiveStartTimeout_ThrowsArgumentException - RecognizeCommand_ParseArguments_MalformedStartTimeout_ThrowsArgumentException - - SilenceTimeoutRecognizerSession_Construct_StartTimeoutOmitted_ArmsTimerWithSilenceTimeout - - SilenceTimeoutRecognizerSession_Construct_StartTimeoutGiven_ArmsTimerWithStartTimeout - - SilenceTimeoutRecognizerSession_FirstResultReceived_ReArmsWithSilenceTimeoutNotStartTimeout - - SilenceTimeoutRecognizerSession_SecondResultReceived_StaysOnSilenceTimeout - - SilenceTimeoutRecognizerSession_IdleTimerFiresBeforeFirstResult_StopsRecognizerAndRaisesTimedOut + - RecognizeCommand_Run_StartTimeoutOmitted_ResolvesToFixedEightSecondDefault + - RecognizeCommand_Run_SilenceTimeoutOmitted_ResolvesToFixedFiveSecondDefault + - RecognizeCommand_Run_StartTimeoutGiven_ResolvesToGivenValue - SilenceTimeoutRecognizerSession_Construct_NonPositiveStartTimeout_ThrowsArgumentOutOfRangeException + - SilenceTimeoutRecognizerSession_GetResultsAsync_StartTimeoutOmitted_ArmsTimerWithIdleTimeout + - SilenceTimeoutRecognizerSession_GetResultsAsync_StartTimeoutGiven_ArmsTimerWithStartTimeout + - SilenceTimeoutRecognizerSession_GetResultsAsync_FirstResultReceived_ReArmsWithIdleTimeoutNotStartTimeout + - SilenceTimeoutRecognizerSession_GetResultsAsync_SecondResultReceived_StaysOnIdleTimeout - id: SpeechCli-RecognitionCommands-VerbosityAndOutput title: >- @@ -186,10 +189,15 @@ sections: - id: SpeechCli-RecognitionCommands-CatalogSeamExtension title: >- - The tool shall resolve a recognition-role model's audio format and construct its - `ISpeechRecognizer` exclusively through the `ICliModelCatalog` seam's - `GetAudioFormat`/`CreateRecognizer` members, which shall reject a descriptor whose - model is not a recognition model with a clean `ArgumentException`. + The `ICliModelCatalog` seam shall expose a `GetAudioFormat` member, for a + recognition-role model's required audio format, and a + `CreateRecognizerEngineAsync` member constructing its `ISpeechRecognizerEngine`, + both rejecting a descriptor whose model is not a recognition model with a clean + `ArgumentException`. The tool shall construct every recognition session exclusively + through `CreateRecognizerEngineAsync`; `GetAudioFormat` is not currently consumed by + any production call site, mirroring `GetPreferredAudioFormat`'s symmetric synthesis + member and reserved for the same future/defensive uses (e.g. a host validating audio + compatibility before composing a session). tags: [subsystem] justification: | `RecognizeCommand` cannot itself distinguish a recognition-capable model without @@ -197,13 +205,14 @@ sections: introducing a second, competing one) keeps a single, narrow, already-established boundary between the CLI and the library's internal model types, letting `RecognizeCommand`'s own logic remain fully testable against a fake catalog with no - dependency on `IRecognitionModel`. + dependency on `IRecognitionModel`. `GetAudioFormat` is retained unconsumed for + symmetry with the synthesis seam and to keep both roles' seams equally capable, + even though `RecognizeCommand` currently only needs `CreateRecognizerEngineAsync`. tests: - SpeechModelCatalogAdapter_GetAudioFormat_SynthesisRoleModel_ThrowsArgumentException - SpeechModelCatalogAdapter_GetAudioFormat_NullDescriptor_ThrowsArgumentNullException - - SpeechModelCatalogAdapter_CreateRecognizer_SynthesisRoleModel_ThrowsArgumentException - - SpeechModelCatalogAdapter_CreateRecognizer_NullDescriptor_ThrowsArgumentNullException - - SpeechModelCatalogAdapter_CreateRecognizer_NullCaptureDevice_ThrowsArgumentNullException + - SpeechModelCatalogAdapter_CreateRecognizerEngineAsync_SynthesisRoleModel_ThrowsArgumentException + - SpeechModelCatalogAdapter_CreateRecognizerEngineAsync_NullDescriptor_ThrowsArgumentNullException - id: SpeechCli-RecognitionCommands-NullGuards title: >- diff --git a/docs/reqstream/speech-cli/synthesis-command-subsystem.yaml b/docs/reqstream/speech-cli/synthesis-command-subsystem.yaml index 4bcacd3..2d3ef39 100644 --- a/docs/reqstream/speech-cli/synthesis-command-subsystem.yaml +++ b/docs/reqstream/speech-cli/synthesis-command-subsystem.yaml @@ -5,8 +5,8 @@ # (speak), decomposing the system-level dispatch requirements in docs/reqstream/speech-cli.yaml. # # This subsystem extends the ModelCommandsSubsystem seam (ICliModelCatalog) with two further -# members - GetPreferredAudioFormat and CreateSynthesizer - rather than introduce a second, -# competing seam, since a resolved model must still be cast to the library's internal +# members - GetPreferredAudioFormat and CreateSynthesizerEngineAsync - rather than introduce a +# second, competing seam, since a resolved model must still be cast to the library's internal # ISynthesisModel before either operation is possible, and that cast belongs in the same # assembly-scoped adapter that already owns the original five members. @@ -40,7 +40,7 @@ sections: - SpeakCommand_RunAsync_UnknownModelId_ThrowsArgumentException - SpeakCommand_RunAsync_WrongRoleModel_ThrowsArgumentException - SpeakCommand_RunAsync_NotDownloadedModel_ThrowsArgumentExceptionWithDownloadHint - - SpeakCommand_RunAsync_Success_DisposesSynthesizerOnce + - SpeakCommand_RunAsync_Success_DisposesEngineAndSessionOnce - SpeechCli_SpeakCommandWithoutModel_Invoked_ReturnsCleanError - SpeechCli_SpeakCommandWithUnknownModel_Invoked_ReturnsCleanError - SpeechCli_SpeakCommandWithWrongRoleModel_Invoked_ReturnsCleanError @@ -64,7 +64,7 @@ sections: starts catches an operator typo or out-of-range value immediately, rather than having it silently ignored (or fail deep inside the synthesis engine) minutes into a run. An unrecognized key is deliberately rejected here even though the library's - own `SpeechSynthesizerFactory.Create` silently ignores one, since an unrecognized + own `SpeechSynthesizerFactory.LoadAsync` silently ignores one, since an unrecognized CLI flag value is almost always an operator mistake that should fail loudly at the command line. tests: @@ -84,7 +84,7 @@ sections: - ParameterBagParser_Resolve_UnrecognizedKey_ThrowsArgumentException - ParameterBagParser_Resolve_NoRawValues_ReturnsEmptyBag - SpeakCommand_ParseArguments_RepeatedParamFlags_AccumulatesInOrder - - SpeakCommand_RunAsync_ValidParam_ForwardsToCreateSynthesizer + - SpeakCommand_RunAsync_ValidParam_ForwardsToCreateSynthesizerEngine - SpeakCommand_RunAsync_InvalidParam_ThrowsArgumentException - id: SpeechCli-SynthesisCommands-NoTags @@ -141,14 +141,14 @@ sections: `DownloadCommand` already established for its own cancellation handling. tests: - SpeakCommand_RunAsync_Canceled_ReportsErrorCleanly - - SpeakCommand_RunAsync_Success_DisposesSynthesizerOnce + - SpeakCommand_RunAsync_Success_DisposesEngineAndSessionOnce - id: SpeechCli-SynthesisCommands-CatalogSeamExtension title: >- The tool shall resolve a synthesis-role model's preferred audio format and construct - its `ISpeechSynthesizer` exclusively through the `ICliModelCatalog` seam's - `GetPreferredAudioFormat`/`CreateSynthesizer` members, which shall reject a - descriptor whose model is not a synthesis model with a clean `ArgumentException`. + its `ISynthesisSession` exclusively through the `ICliModelCatalog` seam's + `GetPreferredAudioFormat`/`CreateSynthesizerEngineAsync` members, which shall reject + a descriptor whose model is not a synthesis model with a clean `ArgumentException`. tags: [subsystem] justification: | `SpeakCommand` cannot itself distinguish a synthesis-capable model without internal @@ -160,9 +160,8 @@ sections: tests: - SpeechModelCatalogAdapter_GetPreferredAudioFormat_RecognitionRoleModel_ThrowsArgumentException - SpeechModelCatalogAdapter_GetPreferredAudioFormat_NullDescriptor_ThrowsArgumentNullException - - SpeechModelCatalogAdapter_CreateSynthesizer_RecognitionRoleModel_ThrowsArgumentException - - SpeechModelCatalogAdapter_CreateSynthesizer_NullDescriptor_ThrowsArgumentNullException - - SpeechModelCatalogAdapter_CreateSynthesizer_NullPlaybackDevice_ThrowsArgumentNullException + - SpeechModelCatalogAdapter_CreateSynthesizerEngineAsync_RecognitionRoleModel_ThrowsArgumentException + - SpeechModelCatalogAdapter_CreateSynthesizerEngineAsync_NullDescriptor_ThrowsArgumentNullException - id: SpeechCli-SynthesisCommands-NullGuards title: >- @@ -178,6 +177,5 @@ sections: - SpeakCommand_RunAsync_NullContext_ThrowsArgumentNullException - SpeakCommand_RunAsync_NullCatalog_ThrowsArgumentNullException - SpeakCommand_RunAsync_NullFactory_ThrowsArgumentNullException - - SpeakCommand_Run_NullContext_ThrowsArgumentNullException - ParameterBagParser_Resolve_NullRawValues_ThrowsArgumentNullException - ParameterBagParser_Resolve_NullDeclaredParameters_ThrowsArgumentNullException diff --git a/docs/reqstream/speech-demo.yaml b/docs/reqstream/speech-demo.yaml index 7a53b10..2edcd88 100644 --- a/docs/reqstream/speech-demo.yaml +++ b/docs/reqstream/speech-demo.yaml @@ -187,6 +187,7 @@ sections: - SpeechDemo-Synthesis-PlayLifecycle - SpeechDemo-Synthesis-StopControl - SpeechDemo-Synthesis-SessionSeam + - SpeechDemo-Synthesis-EngineSessionReuse - SpeechDemo-Synthesis-VoiceSelectionForwarding - SpeechDemo-Synthesis-AutoRefreshOnInstall - SpeechDemo-Synthesis-ResourceLifetime diff --git a/docs/reqstream/speech-demo/recognition-panel-subsystem.yaml b/docs/reqstream/speech-demo/recognition-panel-subsystem.yaml index 9855148..c942330 100644 --- a/docs/reqstream/speech-demo/recognition-panel-subsystem.yaml +++ b/docs/reqstream/speech-demo/recognition-panel-subsystem.yaml @@ -31,35 +31,42 @@ sections: title: >- The RecognitionPanelSubsystem shall start a streaming transcription session for the selected model through the selected capture device when Start is invoked, shall let - a user stop it deterministically at any time, shall reuse the composed recognizer - across Start/Stop cycles rather than recomposing it on every click, and shall - invalidate the cached recognizer - stopping an active session first, even while - actively listening, rather than leaving the panel stuck in a listening state it can - no longer stop - whenever the selected model changes, the selected capture device - changes, or a shared device refresh replaces the underlying capture device. + a user stop it deterministically at any time - including two concurrent Stop + invocations completing without racing to double-dispose - shall reuse the loaded + recognizer engine across Start/Stop cycles rather than reloading it on every click + while creating a fresh single-use session per Start, and shall invalidate the + cached engine or session - stopping an active session first, even while actively + listening, rather than leaving the panel stuck in a listening state it can no + longer stop - whenever the selected model changes (invalidating the engine), the + selected capture device changes (invalidating only the session), or a shared device + refresh replaces the underlying capture device (invalidating only the session). justification: | Streaming speech-to-text is only useful as a start/stop lifecycle a user directly - controls; this is the demonstrable proof that a host can drive the library's - recognizer contract from a UI thread with ordinary property-changed notification - alone. Recomposing the (expensive) recognizer on every click previously reloaded - the model from scratch each turn, contradicting the library's own "construct once, - reuse across Start/Stop" guidance; caching it instead makes repeated Start/Stop - cycles cheap. A model or device change can arrive mid-session (neither property is - guarded at the model level, only disabled in the view), so invalidation must stop - the session first or the panel would be left listening with no way to stop it. A - device refresh can also replace the underlying capture device object without - changing the selected name, so it must invalidate the cache too even though no - selection property actually changed. + controls; this is the demonstrable proof that a host can drive the library's async + engine/session contract from a UI thread with ordinary property-changed + notification alone. Reloading the (expensive) engine on every click previously + reloaded the model from scratch each turn, contradicting the library's own + "construct once, reuse across many sessions" guidance for + `ISpeechRecognizerEngine`; caching it instead makes repeated Start/Stop cycles + cheap, while a fresh `IRecognitionSession` is still created per Start since a + session is single-use. A model or device change can arrive mid-session (neither + property is guarded at the model level, only disabled in the view), so + invalidation must stop the session first or the panel would be left listening with + no way to stop it. A device refresh can also replace the underlying capture device + object without changing the selected name, so it must invalidate the cached session + too even though no selection property actually changed; the cached engine is not + bound to a device and is never invalidated on a device change alone. tests: - RecognitionPanelViewModel_Start_SuccessfulSession_EntersListeningState - RecognitionPanelViewModel_Stop_DuringListening_StopsWithoutDisposingSession - RecognitionPanelViewModel_Stop_NothingListening_IsSafeNoOp - - RecognitionPanelViewModel_StartStopStart_SameSelection_ReusesRecognizer - - RecognitionPanelViewModel_DeviceRefresh_InvalidatesCachedRecognizer - - RecognitionPanelViewModel_SelectedModelChanged_InvalidatesCachedRecognizer - - RecognitionPanelViewModel_SelectedModelChanged_WhileListening_StopsAndInvalidatesRecognizer - - RecognitionPanelViewModel_SelectedCaptureDeviceChanged_InvalidatesCachedRecognizer - - RecognitionPanelViewModel_SelectedCaptureDeviceChanged_WhileListening_StopsAndInvalidatesRecognizer + - RecognitionPanelViewModel_Stop_CalledTwiceConcurrently_BothCompleteWithoutThrowing + - RecognitionPanelViewModel_StartStopStart_SameSelection_ReusesEngine + - RecognitionPanelViewModel_DeviceRefresh_InvalidatesCachedSessionButNotEngine + - RecognitionPanelViewModel_SelectedModelChanged_InvalidatesCachedEngine + - RecognitionPanelViewModel_SelectedModelChanged_WhileListening_StopsAndInvalidatesEngine + - RecognitionPanelViewModel_SelectedCaptureDeviceChanged_InvalidatesSessionButNotEngine + - RecognitionPanelViewModel_SelectedCaptureDeviceChanged_WhileListening_StopsAndInvalidatesSession - id: SpeechDemo-Recognition-TranscriptSequencing title: >- @@ -79,19 +86,23 @@ sections: title: >- The RecognitionPanelSubsystem shall report an honest, explanatory outcome - rather than a crash or a silent no-op - when no recognition model is selected, no capture - device is available, the composed recognizer honestly reports itself unavailable, - or the recognizer faults when actually started, and shall release any recognizer it + device is available, the loaded engine honestly reports itself unavailable, the + session faults when actually started, or an active session itself transitions to + a faulted state after Start completes, and shall release any engine or session it created for a failed attempt. justification: | A user evaluating the library on a machine with no installed model, no microphone, or an engine fault beyond the library's control must see why Start did nothing, matching the library's own "nothing throws at composition" guarantee carried - through to the panel. + through to the panel. A session can also fault asynchronously, after a successful + Start, via `IRecognitionSession.StateChanged`; that must be reported with the same + honesty as a failure discovered synchronously during Start. tests: - RecognitionPanelViewModel_Start_NoModelSelected_ReportsErrorState - RecognitionPanelViewModel_Start_NoCaptureDevice_ReportsErrorState - RecognitionPanelViewModel_Start_RecognizerUnavailable_ReportsErrorStateAndDisposes - RecognitionPanelViewModel_Start_RecognizerStartThrows_ReportsErrorStateAndDisposes + - RecognitionPanelViewModel_StateChanged_SessionTransitionsToFaulted_ReportsErrorState - id: SpeechDemo-Recognition-HonestEmptyCatalogState title: >- @@ -106,24 +117,28 @@ sections: - id: SpeechDemo-Recognition-ResourceLifetime title: >- - The RecognitionPanelSubsystem shall release an active recognizer, unsubscribe from - its events, and unsubscribe from IModelCatalogService.ModelInstalled when the panel - is disposed, idempotently. + The RecognitionPanelSubsystem shall release an active session and the cached + recognizer engine, unsubscribe from the session's StateChanged event, and + unsubscribe from IModelCatalogService.ModelInstalled and the shared + device-selection panel's pre-refresh hook when the panel is disposed, idempotently. justification: | - A recognizer owns unmanaged inference resources and a live capture device; leaking - either when the panel's host shuts down would starve the next session or leave the - device open. The panel also now holds a ModelInstalled subscription for its own - lifetime, which must be released the same way or a disposed panel would keep - refreshing itself in response to later installs. Idempotent disposal keeps a - duplicate Dispose call - however it is triggered by the shell - from faulting. + A recognizer engine and its session own unmanaged inference resources and a live + capture device; leaking either when the panel's host shuts down would starve the + next session or leave the device open. The panel also holds a ModelInstalled + subscription and a registered pre-refresh hook for its own lifetime, which must be + released the same way or a disposed panel would keep refreshing itself in response + to later installs or interfere with a shared device refresh. Idempotent disposal + keeps a duplicate DisposeAsync call - however it is triggered by the shell - from + faulting. tests: - RecognitionPanelViewModel_Dispose_ReleasesActiveSessionWithoutThrowing - id: SpeechDemo-Recognition-SessionSeam title: >- - The RecognitionPanelSubsystem shall compose every speech recognizer it uses through - an injectable seam over the library's recognizer factory, and that seam shall - honestly report a model of the wrong role as unavailable rather than throwing. + The RecognitionPanelSubsystem shall compose every speech recognizer engine it uses + through an injectable seam over the library's recognizer factory, and that seam + shall honestly report a model of the wrong role as unavailable rather than + throwing. justification: | The library's recognizer factory requires an interface only the library's own assemblies can implement, so a demo-owned seam accepting the library's public model @@ -133,9 +148,8 @@ sections: tests: - RecognitionPanelViewModel_Constructor_NullDependency_ThrowsArgumentNullException - RecognizerSessionFactory_Constructor_NullStore_ThrowsArgumentNullException - - RecognizerSessionFactory_Create_NullModel_ThrowsArgumentNullException - - RecognizerSessionFactory_Create_NullDevice_ThrowsArgumentNullException - - RecognizerSessionFactory_Create_ModelNotRecognitionRole_ReturnsUnavailableRecognizer + - RecognizerSessionFactory_LoadAsync_NullModel_ThrowsArgumentNullException + - RecognizerSessionFactory_LoadAsync_ModelNotRecognitionRole_ReturnsUnavailableEngine - id: SpeechDemo-Recognition-AutoRefreshOnInstall title: >- @@ -173,14 +187,17 @@ sections: title: >- The RecognitionPanelSubsystem shall register a pre-refresh hook with the shared device-selection panel at construction that stops an active listening session - before a device refresh proceeds, and shall unregister it on disposal. + before a device refresh proceeds (a safe no-op while idle), and shall unregister it + on disposal. justification: | The recognition panel shares a device-selection panel (and its underlying device factory) with the synthesis panel; clicking "Refresh devices" while this panel is actively listening has a real chance of the backend refusing the refresh because - the capture device is in use. Stop() on this panel is fully synchronous down to - the capture device's own closure, so registering it as a pre-refresh hook lets a - shared refresh proceed deterministically instead of relying on the - AudioDeviceInUseException fallback. + the capture device is in use. Registering the panel's own stop-and-invalidate path + as a pre-refresh hook lets a shared refresh proceed deterministically instead of + relying on the AudioDeviceInUseException fallback, and the hook must also be + provably inert while idle so an unrelated Refresh click never tries to stop a + session that was never created. tests: - RecognitionPanelViewModel_PreRefreshHook_WhileListening_StopsSessionBeforeDeviceRefreshSucceeds + - RecognitionPanelViewModel_PreRefreshHook_WhileIdle_IsNoOpAndDeviceRefreshSucceeds diff --git a/docs/reqstream/speech-demo/synthesis-panel-subsystem/synthesis-panel-view-model.yaml b/docs/reqstream/speech-demo/synthesis-panel-subsystem/synthesis-panel-view-model.yaml index 7a8911d..4ac3c1a 100644 --- a/docs/reqstream/speech-demo/synthesis-panel-subsystem/synthesis-panel-view-model.yaml +++ b/docs/reqstream/speech-demo/synthesis-panel-subsystem/synthesis-panel-view-model.yaml @@ -56,12 +56,14 @@ sections: title: >- The SynthesisPanelSubsystem shall synthesize and speak the entered text through the selected playback device when Play is invoked, and shall report every - lifecycle transition from synthesizing through playing back to idle. + lifecycle transition from synthesizing through playing back to idle, driven by + the ISynthesisSession.StateChanged event rather than ad hoc assignment. justification: | A play button with no feedback leaves a user unsure whether anything is happening. Reporting each transition is the demonstrable proof that a host can - drive the library's synthesis contract from a UI thread with ordinary - property-changed notification alone. + drive the library's async synthesis engine/session contract from a UI thread + with ordinary property-changed notification alone, with the session itself - + not the ViewModel - as the single source of truth for lifecycle state. tests: - SynthesisPanelViewModel_Play_SuccessfulSession_TransitionsThroughLifecycleToIdle @@ -75,13 +77,34 @@ sections: tests: - SynthesisPanelViewModel_Stop_DuringPlayback_CancelsSessionAndReportsStopped + - id: SpeechDemo-Synthesis-EngineSessionReuse + title: >- + The SynthesisPanelSubsystem shall load a synthesizer engine at most once per + selected model/parameter-value combination and create a session at most once + per engine/playback-device combination, reusing both across many Play calls + instead of reloading the model and recreating the session on every Play. + justification: | + A prior design composed a brand-new synthesizer (and therefore reloaded the + model) on every single Play call, even when nothing about the selection had + changed - directly contradicting the library's own "construct once, reuse + across many sessions" guidance for ISpeechSynthesizerEngine and + ISynthesisSession, and making repeated Play clicks needlessly expensive. Lazily + detecting a reload reason on the next Play call (rather than proactively on + every property change) keeps the common case - replaying the same text with the + same voice - cheap, while still reloading the engine when the model or its + built parameter values genuinely change, and recreating only the session (never + the engine) when just the playback device changes. + tests: + - SynthesisPanelViewModel_Play_CalledTwiceWithUnchangedModelAndParameters_ReusesSameSessionWithoutReload + - SynthesisPanelViewModel_Play_ParameterValueChanged_ReloadsEngineAndRecreatesSession + - SynthesisPanelViewModel_Play_PlaybackDeviceChanged_RecreatesSessionButNotEngine + - id: SpeechDemo-Synthesis-HonestUnavailableStates title: >- The SynthesisPanelSubsystem shall report an honest, explanatory outcome - rather than a crash or a silent no-op - when no synthesis model is selected, no - playback device is available, or the composed synthesizer honestly reports - itself unavailable, and shall release any synthesizer it created for a failed - attempt. + playback device is available, or the loaded engine honestly reports itself + unavailable, and shall release any engine it loaded for a failed attempt. justification: | A user evaluating the library on a machine with no installed voice, no speaker, or an engine fault beyond the library's control must see why Play did @@ -122,37 +145,41 @@ sections: title: >- The SynthesisPanelSubsystem shall forward the embedded settings panel's current parameter value bag (Settings.BuildValueBag()) to - ISynthesizerSessionFactory.Create when composing a synthesizer, so a voice - selected through the panel's generic ChoiceParameter control genuinely changes - what is synthesized. + ISynthesizerSessionFactory.LoadAsync when loading a synthesizer engine, so a + voice selected through the panel's generic ChoiceParameter control genuinely + changes what is synthesized. justification: | Prior to this requirement, the demo's ChoiceParameter -> ComboBox UI rendering existed but was not wired to any real consumer: the selected value bag was built and exercised by unit tests but never reached a real synthesis or recognition call. Threading it through to the library's - SpeechSynthesizerFactory.Create (and, from there, to + SpeechSynthesizerFactory.LoadAsync (and, from there, to ISynthesisModel.ResolveSpeakerId) closes that gap for synthesis models such as SherpaOnnxKokoroEnglishSynthesisModel that own real per-voice knowledge. tests: - SynthesisPanelViewModel_Play_ModelDeclaresChoiceParameter_ForwardsValueBagToSessionFactory - - SynthesizerSessionFactory_Create_ModelNotSynthesisRoleWithParameterValues_ReturnsUnavailableSynthesizer + - SynthesizerSessionFactory_LoadAsync_ModelNotSynthesisRoleWithParameterValues_ReturnsUnavailableEngine - id: SpeechDemo-Synthesis-ResourceLifetime title: >- - The SynthesisPanelSubsystem shall implement IDisposable, unsubscribing from - IModelCatalogService.ModelInstalled and defensively disposing an active - synthesizer if one is still held, idempotently. + The SynthesisPanelSubsystem shall implement IAsyncDisposable, unsubscribing + from IModelCatalogService.ModelInstalled and the shared device-selection + panel's pre-refresh hook, stopping and awaiting an in-flight Play if any, and + releasing the cached session and the cached engine, idempotently. justification: | - SynthesisPanelViewModel did not previously implement IDisposable, since it held - no resource or subscription that outlived a single Play call; it now holds a - ModelInstalled subscription for its own lifetime, which must be released the - same way a disposed RecognitionPanelViewModel already releases its own - subscriptions, or a disposed panel would keep refreshing itself in response to - later installs. Idempotent disposal keeps a duplicate Dispose call - however it - is triggered by the shell - from faulting. + SynthesisPanelViewModel did not previously implement any disposable contract, + since it held no resource or subscription that outlived a single Play call; it + now caches an engine and a session across Play calls and holds a + ModelInstalled subscription and a registered pre-refresh hook for its own + lifetime, all of which must be released the same way a disposed + RecognitionPanelViewModel already releases its own resources, or a disposed + panel would keep refreshing itself in response to later installs, leak the + cached engine/session, or interfere with a shared device refresh. Idempotent + disposal keeps a duplicate DisposeAsync call - however it is triggered by the + shell - from faulting. tests: - - SynthesisPanelViewModel_Dispose_UnsubscribesFromModelInstalled_NoRefreshAfterDispose - - SynthesisPanelViewModel_Dispose_NoActiveSynthesizer_IsSafeAndIdempotent + - SynthesisPanelViewModel_DisposeAsync_UnsubscribesFromModelInstalled_NoRefreshAfterDispose + - SynthesisPanelViewModel_DisposeAsync_NoActiveSession_IsSafeAndIdempotent - id: SpeechDemo-Synthesis-ModelSwitchGuard title: >- @@ -174,15 +201,18 @@ sections: title: >- The SynthesisPanelViewModel shall register a pre-refresh hook with the shared device-selection panel at construction that stops an in-flight Play and awaits - its actual completion before a device refresh proceeds, and shall unregister it - on disposal. + its actual completion before a device refresh proceeds (a safe no-op while + idle), and shall unregister it on disposal. justification: | - Unlike the recognition panel's Stop(), this panel's Stop() only requests - cancellation; the playback device is not actually closed until the in-flight - PlayAsync's own finally block runs. Awaiting PlayCommand.ExecutionTask after - requesting Stop() is what makes it safe for a shared "Refresh devices" click to - proceed deterministically instead of relying on the AudioDeviceInUseException - fallback, since the hook does not return until the playback device is genuinely - closed. + Unlike the recognition panel's StopAsync(), this panel's session StopAsync() + only requests cancellation of the in-flight SpeakAsync; the playback device is + not actually closed until the in-flight PlayAsync task itself settles. Awaiting + PlayCommand.ExecutionTask after requesting Stop is what makes it safe for a + shared "Refresh devices" click to proceed deterministically instead of relying + on the AudioDeviceInUseException fallback, since the hook does not return until + the playback device is genuinely closed; the hook must also be provably inert + while idle so an unrelated Refresh click never awaits a Play task that was + never started. tests: - SynthesisPanelViewModel_PreRefreshHook_WhilePlaying_StopsAndAwaitsExecutionTaskBeforeDeviceRefreshSucceeds + - SynthesisPanelViewModel_PreRefreshHook_WhileIdle_IsNoOpAndDeviceRefreshSucceeds diff --git a/docs/reqstream/speech-demo/synthesis-panel-subsystem/synthesizer-session-factory.yaml b/docs/reqstream/speech-demo/synthesis-panel-subsystem/synthesizer-session-factory.yaml index 9886803..1c741d5 100644 --- a/docs/reqstream/speech-demo/synthesis-panel-subsystem/synthesizer-session-factory.yaml +++ b/docs/reqstream/speech-demo/synthesis-panel-subsystem/synthesizer-session-factory.yaml @@ -2,7 +2,7 @@ # Software Unit Requirements for the SpeechDemo SynthesizerSessionFactory # # These requirements describe the design-level behavior of the ISynthesizerSessionFactory / -# SynthesizerSessionFactory unit - the demo-owned synthesizer composition seam over the +# SynthesizerSessionFactory unit - the demo-owned synthesizer-engine composition seam over the # library's SpeechModelStore and SpeechSynthesizerFactory - decomposing the parent # SynthesisPanelSubsystem requirements. ISynthesizerSessionFactory and SynthesizerSessionFactory # are treated as one unit because the interface has no independently observable behavior of its @@ -17,9 +17,9 @@ sections: requirements: - id: SpeechDemo-Synthesis-SessionSeam title: >- - The SynthesisPanelSubsystem shall compose every speech synthesizer it uses - through an injectable seam over the library's synthesizer factory, and that - seam shall honestly report a model of the wrong role as unavailable rather + The SynthesisPanelSubsystem shall compose every speech synthesizer engine it + uses through an injectable seam over the library's synthesizer factory, and + that seam shall honestly report a model of the wrong role as unavailable rather than throwing. justification: | The library's synthesizer factory requires an interface only the library's own @@ -30,6 +30,5 @@ sections: tests: - SynthesisPanelViewModel_Constructor_NullDependency_ThrowsArgumentNullException - SynthesizerSessionFactory_Constructor_NullStore_ThrowsArgumentNullException - - SynthesizerSessionFactory_Create_NullModel_ThrowsArgumentNullException - - SynthesizerSessionFactory_Create_NullDevice_ThrowsArgumentNullException - - SynthesizerSessionFactory_Create_ModelNotSynthesisRole_ReturnsUnavailableSynthesizer + - SynthesizerSessionFactory_LoadAsync_NullModel_ThrowsArgumentNullException + - SynthesizerSessionFactory_LoadAsync_ModelNotSynthesisRole_ReturnsUnavailableEngine diff --git a/docs/reqstream/speech.yaml b/docs/reqstream/speech.yaml index f1c2dfe..6cff08d 100644 --- a/docs/reqstream/speech.yaml +++ b/docs/reqstream/speech.yaml @@ -172,6 +172,7 @@ sections: class ships yet, so end-to-end behavior is proven against a test model. children: - Speech-Recognition-MockableContract + - Speech-Recognition-EngineExclusivity - Speech-Recognition-Composition - Speech-Recognition-StreamingPipeline - Speech-Recognition-FaultContainment @@ -195,7 +196,7 @@ sections: intend as a tag (a typo, an unsupported bracket phrase, an unbalanced bracket) is never silently dropped or rejected. This requirement covers the closed vocabulary and the model-independent Layer 1 parser; Layer 2 per-model rendering and the public - ISpeechSynthesizer streaming/playback pipeline are described by + ISynthesisSession streaming/playback pipeline are described by Speech-Synthesis-StreamingSynthesizer below. children: - Speech-Synthesis-ClosedVocabulary @@ -225,13 +226,15 @@ sections: children: - Speech-Synthesis-Layer2Rendering - Speech-Synthesis-MockableContract + - Speech-Synthesis-EngineExclusivity - Speech-Synthesis-Composition - Speech-Synthesis-VoiceSelection - Speech-Synthesis-ChunkedPipeline + - Speech-Synthesis-SessionOverlapAndLifecycle - Speech-Synthesis-FaultContainment - Speech-Synthesis-UnavailableFallback - - Speech-Models-SpeechModelContract-SynthesisEngineDeclaration + - Speech-Models-SpeechModelContract-SynthesisEngineConfigDeclaration + - Speech-Models-SpeechModelContract-SynthesisCapabilityProfileDeclaration - Speech-Audio-PlaybackFormatDisclosure tests: - - SynthesizeStreamAsync_PlainText_YieldsAudioSegment - - PlayStreamAsync_OrderedSegments_StartsWritesInOrderAndStops + - Speech_SystemIntegration_StreamingSynthesis_TextProducesPlayedAudio diff --git a/docs/reqstream/speech/audio-subsystem/wav-file-audio-capture-device.yaml b/docs/reqstream/speech/audio-subsystem/wav-file-audio-capture-device.yaml index 326e888..dfe7ac5 100644 --- a/docs/reqstream/speech/audio-subsystem/wav-file-audio-capture-device.yaml +++ b/docs/reqstream/speech/audio-subsystem/wav-file-audio-capture-device.yaml @@ -42,8 +42,8 @@ sections: delivery first. justification: | The public IAudioCaptureDevice contract has no "end of stream" concept because - a real live-mic device has no equivalent "end"; a future recognize --input CLI - command needs an honest, additive signal to know when to stop waiting for + a real live-mic device has no equivalent "end"; the shipped recognize --input + CLI command needs an honest, additive signal to know when to stop waiting for further recognition results. tests: - WavFileAudioCaptureDevice_Start_FixtureFile_RaisesEndOfFileReachedExactlyOnceAfterLastFrame diff --git a/docs/reqstream/speech/model-management-subsystem.yaml b/docs/reqstream/speech/model-management-subsystem.yaml index 6b4a6bc..ee757db 100644 --- a/docs/reqstream/speech/model-management-subsystem.yaml +++ b/docs/reqstream/speech/model-management-subsystem.yaml @@ -230,7 +230,8 @@ sections: - Speech-Models-SpeechModelContract-RecognitionNormalizeTextHook - Speech-Models-SpeechModelContract-RecognitionEngineDeclaration - Speech-Models-SpeechModelContract-SynthesisPreferredAudioFormat - - Speech-Models-SpeechModelContract-SynthesisEngineDeclaration + - Speech-Models-SpeechModelContract-SynthesisEngineConfigDeclaration + - Speech-Models-SpeechModelContract-SynthesisCapabilityProfileDeclaration - Speech-Models-SpeechModelContract-LicenseDeclaration - Speech-Models-SpeechModelDescriptorEnums-FixedValueSets tests: @@ -407,11 +408,13 @@ sections: - Speech-Models-SherpaOnnxVitsLibriTtsEnglishSynthesisModel-PreferredAudioFormat - Speech-Models-SherpaOnnxVitsLibriTtsEnglishSynthesisModel-EngineConfigArgumentValidation - Speech-Models-SherpaOnnxVitsLibriTtsEnglishSynthesisModel-InstallAsync + - Speech-Models-SherpaOnnxVitsLibriTtsEnglishSynthesisModel-InstallAsync-ArgumentValidation - Speech-Models-SherpaOnnxVitsLibriTtsEnglishSynthesisModel-ResolveSpeakerId tests: - SherpaOnnxVitsLibriTtsEnglishSynthesisModel_Identity_DeclaresExpectedValues - SherpaOnnxVitsLibriTtsEnglishSynthesisModel_CreateEngineConfig_ResolvesVitsFilesAndConfig - SherpaOnnxVitsLibriTtsEnglishSynthesisModel_InstallAsync_SyntheticArchive_ExtractsFilesAndRemovesArchive + - SherpaOnnxVitsLibriTtsEnglishSynthesisModel_InstallAsync_NullOrEmptyDirectory_ThrowsArgumentException - SherpaOnnxVitsLibriTtsEnglishSynthesisModel_ResolveSpeakerId_ValidValue_ReturnsThatValue - id: Speech-Models-RealKokoroSynthesisModelWithVoiceSelection @@ -432,7 +435,7 @@ sections: own generation script, byte arithmetic against the real voices.bin, and a live model load reporting NumSpeakers = 11, and by extending ISynthesisModel itself with a non-breaking ResolveSpeakerId default-hook member that - SherpaOnnxSpeechSynthesizer.GenerateSegment now calls instead of hard-coding + SherpaOnnxSynthesisSession.GenerateSegmentAsync now calls instead of hard-coding speakerId: 0. children: - Speech-Models-SherpaOnnxKokoroEnglishSynthesisModel-Identity diff --git a/docs/reqstream/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.yaml b/docs/reqstream/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.yaml index 43c1f38..8e2c209 100644 --- a/docs/reqstream/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.yaml +++ b/docs/reqstream/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.yaml @@ -117,7 +117,7 @@ sections: justification: | ISynthesisModel.ResolveSpeakerId is this model's own owned knowledge - the confirmed id2speaker ordering (index = speaker id) that lets - SherpaOnnxSpeechSynthesizer.GenerateSegment turn a demo-selected or + SherpaOnnxSynthesisSession.GenerateSegmentAsync turn a demo-selected or library-consumer-selected voice name into the real speaker id sherpa-onnx's native Generate call needs, replacing the previous hard-coded speakerId: 0. Never throwing on a missing/unrecognized selection keeps synthesis available diff --git a/docs/reqstream/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.yaml b/docs/reqstream/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.yaml index c2ad330..ce12259 100644 --- a/docs/reqstream/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.yaml +++ b/docs/reqstream/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.yaml @@ -108,6 +108,17 @@ sections: tests: - SherpaOnnxVitsLibriTtsEnglishSynthesisModel_InstallAsync_SyntheticArchive_ExtractsFilesAndRemovesArchive + - id: Speech-Models-SherpaOnnxVitsLibriTtsEnglishSynthesisModel-InstallAsync-ArgumentValidation + title: >- + The SherpaOnnxVitsLibriTtsEnglishSynthesisModel class's InstallAsync shall throw + ArgumentException for a null or empty staged-files directory. + justification: | + A programming error (null or empty staged-files directory) should surface + immediately with a clear, cheap-to-diagnose exception rather than an opaque + path-resolution failure deeper in archive extraction. + tests: + - SherpaOnnxVitsLibriTtsEnglishSynthesisModel_InstallAsync_NullOrEmptyDirectory_ThrowsArgumentException + - id: Speech-Models-SherpaOnnxVitsLibriTtsEnglishSynthesisModel-ResolveSpeakerId title: >- The SherpaOnnxVitsLibriTtsEnglishSynthesisModel class shall resolve a numeric @@ -118,7 +129,7 @@ sections: justification: | ISynthesisModel.ResolveSpeakerId is this model's own owned knowledge - reading its numeric speaker parameter directly out of the bag lets - SherpaOnnxSpeechSynthesizer.GenerateSegment turn a demo-selected or + SherpaOnnxSynthesisSession.GenerateSegmentAsync turn a demo-selected or library-consumer-selected numeric speaker index into the real speaker id sherpa-onnx's native Generate call needs, replacing the previous hard-coded speakerId: 0 and closing this model's previously known, accepted, out-of-scope diff --git a/docs/reqstream/speech/model-management-subsystem/speech-model-contract.yaml b/docs/reqstream/speech/model-management-subsystem/speech-model-contract.yaml index a3ed8a5..8c2a559 100644 --- a/docs/reqstream/speech/model-management-subsystem/speech-model-contract.yaml +++ b/docs/reqstream/speech/model-management-subsystem/speech-model-contract.yaml @@ -92,7 +92,7 @@ sections: justification: | Some sherpa-onnx streaming transducer models (empirically confirmed for SherpaOnnxZipformerEnRecognitionModel) produce UPPERCASE, unpunctuated raw - output; SherpaOnnxSpeechRecognizer needs a per-model hook it can call on every + output; SherpaOnnxRecognitionSession needs a per-model hook it can call on every decoded result's text without knowing which model produced it, mirroring the existing synthesis-side NormalizeText hook but kept distinct because the two directions correct fundamentally different things (input text vs. recognized @@ -131,25 +131,29 @@ sections: tests: - ISynthesisModel_PreferredAudioFormat_DeclaredByModel_IsExposed - - id: Speech-Models-SpeechModelContract-SynthesisEngineDeclaration + - id: Speech-Models-SpeechModelContract-SynthesisEngineConfigDeclaration title: >- A synthesis model shall build its offline text-to-speech engine's configuration - from the directory its verified files were installed into, and shall expose a - Layer 2 tag-rendering strategy that defaults to a generically-correct - implementation driven purely by the model's own declared audio-tag support and - parameters. + from the directory its verified files were installed into. justification: | The design makes each model's backing class responsible for the engine configuration for its own model architecture, mirroring the recognition-side requirement above, so that adding a synthesis model is a self-contained, reviewable unit of work that never requires changing the synthesis subsystem. + tests: + - ISynthesisModel_CreateEngineConfig_InstalledDirectory_ResolvesPaths + - ISynthesisModel_CreateEngineConfig_EmptyDirectory_ThrowsArgumentException + + - id: Speech-Models-SpeechModelContract-SynthesisCapabilityProfileDeclaration + title: >- + A synthesis model shall expose a Layer 2 tag-rendering strategy that defaults + to a generically-correct implementation driven purely by the model's own + declared audio-tag support and parameters. + justification: | The default-hook CapabilityProfile member means a model needs zero code to get generically correct Natural Language Audio Tag rendering, and may override it only for bespoke, non-generic behavior. tests: - - SpeechSynthesizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer - - ISynthesisModel_CreateEngineConfig_InstalledDirectory_ResolvesPaths - - ISynthesisModel_CreateEngineConfig_EmptyDirectory_ThrowsArgumentException - ISynthesisModel_CapabilityProfile_DefaultImplementation_ReturnsDefaultProfile - id: Speech-Models-SpeechModelContract-LicenseDeclaration diff --git a/docs/reqstream/speech/recognition-subsystem.yaml b/docs/reqstream/speech/recognition-subsystem.yaml index aa998e2..2b8c676 100644 --- a/docs/reqstream/speech/recognition-subsystem.yaml +++ b/docs/reqstream/speech/recognition-subsystem.yaml @@ -11,117 +11,159 @@ sections: requirements: - id: Speech-Recognition-MockableContract title: >- - The RecognitionSubsystem shall expose a mockable streaming speech-to-text contract + The RecognitionSubsystem shall expose a mockable, async streaming speech-to-text + contract split into a Layer 3 engine (one loaded model, reusable across many + sessions) and a Layer 5 session (bound to one capture device for its entire life), that reports availability, starts and stops recognition, and delivers provisional - and final results, without exposing any inference-engine type. + and final results through an async-enumerable surface, without exposing any + inference-backend type. justification: | This library's "engine backend stays swappable at the public API surface" decision requires that a future non-sherpa-onnx backend can replace the current one without a breaking API change, and hosts must be able to test their transcript handling against the contract with no model, microphone, or native runtime present. + Splitting the mockable surface into an engine and a session lets a host amortize the + expensive model-load step across many device-bound sessions, and lets the + session-level state machine and result delivery be verified independently of engine + composition. children: - - Speech-Recognition-ISpeechRecognizer-AvailabilityFlag - - Speech-Recognition-ISpeechRecognizer-StartStopResultReceived - - Speech-Recognition-ISpeechRecognizer-ResultShape - - Speech-Recognition-ISpeechRecognizer-HotReuseAcrossTurns + - Speech-Recognition-ISpeechRecognizerEngine-AvailabilityFlag + - Speech-Recognition-ISpeechRecognizerEngine-CreateSession + - Speech-Recognition-ISpeechRecognizerEngine-Disposal + - Speech-Recognition-IRecognitionSession-AvailabilityFlag + - Speech-Recognition-IRecognitionSession-StateMachine + - Speech-Recognition-IRecognitionSession-StartStopLifecycle + - Speech-Recognition-IRecognitionSession-ResultDelivery + - Speech-Recognition-IRecognitionSession-Backpressure tests: - - UnavailableSpeechRecognizer_IsAvailable_Read_ReturnsFalse - - SherpaOnnxSpeechRecognizer_Start_Always_SubscribesAndStartsCaptureDevice + - UnavailableSpeechRecognizerEngine_IsAvailable_Read_ReturnsFalse + - SherpaOnnxRecognitionSession_StartAsync_FromCreated_TransitionsToRunning + + - id: Speech-Recognition-EngineExclusivity + title: >- + The RecognitionSubsystem shall permit at most one live session per loaded engine at + a time, failing fast with a documented exception rather than queuing or waiting when + a caller attempts to create a second session before the prior one has fully + released the engine's exclusivity lease. + justification: | + The native recognition backend an engine owns is not safe for two sessions to drive + concurrently; failing fast surfaces a caller's sequencing bug immediately instead of + hiding it behind an unpredictable queued delay. + children: + - Speech-Recognition-EngineExclusivity-SingleActiveSession + - Speech-Recognition-EngineExclusivity-FailFastNotQueued + - Speech-Recognition-EngineExclusivity-DisposalOwnsSession + - Speech-Recognition-RecognitionEngineBusyException-Thrown + - Speech-Recognition-RecognitionEngineBusyException-Constructors + tests: + - SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_SessionAlreadyLeased_ThrowsRecognitionEngineBusyException + - SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_AfterPriorSessionFullyDisposed_ReturnsNewSession - id: Speech-Recognition-Composition title: >- - The RecognitionSubsystem shall compose a speech recognizer for an installed - recognition model and an audio capture device without ever throwing for an ordinary - machine state, while validating any supplied parameterValues up front: silently - ignoring an unrecognized parameter id (an ordinary cross-model-compatibility gap) - but throwing ArgumentException for a recognized parameter's invalid value (a caller - bug, not an ordinary machine state). + The RecognitionSubsystem shall compose, via an async LoadAsync, a speech recognizer + engine for an installed recognition model without ever faulting the returned task + for an ordinary machine state, while validating any supplied parameterValues up + front: silently ignoring an unrecognized parameter id (an ordinary + cross-model-compatibility gap) but faulting with ArgumentException for a recognized + parameter's invalid value (a caller bug, not an ordinary machine state). justification: | Composition happens at application start-up, where a model that has not been - downloaded yet, a machine with no microphone, and a machine without the speech - engine's native runtime are all ordinary states rather than programming errors, per - this library's "nothing throws at composition" decision. A caller explicitly - targeting a parameter it knows this model declares, with a value invalid for it, is - a distinct case - a programming error at the call site - so it must be surfaced - synchronously and loudly rather than folded into the same never-throw contract as - genuine machine-state unavailability. + downloaded yet and a machine without the speech engine's native runtime are both + ordinary states rather than programming errors, per this library's "nothing throws + at composition" decision. A caller explicitly targeting a parameter it knows this + model declares, with a value invalid for it, is a distinct case - a programming + error at the call site - so it must be surfaced through the returned task's fault + rather than folded into the same never-fault contract as genuine machine-state + unavailability. children: - - Speech-Recognition-SpeechRecognizerFactory-RealRecognizer + - Speech-Recognition-SpeechRecognizerFactory-RealEngine - Speech-Recognition-SpeechRecognizerFactory-HonestFallbacks - Speech-Recognition-SpeechRecognizerFactory-EngineLoadFailureFallback - Speech-Recognition-SpeechRecognizerFactory-ArgumentValidation - Speech-Recognition-SpeechRecognizerFactory-ParameterValuesForwarding - Speech-Recognition-SpeechRecognizerFactory-UnrecognizedParameterIdIgnored - Speech-Recognition-SpeechRecognizerFactory-InvalidRecognizedParameterValueThrows - - Speech-Recognition-SpeechRecognizerFactory-ConcurrentPrewarmingSafety - - Speech-Recognition-RecognitionEngine-MockableSeam - - Speech-Recognition-RecognitionEngine-ModelOwnedConfiguration - - Speech-Recognition-RecognitionEngine-SessionEndResetDiscardsBufferedAudio - - Speech-Recognition-RecognitionEngine-TryFlushRecoversTrailingAudio - - Speech-Recognition-RecognitionEngine-PostEndpointWarmupReplay - - Speech-Recognition-RecognitionEngine-RealSpeechAccuracy + - Speech-Recognition-RecognitionBackend-MockableSeam + - Speech-Recognition-RecognitionBackend-ModelOwnedConfiguration + - Speech-Recognition-RecognitionBackend-SessionEndResetDiscardsBufferedAudio + - Speech-Recognition-RecognitionBackend-TryFlushRecoversTrailingAudio + - Speech-Recognition-RecognitionBackend-PostEndpointWarmupReplay + - Speech-Recognition-RecognitionBackend-RealSpeechAccuracy tests: - - SpeechRecognizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealRecognizer - - SpeechRecognizerFactory_Create_ModelNotInstalled_ReturnsUnavailableRecognizer - - SpeechRecognizerFactory_Create_CaptureDeviceUnavailable_ReturnsUnavailableRecognizer - - SpeechRecognizerFactory_Create_UnrecognizedParameterId_ComposesAndReportsInfo - - SpeechRecognizerFactory_Create_RecognizedNumericParameterOutOfRange_Throws + - SpeechRecognizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine + - SpeechRecognizerFactory_LoadAsync_ModelNotInstalled_ReturnsUnavailableEngine + - SpeechRecognizerFactory_LoadAsync_UnrecognizedParameterId_ComposesAndReportsInfo + - SpeechRecognizerFactory_LoadAsync_RecognizedNumericParameterOutOfRange_FaultsWithArgumentException - id: Speech-Recognition-StreamingPipeline title: >- The RecognitionSubsystem shall convert captured audio into the format the selected - model requires and deliver the resulting recognition results while running. + model requires and deliver the resulting recognition results through + GetResultsAsync while the session runs. justification: | A capture device resolves whatever channel count and sample rate its hardware defaults to, while a recognition model requires a single fixed mono rate. Without that conversion the library could only recognize speech on machines whose microphone happened to match the model, which is not an acceptable cross-platform promise. children: - - Speech-Recognition-SherpaOnnxSpeechRecognizer-FormatConversion - - Speech-Recognition-SherpaOnnxSpeechRecognizer-ResultDelivery - - Speech-Recognition-SherpaOnnxSpeechRecognizer-TextNormalization - - Speech-Recognition-SherpaOnnxSpeechRecognizer-LifecycleDrain + - Speech-Recognition-SherpaOnnxRecognitionSession-FormatConversion + - Speech-Recognition-SherpaOnnxRecognitionSession-TextNormalization + - Speech-Recognition-RecognitionResultBuffer-TwoTierBackpressure - Speech-Recognition-AudioFrameResampler-Downmix - Speech-Recognition-AudioFrameResampler-RateConversion - Speech-Recognition-AudioFrameResampler-BoundaryHandling tests: - - SherpaOnnxSpeechRecognizer_FrameCaptured_StereoAtHigherRate_FeedsResampledMonoToEngine - - SherpaOnnxSpeechRecognizer_FrameCaptured_EngineDecodesResults_RaisesResultReceivedInOrder + - SherpaOnnxRecognitionSession_FrameCaptured_StereoAtModelRate_FeedsDownmixedMonoToBackend + - SherpaOnnxRecognitionSession_FrameCaptured_FinalResult_AppliesModelNormalizeTextWithIsFinalTrue - id: Speech-Recognition-FaultContainment title: >- The RecognitionSubsystem shall contain and report faults raised while recognizing - captured audio, rather than propagating them into the audio capture path or ending - the recognition session. + captured audio or driving the native recognition backend, surfacing a faulted + session through GetResultsAsync as a documented exception rather than propagating + the underlying fault into the audio capture path or silently ending the session. justification: | Captured audio is delivered on a high-priority audio callback thread where an - escaping exception tears down the audio stream. A single bad audio block, a single - engine fault, or a single misbehaving host handler must never silence the - microphone or lose the rest of the session. + escaping exception tears down the audio stream, and a stuck native call must never + hang an awaiting caller indefinitely. A single bad audio block, a single backend + fault, a lost capture device, or a single misbehaving host handler must never + silence the microphone, hang the session, or lose the rest of the session's + recognized text. children: - - Speech-Recognition-SherpaOnnxSpeechRecognizer-FaultContainment - - Speech-Recognition-SherpaOnnxSpeechRecognizer-ResetFailureContainment - - Speech-Recognition-SherpaOnnxSpeechRecognizer-FlushFailureContainment - - Speech-Recognition-SherpaOnnxSpeechRecognizer-CaptureStartFailure + - Speech-Recognition-SherpaOnnxRecognitionSession-DeviceLostFault + - Speech-Recognition-SherpaOnnxRecognitionSession-NativeCallAbandonIntegration + - Speech-Recognition-RecognitionSessionFaultedException-Thrown + - Speech-Recognition-DedicatedWorker-LongRunningDedicatedThread + - Speech-Recognition-DedicatedWorker-CooperativeCancellation + - Speech-Recognition-DedicatedWorker-AbandonAfterTimeout tests: - - SherpaOnnxSpeechRecognizer_ResultReceived_HandlerThrows_ReportsFaultAndDoesNotRethrow - - SherpaOnnxSpeechRecognizer_FrameCaptured_EngineThrows_ReportsFaultAndKeepsRunning + - SherpaOnnxRecognitionSession_DeviceLostMidSession_TransitionsToFaulted + - SherpaOnnxRecognitionSession_NativeCallExceedsAbandonTimeout_TaskCompletesAndDiagnosticsReportsWarning + - SherpaOnnxRecognitionSession_GetResultsAsync_SessionFaulted_ThrowsRecognitionSessionFaultedException - id: Speech-Recognition-UnavailableFallback title: >- - The RecognitionSubsystem shall provide an honest unavailable recognizer that reports - unavailability rather than throwing, and throws a documented exception only when an - operational member is actually invoked. + The RecognitionSubsystem shall provide an honest unavailable engine and an honest + unavailable session that report unavailability rather than faulting or throwing at + composition, and throw a documented exception only when an operational member is + actually invoked. justification: | Per this library's "nothing throws at composition" decision, obtaining and holding - a fallback recognizer must never throw; only misuse of an already unavailable - recognizer is a programming error that should surface as an exception. + a fallback engine or session must never throw or fault; only misuse of an already + unavailable engine or session is a programming error that should surface as an + exception. children: - - Speech-Recognition-UnavailableSpeechRecognizer-ReportsUnavailable - - Speech-Recognition-UnavailableSpeechRecognizer-OperationsThrow - - Speech-Recognition-UnavailableSpeechRecognizer-SafeNoOps + - Speech-Recognition-UnavailableSpeechRecognizerEngine-ReportsUnavailable + - Speech-Recognition-UnavailableSpeechRecognizerEngine-CreateSessionReturnsFallback + - Speech-Recognition-UnavailableSpeechRecognizerEngine-DisposalIdempotent + - Speech-Recognition-UnavailableRecognitionSession-ReportsUnavailable + - Speech-Recognition-UnavailableRecognitionSession-OperationsThrow + - Speech-Recognition-UnavailableRecognitionSession-SafeNoOps - Speech-Recognition-SpeechRecognizerUnavailableException-Thrown + - Speech-Recognition-RecognitionBackend-UnavailableDeviceFallback tests: - - UnavailableSpeechRecognizer_IsAvailable_Read_ReturnsFalse - - UnavailableSpeechRecognizer_Start_Always_ThrowsSpeechRecognizerUnavailableException - - UnavailableSpeechRecognizer_Stop_Always_ThrowsSpeechRecognizerUnavailableException + - UnavailableSpeechRecognizerEngine_IsAvailable_Read_ReturnsFalse + - UnavailableSpeechRecognizerEngine_CreateSessionAsync_Always_ReturnsUnavailableSession + - UnavailableRecognitionSession_StartAsync_Always_ThrowsSpeechRecognizerUnavailableException + - UnavailableRecognitionSession_StopAsync_Always_IsSafeNoOp diff --git a/docs/reqstream/speech/recognition-subsystem/dedicated-worker.yaml b/docs/reqstream/speech/recognition-subsystem/dedicated-worker.yaml new file mode 100644 index 0000000..53d69f7 --- /dev/null +++ b/docs/reqstream/speech/recognition-subsystem/dedicated-worker.yaml @@ -0,0 +1,53 @@ +--- +# Software Unit Requirements for DedicatedWorker +# +# These requirements describe the design-level behavior of the internal DedicatedWorker unit, +# decomposing the parent RecognitionSubsystem requirements. + +sections: + - title: Speech Requirements + sections: + - title: RecognitionSubsystem Requirements + sections: + - title: DedicatedWorker Requirements + requirements: + - id: Speech-Recognition-DedicatedWorker-LongRunningDedicatedThread + title: >- + The DedicatedWorker class shall run each native call on a dedicated, + long-running task rather than a thread-pool task, so a stuck native call can + never starve the shared thread pool. + justification: | + Every call into the sherpa-onnx native runtime (decode, reset, flush) is a + potentially blocking, non-cooperative native call; running it on an ordinary + pooled task risks the pool's limited thread budget being consumed by stuck + native calls, starving unrelated work across the whole process. + tests: + - DedicatedWorker_Run_UsesLongRunningTaskCreationOption + + - id: Speech-Recognition-DedicatedWorker-CooperativeCancellation + title: >- + The DedicatedWorker class shall complete promptly when a cooperative delegate + observes and honors cancellation. + justification: | + A delegate that checks its own cancellation token and returns in response to it + is the common, well-behaved case, and must not pay the cost of the abandon + timeout merely because it was asked to cancel. + tests: + - DedicatedWorker_Run_CooperativeCancellation_CompletesPromptly + + - id: Speech-Recognition-DedicatedWorker-AbandonAfterTimeout + title: >- + The DedicatedWorker class shall abandon a non-cooperative delegate that ignores + cancellation after a configurable timeout, letting its caller's await complete + while the native call itself keeps running orphaned in the background, and + shall report a Warning diagnostic naming the abandonment. + justification: | + A native call can block indefinitely despite a cancellation request being + honored at the managed boundary; without a bounded abandon timeout, a single + stuck native call would hang its caller (for example session teardown) + indefinitely. Abandoning the await, rather than the underlying native call + itself (which cannot be forcibly terminated), is the only way to bound the + caller's wait, so the abandonment must be surfaced as a diagnostic rather than + silently swallowed. + tests: + - DedicatedWorker_Run_NonCooperativeDelegate_AbandonsAfterTimeoutAndReportsDiagnostics diff --git a/docs/reqstream/speech/recognition-subsystem/i-recognition-session.yaml b/docs/reqstream/speech/recognition-subsystem/i-recognition-session.yaml new file mode 100644 index 0000000..a0376b1 --- /dev/null +++ b/docs/reqstream/speech/recognition-subsystem/i-recognition-session.yaml @@ -0,0 +1,118 @@ +--- +# Software Unit Requirements for IRecognitionSession +# +# Groups requirements for IRecognitionSession together with its adjacent value/event types +# (RecognitionSessionState, SessionStateChangedEventArgs, RecognitionSessionFaultedException, +# SpeechRecognitionResult, SpeechRecognitionEvent), decomposing the parent RecognitionSubsystem +# requirements. These supporting types carry no behavior independent of the session contract they +# describe, per design-documentation.md's allowance for small supporting types documented inline. + +sections: + - title: Speech Requirements + sections: + - title: RecognitionSubsystem Requirements + sections: + - title: IRecognitionSession Requirements + requirements: + - id: Speech-Recognition-IRecognitionSession-AvailabilityFlag + title: >- + The IRecognitionSession interface shall expose an IsAvailable flag that never + throws to read. + justification: | + Callers must be able to check availability before invoking an operational + member, without risk of the check itself failing. + tests: + - UnavailableRecognitionSession_IsAvailable_Read_ReturnsFalse + + - id: Speech-Recognition-IRecognitionSession-StateMachine + title: >- + The IRecognitionSession interface shall expose a State property and a + StateChanged event that together report every RecognitionSessionState + transition, in order, as the session makes only the documented forward-only + transitions (Created, Starting, Running, Stopping, Stopped, Disposing, + Disposed, with Faulted reachable from Starting, Running, or Stopping). + justification: | + A host rendering session status, and a test asserting on session lifecycle, + both need a single authoritative, orderable signal of what the session is doing + rather than inferring it from the side effects of Start/Stop/GetResults calls. + tests: + - SherpaOnnxRecognitionSession_StartAsync_FromCreated_TransitionsToRunning + - SherpaOnnxRecognitionSession_StartAsync_FromStopped_ThrowsInvalidOperationException + - SherpaOnnxRecognitionSession_StateChanged_EmitsEveryTransitionInOrder + - SherpaOnnxRecognitionSession_DeviceLostMidSession_TransitionsToFaulted + + - id: Speech-Recognition-IRecognitionSession-StartStopLifecycle + title: >- + The IRecognitionSession interface shall define async StartAsync/StopAsync + lifecycle methods and an async disposal that releases the resources a running + session holds. StopAsync shall finalize and deliver any trailing audio + accepted but not yet decoded before it returns, rather than carrying it over + into (or losing it between) sessions, and shall be safe to call concurrently or + repeatedly. + justification: | + A mockable streaming-recognition contract lets hosts and tests depend on + recognition behavior without tying themselves to a specific inference backend, + while still giving them a defined point at which backend resources are + released. A streaming backend cannot decode the tail of an utterance released + with no trailing silence without audio it will now never receive (for example a + push-to-talk release), so without a defined finalization step that trailing + audio would be silently lost rather than delivered as one last result. + tests: + - UnavailableRecognitionSession_StartAsync_Always_ThrowsSpeechRecognizerUnavailableException + - UnavailableRecognitionSession_StopAsync_Always_IsSafeNoOp + - UnavailableRecognitionSession_DisposeAsync_CalledTwice_DoesNotThrow + - SherpaOnnxRecognitionSession_StopAsync_FlushesTrailingResultsBeforeCompleting + - SherpaOnnxRecognitionSession_StopAsync_CalledConcurrentlyTwice_BothCompleteOnceStopped + - SherpaOnnxRecognitionSession_StopAsync_BackendResetFails_CompletesAndReportsFault + + - id: Speech-Recognition-IRecognitionSession-ResultDelivery + title: >- + The IRecognitionSession interface shall expose an async-enumerable + GetResultsAsync that streams every provisional and final recognition result + produced while the session runs, is single-consumer (rejecting a second + concurrent enumeration), ends an individual enumeration without stopping the + session when its own cancellation token is cancelled, and throws + RecognitionSessionFaultedException from the enumerator when the session + transitions to Faulted while being enumerated. + justification: | + The design requires streaming recognition that emits "progressive provisional + and final results as audio arrives" through a single-consumer, async-native + surface, so a host must be able to tell from a single result whether to replace + an in-progress transcript line or commit it, and must learn promptly, through + the same enumeration, if the session can no longer produce results. + tests: + - UnavailableRecognitionSession_GetResultsAsync_Always_ThrowsSpeechRecognizerUnavailableException + - SherpaOnnxRecognitionSession_GetResultsAsync_CalledConcurrently_ThrowsInvalidOperationException + - SherpaOnnxRecognitionSession_GetResultsAsync_CancelledToken_EndsEnumerationWithoutStoppingSession + - SherpaOnnxRecognitionSession_GetResultsAsync_SessionFaulted_ThrowsRecognitionSessionFaultedException + - SherpaOnnxRecognitionSession_FrameCaptured_FinalResult_AppliesModelNormalizeTextWithIsFinalTrue + + - id: Speech-Recognition-IRecognitionSession-Backpressure + title: >- + A session's GetResultsAsync shall apply bounded backpressure to a slow consumer + by coalescing provisional results down to the latest one while never dropping a + final result under the documented byte cap. + justification: | + A host transcript consumer that is temporarily slower than the recognition + backend must never cause unbounded memory growth, but a provisional result is, + by definition, superseded by the next one, so only the latest provisional need + survive; a final result is a committed transcript line and must never be + silently dropped merely because the consumer is behind. + tests: + - SherpaOnnxRecognitionSession_GetResultsAsync_SlowConsumer_CoalescesProvisionalResults + - SherpaOnnxRecognitionSession_GetResultsAsync_SlowConsumer_NeverDropsFinalResultsUnderByteCap + + - title: RecognitionSessionFaultedException Requirements + requirements: + - id: Speech-Recognition-RecognitionSessionFaultedException-Thrown + title: >- + The RecognitionSessionFaultedException class shall be thrown from + GetResultsAsync's enumerator, and only from it, when the session transitions to + RecognitionSessionState.Faulted while being enumerated. + justification: | + A consumer enumerating results needs an unambiguous, typed signal that the + session itself has failed, distinct from an ordinary end of enumeration or a + caller-initiated cancellation, so it can decide whether to recreate a session + rather than silently treating a faulted session as merely finished. + tests: + - SherpaOnnxRecognitionSession_GetResultsAsync_SessionFaulted_ThrowsRecognitionSessionFaultedException diff --git a/docs/reqstream/speech/recognition-subsystem/i-speech-recognizer-engine.yaml b/docs/reqstream/speech/recognition-subsystem/i-speech-recognizer-engine.yaml new file mode 100644 index 0000000..dde568f --- /dev/null +++ b/docs/reqstream/speech/recognition-subsystem/i-speech-recognizer-engine.yaml @@ -0,0 +1,72 @@ +--- +# Software Unit Requirements for ISpeechRecognizerEngine +# +# These requirements describe the design-level behavior of the ISpeechRecognizerEngine contract, +# decomposing the parent RecognitionSubsystem requirements. Engine exclusivity/lease behavior and +# RecognitionEngineBusyException are documented separately in recognition-session-lease.yaml, +# since that behavior is a cross-cutting policy shared with SherpaOnnxSpeechRecognizerEngine. + +sections: + - title: Speech Requirements + sections: + - title: RecognitionSubsystem Requirements + sections: + - title: ISpeechRecognizerEngine Requirements + requirements: + - id: Speech-Recognition-ISpeechRecognizerEngine-AvailabilityFlag + title: >- + The ISpeechRecognizerEngine interface shall expose an IsAvailable flag that + never throws to read. + justification: | + Callers must be able to check availability before invoking an operational + member, without risk of the check itself failing. + tests: + - UnavailableSpeechRecognizerEngine_IsAvailable_Read_ReturnsFalse + - SpeechRecognizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine + + - id: Speech-Recognition-ISpeechRecognizerEngine-CreateSession + title: >- + The ISpeechRecognizerEngine interface shall expose an async CreateSessionAsync + method that binds a supplied capture device to a new IRecognitionSession for + the session's entire life, rejecting a null device and honoring caller + cancellation. + justification: | + A mockable Layer 3 engine contract lets hosts and tests depend on recognition + behavior without tying themselves to a specific inference backend, while + keeping the one-device-per-session binding explicit at the point a session is + created, per this library's Engine/Session split. + tests: + - UnavailableSpeechRecognizerEngine_CreateSessionAsync_Always_ReturnsUnavailableSession + - UnavailableSpeechRecognizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException + - SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException + - SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_NoActiveSession_ReturnsSession + - SpeechRecognizerFactory_LoadAsync_CancelledToken_FaultsWithOperationCanceledException + + - id: Speech-Recognition-ISpeechRecognizerEngine-Disposal + title: >- + The ISpeechRecognizerEngine interface shall expose an idempotent async + DisposeAsync that disposes any still-active session it created before + releasing the engine's own native resources. + justification: | + An engine owns the native-backed recognition backend across many sessions, so + disposing the engine while a session it created is still active must not leak + that session's resources, and repeated disposal from multiple owners must stay + safe. + tests: + - UnavailableSpeechRecognizerEngine_DisposeAsync_CalledTwice_DoesNotThrow + - SherpaOnnxSpeechRecognizerEngine_DisposeAsync_WithActiveSession_DisposesSessionFirst + + - id: Speech-Recognition-RecognitionBackend-UnavailableDeviceFallback + title: >- + SherpaOnnxSpeechRecognizerEngine's CreateSessionAsync shall report a Warning + diagnostic and return UnavailableRecognitionSession.Instance, without attempting + to lease its backend, when the supplied capture device itself reports + unavailable. + justification: | + No microphone (or no working audio backend) is an ordinary machine state, + exactly like a model not being installed - there is nothing to stream, so this + engine returns the same honest fallback used elsewhere in this library rather + than a session doomed to fail the instant it is started, while still telling an + observer why via the Warning diagnostic. + tests: + - SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_DeviceUnavailable_ReturnsFallbackAndReportsWarning diff --git a/docs/reqstream/speech/recognition-subsystem/i-speech-recognizer.yaml b/docs/reqstream/speech/recognition-subsystem/i-speech-recognizer.yaml deleted file mode 100644 index 426ea8b..0000000 --- a/docs/reqstream/speech/recognition-subsystem/i-speech-recognizer.yaml +++ /dev/null @@ -1,74 +0,0 @@ ---- -# Software Unit Requirements for ISpeechRecognizer -# -# These requirements describe the design-level behavior of the ISpeechRecognizer contract and its -# adjacent result/event value types, decomposing the parent RecognitionSubsystem requirements. -# The three types are documented together since the result and event types carry no behavior -# independent of the contract that produces them. - -sections: - - title: Speech Requirements - sections: - - title: RecognitionSubsystem Requirements - sections: - - title: ISpeechRecognizer Requirements - requirements: - - id: Speech-Recognition-ISpeechRecognizer-AvailabilityFlag - title: >- - The ISpeechRecognizer interface shall expose an IsAvailable flag that never - throws to read. - justification: | - Callers must be able to check availability before invoking an operational - member, without risk of the check itself failing. - tests: - - UnavailableSpeechRecognizer_IsAvailable_Read_ReturnsFalse - - SpeechRecognizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealRecognizer - - - id: Speech-Recognition-ISpeechRecognizer-StartStopResultReceived - title: >- - The ISpeechRecognizer interface shall define Start/Stop lifecycle methods, a - ResultReceived event for delivering recognition results while running, and - disposal of the resources a running recognizer holds. Stop shall finalize and - deliver any trailing audio accepted but not yet decoded before it returns, rather - than carrying it over into (or losing it between) sessions. - justification: | - A mockable streaming-recognition contract lets hosts and tests depend on - recognition behavior without tying themselves to a specific inference engine, - while still giving them a defined point at which engine resources are released. - A streaming engine cannot decode the tail of an utterance released with no - trailing silence without audio it will now never receive (for example a - push-to-talk release), so without a defined finalization step that trailing - audio would be silently lost rather than delivered as one last result. - tests: - - UnavailableSpeechRecognizer_Start_Always_ThrowsSpeechRecognizerUnavailableException - - UnavailableSpeechRecognizer_Stop_Always_ThrowsSpeechRecognizerUnavailableException - - SherpaOnnxSpeechRecognizer_Start_Always_SubscribesAndStartsCaptureDevice - - SherpaOnnxSpeechRecognizer_Stop_WhileRunning_UnsubscribesAndStopsCaptureDevice - - SherpaOnnxSpeechRecognizer_Stop_EngineHasFlushableTrailingAudio_RaisesFlushedFinalResult - - SherpaOnnxSpeechRecognizer_Dispose_CalledTwice_StopsAndDisposesEngineOnce - - - id: Speech-Recognition-ISpeechRecognizer-ResultShape - title: >- - A recognition result shall carry the full recognized text of the current - utterance together with whether that text is provisional or final. - justification: | - The design requires streaming recognition that emits "progressive - provisional and final results as audio arrives", so a host must be able to tell - from a single result whether to replace an in-progress transcript line or commit - it, without accumulating fragments itself. - tests: - - SherpaOnnxSpeechRecognizer_FrameCaptured_EngineDecodesResults_RaisesResultReceivedInOrder - - - id: Speech-Recognition-ISpeechRecognizer-HotReuseAcrossTurns - title: >- - The ISpeechRecognizer interface shall support many independent Start/Stop - cycles on the same instance without requiring the recognizer to be disposed and - reconstructed between cycles. - justification: | - Obtaining a recognizer from SpeechRecognizerFactory is the expensive step - it - loads the model into native memory - while Start/Stop are cheap. A host doing - repeated, low-latency recognition (for example, many turns of a voice - conversation) must be able to construct one recognizer once and reuse it across - turns rather than reloading the model per turn. - tests: - - SherpaOnnxSpeechRecognizer_MultipleStartStopCycles_ReusesSameInstanceWithoutReconstruction diff --git a/docs/reqstream/speech/recognition-subsystem/recognition-session-lease.yaml b/docs/reqstream/speech/recognition-subsystem/recognition-session-lease.yaml new file mode 100644 index 0000000..5a91849 --- /dev/null +++ b/docs/reqstream/speech/recognition-subsystem/recognition-session-lease.yaml @@ -0,0 +1,90 @@ +--- +# Software Unit Requirements for engine exclusivity / lease behavior +# +# These requirements describe the single-active-session lease policy shared by +# ISpeechRecognizerEngine implementations (concretely verified against +# SherpaOnnxSpeechRecognizerEngine) and RecognitionEngineBusyException, decomposing the parent +# RecognitionSubsystem requirements. Documented as a standalone file because this fail-fast +# exclusivity policy is a cross-cutting design decision independent of any single unit's other +# behavior. + +sections: + - title: Speech Requirements + sections: + - title: RecognitionSubsystem Requirements + sections: + - title: Engine Exclusivity Requirements + requirements: + - id: Speech-Recognition-EngineExclusivity-SingleActiveSession + title: >- + An ISpeechRecognizerEngine implementation shall permit at most one live session + created from it at a time, returning the same kind of new session once a prior + session created from it has fully completed its own disposal. + justification: | + The native recognition backend an engine owns is not safe for two sessions to + drive concurrently, so the engine - not each session - is the natural place to + enforce that exclusivity; reuse across sequential sessions (rather than one + session per engine lifetime) is still required to keep the expensive model-load + step amortized across many turns. + tests: + - SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_NoActiveSession_ReturnsSession + - SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_AfterPriorSessionFullyDisposed_ReturnsNewSession + + - id: Speech-Recognition-EngineExclusivity-FailFastNotQueued + title: >- + An ISpeechRecognizerEngine implementation's CreateSessionAsync shall fail fast + with RecognitionEngineBusyException, rather than queuing or waiting, when a + prior session created from it already holds the engine's exclusivity lease - + including while that prior session is still tearing down via its own + DisposeAsync. + justification: | + A caller that attempts to create a second session before releasing the first + has made a sequencing error; silently queuing the request would hide that bug + behind an unpredictable delay instead of surfacing it immediately, and the lease + is held for the whole disposal window because disposal itself still drives the + native backend. + tests: + - SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_SessionAlreadyLeased_ThrowsRecognitionEngineBusyException + - SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_PriorSessionDisposing_ThrowsRecognitionEngineBusyException + + - id: Speech-Recognition-EngineExclusivity-DisposalOwnsSession + title: >- + An ISpeechRecognizerEngine implementation's DisposeAsync shall dispose any + session it still holds the lease for before releasing the engine's own native + resources. + justification: | + A host that disposes the engine without first disposing an active session it + created must not leak that session's unmanaged inference resources or leave the + lease held with no way to release it. + tests: + - SherpaOnnxSpeechRecognizerEngine_DisposeAsync_WithActiveSession_DisposesSessionFirst + + - title: RecognitionEngineBusyException Requirements + requirements: + - id: Speech-Recognition-RecognitionEngineBusyException-Thrown + title: >- + The RecognitionEngineBusyException class shall be thrown, and only be thrown, + from CreateSessionAsync when the engine's single-session exclusivity lease is + already held. + justification: | + A consumer calling CreateSessionAsync needs an unambiguous, typed signal that + its failure is caused by a currently-held lease rather than a null argument, + cancellation, or unrelated composition failure, so it can decide whether to + retry after releasing the prior session. + tests: + - SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_SessionAlreadyLeased_ThrowsRecognitionEngineBusyException + - SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_PriorSessionDisposing_ThrowsRecognitionEngineBusyException + + - id: Speech-Recognition-RecognitionEngineBusyException-Constructors + title: >- + The RecognitionEngineBusyException class shall provide standard exception + constructors (default, message, message with inner exception). + justification: | + Standard exception conformance lets callers construct, catch, and inspect this + exception like any other .NET exception type, mirroring the identical + constructor pattern used by this library's other custom exception types (for + example AudioDeviceInUseException). + tests: + - RecognitionEngineBusyException_Constructor_Default_HasNonEmptyMessage + - RecognitionEngineBusyException_Constructor_WithMessage_ExposesMessage + - RecognitionEngineBusyException_Constructor_WithInnerException_ExposesBoth diff --git a/docs/reqstream/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.yaml b/docs/reqstream/speech/recognition-subsystem/sherpa-onnx-recognition-session.yaml similarity index 55% rename from docs/reqstream/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.yaml rename to docs/reqstream/speech/recognition-subsystem/sherpa-onnx-recognition-session.yaml index d53e677..1f566dc 100644 --- a/docs/reqstream/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.yaml +++ b/docs/reqstream/speech/recognition-subsystem/sherpa-onnx-recognition-session.yaml @@ -1,144 +1,83 @@ --- -# Software Unit Requirements for the streaming recognizer implementation +# Software Unit Requirements for the streaming recognition session implementation # -# Groups requirements for SherpaOnnxSpeechRecognizer together with the recognition-engine seam -# and the audio-format converter it depends on, mirroring how port-audio.yaml groups a cohesive -# implementation unit. Each unit's requirement subset is called out in its own subsection below so -# every unit still has requirements tracing to it individually. +# Groups requirements for SherpaOnnxRecognitionSession together with the recognition-backend seam +# (IRecognitionBackend/IRecognitionBackendFactory, concretely SherpaOnnxRecognitionEngine), the +# audio-format converter, and the result-buffering backpressure policy it depends on, mirroring +# how port-audio.yaml groups a cohesive implementation unit. Each unit's requirement subset is +# called out in its own subsection below so every unit still has requirements tracing to it +# individually. DedicatedWorker's cooperative-cancel-then-abandon policy is documented separately +# in dedicated-worker.yaml, since that policy is a standalone, independently reusable concern. sections: - title: Speech Requirements sections: - title: RecognitionSubsystem Requirements sections: - - title: SherpaOnnxSpeechRecognizer Requirements + - title: SherpaOnnxRecognitionSession Requirements requirements: - - id: Speech-Recognition-SherpaOnnxSpeechRecognizer-FormatConversion + - id: Speech-Recognition-SherpaOnnxRecognitionSession-FormatConversion title: >- - The SherpaOnnxSpeechRecognizer class shall convert each captured audio block + The SherpaOnnxRecognitionSession class shall convert each captured audio block from the capture device's reported channel count and sample rate into - single-channel audio at the rate the selected model requires. + single-channel audio at the rate the selected model requires before feeding it + to the recognition backend. justification: | Desktop microphones commonly capture stereo audio at 44.1 or 48 kHz while recognition models require one fixed mono rate. Without this conversion, recognition would only work on machines whose microphone happened to match the model, which contradicts the library's cross-platform promise. tests: - - SherpaOnnxSpeechRecognizer_FrameCaptured_StereoAtHigherRate_FeedsResampledMonoToEngine - - SherpaOnnxSpeechRecognizer_Constructor_DeviceReportsUnusableFormat_FallsBackToPassThrough + - SherpaOnnxRecognitionSession_FrameCaptured_StereoAtModelRate_FeedsDownmixedMonoToBackend - - id: Speech-Recognition-SherpaOnnxSpeechRecognizer-ResultDelivery + - id: Speech-Recognition-SherpaOnnxRecognitionSession-TextNormalization title: >- - The SherpaOnnxSpeechRecognizer class shall deliver every recognition result - produced from captured audio, in the order produced, on a thread other than the - audio capture callback thread. - justification: | - Recognition work performed on the audio callback thread causes capture dropouts, - and a host rendering a live transcript needs results in the order they were - produced for the transcript to read correctly. - tests: - - SherpaOnnxSpeechRecognizer_FrameCaptured_EngineDecodesResults_RaisesResultReceivedInOrder - - - id: Speech-Recognition-SherpaOnnxSpeechRecognizer-TextNormalization - title: >- - The SherpaOnnxSpeechRecognizer class shall require a non-null owning - IRecognitionModel and shall pass every decoded result's text through that - model's IRecognitionModel.NormalizeText(text, isFinal) before raising it, - supplying the result's own IsFinal flag as isFinal. + The SherpaOnnxRecognitionSession class shall pass every decoded result's text + through its owning model's IRecognitionModel.NormalizeText(text, isFinal) + before delivering it through GetResultsAsync, supplying the result's own + IsFinal flag as isFinal. justification: | Some recognition models' raw output is UPPERCASE and unpunctuated (empirically confirmed for the Zipformer model); consumers must see readable, cased, - punctuated text rather than the engine's raw output, and the recognizer is the + punctuated text rather than the backend's raw output, and the session is the one place every result passes through regardless of which model produced it. tests: - - SherpaOnnxSpeechRecognizer_Constructor_NullModel_ThrowsArgumentNullException - - SherpaOnnxSpeechRecognizer_FrameCaptured_FinalResult_AppliesModelNormalizeTextWithIsFinalTrue - - SherpaOnnxSpeechRecognizer_FrameCaptured_ProvisionalResult_AppliesModelNormalizeTextWithIsFinalFalse + - SherpaOnnxRecognitionSession_FrameCaptured_FinalResult_AppliesModelNormalizeTextWithIsFinalTrue - - id: Speech-Recognition-SherpaOnnxSpeechRecognizer-LifecycleDrain + - id: Speech-Recognition-SherpaOnnxRecognitionSession-DeviceLostFault title: >- - The SherpaOnnxSpeechRecognizer class shall start and stop audio capture with the - recognition session, deliver every result derived from audio captured before a - stop request before that request completes, and release its engine resources on - disposal. + The SherpaOnnxRecognitionSession class shall transition to + RecognitionSessionState.Faulted when its bound capture device is lost mid + session, rather than silently stopping or hanging a concurrent + GetResultsAsync enumeration. justification: | - A host that stops recognition must be able to rely on having received the whole - transcript, and must not leak native inference resources; without a defined - drain point the tail of an utterance could be silently lost. + A device that disappears mid-session (for example unplugged, or its driver + crashes) is a genuine runtime failure a host must learn about promptly and + unambiguously through the session's own state machine, exactly like a backend + fault. tests: - - SherpaOnnxSpeechRecognizer_Start_Always_SubscribesAndStartsCaptureDevice - - SherpaOnnxSpeechRecognizer_Start_AlreadyRunning_IsNoOp - - SherpaOnnxSpeechRecognizer_Stop_WhileRunning_UnsubscribesAndStopsCaptureDevice - - SherpaOnnxSpeechRecognizer_Stop_NotRunning_IsNoOp - - SherpaOnnxSpeechRecognizer_Stop_EngineHasFlushableTrailingAudio_RaisesFlushedFinalResult - - SherpaOnnxSpeechRecognizer_Stop_EngineHasNothingToFlush_RaisesNoExtraResult - - SherpaOnnxSpeechRecognizer_Dispose_EngineHasFlushableTrailingAudio_RaisesFlushedFinalResult - - SherpaOnnxSpeechRecognizer_Dispose_CalledTwice_StopsAndDisposesEngineOnce - - SherpaOnnxSpeechRecognizer_Start_AfterDispose_ThrowsObjectDisposedException + - SherpaOnnxRecognitionSession_DeviceLostMidSession_TransitionsToFaulted - - id: Speech-Recognition-SherpaOnnxSpeechRecognizer-FaultContainment + - id: Speech-Recognition-SherpaOnnxRecognitionSession-NativeCallAbandonIntegration title: >- - The SherpaOnnxSpeechRecognizer class shall report, and continue running after, - any fault raised while recognizing a captured audio block or while delivering a - result to a host handler. + The SherpaOnnxRecognitionSession class shall route every call into the + recognition backend through the cooperative-cancel-then-abandon policy (see + dedicated-worker.yaml), so a single native call that ignores cancellation + cannot hang the session's await indefinitely, and shall report a Warning + diagnostic through its own diagnostics sink when abandonment occurs. justification: | - Captured audio arrives on a high-priority audio callback thread where an - escaping exception tears down the audio stream, so one bad block or one - misbehaving host handler must never silence the microphone or end the session. + The sherpa-onnx native runtime offers no cooperative cancellation of its own; + without this integration a single stuck native decode/reset/flush call would + hang the awaiting caller (for example StopAsync during teardown) indefinitely. tests: - - SherpaOnnxSpeechRecognizer_FrameCaptured_EngineThrows_ReportsFaultAndKeepsRunning - - SherpaOnnxSpeechRecognizer_ResultReceived_HandlerThrows_ReportsFaultAndDoesNotRethrow - - SherpaOnnxSpeechRecognizer_FrameCaptured_EmptyBlock_IsIgnored + - SherpaOnnxRecognitionSession_NativeCallExceedsAbandonTimeout_TaskCompletesAndDiagnosticsReportsWarning - - id: Speech-Recognition-SherpaOnnxSpeechRecognizer-ResetFailureContainment - title: >- - The SherpaOnnxSpeechRecognizer class shall report, rather than propagate, a - fault raised while resetting the recognition engine during Stop()/Dispose() - teardown and still complete that teardown; a failed reset during Stop() shall - still permit a later Start(), while Dispose() remains terminal regardless of - whether its reset succeeds. - justification: | - Resetting the engine crosses the native decoder boundary during teardown, so a - fault there must not prevent Stop()/Dispose() from completing, and must not - block a host from restarting recognition after a failed Stop(), even though the - engine's state cannot be guaranteed clean in that one failure case. Dispose() - is unconditionally terminal, so no restart guarantee applies to it. - tests: - - SherpaOnnxSpeechRecognizer_Stop_EngineResetFails_CompletesReportsFaultAndPermitsRestart - - - id: Speech-Recognition-SherpaOnnxSpeechRecognizer-FlushFailureContainment - title: >- - The SherpaOnnxSpeechRecognizer class shall report, rather than propagate, a - fault raised while flushing the recognition engine's trailing audio during - Stop()/Dispose() teardown, still complete that teardown, and still reset the - engine afterward. - justification: | - Flushing trailing audio (see Speech-Recognition-RecognitionEngine- - TryFlushRecoversTrailingAudio) crosses the native decoder boundary during - teardown, exactly like the reset that follows it, so a fault there must not - prevent Stop()/Dispose() from completing or from still resetting the engine for - the next session. - tests: - - SherpaOnnxSpeechRecognizer_Stop_EngineFlushFails_CompletesResetsEngineAndReportsFault - - - id: Speech-Recognition-SherpaOnnxSpeechRecognizer-CaptureStartFailure - title: >- - The SherpaOnnxSpeechRecognizer class shall throw the documented recognizer - exception when the capture device it was composed with fails to start. - justification: | - A device that reported itself available but then failed on first use is a - genuine failure the caller asked for, so it must surface immediately rather than - leaving the host believing recognition is running. - tests: - - SherpaOnnxSpeechRecognizer_Start_CaptureDeviceFails_ThrowsSpeechRecognizerUnavailableException - - SherpaOnnxSpeechRecognizer_Start_CaptureDeviceFails_ResetsEngineDuringRollback - - - title: RecognitionEngine Requirements + - title: IRecognitionBackend Requirements requirements: - - id: Speech-Recognition-RecognitionEngine-MockableSeam + - id: Speech-Recognition-RecognitionBackend-MockableSeam title: >- - The recognition-engine seam shall isolate all speech-inference interop so the - recognition pipeline's threading, audio conversion, and result delivery can be + The recognition-backend seam shall isolate all speech-inference interop so the + recognition session's threading, audio conversion, and result delivery can be verified without a downloaded model or an installed speech-engine native runtime. justification: | @@ -146,24 +85,24 @@ sections: platform-specific native inference binary available, so the recognition logic must be verifiable entirely in pure managed tests. tests: - - SpeechRecognizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealRecognizer - - SherpaOnnxSpeechRecognizer_FrameCaptured_EngineDecodesResults_RaisesResultReceivedInOrder + - SherpaOnnxRecognitionSession_StartAsync_FromCreated_TransitionsToRunning + - SherpaOnnxRecognitionSession_FrameCaptured_FinalResult_AppliesModelNormalizeTextWithIsFinalTrue - - id: Speech-Recognition-RecognitionEngine-ModelOwnedConfiguration + - id: Speech-Recognition-RecognitionBackend-ModelOwnedConfiguration title: >- - The recognition-engine seam shall load an engine using only the configuration + The recognition-backend seam shall load a backend using only the configuration and input rate the selected model itself declares. justification: | The design makes each model's backing class responsible for its own engine - configuration, so that adding a new model is a self-contained, reviewable unit of - work that never requires changing the recognition subsystem. + configuration, so that adding a new model is a self-contained, reviewable unit + of work that never requires changing the recognition subsystem. tests: - - SpeechRecognizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealRecognizer + - SpeechRecognizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine - IRecognitionModel_CreateEngineConfig_InstalledDirectory_ResolvesPathsAndSampleRate - - id: Speech-Recognition-RecognitionEngine-SessionEndResetDiscardsBufferedAudio + - id: Speech-Recognition-RecognitionBackend-SessionEndResetDiscardsBufferedAudio title: >- - The recognition-engine seam's session-end Reset() shall discard audio already + The recognition-backend seam's session-end Reset() shall discard audio already accepted but not yet decoded, not just the decoder's current hypothesis, so an abandoned utterance cannot bleed into the next session's decoding. justification: | @@ -172,21 +111,21 @@ sections: audio decodes into the next session as soon as any audio - even silence - supplies the missing future context, so an abandoned utterance (for example a push-to-talk release with no trailing silence) would otherwise bleed into the - next Start(). + next StartAsync(). tests: - SherpaOnnxRecognitionEngine_Reset_AbandonedUtteranceWithNoTrailingSilence_DoesNotBleedIntoNextSession - - id: Speech-Recognition-RecognitionEngine-TryFlushRecoversTrailingAudio + - id: Speech-Recognition-RecognitionBackend-TryFlushRecoversTrailingAudio title: >- - The recognition-engine seam's TryFlush() shall finalize and decode audio already - accepted but not yet decoded, delivering it as a final result rather than - losing it, before a session-end Reset() discards whatever remains. + The recognition-backend seam's TryFlush() shall finalize and decode audio + already accepted but not yet decoded, delivering it as a final result rather + than losing it, before a session-end Reset() discards whatever remains. justification: | A streaming transducer sometimes cannot decode the tail of an utterance without more audio a caller who has just stopped will never supply - for example a push-to-talk release with no trailing silence. Without a way to finalize that tail, the only alternative is to discard it unread (see - Speech-Recognition-RecognitionEngine-SessionEndResetDiscardsBufferedAudio), + Speech-Recognition-RecognitionBackend-SessionEndResetDiscardsBufferedAudio), silently losing the last word a user spoke. The native engine's own InputFinished mechanism exists precisely to force a decode of buffered audio without needing genuine future context, so this recovers that audio instead of @@ -194,9 +133,9 @@ sections: tests: - SherpaOnnxRecognitionEngine_TryFlush_AbandonedUtteranceWithNoTrailingSilence_RecoversTrailingWords - - id: Speech-Recognition-RecognitionEngine-PostEndpointWarmupReplay + - id: Speech-Recognition-RecognitionBackend-PostEndpointWarmupReplay title: >- - The recognition-engine seam shall not lose genuinely spoken audio that falls + The recognition-backend seam shall not lose genuinely spoken audio that falls within a model's post-reset encoder warm-up period, for any model that declares a positive IRecognitionModel.PostEndpointWarmupWindowMs, by buffering a rolling window of pre-endpoint audio and silently replaying it into the freshly reset @@ -206,8 +145,8 @@ sections: fires on pure/near-silence with no genuine text recognized yet since the last reset shall not replay, dropping the buffered audio instead, since replaying it into a freshly reset (encoder-cold) stream risks corrupting the next genuinely - recognized segment; models that declare the default 0 (disabled) window shall be - entirely unaffected, with no buffer allocated and no replay logic invoked. + recognized segment; models that declare the default 0 (disabled) window shall + be entirely unaffected, with no buffer allocated and no replay logic invoked. justification: | SherpaOnnxNemotronStreamingEnRecognitionModel's endpoint detector was confirmed, on two real user recordings, to fire as a false positive mid-utterance; the @@ -240,33 +179,48 @@ sections: - SherpaOnnxRecognitionEngine_PostEndpointWarmupWindowMsEnabled_GracePeriodSuppressesImmediateReTrigger - SherpaOnnxRecognitionEngine_PostEndpointWarmupWindowMsEnabled_SilenceOnlyEndpoint_NeverArmsReplay - SherpaOnnxRecognitionEngine_PostEndpointWarmupWindowMsEnabled_ReplayEligibility_GatedOnHasRecognizedTextSinceReset - - SherpaOnnxNemotronStreamingEnRecognitionModel_PostEndpointWarmupWindowMs_Is800 - - SherpaOnnxZipformerEnRecognitionModel_PostEndpointWarmupWindowMs_IsDisabledDefault - SherpaOnnxRecognitionEngine_PostEndpointWarmupWindowMsEnabled_TryFlushAfterReplayWithNoNewAudio_ReportsNothing - SherpaOnnxRecognitionEngine_PostEndpointWarmupWindowMsEnabled_TryDecodeAfterReplayWithNoNewAudio_ReportsNothing - - id: Speech-Recognition-RecognitionEngine-RealSpeechAccuracy + - id: Speech-Recognition-RecognitionBackend-RealSpeechAccuracy title: >- - The recognition-engine seam shall transcribe a clear, single-speaker, + The recognition-backend seam shall transcribe a clear, single-speaker, studio-quality real human speech recording into text whose Word Error Rate against the recording's known-correct ground-truth transcript does not exceed a documented tolerance, for every real, installed recognition model. justification: | Every existing automated test for this seam either replaces the real engine with a fake, or drives the real engine with silence and a synthesized sine-wave tone - purely to exercise buffer/bookkeeping logic (see the RecognitionEngine - Requirements above) - none of them ever assert on actual transcribed text - content. That left a genuine, previously unaddressed coverage gap: nothing - proved the real, shipped models transcribe real speech correctly. A tolerant - Word-Error-Rate comparison, not an exact string match, is required because a - real model's raw output legitimately differs from the ground truth in ways that - do not reflect a transcription failure - for example Tennyson's archaic spelling - "crost" (for "crossed") in the "Crossing the Bar" fixture recording may be - rendered as either spelling by a modern model's vocabulary. + purely to exercise buffer/bookkeeping logic (see the requirements above) - none + of them ever assert on actual transcribed text content. That left a genuine, + previously unaddressed coverage gap: nothing proved the real, shipped models + transcribe real speech correctly. A tolerant Word-Error-Rate comparison, not an + exact string match, is required because a real model's raw output legitimately + differs from the ground truth in ways that do not reflect a transcription + failure - for example Tennyson's archaic spelling "crost" (for "crossed") in the + "Crossing the Bar" fixture recording may be rendered as either spelling by a + modern model's vocabulary. tests: - SherpaOnnxRecognitionEngine_Transcribe_RealCrossingTheBarRecording_WordErrorRateBelowTolerance - SherpaOnnxRecognitionEngine_Transcribe_RealCrossingTheBarRecording_NemotronWordErrorRateBelowTolerance + - title: RecognitionResultBuffer Requirements + requirements: + - id: Speech-Recognition-RecognitionResultBuffer-TwoTierBackpressure + title: >- + The RecognitionResultBuffer class shall coalesce queued provisional results + down to a single latest-provisional slot while queuing final results in a + bounded, drop-nothing FIFO up to a documented byte cap, so GetResultsAsync can + apply the documented backpressure policy without dropping a final result. + justification: | + A session's result-delivery pipeline must never grow without bound merely + because a consumer is temporarily slower than the recognition backend, but a + provisional result's whole purpose is to be superseded, while a final result is + a committed transcript line that must never be silently dropped. + tests: + - SherpaOnnxRecognitionSession_GetResultsAsync_SlowConsumer_CoalescesProvisionalResults + - SherpaOnnxRecognitionSession_GetResultsAsync_SlowConsumer_NeverDropsFinalResultsUnderByteCap + - title: AudioFrameResampler Requirements requirements: - id: Speech-Recognition-AudioFrameResampler-Downmix diff --git a/docs/reqstream/speech/recognition-subsystem/speech-recognizer-factory.yaml b/docs/reqstream/speech/recognition-subsystem/speech-recognizer-factory.yaml index 6f78c8e..4669ad4 100644 --- a/docs/reqstream/speech/recognition-subsystem/speech-recognizer-factory.yaml +++ b/docs/reqstream/speech/recognition-subsystem/speech-recognizer-factory.yaml @@ -11,141 +11,117 @@ sections: sections: - title: SpeechRecognizerFactory Requirements requirements: - - id: Speech-Recognition-SpeechRecognizerFactory-RealRecognizer + - id: Speech-Recognition-SpeechRecognizerFactory-RealEngine title: >- - The SpeechRecognizerFactory class shall return a working speech recognizer when - the requested model is installed, declares the recognition role, its engine - loads, and the supplied capture device is available. + The SpeechRecognizerFactory class shall return a working ISpeechRecognizerEngine + when the requested model is installed, declares the recognition role, and its + recognition backend loads. justification: | A single composition entry point keeps all "can this machine recognize speech right now?" logic in one reviewable place, so a host never has to reason about - model installation, device availability, and engine loading separately. + model installation and backend loading separately. tests: - - SpeechRecognizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealRecognizer - - SpeechRecognizerFactory_Create_WithStoreModelInstalledAndDeviceAvailable_ReturnsRealRecognizer - - SpeechRecognizerFactory_Create_WithCatalogModelInstalledAndDeviceAvailable_ReturnsRealRecognizer + - SpeechRecognizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine + - SpeechRecognizerFactory_LoadAsync_WithStoreModelInstalled_ReturnsRealEngine + - SpeechRecognizerFactory_LoadAsync_WithCatalogModelInstalled_ReturnsRealEngine - id: Speech-Recognition-SpeechRecognizerFactory-HonestFallbacks title: >- - The SpeechRecognizerFactory class shall return an honest unavailable recognizer, - without throwing, when the requested model is not installed, the model does not - declare the recognition role, or the supplied capture device is unavailable. + The SpeechRecognizerFactory class shall return an honest unavailable engine, + without faulting the returned task, when the requested model is not installed + or the model does not declare the recognition role. justification: | Per this library's "nothing throws at composition" decision, a model that has - not been downloaded yet and a machine with no microphone are ordinary states at - application start-up, not programming errors. + not been downloaded yet is an ordinary state at application start-up, not a + programming error. tests: - - SpeechRecognizerFactory_Create_ModelNotInstalled_ReturnsUnavailableRecognizer - - SpeechRecognizerFactory_Create_CaptureDeviceUnavailable_ReturnsUnavailableRecognizer - - SpeechRecognizerFactory_Create_ModelRoleIsNotRecognition_ReturnsUnavailableRecognizer - - SpeechRecognizerFactory_Create_WithStoreModelNotInstalled_ReturnsUnavailableRecognizer - - SpeechRecognizerFactory_Create_WithCatalogModelNotInstalled_ReturnsUnavailableRecognizer + - SpeechRecognizerFactory_LoadAsync_ModelNotInstalled_ReturnsUnavailableEngine + - SpeechRecognizerFactory_LoadAsync_ModelRoleIsNotRecognition_ReturnsUnavailableEngine + - SpeechRecognizerFactory_LoadAsync_WithStoreModelNotInstalled_ReturnsUnavailableEngine + - SpeechRecognizerFactory_LoadAsync_WithCatalogModelNotInstalled_ReturnsUnavailableEngine - id: Speech-Recognition-SpeechRecognizerFactory-EngineLoadFailureFallback title: >- - The SpeechRecognizerFactory class shall return an honest unavailable recognizer, - and report the reason, when the recognition engine for an installed model cannot - be loaded. + The SpeechRecognizerFactory class shall return an honest unavailable engine, + and report the reason, when the recognition backend for an installed model + cannot be loaded, without faulting the returned task. justification: | - The design requires that "a model whose native runtime is absent must - degrade the same honest way as a missing model file, never crash". An engine - that cannot load because of a missing native binary, an unsupported platform, or + The design requires that "a model whose native runtime is absent must degrade + the same honest way as a missing model file, never crash". A backend that + cannot load because of a missing native binary, an unsupported platform, or corrupt model files must therefore not fail application start-up. tests: - - SpeechRecognizerFactory_Create_EngineLoadFails_ReturnsUnavailableRecognizerAndDoesNotThrow + - SpeechRecognizerFactory_LoadAsync_EngineLoadFails_ReturnsUnavailableEngineAndDoesNotFaultTask - id: Speech-Recognition-SpeechRecognizerFactory-ArgumentValidation title: >- - The SpeechRecognizerFactory class shall reject a null model, a null - SpeechModelStore, a null SpeechModelCatalog, or a null capture device. + The SpeechRecognizerFactory class shall fault LoadAsync's returned task with + ArgumentNullException for a null model, a null SpeechModelStore, or a null + SpeechModelCatalog, and shall fault it with OperationCanceledException when its + cancellation token is cancelled. justification: | - A null model, store, catalog, or capture device is a programming error at the - call site, not an ordinary "unavailable" machine state, and must fail - immediately with a clear exception rather than being silently treated as - unavailable. + A null model, store, or catalog is a programming error at the call site, not an + ordinary "unavailable" machine state, and caller cancellation is a caller + decision - both must fail the returned task immediately with a clear exception + rather than being silently treated as unavailable. tests: - - SpeechRecognizerFactory_Create_NullModel_ThrowsArgumentNullException - - SpeechRecognizerFactory_Create_NullCaptureDevice_ThrowsArgumentNullException - - SpeechRecognizerFactory_Create_WithStoreNullModel_ThrowsArgumentNullException - - SpeechRecognizerFactory_Create_WithStoreNullStore_ThrowsArgumentNullException - - SpeechRecognizerFactory_Create_WithStoreNullCaptureDevice_ThrowsArgumentNullException - - SpeechRecognizerFactory_Create_WithCatalogNullModel_ThrowsArgumentNullException - - SpeechRecognizerFactory_Create_WithCatalogNullCatalog_ThrowsArgumentNullException - - SpeechRecognizerFactory_Create_WithCatalogNullCaptureDevice_ThrowsArgumentNullException + - SpeechRecognizerFactory_LoadAsync_NullModel_FaultsWithArgumentNullException + - SpeechRecognizerFactory_LoadAsync_WithStoreNullModel_FaultsWithArgumentNullException + - SpeechRecognizerFactory_LoadAsync_WithStoreNullStore_FaultsWithArgumentNullException + - SpeechRecognizerFactory_LoadAsync_WithCatalogNullModel_FaultsWithArgumentNullException + - SpeechRecognizerFactory_LoadAsync_WithCatalogNullCatalog_FaultsWithArgumentNullException + - SpeechRecognizerFactory_LoadAsync_CancelledToken_FaultsWithOperationCanceledException - id: Speech-Recognition-SpeechRecognizerFactory-ParameterValuesForwarding title: >- The SpeechRecognizerFactory class shall forward an optional parameterValues - bag to the recognition engine construction path, defaulting to null for a + bag to the recognition backend construction path, defaulting to null for a caller that does not supply one, so a model with real tunable-parameter knowledge can later interpret a genuine parameter value from it. justification: | - Prior to this requirement, no seam existed to carry a session-level tunable - value (for example a selected recognition language) from a host's - ISpeechModel.Parameters selection through to a recognition model's engine - configuration; adding this argument as optional with a null default keeps every - existing call site source-compatible while unblocking real parameter - interpretation end-to-end for a future tunable recognition model, mirroring the - equivalent SpeechSynthesizerFactory capability already added for voice - selection. + This seam carries a session-level tunable value (for example a selected + recognition language) from a host's ISpeechModel.Parameters selection through to + a recognition model's engine configuration, mirroring the equivalent + SpeechSynthesizerFactory capability already added for voice selection. tests: - - SpeechRecognizerFactory_Create_ParameterValuesSupplied_ReachesModelCreateEngineConfig - - SpeechRecognizerFactory_Create_WithStoreParameterValuesSupplied_ReachesModelCreateEngineConfig - - SpeechRecognizerFactory_Create_WithCatalogParameterValuesSupplied_ReachesModelCreateEngineConfig + - SpeechRecognizerFactory_LoadAsync_ParameterValuesSupplied_ReachesModelCreateEngineConfig + - SpeechRecognizerFactory_LoadAsync_WithStoreParameterValuesSupplied_ReachesModelCreateEngineConfig + - SpeechRecognizerFactory_LoadAsync_WithCatalogParameterValuesSupplied_ReachesModelCreateEngineConfig - id: Speech-Recognition-SpeechRecognizerFactory-UnrecognizedParameterIdIgnored title: >- The SpeechRecognizerFactory class shall silently ignore a supplied parameterValues key that names a parameter not declared by the requested model, - still composing a real recognizer, while reporting an Info-level diagnostic - naming the ignored key. + still composing a real engine, while reporting an Info-level diagnostic naming + the ignored key. justification: | A host reusing one settings dictionary across different recognition models must not break composition just because a given model does not declare a parameter another model had; this is a deliberate, documented - cross-model-compatibility contract, not a caller bug, so it must never throw - - only be surfaced as an informational diagnostic for a host that wired up a - sink. + cross-model-compatibility contract, not a caller bug, so it must never fault the + returned task - only be surfaced as an informational diagnostic for a host that + wired up a sink. tests: - - SpeechRecognizerFactory_Create_UnrecognizedParameterId_ComposesAndReportsInfo + - SpeechRecognizerFactory_LoadAsync_UnrecognizedParameterId_ComposesAndReportsInfo - id: Speech-Recognition-SpeechRecognizerFactory-InvalidRecognizedParameterValueThrows title: >- - The SpeechRecognizerFactory class shall throw ArgumentException synchronously - from Create() - before installed/role/device/engine checks run - when - parameterValues supplies a value for a parameter the requested model does - declare, but the value is invalid for that parameter (wrong CLR type, outside - [Minimum, Maximum] or non-integral for a NumericParameter declared IsInteger, - or not a declared ChoiceParameterOption.Value for a ChoiceParameter, or not a - bool for a BooleanParameter), naming the parameter id, the model id, and the - reason the value is invalid. + The SpeechRecognizerFactory class shall fault LoadAsync's returned task with + ArgumentException when parameterValues supplies a value for a parameter the + requested model does declare, but the value is invalid for that parameter + (wrong CLR type, outside [Minimum, Maximum] or non-integral for a + NumericParameter declared IsInteger, or not a declared + ChoiceParameterOption.Value for a ChoiceParameter, or not a bool for a + BooleanParameter), naming the parameter id, the model id, and the reason the + value is invalid. justification: | - This is a deliberate, user-approved breaking change from this library's - earlier silent-default behavior for this case: a host explicitly targeting a - parameter it knows this model declares, with a value invalid for it, is a - caller bug rather than a legitimate cross-model compatibility gap, and must be - surfaced immediately and loudly - synchronously from Create() - rather than - silently substituted and discovered much later from confusing recognizer - behavior. + A host explicitly targeting a parameter it knows this model declares, with a + value invalid for it, is a caller bug rather than a legitimate cross-model + compatibility gap, and must be surfaced loudly through the returned task's fault + rather than silently substituted and discovered much later from confusing + engine behavior. tests: - - SpeechRecognizerFactory_Create_RecognizedNumericParameterOutOfRange_Throws - - SpeechRecognizerFactory_Create_RecognizedNumericParameterWrongType_Throws - - SpeechRecognizerFactory_Create_RecognizedChoiceParameterInvalidOption_Throws - - SpeechRecognizerFactory_Create_RecognizedBooleanParameterWrongType_Throws - - - id: Speech-Recognition-SpeechRecognizerFactory-ConcurrentPrewarmingSafety - title: >- - A host may call SpeechRecognizerFactory.Create concurrently from a background - task while other unrelated work proceeds, since the factory itself holds no - state; doing so is only safe with respect to a caller-supplied diagnostics sink - when that sink is itself safe for concurrent use from multiple threads. - justification: | - This factory's model-load work is expensive, so a host doing repeated, - low-latency recognition (for example, a voice-conversation "ask" command) may - want to pre-warm the next turn's recognizer on a background task while other - work - such as speaking the current turn's prompt - proceeds concurrently. This - is only safe with respect to the factory's own (stateless) implementation; a - non-thread-safe diagnostics sink shared with other concurrent work remains the - caller's responsibility to synchronize. - tests: - - AskCommand_Run_PrewarmsRecognizerConcurrentlyWithPlayback_CreatesRecognizerBeforePlaybackCompletes + - SpeechRecognizerFactory_LoadAsync_RecognizedNumericParameterOutOfRange_FaultsWithArgumentException + - SpeechRecognizerFactory_LoadAsync_RecognizedNumericParameterWrongType_FaultsWithArgumentException + - SpeechRecognizerFactory_LoadAsync_RecognizedChoiceParameterInvalidOption_FaultsWithArgumentException + - SpeechRecognizerFactory_LoadAsync_RecognizedBooleanParameterWrongType_FaultsWithArgumentException diff --git a/docs/reqstream/speech/recognition-subsystem/unavailable-recognition-session.yaml b/docs/reqstream/speech/recognition-subsystem/unavailable-recognition-session.yaml new file mode 100644 index 0000000..295e3bd --- /dev/null +++ b/docs/reqstream/speech/recognition-subsystem/unavailable-recognition-session.yaml @@ -0,0 +1,48 @@ +--- +# Software Unit Requirements for UnavailableRecognitionSession +# +# These requirements describe the design-level behavior of the UnavailableRecognitionSession +# unit, decomposing the parent RecognitionSubsystem requirements. + +sections: + - title: Speech Requirements + sections: + - title: RecognitionSubsystem Requirements + sections: + - title: UnavailableRecognitionSession Requirements + requirements: + - id: Speech-Recognition-UnavailableRecognitionSession-ReportsUnavailable + title: >- + The UnavailableRecognitionSession class shall report IsAvailable as false and + never throw merely for being obtained or held. + justification: | + A caller must always be able to safely obtain and hold a fallback session, per + this library's "nothing throws at composition" decision. + tests: + - UnavailableRecognitionSession_IsAvailable_Read_ReturnsFalse + + - id: Speech-Recognition-UnavailableRecognitionSession-OperationsThrow + title: >- + The UnavailableRecognitionSession class shall throw + SpeechRecognizerUnavailableException when StartAsync or GetResultsAsync is + invoked. + justification: | + A caller that ignores the IsAvailable flag and invokes an operational member + has made a programming error that must surface immediately rather than silently + recognizing nothing. + tests: + - UnavailableRecognitionSession_StartAsync_Always_ThrowsSpeechRecognizerUnavailableException + - UnavailableRecognitionSession_GetResultsAsync_Always_ThrowsSpeechRecognizerUnavailableException + + - id: Speech-Recognition-UnavailableRecognitionSession-SafeNoOps + title: >- + The UnavailableRecognitionSession class shall treat StopAsync and disposal as + safe, idempotent no-ops that never throw and never invalidate the shared + fallback. + justification: | + A host that calls StopAsync unconditionally during teardown and wraps its + session in a disposal scope must be able to run that same code unchanged on a + machine where recognition is unavailable. + tests: + - UnavailableRecognitionSession_StopAsync_Always_IsSafeNoOp + - UnavailableRecognitionSession_DisposeAsync_CalledTwice_DoesNotThrow diff --git a/docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer-engine.yaml b/docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer-engine.yaml new file mode 100644 index 0000000..1a8fec0 --- /dev/null +++ b/docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer-engine.yaml @@ -0,0 +1,68 @@ +--- +# Software Unit Requirements for UnavailableSpeechRecognizerEngine +# +# Groups requirements for UnavailableSpeechRecognizerEngine and SpeechRecognizerUnavailableException +# into a single file, per design-documentation.md's allowance for small supporting types +# documented inline within their subsystem's design doc. Each unit's requirement subset is called +# out in its own subsection below so every unit still has requirements tracing to it individually. + +sections: + - title: Speech Requirements + sections: + - title: RecognitionSubsystem Requirements + sections: + - title: UnavailableSpeechRecognizerEngine Requirements + requirements: + - id: Speech-Recognition-UnavailableSpeechRecognizerEngine-ReportsUnavailable + title: >- + The UnavailableSpeechRecognizerEngine class shall report IsAvailable as false + and never throw merely for being obtained or held. + justification: | + A caller must always be able to safely obtain and hold a fallback engine, per + this library's "nothing throws at composition" decision. + tests: + - UnavailableSpeechRecognizerEngine_IsAvailable_Read_ReturnsFalse + + - id: Speech-Recognition-UnavailableSpeechRecognizerEngine-CreateSessionReturnsFallback + title: >- + The UnavailableSpeechRecognizerEngine class shall return + UnavailableRecognitionSession.Instance from CreateSessionAsync, without + faulting the returned task, for any non-null device with a non-cancelled token, + while still rejecting a null device with ArgumentNullException and an + already-cancelled token with OperationCanceledException. + justification: | + A caller composing against this fallback engine must be able to reuse the same + create-session call unchanged regardless of availability; only a programming + error (a null device argument or an already-cancelled token) should surface as + a fault, matching the ISpeechRecognizerEngine.CreateSessionAsync contract every + implementation - including this fallback - must honor. + tests: + - UnavailableSpeechRecognizerEngine_CreateSessionAsync_Always_ReturnsUnavailableSession + - UnavailableSpeechRecognizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException + - UnavailableSpeechRecognizerEngine_CreateSessionAsync_CancelledToken_ThrowsOperationCanceledException + + - id: Speech-Recognition-UnavailableSpeechRecognizerEngine-DisposalIdempotent + title: >- + The UnavailableSpeechRecognizerEngine class shall treat disposal as a safe, + idempotent no-op that never throws and never invalidates the shared fallback. + justification: | + A host that wraps every engine it obtains in a disposal scope must be able to + run that same code unchanged on a machine where recognition is unavailable, + including disposing the shared singleton fallback more than once. + tests: + - UnavailableSpeechRecognizerEngine_DisposeAsync_CalledTwice_DoesNotThrow + + - title: SpeechRecognizerUnavailableException Requirements + requirements: + - id: Speech-Recognition-SpeechRecognizerUnavailableException-Thrown + title: >- + The SpeechRecognizerUnavailableException class shall provide standard exception + constructors (default, message, message with inner exception). + justification: | + Standard exception conformance lets callers construct, catch, and inspect this + exception like any other .NET exception type, including wrapping an underlying + capture-device failure as the inner exception. + tests: + - SpeechRecognizerUnavailableException_Constructor_WithMessage_ExposesMessage + - SpeechRecognizerUnavailableException_Constructor_WithInnerException_ExposesBoth + - SpeechRecognizerUnavailableException_Constructor_Default_HasNonEmptyMessage diff --git a/docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer.yaml b/docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer.yaml deleted file mode 100644 index 35c4459..0000000 --- a/docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer.yaml +++ /dev/null @@ -1,63 +0,0 @@ ---- -# Software Unit Requirements for UnavailableSpeechRecognizer -# -# Groups requirements for UnavailableSpeechRecognizer and SpeechRecognizerUnavailableException -# into a single file, per design-documentation.md's allowance for small supporting types -# documented inline within their subsystem's design doc. Each unit's requirement subset is called -# out in its own subsection below so every unit still has requirements tracing to it individually. - -sections: - - title: Speech Requirements - sections: - - title: RecognitionSubsystem Requirements - sections: - - title: UnavailableSpeechRecognizer Requirements - requirements: - - id: Speech-Recognition-UnavailableSpeechRecognizer-ReportsUnavailable - title: >- - The UnavailableSpeechRecognizer class shall report IsAvailable as false and - never throw merely for being obtained or held. - justification: | - A caller must always be able to safely obtain and hold a fallback recognizer, - per this library's "nothing throws at composition" decision. - tests: - - UnavailableSpeechRecognizer_IsAvailable_Read_ReturnsFalse - - - id: Speech-Recognition-UnavailableSpeechRecognizer-OperationsThrow - title: >- - The UnavailableSpeechRecognizer class shall throw - SpeechRecognizerUnavailableException when Start or Stop is invoked. - justification: | - A caller that ignores the IsAvailable flag and invokes an operational member - has made a programming error that must surface immediately rather than silently - recognizing nothing. - tests: - - UnavailableSpeechRecognizer_Start_Always_ThrowsSpeechRecognizerUnavailableException - - UnavailableSpeechRecognizer_Stop_Always_ThrowsSpeechRecognizerUnavailableException - - - id: Speech-Recognition-UnavailableSpeechRecognizer-SafeNoOps - title: >- - The UnavailableSpeechRecognizer class shall treat result-event subscription and - disposal as safe no-ops that never throw and never invalidate the shared - fallback. - justification: | - A host that wires up its transcript handler and wraps its recognizer in a - disposal scope must be able to run that same code unchanged on a machine where - recognition is unavailable. - tests: - - UnavailableSpeechRecognizer_SubscriptionAndDispose_Always_AreSafeNoOps - - - title: SpeechRecognizerUnavailableException Requirements - requirements: - - id: Speech-Recognition-SpeechRecognizerUnavailableException-Thrown - title: >- - The SpeechRecognizerUnavailableException class shall provide standard exception - constructors (default, message, message with inner exception). - justification: | - Standard exception conformance lets callers construct, catch, and inspect this - exception like any other .NET exception type, including wrapping an underlying - capture-device failure as the inner exception. - tests: - - SpeechRecognizerUnavailableException_Constructor_WithMessage_ExposesMessage - - SpeechRecognizerUnavailableException_Constructor_WithInnerException_ExposesBoth - - SpeechRecognizerUnavailableException_Constructor_Default_HasNonEmptyMessage diff --git a/docs/reqstream/speech/synthesis-subsystem.yaml b/docs/reqstream/speech/synthesis-subsystem.yaml index 242e49e..de1d4d9 100644 --- a/docs/reqstream/speech/synthesis-subsystem.yaml +++ b/docs/reqstream/speech/synthesis-subsystem.yaml @@ -5,9 +5,14 @@ # decomposing the system-level Natural Language Audio Tag requirement in docs/reqstream/speech.yaml. # # Sub-phase 4a covers the closed tag vocabulary and the model-independent Layer 1 parser -# (AudioTagCatalog/AudioTagParser). Sub-phase 4b, described below, adds Layer 2 per-model -# rendering into a SpeechPlan, the chunked/streaming ISpeechSynthesizer pipeline, the real -# sherpa-onnx synthesis engine, and playback. +# (AudioTagCatalog/AudioTagParser). Sub-phase 4b adds Layer 2 per-model rendering into a +# SpeechPlan, the chunked synthesis pipeline, the real sherpa-onnx synthesis backend, and +# playback. A later redesign replaces the original synchronous ISpeechSynthesizer contract with +# an asynchronous Engine/Session split: ISpeechSynthesizerEngine (Layer 3, a loaded model) hosts +# at most one leased ISynthesisSession (Layer 5, bound to one playback device) at a time, adding +# session exclusivity, a lifecycle state machine, an overlap-not-allowed rule for concurrent +# operations on one session, a terminal Faulted state, and a dedicated-worker cancellation policy +# mirroring the identical recognition-direction mechanism. sections: - title: Speech Requirements @@ -85,68 +90,100 @@ sections: - id: Speech-Synthesis-MockableContract title: >- - The SynthesisSubsystem shall expose a mockable chunked/streaming text-to-speech - contract that reports availability, synthesizes and plays speech, and supports - deterministic cancellation, without exposing any inference-engine type. + The SynthesisSubsystem shall expose a mockable Engine/Session text-to-speech + contract (ISpeechSynthesizerEngine for a loaded model, ISynthesisSession bound to + one device) that reports availability, synthesizes and plays speech, and supports + deterministic cancellation, without exposing any inference backend type. justification: | This library's "engine backend stays swappable at the public API surface" decision requires that a future non-sherpa-onnx backend can replace the current one without a breaking API change, and hosts must be able to test their playback handling against the contract with no model, speakers, or native runtime present. + Splitting a Layer 3 engine from a Layer 5 session lets one loaded model be reused to + bind multiple sessions over its lifetime, one at a time, without reloading. children: - - Speech-Synthesis-ISpeechSynthesizer-AvailabilityFlag - - Speech-Synthesis-ISpeechSynthesizer-StreamingContract - - Speech-Synthesis-ISpeechSynthesizer-UnavailableOperationsThrow - - Speech-Synthesis-ISpeechSynthesizer-SynthesizedSpeechShape - - Speech-Synthesis-ISpeechSynthesizer-HotReuseAcrossTurns + - Speech-Synthesis-ISpeechSynthesizerEngine-AvailabilityFlag + - Speech-Synthesis-ISpeechSynthesizerEngine-SessionCreationContract + - Speech-Synthesis-ISpeechSynthesizerEngine-UnavailableOperationsThrow + - Speech-Synthesis-ISpeechSynthesizerEngine-SynthesizedSpeechShape + - Speech-Synthesis-ISynthesisSession-AvailabilityFlag + - Speech-Synthesis-ISynthesisSession-StateMachine + - Speech-Synthesis-ISynthesisSession-OverlapNotAllowed + - Speech-Synthesis-ISynthesisSession-StreamingContract + - Speech-Synthesis-ISynthesisSession-UnavailableOperationsThrow + - Speech-Synthesis-ISynthesisSession-HotReuseAcrossTurns tests: - - UnavailableSpeechSynthesizer_IsAvailable_Read_ReturnsFalse - - SynthesizeStreamAsync_PlainText_YieldsAudioSegment + - UnavailableSpeechSynthesizerEngine_IsAvailable_Read_ReturnsFalse + - SherpaOnnxSynthesisSession_SynthesizeAsync_PlainText_YieldsAudioSegment + + - id: Speech-Synthesis-EngineExclusivity + title: >- + The SynthesisSubsystem shall allow an ISpeechSynthesizerEngine to host at most one + live ISynthesisSession at a time, failing a concurrent session request fast with a + documented exception rather than queuing it, sharing the engine, or silently + misbehaving, and shall dispose an active leased session before disposing the engine + itself. + justification: | + A single loaded model (Layer 3 engine) owns one native inference context; two + concurrently live sessions bound to two different playback devices would race + against that same native context and corrupt either or both synthesis streams. + Failing fast gives a caller an unambiguous signal to dispose its current session + before requesting another, and disposing the active session first guarantees no + in-flight native call outlives the engine that owns its backend. + children: + - Speech-Synthesis-SherpaOnnxSpeechSynthesizerEngine-SessionExclusivity + - Speech-Synthesis-SherpaOnnxSpeechSynthesizerEngine-DisposalOrdering + - Speech-Synthesis-SynthesisEngineBusyException-Thrown + tests: + - SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_LeaseAlreadyHeld_ThrowsSynthesisEngineBusyException + - SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_AfterPriorSessionDisposed_SucceedsAgain + - SherpaOnnxSpeechSynthesizerEngine_DisposeAsync_WithActiveLeasedSession_DisposesSessionFirst - id: Speech-Synthesis-Composition title: >- - The SynthesisSubsystem shall compose a speech synthesizer for an installed - synthesis model and an audio playback device without ever throwing for an ordinary - machine state, while validating any supplied parameterValues up front: silently - ignoring an unrecognized parameter id (an ordinary cross-model-compatibility gap) - but throwing ArgumentException for a recognized parameter's invalid value (a caller - bug, not an ordinary machine state). + The SynthesisSubsystem shall asynchronously compose a speech synthesis engine for + an installed synthesis model without ever throwing for an ordinary machine state, + while validating any supplied parameterValues up front: silently ignoring an + unrecognized parameter id (an ordinary cross-model-compatibility gap) but throwing + ArgumentException for a recognized parameter's invalid value (a caller bug, not an + ordinary machine state). justification: | Composition happens at application start-up, where a model that has not been - downloaded yet, a machine with no speakers, and a machine without the speech - engine's native runtime are all ordinary states rather than programming errors, per - this library's "nothing throws at composition" decision. A caller explicitly - targeting a parameter it knows this model declares, with a value invalid for it, is - a distinct case - a programming error at the call site - so it must be surfaced - synchronously and loudly rather than folded into the same never-throw contract as - genuine machine-state unavailability. + downloaded yet and a machine without the speech engine's native runtime are both + ordinary states rather than programming errors, per this library's "nothing throws + at composition" decision. Composition no longer depends on playback-device + availability, since LoadAsync produces a Layer 3 engine not yet bound to any + device. A caller explicitly targeting a parameter it knows this model declares, + with a value invalid for it, is a distinct case - a programming error at the call + site - so it must be surfaced synchronously and loudly rather than folded into the + same never-throw contract as genuine machine-state unavailability. children: - - Speech-Synthesis-SpeechSynthesizerFactory-RealSynthesizer + - Speech-Synthesis-SpeechSynthesizerFactory-RealEngine - Speech-Synthesis-SpeechSynthesizerFactory-HonestFallbacks - Speech-Synthesis-SpeechSynthesizerFactory-EngineLoadFailureFallback - - Speech-Synthesis-SpeechSynthesizerFactory-ArgumentValidation + - Speech-Synthesis-SpeechSynthesizerFactory-ArgumentValidationAndCancellation - Speech-Synthesis-SpeechSynthesizerFactory-UnrecognizedParameterIdIgnored - Speech-Synthesis-SpeechSynthesizerFactory-InvalidRecognizedParameterValueThrows - - Speech-Synthesis-SynthesisEngine-MockableSeam - - Speech-Synthesis-SynthesisEngine-ModelOwnedConfiguration + - Speech-Synthesis-SynthesisBackend-MockableSeam + - Speech-Synthesis-SynthesisBackend-ModelOwnedConfiguration tests: - - SpeechSynthesizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer - - SpeechSynthesizerFactory_Create_ModelNotInstalled_ReturnsUnavailableSynthesizer - - SpeechSynthesizerFactory_Create_PlaybackDeviceUnavailable_ReturnsUnavailableSynthesizer - - SpeechSynthesizerFactory_Create_UnrecognizedParameterId_ComposesAndReportsInfo - - SpeechSynthesizerFactory_Create_RecognizedNumericParameterOutOfRange_Throws + - SpeechSynthesizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine + - SpeechSynthesizerFactory_LoadAsync_ModelNotInstalled_ReturnsUnavailableEngine + - SpeechSynthesizerFactory_LoadAsync_CancelledToken_ThrowsOperationCanceledException + - SpeechSynthesizerFactory_LoadAsync_UnrecognizedParameterId_ComposesAndReportsInfo + - SpeechSynthesizerFactory_LoadAsync_RecognizedNumericParameterOutOfRange_Throws - id: Speech-Synthesis-VoiceSelection title: >- The SynthesisSubsystem shall thread an optional, session-level parameterValues bag - from SpeechSynthesizerFactory.Create through to SherpaOnnxSpeechSynthesizer, which + from SpeechSynthesizerFactory.LoadAsync through to SherpaOnnxSynthesisSession, which shall resolve a real speaker id from it via ISynthesisModel.ResolveSpeakerId once per synthesized segment, replacing a previously hard-coded speakerId: 0, without regressing the independent, per-segment Natural Language Audio Tag speed/volume override mechanism. justification: | Phase 7b shipped the library's first real ISynthesisModel but left voice/speaker - selection entirely out of scope: GenerateSegment hard-coded speakerId: 0 + selection entirely out of scope: segment generation hard-coded speakerId: 0 regardless of any model-declared voice parameter, so a model such as SherpaOnnxKokoroEnglishSynthesisModel's 11-voice ChoiceParameter had no real path to change synthesized output. This requirement closes that gap for any model that @@ -154,21 +191,21 @@ sections: per-segment speed/volume convention correctly independent of each other. children: - Speech-Synthesis-SpeechSynthesizerFactory-ParameterValuesForwarding - - Speech-Synthesis-SherpaOnnxSpeechSynthesizer-SpeakerSelection + - Speech-Synthesis-SherpaOnnxSynthesisSession-SpeakerSelection tests: - - SpeechSynthesizerFactory_Create_ParameterValuesSupplied_ForwardedToSynthesizer - - SynthesizeStreamAsync_NoParameterValues_ResolvesDefaultSpeakerIdFromModel - - SynthesizeStreamAsync_ParameterValuesSupplied_ResolvesSpeakerIdFromBag - - SynthesizeStreamAsync_ParameterValuesSuppliedAlongsideSpeedTag_BothMechanismsApplyIndependently + - SpeechSynthesizerFactory_LoadAsync_ParameterValuesSupplied_ForwardedToEngine + - SherpaOnnxSynthesisSession_SynthesizeAsync_NoParameterValues_ResolvesDefaultSpeakerIdFromModel + - SherpaOnnxSynthesisSession_SynthesizeAsync_ParameterValuesSupplied_ResolvesSpeakerIdFromBag + - SherpaOnnxSynthesisSession_SynthesizeAsync_ParameterValuesSuppliedAlongsideSpeedTag_BothMechanismsApplyIndependently - id: Speech-Synthesis-ChunkedPipeline title: >- The SynthesisSubsystem shall chunk synthesis-ready text into sentence/clause-sized pieces and synthesize a later chunk while an earlier chunk is still playing. justification: | - The design requires low-latency streaming synthesis where playback of an early - chunk begins while later chunks are still being synthesized, rather than a caller - waiting for an entire utterance's inference to complete before hearing anything. + The design requires low-latency synthesis where playback of an early chunk begins + while later chunks are still being synthesized, rather than a caller waiting for an + entire utterance's inference to complete before hearing anything. children: - Speech-Synthesis-SentenceChunker-BoundarySplitting - Speech-Synthesis-SentenceChunker-BoundaryHandling @@ -176,50 +213,94 @@ sections: - Speech-Synthesis-SentenceChunker-ConsecutiveBoundaryMerging - Speech-Synthesis-SentenceChunker-DegeneratePunctuationMerging - Speech-Synthesis-SentenceChunker-EllipsisMetadata - - Speech-Synthesis-SherpaOnnxSpeechSynthesizer-ChunkedPipeline - - Speech-Synthesis-SherpaOnnxSpeechSynthesizer-PipelinedPlayback - - Speech-Synthesis-SherpaOnnxSpeechSynthesizer-GenuinePlaybackDrainWait + - Speech-Synthesis-SherpaOnnxSynthesisSession-ChunkedPipeline + - Speech-Synthesis-SherpaOnnxSynthesisSession-FullFidelitySegmentList + - Speech-Synthesis-SherpaOnnxSynthesisSession-PipelinedPlayback + - Speech-Synthesis-SherpaOnnxSynthesisSession-GenuinePlaybackDrainWait - Speech-Synthesis-PlaybackAudioResampler-RateConversion - Speech-Synthesis-PlaybackAudioResampler-Upmix - Speech-Synthesis-PlaybackAudioResampler-Composition tests: - SentenceChunker_Chunk_MultipleSentences_SplitsOnPrimaryBoundaries - - PlayStreamAsync_OrderedSegments_StartsWritesInOrderAndStops - - SynthesizeStreamAsync_LongMultiSentenceInput_ProducesOrderedSegmentsSequentially + - SherpaOnnxSynthesisSession_SpeakAsync_PlainText_StartsWritesAndStopsDevice + - SherpaOnnxSynthesisSession_SynthesizeAsync_LongMultiSentenceInput_ProducesOrderedSegmentsSequentially + - SherpaOnnxSynthesisSession_SynthesizeAsync_ReturnsFullFidelitySegmentListIncludingSilence + + - id: Speech-Synthesis-SessionOverlapAndLifecycle + title: >- + The SynthesisSubsystem shall reject an overlapping SpeakAsync/SynthesizeAsync call + on a session already performing one, raise an observable lifecycle state machine + over each operation, and make a faulted session's terminal state unrecoverable. + justification: | + Each ISynthesisSession drives exactly one in-flight native synthesis call and one + playback stream at a time; the design document's state diagram documents + Starting/Running/Stopping as denoting one in-flight operation rather than a + continuous stream, so a host can correctly render a "speaking" indicator and must + be told, synchronously and unambiguously, when it has attempted to start a second + operation before the first has finished, or when it must discard and replace a + session whose underlying native state is no longer trustworthy after a fault. + children: + - Speech-Synthesis-SherpaOnnxSynthesisSession-OverlapNotAllowed + - Speech-Synthesis-SherpaOnnxSynthesisSession-StateMachine + - Speech-Synthesis-SherpaOnnxSynthesisSession-HotReuse + - Speech-Synthesis-SherpaOnnxSynthesisSession-FaultedIsTerminal + - Speech-Synthesis-SynthesisSessionFaultedException-Thrown + tests: + - SherpaOnnxSynthesisSession_SpeakAsync_CalledWhileAlreadySpeaking_ThrowsInvalidOperationException + - SherpaOnnxSynthesisSession_StateChanged_OneSuccessfulOperation_RaisesExpectedTransitionsInOrder + - SherpaOnnxSynthesisSession_SpeakAsync_CalledTwiceOnSameInstance_ReusesSameInstanceWithoutReconstruction + - SherpaOnnxSynthesisSession_SynthesizeAsync_AfterFault_ThrowsSynthesisSessionFaultedException - id: Speech-Synthesis-FaultContainment title: >- The SynthesisSubsystem shall contain and report faults raised while synthesizing or playing speech, and shall support deterministic cancellation of an in-flight speak - session, rather than hanging or crashing the process. + operation - including a dedicated-worker cooperative-cancel-then-abandon policy for + a native call that does not itself observe cancellation promptly - rather than + hanging or crashing the process. justification: | A model that fails mid-utterance, a dropped playback device, or a user interrupting - speech (barge-in) are all ordinary events during a long-running synthesis session - and must never leave the caller's task hanging or tear down the process, per - this library's fault-containment requirements mirrored from the recognition - direction. + speech (barge-in) are all ordinary events during a synthesis operation and must + never leave the caller's task hanging or tear down the process, per this library's + fault-containment requirements mirrored from the recognition direction. A native + call is not guaranteed to honor .NET cancellation tokens at all, so the dedicated + worker must bound how long a caller can be blocked by one. children: - - Speech-Synthesis-SherpaOnnxSpeechSynthesizer-FaultContainment - - Speech-Synthesis-SherpaOnnxSpeechSynthesizer-CancellationAndLifecycle + - Speech-Synthesis-SherpaOnnxSynthesisSession-FaultContainment + - Speech-Synthesis-SherpaOnnxSynthesisSession-CancellationAndLifecycle + - Speech-Synthesis-DedicatedWorker-CooperativeCancellation + - Speech-Synthesis-DedicatedWorker-NonCooperativeAbandonment + - Speech-Synthesis-DedicatedWorker-LongRunningExecution tests: - - SynthesizeStreamAsync_EngineThrows_ReportsFaultAndPropagatesToCaller - - Stop_WhileSpeaking_CancelsInFlightSessionOnlyAfterInFlightGenerateReturns + - SherpaOnnxSynthesisSession_SynthesizeAsync_BackendThrows_ReportsFaultAndTransitionsToFaulted + - SherpaOnnxSynthesisSession_StopAsync_WhileSpeaking_DoesNotCompleteUntilInFlightGenerateReturns + - DedicatedWorker_Run_CooperativeCancellation_CompletesPromptly + - DedicatedWorker_Run_NonCooperativeDelegate_AbandonsAfterTimeoutAndReportsDiagnostics - id: Speech-Synthesis-UnavailableFallback title: >- - The SynthesisSubsystem shall provide an honest unavailable synthesizer that reports - unavailability rather than throwing, and throws a documented exception only when an - operational member is actually invoked. + The SynthesisSubsystem shall provide an honest unavailable engine and an honest + unavailable session that report unavailability rather than throwing merely for + being obtained or held, and throw a documented exception only when an operational + member is actually invoked. justification: | - Per this library's "nothing throws at composition" decision, obtaining and - holding a fallback synthesizer must never throw; only misuse of an already - unavailable synthesizer is a programming error that should surface as an exception. + Per this library's "nothing throws at composition" decision, obtaining and holding + a fallback engine or session must never throw; only misuse of an already + unavailable engine or session is a programming error that should surface as an + exception. Binding a device to an already unavailable engine via CreateSessionAsync + is itself an ordinary (if useless) composition rather than an error. children: - - Speech-Synthesis-UnavailableSpeechSynthesizer-ReportsUnavailable - - Speech-Synthesis-UnavailableSpeechSynthesizer-OperationsThrow - - Speech-Synthesis-UnavailableSpeechSynthesizer-SafeDispose + - Speech-Synthesis-UnavailableSpeechSynthesizerEngine-ReportsUnavailable + - Speech-Synthesis-UnavailableSpeechSynthesizerEngine-CreateSessionSucceeds + - Speech-Synthesis-UnavailableSpeechSynthesizerEngine-OperationsThrow + - Speech-Synthesis-UnavailableSpeechSynthesizerEngine-SafeDispose + - Speech-Synthesis-UnavailableSynthesisSession-ReportsUnavailable + - Speech-Synthesis-UnavailableSynthesisSession-OperationsThrow + - Speech-Synthesis-UnavailableSynthesisSession-SafeDispose - Speech-Synthesis-SpeechSynthesizerUnavailableException-Thrown tests: - - UnavailableSpeechSynthesizer_IsAvailable_Read_ReturnsFalse - - UnavailableSpeechSynthesizer_SynthesizeStreamAsync_Always_ThrowsSpeechSynthesizerUnavailableException - - UnavailableSpeechSynthesizer_Stop_Always_ThrowsSpeechSynthesizerUnavailableException + - UnavailableSpeechSynthesizerEngine_IsAvailable_Read_ReturnsFalse + - UnavailableSpeechSynthesizerEngine_CreateSessionAsync_Always_ReturnsUnavailableSession + - UnavailableSpeechSynthesizerEngine_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException + - UnavailableSynthesisSession_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException + - UnavailableSynthesisSession_State_Read_ReturnsCreated diff --git a/docs/reqstream/speech/synthesis-subsystem/dedicated-worker.yaml b/docs/reqstream/speech/synthesis-subsystem/dedicated-worker.yaml new file mode 100644 index 0000000..ea09dce --- /dev/null +++ b/docs/reqstream/speech/synthesis-subsystem/dedicated-worker.yaml @@ -0,0 +1,57 @@ +--- +# Software Unit Requirements for DedicatedWorker (synthesis's own copy) +# +# Describes the design-level behavior of the internal DedicatedWorker helper that runs a native +# synthesis call on a dedicated, long-running thread and applies a cooperative-cancel-then-abandon +# policy, so a native call that ignores its cancellation token can never hang the caller +# indefinitely. This is synthesis's own copy of the identical recognition-direction helper, kept +# as a separate, independently testable type per this library's internal-class-per-subsystem +# convention rather than a shared cross-subsystem dependency. + +sections: + - title: Speech Requirements + sections: + - title: SynthesisSubsystem Requirements + sections: + - title: DedicatedWorker Requirements + requirements: + - id: Speech-Synthesis-DedicatedWorker-CooperativeCancellation + title: >- + The DedicatedWorker class shall complete promptly once a cooperative delegate - + one that observes and honors its supplied cancellation token - responds to + cancellation, without waiting out its abandon timeout. + justification: | + Most native calls can be made to observe cancellation promptly; for these, the + worker must return control to its caller as soon as the delegate itself + finishes unwinding, rather than imposing a fixed worst-case delay on every + cancellation regardless of how quickly the delegate actually responds. + tests: + - DedicatedWorker_Run_CooperativeCancellation_CompletesPromptly + + - id: Speech-Synthesis-DedicatedWorker-NonCooperativeAbandonment + title: >- + The DedicatedWorker class shall abandon a non-cooperative delegate - one that + ignores its supplied cancellation token - once a configurable abandon timeout + elapses, reporting a Warning-level diagnostic naming the abandonment, and shall + still return control to its caller as cancelled rather than hanging + indefinitely. + justification: | + A native call is not guaranteed to honor .NET cancellation tokens at all; the + worker must bound how long a caller can be blocked by such a call, trading an + orphaned background thread (reported so a host can investigate/monitor it) for + guaranteed forward progress of the caller, per this library's fault-containment + requirements. + tests: + - DedicatedWorker_Run_NonCooperativeDelegate_AbandonsAfterTimeoutAndReportsDiagnostics + + - id: Speech-Synthesis-DedicatedWorker-LongRunningExecution + title: >- + The DedicatedWorker class shall always execute its delegate on a + TaskCreationOptions.LongRunning task, never on a pooled thread-pool thread. + justification: | + A native synthesis call can legitimately run for a long time (an entire + utterance's inference); running it on a pooled thread risks starving the + thread pool for every other unrelated piece of work sharing it, so the worker + must always request a dedicated, long-running thread instead. + tests: + - DedicatedWorker_Run_UsesLongRunningTaskCreationOption diff --git a/docs/reqstream/speech/synthesis-subsystem/i-speech-synthesizer-engine.yaml b/docs/reqstream/speech/synthesis-subsystem/i-speech-synthesizer-engine.yaml new file mode 100644 index 0000000..8fd2f43 --- /dev/null +++ b/docs/reqstream/speech/synthesis-subsystem/i-speech-synthesizer-engine.yaml @@ -0,0 +1,75 @@ +--- +# Software Unit Requirements for ISpeechSynthesizerEngine +# +# These requirements describe the design-level behavior of the ISpeechSynthesizerEngine contract +# (Layer 3, a loaded model) and its adjacent SynthesizedSpeech value type, decomposing the parent +# SynthesisSubsystem requirements. Splits the former ISpeechSynthesizer requirements between this +# file and i-synthesis-session.yaml to match the async Engine/Session redesign, mirroring +# i-speech-recognizer-engine.yaml's identical split. + +sections: + - title: Speech Requirements + sections: + - title: SynthesisSubsystem Requirements + sections: + - title: ISpeechSynthesizerEngine Requirements + requirements: + - id: Speech-Synthesis-ISpeechSynthesizerEngine-AvailabilityFlag + title: >- + The ISpeechSynthesizerEngine interface shall expose an IsAvailable flag that + never throws to read. + justification: | + Callers must be able to check availability before invoking an operational + member, without risk of the check itself failing. + tests: + - UnavailableSpeechSynthesizerEngine_IsAvailable_Read_ReturnsFalse + - SherpaOnnxSpeechSynthesizerEngine_IsAvailable_Always_ReturnsTrue + + - id: Speech-Synthesis-ISpeechSynthesizerEngine-SessionCreationContract + title: >- + The ISpeechSynthesizerEngine interface shall define an asynchronous + CreateSessionAsync method that binds a single ISynthesisSession to one + supplied audio playback device, plus one-shot SpeakAsync/SynthesizeAsync + convenience overloads that each create and dispose a session internally, and + disposal of the resources an engine holds, without exposing any inference + backend or native audio type. + justification: | + This library's "engine backend stays swappable at the public API surface" + decision requires that a future non-sherpa-onnx backend can replace the current + one without a breaking API change. Separating a Layer 3 engine (the loaded + model) from a Layer 5 session (bound to one device) lets a single engine host + exactly one in-flight session while still offering a convenient one-shot API for + the common case of a single speak call against a single device. + tests: + - SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_NoLeaseHeld_ReturnsRealSession + - SherpaOnnxSpeechSynthesizerEngine_SpeakAsync_CalledTwice_CreatesAndDisposesASessionEachTime + - SherpaOnnxSpeechSynthesizerEngine_SynthesizeAsync_NoDeviceSupplied_ReturnsSegments + + - id: Speech-Synthesis-ISpeechSynthesizerEngine-UnavailableOperationsThrow + title: >- + Every operational member of the ISpeechSynthesizerEngine contract + (SpeakAsync, SynthesizeAsync) shall throw a documented exception when invoked + on an unavailable engine, while CreateSessionAsync shall still succeed and + return the shared unavailable session rather than throwing. + justification: | + Hosts must be able to test their playback logic against the contract with no + model, no speakers, and no native runtime present, so an unavailable engine's + one-shot operational members must fail loudly and consistently rather than + silently doing nothing, while binding a device to an already unavailable engine + is an ordinary (if useless) composition rather than an error. + tests: + - UnavailableSpeechSynthesizerEngine_CreateSessionAsync_Always_ReturnsUnavailableSession + - UnavailableSpeechSynthesizerEngine_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException + - UnavailableSpeechSynthesizerEngine_SynthesizeAsync_Always_ThrowsSpeechSynthesizerUnavailableException + + - id: Speech-Synthesis-ISpeechSynthesizerEngine-SynthesizedSpeechShape + title: >- + A synthesized speech segment shall carry its normalized audio samples, the rate + they were produced at, and the real pre/post silence to play alongside them. + justification: | + The design requires chunked, low-latency synthesis in which each segment is + played back independently and in order; a caller playing segments as they + arrive needs each segment's own silence and rate to play it correctly without + consulting any other segment. + tests: + - SherpaOnnxSynthesisSession_SynthesizeAsync_ReturnsFullFidelitySegmentListIncludingSilence diff --git a/docs/reqstream/speech/synthesis-subsystem/i-speech-synthesizer.yaml b/docs/reqstream/speech/synthesis-subsystem/i-speech-synthesizer.yaml deleted file mode 100644 index 2f3d4a4..0000000 --- a/docs/reqstream/speech/synthesis-subsystem/i-speech-synthesizer.yaml +++ /dev/null @@ -1,83 +0,0 @@ ---- -# Software Unit Requirements for ISpeechSynthesizer -# -# These requirements describe the design-level behavior of the ISpeechSynthesizer contract and -# its adjacent SynthesizedSpeech value type, decomposing the parent SynthesisSubsystem -# requirements. Mirrors i-speech-recognizer.yaml's identical Phase 3 pattern. - -sections: - - title: Speech Requirements - sections: - - title: SynthesisSubsystem Requirements - sections: - - title: ISpeechSynthesizer Requirements - requirements: - - id: Speech-Synthesis-ISpeechSynthesizer-AvailabilityFlag - title: >- - The ISpeechSynthesizer interface shall expose an IsAvailable flag that never - throws to read. - justification: | - Callers must be able to check availability before invoking an operational - member, without risk of the check itself failing. - tests: - - UnavailableSpeechSynthesizer_IsAvailable_Read_ReturnsFalse - - IsAvailable_Always_ReturnsTrue - - - id: Speech-Synthesis-ISpeechSynthesizer-StreamingContract - title: >- - The ISpeechSynthesizer interface shall define a chunked/streaming - synthesize-and-play contract (SynthesizeStreamAsync, PlayStreamAsync, - SpeakAsync), a Stop method to cancel an in-flight session deterministically, and - disposal of the resources a synthesizer holds, without exposing any inference - engine or native audio type. - justification: | - This library's "engine backend stays swappable at the public API surface" - decision requires that a future non-sherpa-onnx backend can replace the current - one without a breaking API change. - tests: - - SynthesizeStreamAsync_PlainText_YieldsAudioSegment - - Stop_WhileSpeaking_CancelsInFlightSessionOnlyAfterInFlightGenerateReturns - - Stop_NoSessionInFlight_IsNoOp - - - id: Speech-Synthesis-ISpeechSynthesizer-UnavailableOperationsThrow - title: >- - Every operational member of the ISpeechSynthesizer contract - (SynthesizeStreamAsync, PlayStreamAsync, SpeakAsync, Stop) shall throw a - documented exception when invoked on an unavailable synthesizer, rather than - silently no-op-ing or misbehaving. - justification: | - Hosts must be able to test their playback logic against the contract with no - model, no speakers, and no native runtime present, so an unavailable - synthesizer's operational members must fail loudly and consistently rather - than silently doing nothing. - tests: - - UnavailableSpeechSynthesizer_SynthesizeStreamAsync_Always_ThrowsSpeechSynthesizerUnavailableException - - UnavailableSpeechSynthesizer_PlayStreamAsync_Always_ThrowsSpeechSynthesizerUnavailableException - - UnavailableSpeechSynthesizer_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException - - UnavailableSpeechSynthesizer_Stop_Always_ThrowsSpeechSynthesizerUnavailableException - - - id: Speech-Synthesis-ISpeechSynthesizer-SynthesizedSpeechShape - title: >- - A synthesized speech segment shall carry its normalized audio samples, the rate - they were produced at, and the real pre/post silence to play alongside them. - justification: | - The design requires chunked, low-latency streaming synthesis and playback - in which each segment is played back independently and in order; a caller - playing segments as they arrive needs each segment's own silence and rate to - play it correctly without consulting any other segment. - tests: - - SynthesizeStreamAsync_TextWithPauseTag_YieldsSilenceSegmentWithNoEngineCall - - - id: Speech-Synthesis-ISpeechSynthesizer-HotReuseAcrossTurns - title: >- - The ISpeechSynthesizer interface shall support many independent - synthesize/play sessions on the same instance without requiring the synthesizer - to be disposed and reconstructed between sessions. - justification: | - Obtaining a synthesizer from SpeechSynthesizerFactory is the expensive step - it - loads the model into native memory - while a synthesize/play session is cheap. - A host doing repeated, low-latency synthesis (for example, many turns of a - voice conversation) must be able to construct one synthesizer once and reuse it - across turns rather than reloading the model per turn. - tests: - - PlayStreamAsync_CalledTwiceOnSameInstance_ReusesSameInstanceWithoutReconstruction diff --git a/docs/reqstream/speech/synthesis-subsystem/i-synthesis-session.yaml b/docs/reqstream/speech/synthesis-subsystem/i-synthesis-session.yaml new file mode 100644 index 0000000..2d2f743 --- /dev/null +++ b/docs/reqstream/speech/synthesis-subsystem/i-synthesis-session.yaml @@ -0,0 +1,115 @@ +--- +# Software Unit Requirements for ISynthesisSession +# +# These requirements describe the design-level behavior of the ISynthesisSession contract +# (Layer 5, bound to one device), its SynthesisSessionState/SessionStateChangedEventArgs +# supporting types, and the overlap-not-allowed rule, decomposing the parent SynthesisSubsystem +# requirements. Splits the former ISpeechSynthesizer requirements between this file and +# i-speech-synthesizer-engine.yaml to match the async Engine/Session redesign, mirroring +# i-synthesis-session.yaml's recognition-direction counterpart. + +sections: + - title: Speech Requirements + sections: + - title: SynthesisSubsystem Requirements + sections: + - title: ISynthesisSession Requirements + requirements: + - id: Speech-Synthesis-ISynthesisSession-AvailabilityFlag + title: >- + The ISynthesisSession interface shall expose an IsAvailable flag that never + throws to read, reporting false for the shared unavailable session and false + once a real session has faulted or been disposed. + justification: | + Callers must be able to check availability before invoking an operational + member, without risk of the check itself failing, and a session's availability + must honestly reflect that it can no longer perform a speak/synthesize + operation once it has faulted or been disposed. + tests: + - UnavailableSynthesisSession_IsAvailable_Read_ReturnsFalse + - SherpaOnnxSynthesisSession_IsAvailable_BeforeAndAfterDispose_ReflectsLifecycle + + - id: Speech-Synthesis-ISynthesisSession-StateMachine + title: >- + The ISynthesisSession interface shall expose a State property and a + StateChanged event reporting a SynthesisSessionState lifecycle of Created, + Starting, Running, Stopping, Stopped, and the terminal Faulted, where + Starting/Running/Stopping denote one in-flight Speak/Synthesize operation + rather than a continuous stream, raising the documented transition sequence for + both a successful and a faulting operation, and isolating a subscriber's + handler exception rather than propagating it. + justification: | + A host rendering UI state (for example a "speaking" indicator) needs an + authoritative, observable state machine rather than inferring state from + whether an awaited task has completed, and a misbehaving subscriber must never + be able to destabilize the session it is observing. + tests: + - UnavailableSynthesisSession_State_Read_ReturnsCreated + - UnavailableSynthesisSession_StateChanged_SubscribeAndUnsubscribe_DoesNotThrow + - SherpaOnnxSynthesisSession_StateChanged_OneSuccessfulOperation_RaisesExpectedTransitionsInOrder + - SherpaOnnxSynthesisSession_StateChanged_HandlerThrows_IsIsolatedAndDoesNotPropagate + + - id: Speech-Synthesis-ISynthesisSession-OverlapNotAllowed + title: >- + The ISynthesisSession interface shall reject a SpeakAsync or SynthesizeAsync + call made while an earlier call on the same session is still in flight, by + throwing InvalidOperationException immediately, rather than queuing it or + running both concurrently. + justification: | + A session owns exactly one bound playback device and one in-flight native + inference call at a time; allowing a second overlapping call would either + corrupt playback ordering or race two calls against the same native resources. + Rejecting the second call immediately gives a caller a clear, synchronous signal + to await the first operation (or call StopAsync) before starting another, rather + than silently misbehaving. + tests: + - SherpaOnnxSynthesisSession_SpeakAsync_CalledWhileAlreadySpeaking_ThrowsInvalidOperationException + - SherpaOnnxSynthesisSession_SynthesizeAsync_CalledWhileAlreadySpeaking_ThrowsInvalidOperationException + + - id: Speech-Synthesis-ISynthesisSession-StreamingContract + title: >- + The ISynthesisSession interface shall define SpeakAsync/SynthesizeAsync + operations, a StopAsync method to cancel an in-flight operation + deterministically, and disposal of the resources a session holds, without + exposing any inference backend or native audio type. + justification: | + This library's "engine backend stays swappable at the public API surface" + decision requires that a future non-sherpa-onnx backend can replace the current + one without a breaking API change. + tests: + - SherpaOnnxSynthesisSession_SynthesizeAsync_PlainText_YieldsAudioSegment + - SherpaOnnxSynthesisSession_StopAsync_WhileSpeaking_DoesNotCompleteUntilInFlightGenerateReturns + - SherpaOnnxSynthesisSession_StopAsync_NoOperationInFlight_IsNoOp + + - id: Speech-Synthesis-ISynthesisSession-UnavailableOperationsThrow + title: >- + Every operational member of the ISynthesisSession contract (SpeakAsync, + SynthesizeAsync) shall throw a documented exception when invoked on an + unavailable session, rather than silently no-op-ing or misbehaving, while + StopAsync and disposal remain safe no-ops. + justification: | + Hosts must be able to test their playback logic against the contract with no + model, no speakers, and no native runtime present, so an unavailable session's + operational members must fail loudly and consistently rather than silently + doing nothing. + tests: + - UnavailableSynthesisSession_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException + - UnavailableSynthesisSession_SpeakAsync_NullText_ThrowsArgumentNullException + - UnavailableSynthesisSession_SynthesizeAsync_Always_ThrowsSpeechSynthesizerUnavailableException + - UnavailableSynthesisSession_SynthesizeAsync_NullText_ThrowsArgumentNullException + - UnavailableSynthesisSession_StopAsync_NoSessionInFlight_IsNoOp + - UnavailableSynthesisSession_DisposeAsync_CalledTwice_DoesNotThrow + + - id: Speech-Synthesis-ISynthesisSession-HotReuseAcrossTurns + title: >- + The ISynthesisSession interface shall support many independent, non-overlapping + speak/synthesize operations on the same session instance without requiring the + session to be disposed and reconstructed between operations. + justification: | + Obtaining an engine from SpeechSynthesizerFactory is the expensive step - it + loads the model into native memory - while a session bound to a device is cheap + to create and reuse. A host doing repeated, low-latency synthesis (for example, + many turns of a voice conversation) must be able to construct one session once + and reuse it across turns rather than reloading the model per turn. + tests: + - SherpaOnnxSynthesisSession_SpeakAsync_CalledTwiceOnSameInstance_ReusesSameInstanceWithoutReconstruction diff --git a/docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.yaml b/docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.yaml similarity index 62% rename from docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.yaml rename to docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.yaml index c2a94d1..d6b7844 100644 --- a/docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.yaml +++ b/docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.yaml @@ -1,92 +1,177 @@ --- -# Software Unit Requirements for the streaming synthesizer implementation +# Software Unit Requirements for the real synthesis session implementation # -# Groups requirements for SherpaOnnxSpeechSynthesizer together with the synthesis-engine seam, -# the Layer 2 rendering pipeline (IModelCapabilityProfile/DefaultModelCapabilityProfile, -# SpeechSegment/SpeechPlan), the SentenceChunker, and the PlaybackAudioResampler, mirroring how +# Groups requirements for SherpaOnnxSynthesisSession together with the synthesis-backend seam +# (ISynthesisBackend/ISynthesisBackendFactory), SynthesisSessionFaultedException, the Layer 2 +# rendering pipeline (IModelCapabilityProfile/DefaultModelCapabilityProfile, SpeechSegment/ +# SpeechPlan), the SentenceChunker, and the PlaybackAudioResampler, mirroring how # sherpa-onnx-speech-recognizer.yaml groups a cohesive implementation unit. Each unit's # requirement subset is called out in its own subsection below so every unit still has -# requirements tracing to it individually. +# requirements tracing to it individually. Renamed/restructured from the former +# sherpa-onnx-speech-synthesizer.yaml for the async Engine/Session redesign: the former +# "SherpaOnnxSpeechSynthesizer Requirements" section becomes "SherpaOnnxSynthesisSession +# Requirements" (adding overlap-rule and faulted-terminal-state requirements), and the former +# "SynthesisEngine Requirements" section becomes "SynthesisBackend Requirements". sections: - title: Speech Requirements sections: - title: SynthesisSubsystem Requirements sections: - - title: SherpaOnnxSpeechSynthesizer Requirements + - title: SherpaOnnxSynthesisSession Requirements requirements: - - id: Speech-Synthesis-SherpaOnnxSpeechSynthesizer-ChunkedPipeline + - id: Speech-Synthesis-SherpaOnnxSynthesisSession-ChunkedPipeline title: >- - The SherpaOnnxSpeechSynthesizer class shall normalize, tag-parse, and Layer 2 - render input text into an ordered SpeechPlan, then synthesize and yield each - resulting segment as it becomes ready rather than waiting for the whole plan to - finish. + The SherpaOnnxSynthesisSession class shall normalize, tag-parse, and Layer 2 + render input text into an ordered SpeechPlan, then synthesize each resulting + segment in order, making each one available to playback as soon as it is ready + rather than waiting for the whole plan to finish synthesizing. SynthesizeAsync + returns the complete, ordered segment list only once every segment has been + synthesized; this chunked pipelining is an internal synthesis/playback + optimization, not a streaming result contract exposed to SynthesizeAsync's + caller. justification: | - The design requires chunked, low-latency streaming synthesis so playback of - an early sentence can start while a later sentence is still being synthesized, - rather than a host waiting for an entire utterance's inference to complete - before hearing anything. + The design requires chunked, low-latency synthesis so playback of an early + sentence can start while a later sentence is still being synthesized, rather + than a host waiting for an entire utterance's inference to complete before + hearing anything. SpeakAsync relies on this internal pipelining directly; + SynthesizeAsync still returns a single materialized, full-fidelity segment list + (see Speech-Synthesis-SherpaOnnxSynthesisSession-FullFidelitySegmentList) since a + caller inspecting or re-encoding the complete synthesized output needs every + segment together, not an incremental result surface. tests: - - SynthesizeStreamAsync_PlainText_YieldsAudioSegment - - SynthesizeStreamAsync_TextWithPauseTag_YieldsSilenceSegmentWithNoEngineCall + - SherpaOnnxSynthesisSession_SynthesizeAsync_PlainText_YieldsAudioSegment + - SherpaOnnxSynthesisSession_SynthesizeAsync_LongMultiSentenceInput_ProducesOrderedSegmentsSequentially - - id: Speech-Synthesis-SherpaOnnxSpeechSynthesizer-PipelinedPlayback + - id: Speech-Synthesis-SherpaOnnxSynthesisSession-FullFidelitySegmentList title: >- - The SherpaOnnxSpeechSynthesizer class shall start the playback device once, play + SynthesizeAsync shall return the complete, ordered list of every synthesized + speech segment for the input text, including pure-silence pause segments, + rather than only speech-bearing segments. + justification: | + A caller that wants the full fidelity of the rendered SpeechPlan - for example + to re-encode or inspect the synthesized output rather than merely hearing it - + must see every segment the rendering pipeline produced, including a pause tag's + silence, not a filtered subset. + tests: + - SherpaOnnxSynthesisSession_SynthesizeAsync_ReturnsFullFidelitySegmentListIncludingSilence + + - id: Speech-Synthesis-SherpaOnnxSynthesisSession-PipelinedPlayback + title: >- + The SherpaOnnxSynthesisSession class shall start the playback device once, play each segment's pre-silence, resampled audio, and post-silence in order, and stop - the device once playback of the stream ends or faults. + the device once playback of the operation ends or faults. justification: | - This library's synthesize-while-playing-previous chunk behavior requires that + This library's synthesize-while-playing-previous-chunk behavior requires that the playback device is driven continuously across segments rather than being restarted per segment, while still guaranteeing the device is released when the - session ends for any reason. + operation ends for any reason. tests: - - PlayStreamAsync_OrderedSegments_StartsWritesInOrderAndStops - - PlayStreamAsync_PlaybackDeviceWriteThrows_PropagatesAndStillStopsDevice + - SherpaOnnxSynthesisSession_SpeakAsync_PlainText_StartsWritesAndStopsDevice + - SherpaOnnxSynthesisSession_SpeakAsync_PlaybackDeviceWriteThrows_PropagatesAndStillStopsDevice + - SherpaOnnxSynthesisSession_SpeakAsync_PlaybackDeviceStartThrows_StillCallsStop - - id: Speech-Synthesis-SherpaOnnxSpeechSynthesizer-GenuinePlaybackDrainWait + - id: Speech-Synthesis-SherpaOnnxSynthesisSession-GenuinePlaybackDrainWait title: >- - The SherpaOnnxSpeechSynthesizer class shall wait for the playback device to + The SherpaOnnxSynthesisSession class shall wait for the playback device to report a fully drained pending-sample count, plus a small safety margin, before - stopping the device at the end of a PlayStreamAsync session, and shall exit that + stopping the device at the end of a SpeakAsync operation, and shall exit that wait promptly on cancellation rather than waiting out the full drain. justification: | Confirmed via user testing: pressing "Play" in the demo's TTS panel flashed the status from "Idle" to "Playing" for about a quarter second before flashing back - to "Idle" with no audible sound, on every model. Root cause was that - PlayStreamAsync treated "every segment enqueued" as "finished playing" and + to "Idle" with no audible sound, on every model. Root cause was that the + playback loop treated "every segment enqueued" as "finished playing" and immediately stopped the playback device (discarding whatever audio the hardware had not yet actually rendered) as soon as the fire-and-forget Write calls returned, rather than once the hardware had genuinely finished. Waiting for IAudioPlaybackDevice.PendingSampleCount to reach zero closes that gap without regressing existing cancellation behavior. tests: - - PlayStreamAsync_PlaybackDeviceReportsPendingSamples_WaitsForDrainBeforeStopping + - SherpaOnnxSynthesisSession_SpeakAsync_PlaybackDeviceReportsPendingSamples_WaitsForDrainBeforeStopping - - id: Speech-Synthesis-SherpaOnnxSpeechSynthesizer-FaultContainment + - id: Speech-Synthesis-SherpaOnnxSynthesisSession-OverlapNotAllowed title: >- - The SherpaOnnxSpeechSynthesizer class shall surface a synthesis-engine fault or + The SherpaOnnxSynthesisSession class shall throw InvalidOperationException + synchronously when SpeakAsync or SynthesizeAsync is called while an earlier call + on the same session is still in flight. + justification: | + A session drives exactly one in-flight native synthesis call and one playback + stream at a time; a second overlapping call would race the same backend and + playback device. Rejecting the overlapping call immediately, rather than + queuing or interleaving it, keeps the session's behavior simple and + deterministic for callers. + tests: + - SherpaOnnxSynthesisSession_SpeakAsync_CalledWhileAlreadySpeaking_ThrowsInvalidOperationException + - SherpaOnnxSynthesisSession_SynthesizeAsync_CalledWhileAlreadySpeaking_ThrowsInvalidOperationException + + - id: Speech-Synthesis-SherpaOnnxSynthesisSession-StateMachine + title: >- + The SherpaOnnxSynthesisSession class shall raise StateChanged with the + documented SynthesisSessionState transition sequence for a successful operation + (Created, Starting, Running, Stopping, Stopped) and for a faulting operation + (ending in Faulted), isolating a subscriber's handler exception rather than + propagating it or destabilizing the session. + justification: | + A host rendering UI state needs an authoritative, observable state machine, and + a misbehaving subscriber must never be able to break the session it is merely + observing. + tests: + - SherpaOnnxSynthesisSession_StateChanged_OneSuccessfulOperation_RaisesExpectedTransitionsInOrder + - SherpaOnnxSynthesisSession_StateChanged_HandlerThrows_IsIsolatedAndDoesNotPropagate + + - id: Speech-Synthesis-SherpaOnnxSynthesisSession-HotReuse + title: >- + The SherpaOnnxSynthesisSession class shall support repeated, non-overlapping + SpeakAsync/SynthesizeAsync calls on the same instance without requiring + reconstruction between calls. + justification: | + A host doing repeated, low-latency synthesis (for example, many turns of a + voice conversation) must be able to reuse one session across turns rather than + recreating it per turn. + tests: + - SherpaOnnxSynthesisSession_SpeakAsync_CalledTwiceOnSameInstance_ReusesSameInstanceWithoutReconstruction + + - id: Speech-Synthesis-SherpaOnnxSynthesisSession-FaultedIsTerminal + title: >- + The SherpaOnnxSynthesisSession class shall transition to the terminal Faulted + state and report the fault through diagnostics when a non-cancellation failure + occurs during SpeakAsync/SynthesizeAsync, and shall throw + SynthesisSessionFaultedException wrapping that same fault from every subsequent + call once faulted, rather than attempting to recover or silently resetting. + justification: | + Once a session has faulted, its underlying native state is no longer trusted to + be consistent; allowing further calls to silently retry could produce corrupted + or misleading output. A caller must dispose a faulted session and obtain a new + one rather than being misled into believing recovery occurred. + tests: + - SherpaOnnxSynthesisSession_SynthesizeAsync_BackendThrows_ReportsFaultAndTransitionsToFaulted + - SherpaOnnxSynthesisSession_SynthesizeAsync_AfterFault_ThrowsSynthesisSessionFaultedException + + - id: Speech-Synthesis-SherpaOnnxSynthesisSession-FaultContainment + title: >- + The SherpaOnnxSynthesisSession class shall surface a synthesis-backend fault or an unavailable playback device honestly to the caller rather than hanging or crashing the process. justification: | A model that fails mid-utterance or a playback device that drops must fail the - caller's awaited task promptly so a host can report the failure, per - this library's "fail honestly, never crash or hang" requirement for the - synthesis pipeline. + caller's awaited task promptly so a host can report the failure, per this + library's "fail honestly, never crash or hang" requirement for the synthesis + pipeline. tests: - - SynthesizeStreamAsync_EngineThrows_ReportsFaultAndPropagatesToCaller - - PlayStreamAsync_PlaybackDeviceUnavailable_ThrowsRatherThanHanging + - SherpaOnnxSynthesisSession_SynthesizeAsync_BackendThrows_ReportsFaultAndTransitionsToFaulted + - SherpaOnnxSynthesisSession_SpeakAsync_PlaybackDeviceUnavailable_ThrowsRatherThanHanging - - id: Speech-Synthesis-SherpaOnnxSpeechSynthesizer-SpeakerSelection + - id: Speech-Synthesis-SherpaOnnxSynthesisSession-SpeakerSelection title: >- - The SherpaOnnxSpeechSynthesizer class shall resolve the speaker id passed to + The SherpaOnnxSynthesisSession class shall resolve the speaker id passed to Generate by calling ISynthesisModel.ResolveSpeakerId against the constructor-supplied parameterValues bag once per synthesized segment, rather than a hard-coded speakerId: 0, and this resolution shall coexist correctly with the independent, per-segment ParameterOverrides speed/volume mechanism (Natural Language Audio Tags) without either mechanism regressing the other. justification: | - Prior to this requirement, GenerateSegment hard-coded speakerId: 0 for every + Prior to this requirement, segment generation hard-coded speakerId: 0 for every synthesis model, so a model's declared voice-selection parameter (for example SherpaOnnxKokoroEnglishSynthesisModel's 11-voice ChoiceParameter) had no real effect on synthesized output even when a host selected a non-default voice. @@ -95,73 +180,82 @@ sections: whole session while a [fast]/[loud] tag applies only to the segment it annotates - and both must keep working correctly at the same time. tests: - - SynthesizeStreamAsync_NoParameterValues_ResolvesDefaultSpeakerIdFromModel - - SynthesizeStreamAsync_ParameterValuesSupplied_ResolvesSpeakerIdFromBag - - SynthesizeStreamAsync_ParameterValuesSuppliedAlongsideSpeedTag_BothMechanismsApplyIndependently + - SherpaOnnxSynthesisSession_SynthesizeAsync_NoParameterValues_ResolvesDefaultSpeakerIdFromModel + - SherpaOnnxSynthesisSession_SynthesizeAsync_ParameterValuesSupplied_ResolvesSpeakerIdFromBag + - SherpaOnnxSynthesisSession_SynthesizeAsync_ParameterValuesSuppliedAlongsideSpeedTag_BothMechanismsApplyIndependently - - id: Speech-Synthesis-SherpaOnnxSpeechSynthesizer-CancellationAndLifecycle + - id: Speech-Synthesis-SherpaOnnxSynthesisSession-CancellationAndLifecycle title: >- - The SherpaOnnxSpeechSynthesizer class shall cancel an in-flight SpeakAsync - session deterministically when Stop is called, treat Stop as a safe no-op when - idle, release its engine resources on disposal, never return control from - SynthesizeStreamAsync (on normal completion, cancellation, or any other - exception) until the background producer task - and any native engine call it - may still be mid-way through - has genuinely finished, and never hang doing so - even when enumeration is abandoned for a reason unrelated to cancellation. + The SherpaOnnxSynthesisSession class shall cancel an in-flight + SpeakAsync/SynthesizeAsync operation deterministically when StopAsync is + called, treat StopAsync as a safe no-op when idle, release its lease and + playback-device reference on disposal exactly once across repeated DisposeAsync + calls, and never return control from an in-flight operation until the + background producer - and any native backend call it may still be mid-way + through - has genuinely finished, even when enumeration/awaiting is abandoned + for a reason unrelated to cancellation. justification: | A host must be able to interrupt speech immediately in response to a user action (e.g. barge-in), without the cancellation racing or hanging, and must not leak native inference resources across repeated speak/stop cycles. A confirmed - AccessViolationException crash occurred because the cancellation path let - SynthesizeStreamCore return control to its caller while the producer task was + AccessViolationException crash occurred in the predecessor design because the + cancellation path let control return to the caller while the producer task was still genuinely executing a native Generate call on a background thread; the - caller then disposed the synthesizer's owned engine, and the still-running - native call touched freed native memory. The producer task must therefore never - be orphaned on any exit path. Unconditionally awaiting the producer task alone - is not sufficient, though: the producer writes into a bounded channel that only - unblocks a full write when either the reader keeps draining or the write's own - token is cancelled, so abandoning enumeration for a reason that never cancels - the caller's token - for example a consumer's own await foreach body throwing an - unrelated exception - could otherwise hang the await forever once the channel - fills up. The producer must therefore observe its own, always-cancelled - cancellation token on every abandonment path, distinct from the caller's token. + caller then disposed the owning engine, and the still-running native call + touched freed native memory. DedicatedWorker closes this gap by always awaiting + the in-flight native call's genuine completion, cooperative or abandoned, before + returning control on any exit path. + tests: + - SherpaOnnxSynthesisSession_StopAsync_WhileSpeaking_DoesNotCompleteUntilInFlightGenerateReturns + - SherpaOnnxSynthesisSession_StopAsync_NoOperationInFlight_IsNoOp + - SherpaOnnxSynthesisSession_DisposeAsync_CalledTwice_ReleasesLeaseOnce + - SherpaOnnxSynthesisSession_SynthesizeAsync_AfterDispose_ThrowsObjectDisposedException + - SherpaOnnxSynthesisSession_IsAvailable_BeforeAndAfterDispose_ReflectsLifecycle + + - title: SynthesisSessionFaultedException Requirements + requirements: + - id: Speech-Synthesis-SynthesisSessionFaultedException-Thrown + title: >- + The SynthesisSessionFaultedException class shall be the documented exception + type thrown by SherpaOnnxSynthesisSession for every call made after it has + transitioned to the terminal Faulted state, wrapping the original fault as its + inner exception. + justification: | + A distinct, documented exception type lets a caller distinguish "this session + has already faulted and must be replaced" from an ordinary new failure, and the + wrapped inner exception preserves the original root cause for diagnostics. tests: - - Stop_WhileSpeaking_CancelsInFlightSessionOnlyAfterInFlightGenerateReturns - - SynthesizeStreamAsync_CancelledMidGenerate_AwaitsProducerBeforeEnumerationCompletesAndDisposalIsSafe - - SynthesizeStreamAsync_EnumerationAbandonedWithoutCancellation_DisposesPromptlyInsteadOfHanging - - Stop_NoSessionInFlight_IsNoOp - - Dispose_CalledTwice_DisposesEngineOnce - - SynthesizeStreamAsync_AfterDispose_ThrowsObjectDisposedException - - IsAvailable_Always_ReturnsTrue + - SherpaOnnxSynthesisSession_SynthesizeAsync_AfterFault_ThrowsSynthesisSessionFaultedException - - title: SynthesisEngine Requirements + - title: SynthesisBackend Requirements requirements: - - id: Speech-Synthesis-SynthesisEngine-MockableSeam + - id: Speech-Synthesis-SynthesisBackend-MockableSeam title: >- - The synthesis-engine seam shall isolate all speech-inference interop so the - synthesis pipeline's chunking, Layer 2 rendering, and playback can be verified - without a downloaded model or an installed speech-engine native runtime. + The ISynthesisBackend/ISynthesisBackendFactory seam shall isolate all + speech-inference interop so the synthesis pipeline's chunking, Layer 2 + rendering, session lifecycle, and playback can be verified without a downloaded + model or an installed speech-engine native runtime. justification: | CI runners cannot be assumed to have a multi-hundred-megabyte speech model or a platform-specific native inference binary available, so the synthesis logic must be verifiable entirely in pure managed tests. tests: - - SpeechSynthesizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer - - SynthesizeStreamAsync_PlainText_YieldsAudioSegment + - SpeechSynthesizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine + - SherpaOnnxSynthesisSession_SynthesizeAsync_PlainText_YieldsAudioSegment - - id: Speech-Synthesis-SynthesisEngine-ModelOwnedConfiguration + - id: Speech-Synthesis-SynthesisBackend-ModelOwnedConfiguration title: >- - The synthesis-engine seam shall load an engine using only the configuration the + The ISynthesisBackend seam shall load a backend using only the configuration the selected model itself declares, reusing the already-referenced sherpa-onnx package rather than adding a new native dependency. justification: | - The design makes each model's backing class responsible for its own engine + The design makes each model's backing class responsible for its own backend configuration, so that adding a new synthesis model is a self-contained, reviewable unit of work that never requires changing the synthesis subsystem or its native package reference. tests: - ISynthesisModel_CreateEngineConfig_InstalledDirectory_ResolvesPaths - - SpeechSynthesizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer + - SpeechSynthesizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine - title: Layer2Rendering Requirements requirements: @@ -199,10 +293,9 @@ sections: justification: | This library's conservative risk mitigation for unsupported tags requires that a model without native tag support still gets a reasonable, non-crashing - approximation of pace/volume - intent, while intentionally conservative scope means only the tags with a clear - numeric convention are mapped; every other tag safely degrades to plain - narration instead of causing an error. + approximation of pace/volume intent, while intentionally conservative scope + means only the tags with a clear numeric convention are mapped; every other tag + safely degrades to plain narration instead of causing an error. tests: - DefaultModelCapabilityProfile_Render_ParameterMappedSupport_FastTag_MapsToSpeedParameter - DefaultModelCapabilityProfile_Render_ParameterMappedSupport_EmotionTag_StripsWithNoOverride diff --git a/docs/reqstream/speech/synthesis-subsystem/speech-synthesizer-factory.yaml b/docs/reqstream/speech/synthesis-subsystem/speech-synthesizer-factory.yaml index b870410..c696f65 100644 --- a/docs/reqstream/speech/synthesis-subsystem/speech-synthesizer-factory.yaml +++ b/docs/reqstream/speech/synthesis-subsystem/speech-synthesizer-factory.yaml @@ -2,8 +2,10 @@ # Software Unit Requirements for SpeechSynthesizerFactory # # These requirements describe the design-level behavior of the SpeechSynthesizerFactory unit, -# decomposing the parent SynthesisSubsystem requirements. Mirrors -# speech-recognizer-factory.yaml's identical Phase 3 pattern. +# decomposing the parent SynthesisSubsystem requirements. Updated for the async Engine/Session +# redesign: Create() became the asynchronous LoadAsync(), returning Task +# with no playback-device parameter, since device binding now happens later at +# CreateSessionAsync. Mirrors speech-recognizer-factory.yaml's identical Phase 3 pattern. sections: - title: Speech Requirements @@ -12,91 +14,91 @@ sections: sections: - title: SpeechSynthesizerFactory Requirements requirements: - - id: Speech-Synthesis-SpeechSynthesizerFactory-RealSynthesizer + - id: Speech-Synthesis-SpeechSynthesizerFactory-RealEngine title: >- - The SpeechSynthesizerFactory class shall return a working speech synthesizer - when the requested model is installed, declares the synthesis role, its engine - loads, and the supplied playback device is available. + The SpeechSynthesizerFactory class shall asynchronously return a working + ISpeechSynthesizerEngine when the requested model is installed, declares the + synthesis role, and its backend loads. justification: | A single composition entry point keeps all "can this machine speak right now?" logic in one reviewable place, so a host never has to reason about model - installation, device availability, and engine loading separately. + installation and backend loading separately. Composition no longer depends on + playback-device availability, since LoadAsync produces a Layer 3 engine that is + not yet bound to any device. tests: - - SpeechSynthesizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer - - SpeechSynthesizerFactory_Create_WithStoreModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer - - SpeechSynthesizerFactory_Create_WithCatalogModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer + - SpeechSynthesizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine + - SpeechSynthesizerFactory_LoadAsync_WithStoreModelInstalled_ReturnsRealEngine + - SpeechSynthesizerFactory_LoadAsync_WithCatalogModelInstalled_ReturnsRealEngine - id: Speech-Synthesis-SpeechSynthesizerFactory-HonestFallbacks title: >- - The SpeechSynthesizerFactory class shall return an honest unavailable - synthesizer, without throwing, when the requested model is not installed, the - model does not declare the synthesis role, or the supplied playback device is - unavailable. + The SpeechSynthesizerFactory class shall asynchronously return an honest + unavailable engine, without throwing, when the requested model is not + installed, or the model does not declare the synthesis role. justification: | Per this library's "nothing throws at composition" decision, a model that has - not been downloaded yet and a machine with no speakers are ordinary states at - application start-up, not programming errors. + not been downloaded yet is an ordinary state at application start-up, not a + programming error. tests: - - SpeechSynthesizerFactory_Create_ModelNotInstalled_ReturnsUnavailableSynthesizer - - SpeechSynthesizerFactory_Create_PlaybackDeviceUnavailable_ReturnsUnavailableSynthesizer - - SpeechSynthesizerFactory_Create_ModelRoleIsNotSynthesis_ReturnsUnavailableSynthesizer - - SpeechSynthesizerFactory_Create_WithStoreModelNotInstalled_ReturnsUnavailableSynthesizer - - SpeechSynthesizerFactory_Create_WithCatalogModelNotInstalled_ReturnsUnavailableSynthesizer + - SpeechSynthesizerFactory_LoadAsync_ModelNotInstalled_ReturnsUnavailableEngine + - SpeechSynthesizerFactory_LoadAsync_ModelRoleIsNotSynthesis_ReturnsUnavailableEngine + - SpeechSynthesizerFactory_LoadAsync_WithStoreModelNotInstalled_ReturnsUnavailableEngine + - SpeechSynthesizerFactory_LoadAsync_WithCatalogModelNotInstalled_ReturnsUnavailableEngine - id: Speech-Synthesis-SpeechSynthesizerFactory-EngineLoadFailureFallback title: >- - The SpeechSynthesizerFactory class shall return an honest unavailable - synthesizer, and report the reason, when the synthesis engine for an installed - model cannot be loaded. + The SpeechSynthesizerFactory class shall return an honest unavailable engine, + and report the reason, when the synthesis backend for an installed model cannot + be loaded. justification: | The design requires that "a model whose native runtime is absent must - degrade the same honest way as a missing model file, never crash". An engine + degrade the same honest way as a missing model file, never crash". A backend that cannot load because of a missing native binary, an unsupported platform, or corrupt model files must therefore not fail application start-up. tests: - - SpeechSynthesizerFactory_Create_EngineLoadFails_ReturnsUnavailableSynthesizerAndDoesNotThrow + - SpeechSynthesizerFactory_LoadAsync_EngineLoadFails_ReturnsUnavailableEngineAndDoesNotThrow - - id: Speech-Synthesis-SpeechSynthesizerFactory-ArgumentValidation + - id: Speech-Synthesis-SpeechSynthesizerFactory-ArgumentValidationAndCancellation title: >- The SpeechSynthesizerFactory class shall reject a null model, a null - SpeechModelStore, a null SpeechModelCatalog, or a null playback device. + SpeechModelStore, or a null SpeechModelCatalog, and shall throw + OperationCanceledException when the supplied cancellation token is already + cancelled. justification: | - A null model, store, catalog, or playback device is a programming error at the - call site, not an ordinary "unavailable" machine state, and must fail - immediately with a clear exception rather than being silently treated as - unavailable. + A null model, store, or catalog is a programming error at the call site, not an + ordinary "unavailable" machine state, and must fail immediately with a clear + exception rather than being silently treated as unavailable. An already + cancelled token means the caller no longer wants the result, so LoadAsync must + honor it rather than performing the (potentially slow) load anyway. tests: - - SpeechSynthesizerFactory_Create_NullModel_ThrowsArgumentNullException - - SpeechSynthesizerFactory_Create_NullPlaybackDevice_ThrowsArgumentNullException - - SpeechSynthesizerFactory_Create_WithStoreNullModel_ThrowsArgumentNullException - - SpeechSynthesizerFactory_Create_WithStoreNullStore_ThrowsArgumentNullException - - SpeechSynthesizerFactory_Create_WithStoreNullPlaybackDevice_ThrowsArgumentNullException - - SpeechSynthesizerFactory_Create_WithCatalogNullModel_ThrowsArgumentNullException - - SpeechSynthesizerFactory_Create_WithCatalogNullCatalog_ThrowsArgumentNullException - - SpeechSynthesizerFactory_Create_WithCatalogNullPlaybackDevice_ThrowsArgumentNullException + - SpeechSynthesizerFactory_LoadAsync_NullModel_ThrowsArgumentNullException + - SpeechSynthesizerFactory_LoadAsync_CancelledToken_ThrowsOperationCanceledException + - SpeechSynthesizerFactory_LoadAsync_WithStoreNullModel_ThrowsArgumentNullException + - SpeechSynthesizerFactory_LoadAsync_WithStoreNullStore_ThrowsArgumentNullException + - SpeechSynthesizerFactory_LoadAsync_WithCatalogNullModel_ThrowsArgumentNullException + - SpeechSynthesizerFactory_LoadAsync_WithCatalogNullCatalog_ThrowsArgumentNullException - id: Speech-Synthesis-SpeechSynthesizerFactory-ParameterValuesForwarding title: >- The SpeechSynthesizerFactory class shall forward an optional parameterValues - bag unchanged to the constructed synthesizer, defaulting to null for a caller - that does not supply one, so a model with real per-voice knowledge can later - resolve a genuine speaker id from it. + bag unchanged to the constructed engine, defaulting to null for a caller that + does not supply one, so a model with real per-voice knowledge can later resolve + a genuine speaker id from it once a session is created. justification: | Prior to this requirement, no seam existed to carry a session-level selected voice from a host's ISpeechModel.Parameters selection through to - SherpaOnnxSpeechSynthesizer; adding this argument as optional with a null - default keeps every existing call site source-compatible while unblocking real - voice selection end-to-end for models such as - SherpaOnnxKokoroEnglishSynthesisModel. + SherpaOnnxSynthesisSession; adding this argument as optional with a null default + keeps every existing call site source-compatible while unblocking real voice + selection end-to-end for models such as SherpaOnnxKokoroEnglishSynthesisModel. tests: - - SpeechSynthesizerFactory_Create_ParameterValuesSupplied_ForwardedToSynthesizer + - SpeechSynthesizerFactory_LoadAsync_ParameterValuesSupplied_ForwardedToEngine - id: Speech-Synthesis-SpeechSynthesizerFactory-UnrecognizedParameterIdIgnored title: >- The SpeechSynthesizerFactory class shall silently ignore a supplied parameterValues key that names a parameter not declared by the requested model, - still composing a real synthesizer, while reporting an Info-level diagnostic - naming the ignored key. + still composing a real engine, while reporting an Info-level diagnostic naming + the ignored key. justification: | A host reusing one settings dictionary across different synthesis models must not break composition just because a given model does not declare a parameter @@ -104,12 +106,12 @@ sections: contract, not a caller bug, so it must never throw - only be surfaced as an informational diagnostic for a host that wired up a sink. tests: - - SpeechSynthesizerFactory_Create_UnrecognizedParameterId_ComposesAndReportsInfo + - SpeechSynthesizerFactory_LoadAsync_UnrecognizedParameterId_ComposesAndReportsInfo - id: Speech-Synthesis-SpeechSynthesizerFactory-InvalidRecognizedParameterValueThrows title: >- The SpeechSynthesizerFactory class shall throw ArgumentException synchronously - from Create() - before installed/role/device/engine checks run - when + from LoadAsync() - before installed/role/backend checks run - when parameterValues supplies a value for a parameter the requested model does declare, but the value is invalid for that parameter (wrong CLR type, outside [Minimum, Maximum] or non-integral for a NumericParameter declared IsInteger, @@ -117,17 +119,15 @@ sections: bool for a BooleanParameter), naming the parameter id, the model id, and the reason the value is invalid. justification: | - This is a deliberate, user-approved breaking change from this library's - earlier silent-default behavior for this case: a host explicitly targeting a - parameter it knows this model declares, with a value invalid for it, is a - caller bug rather than a legitimate cross-model compatibility gap, and must be - surfaced immediately and loudly - synchronously from Create() - rather than - silently substituted the first time ResolveSpeakerId ran per segment. This - validation happens once, up front, at Create(); it does not change - ResolveSpeakerId's or ResolveOverrideRatios's own existing never-throw, - per-segment runtime contract. + A host explicitly targeting a parameter it knows this model declares, with a + value invalid for it, is a caller bug rather than a legitimate cross-model + compatibility gap, and must be surfaced immediately and loudly - synchronously + from LoadAsync() - rather than silently substituted the first time + ResolveSpeakerId ran per segment. This validation happens once, up front, at + LoadAsync(); it does not change ResolveSpeakerId's or ResolveOverrideRatios's + own existing never-throw, per-segment runtime contract. tests: - - SpeechSynthesizerFactory_Create_RecognizedNumericParameterOutOfRange_Throws - - SpeechSynthesizerFactory_Create_RecognizedNumericParameterWrongType_Throws - - SpeechSynthesizerFactory_Create_RecognizedIntegerParameterFractionalValue_Throws - - SpeechSynthesizerFactory_Create_RecognizedChoiceParameterInvalidOption_Throws + - SpeechSynthesizerFactory_LoadAsync_RecognizedNumericParameterOutOfRange_Throws + - SpeechSynthesizerFactory_LoadAsync_RecognizedNumericParameterWrongType_Throws + - SpeechSynthesizerFactory_LoadAsync_RecognizedIntegerParameterFractionalValue_Throws + - SpeechSynthesizerFactory_LoadAsync_RecognizedChoiceParameterInvalidOption_Throws diff --git a/docs/reqstream/speech/synthesis-subsystem/synthesis-session-lease.yaml b/docs/reqstream/speech/synthesis-subsystem/synthesis-session-lease.yaml new file mode 100644 index 0000000..c7ce806 --- /dev/null +++ b/docs/reqstream/speech/synthesis-subsystem/synthesis-session-lease.yaml @@ -0,0 +1,69 @@ +--- +# Software Unit Requirements for engine session exclusivity (lease) and SynthesisEngineBusyException +# +# Describes the design-level behavior of SherpaOnnxSpeechSynthesizerEngine's own session-creation +# and lease-tracking logic: an engine hosts at most one live ISynthesisSession at a time, and +# SynthesisEngineBusyException is the documented signal for a caller that requests a second +# session while the first is still live. Mirrors the recognition direction's equivalent lease +# requirements file for the async Engine/Session redesign. + +sections: + - title: Speech Requirements + sections: + - title: SynthesisSubsystem Requirements + sections: + - title: SherpaOnnxSpeechSynthesizerEngine Requirements + requirements: + - id: Speech-Synthesis-SherpaOnnxSpeechSynthesizerEngine-SessionExclusivity + title: >- + The SherpaOnnxSpeechSynthesizerEngine class shall allow at most one live + ISynthesisSession to be leased from it at a time, throwing + SynthesisEngineBusyException synchronously from CreateSessionAsync when a + session is requested while an earlier leased session has not yet been disposed, + and succeeding again once that session is disposed. + justification: | + A single loaded model (Layer 3 engine) owns one native inference context; two + concurrently live sessions bound to two different playback devices would race + against that same native context and corrupt either or both synthesis streams. + Failing the second CreateSessionAsync call fast and synchronously, rather than + queuing or silently sharing the engine, gives a caller an unambiguous signal to + dispose its current session before requesting another. + children: + - Speech-Synthesis-ISpeechSynthesizerEngine-SessionCreationContract + tests: + - SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_NoLeaseHeld_ReturnsRealSession + - SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_LeaseAlreadyHeld_ThrowsSynthesisEngineBusyException + - SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_AfterPriorSessionDisposed_SucceedsAgain + - SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException + + - id: Speech-Synthesis-SherpaOnnxSpeechSynthesizerEngine-DisposalOrdering + title: >- + The SherpaOnnxSpeechSynthesizerEngine class shall dispose its active leased + session, if any, before disposing its owned backend, and shall dispose its + backend exactly once across repeated DisposeAsync calls. + justification: | + Disposing the backend (and its native inference context) while a leased session + may still be mid-operation would leave the session holding a reference to + already-freed native resources; disposing the session first guarantees any + in-flight native call has genuinely returned before the backend itself is torn + down. Repeated disposal must remain a safe no-op, matching every other + disposable type in this library. + tests: + - SherpaOnnxSpeechSynthesizerEngine_DisposeAsync_WithActiveLeasedSession_DisposesSessionFirst + - SherpaOnnxSpeechSynthesizerEngine_DisposeAsync_CalledTwice_DisposesBackendOnce + + - title: SynthesisEngineBusyException Requirements + requirements: + - id: Speech-Synthesis-SynthesisEngineBusyException-Thrown + title: >- + The SynthesisEngineBusyException class shall be the documented exception type + thrown by CreateSessionAsync when a session is requested while an earlier + leased session is still live, and shall provide standard exception + constructors. + justification: | + A distinct, documented exception type lets a caller distinguish "the engine is + busy with another session" from any other failure mode, and lets it decide + whether to retry after disposing its current session rather than treating the + failure as fatal. + tests: + - SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_LeaseAlreadyHeld_ThrowsSynthesisEngineBusyException diff --git a/docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.yaml b/docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.yaml new file mode 100644 index 0000000..6ce2862 --- /dev/null +++ b/docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.yaml @@ -0,0 +1,76 @@ +--- +# Software Unit Requirements for UnavailableSpeechSynthesizerEngine +# +# Groups requirements for UnavailableSpeechSynthesizerEngine and +# SpeechSynthesizerUnavailableException into a single file, per design-documentation.md's +# allowance for small supporting types documented inline within their subsystem's design doc. +# Splits the former unavailable-speech-synthesizer.yaml between this file and +# unavailable-synthesis-session.yaml to match the async Engine/Session redesign. + +sections: + - title: Speech Requirements + sections: + - title: SynthesisSubsystem Requirements + sections: + - title: UnavailableSpeechSynthesizerEngine Requirements + requirements: + - id: Speech-Synthesis-UnavailableSpeechSynthesizerEngine-ReportsUnavailable + title: >- + The UnavailableSpeechSynthesizerEngine class shall report IsAvailable as false + and never throw merely for being obtained or held. + justification: | + A caller must always be able to safely obtain and hold a fallback engine, per + this library's "nothing throws at composition" decision. + tests: + - UnavailableSpeechSynthesizerEngine_IsAvailable_Read_ReturnsFalse + + - id: Speech-Synthesis-UnavailableSpeechSynthesizerEngine-CreateSessionSucceeds + title: >- + The UnavailableSpeechSynthesizerEngine class shall succeed when + CreateSessionAsync is invoked with a non-null device, returning the shared + UnavailableSynthesisSession, while rejecting a null device. + justification: | + Binding a device to an already unavailable engine is an ordinary (if useless) + composition, not an error; only a null device argument is a programming error + at the call site. + tests: + - UnavailableSpeechSynthesizerEngine_CreateSessionAsync_Always_ReturnsUnavailableSession + - UnavailableSpeechSynthesizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException + + - id: Speech-Synthesis-UnavailableSpeechSynthesizerEngine-OperationsThrow + title: >- + The UnavailableSpeechSynthesizerEngine class shall throw + SpeechSynthesizerUnavailableException when SpeakAsync or SynthesizeAsync is + invoked. + justification: | + A caller that ignores the IsAvailable flag and invokes a one-shot operational + member has made a programming error that must surface immediately rather than + silently speaking nothing. + tests: + - UnavailableSpeechSynthesizerEngine_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException + - UnavailableSpeechSynthesizerEngine_SynthesizeAsync_Always_ThrowsSpeechSynthesizerUnavailableException + + - id: Speech-Synthesis-UnavailableSpeechSynthesizerEngine-SafeDispose + title: >- + The UnavailableSpeechSynthesizerEngine class shall treat disposal as a safe + no-op that never throws and never invalidates the shared fallback. + justification: | + A host that wraps its engine in a disposal scope must be able to run that same + code unchanged on a machine where synthesis is unavailable. + tests: + - UnavailableSpeechSynthesizerEngine_DisposeAsync_CalledTwice_DoesNotThrow + + - title: SpeechSynthesizerUnavailableException Requirements + requirements: + - id: Speech-Synthesis-SpeechSynthesizerUnavailableException-Thrown + title: >- + The SpeechSynthesizerUnavailableException class shall provide standard + exception constructors (default, message, message with inner exception). + justification: | + Standard exception conformance lets callers construct, catch, and inspect this + exception like any other .NET exception type, including wrapping an underlying + playback-device failure as the inner exception. + tests: + - SpeechSynthesizerUnavailableException_Constructor_WithMessage_ExposesMessage + - SpeechSynthesizerUnavailableException_Constructor_WithInnerException_ExposesBoth + - SpeechSynthesizerUnavailableException_Constructor_Default_HasNonEmptyMessage diff --git a/docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer.yaml b/docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer.yaml deleted file mode 100644 index b9bc01b..0000000 --- a/docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer.yaml +++ /dev/null @@ -1,66 +0,0 @@ ---- -# Software Unit Requirements for UnavailableSpeechSynthesizer -# -# Groups requirements for UnavailableSpeechSynthesizer and SpeechSynthesizerUnavailableException -# into a single file, per design-documentation.md's allowance for small supporting types -# documented inline within their subsystem's design doc, mirroring -# unavailable-speech-recognizer.yaml's identical Phase 3 pattern. Each unit's requirement subset -# is called out in its own subsection below so every unit still has requirements tracing to it -# individually. - -sections: - - title: Speech Requirements - sections: - - title: SynthesisSubsystem Requirements - sections: - - title: UnavailableSpeechSynthesizer Requirements - requirements: - - id: Speech-Synthesis-UnavailableSpeechSynthesizer-ReportsUnavailable - title: >- - The UnavailableSpeechSynthesizer class shall report IsAvailable as false and - never throw merely for being obtained or held. - justification: | - A caller must always be able to safely obtain and hold a fallback synthesizer, - per this library's "nothing throws at composition" decision. - tests: - - UnavailableSpeechSynthesizer_IsAvailable_Read_ReturnsFalse - - - id: Speech-Synthesis-UnavailableSpeechSynthesizer-OperationsThrow - title: >- - The UnavailableSpeechSynthesizer class shall throw - SpeechSynthesizerUnavailableException when SynthesizeStreamAsync, - PlayStreamAsync, SpeakAsync, or Stop is invoked. - justification: | - A caller that ignores the IsAvailable flag and invokes an operational member - has made a programming error that must surface immediately rather than silently - speaking nothing. - tests: - - UnavailableSpeechSynthesizer_SynthesizeStreamAsync_Always_ThrowsSpeechSynthesizerUnavailableException - - UnavailableSpeechSynthesizer_PlayStreamAsync_Always_ThrowsSpeechSynthesizerUnavailableException - - UnavailableSpeechSynthesizer_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException - - UnavailableSpeechSynthesizer_Stop_Always_ThrowsSpeechSynthesizerUnavailableException - - - id: Speech-Synthesis-UnavailableSpeechSynthesizer-SafeDispose - title: >- - The UnavailableSpeechSynthesizer class shall treat disposal as a safe no-op - that never throws and never invalidates the shared fallback. - justification: | - A host that wraps its synthesizer in a disposal scope must be able to run that - same code unchanged on a machine where synthesis is unavailable. - tests: - - UnavailableSpeechSynthesizer_Dispose_CalledTwice_DoesNotThrow - - - title: SpeechSynthesizerUnavailableException Requirements - requirements: - - id: Speech-Synthesis-SpeechSynthesizerUnavailableException-Thrown - title: >- - The SpeechSynthesizerUnavailableException class shall provide standard - exception constructors (default, message, message with inner exception). - justification: | - Standard exception conformance lets callers construct, catch, and inspect this - exception like any other .NET exception type, including wrapping an underlying - playback-device failure as the inner exception. - tests: - - SpeechSynthesizerUnavailableException_Constructor_WithMessage_ExposesMessage - - SpeechSynthesizerUnavailableException_Constructor_WithInnerException_ExposesBoth - - SpeechSynthesizerUnavailableException_Constructor_Default_HasNonEmptyMessage diff --git a/docs/reqstream/speech/synthesis-subsystem/unavailable-synthesis-session.yaml b/docs/reqstream/speech/synthesis-subsystem/unavailable-synthesis-session.yaml new file mode 100644 index 0000000..28cfc68 --- /dev/null +++ b/docs/reqstream/speech/synthesis-subsystem/unavailable-synthesis-session.yaml @@ -0,0 +1,54 @@ +--- +# Software Unit Requirements for UnavailableSynthesisSession +# +# Splits the former unavailable-speech-synthesizer.yaml between this file and +# unavailable-speech-synthesizer-engine.yaml to match the async Engine/Session redesign. + +sections: + - title: Speech Requirements + sections: + - title: SynthesisSubsystem Requirements + sections: + - title: UnavailableSynthesisSession Requirements + requirements: + - id: Speech-Synthesis-UnavailableSynthesisSession-ReportsUnavailable + title: >- + The UnavailableSynthesisSession class shall report IsAvailable as false and + State as SynthesisSessionState.Created at all times, never throwing merely for + being obtained or held. + justification: | + A caller must always be able to safely obtain and hold a fallback session, per + this library's "nothing throws at composition" decision, and a session that + never performs a real operation has no meaningful state transitions to report. + tests: + - UnavailableSynthesisSession_IsAvailable_Read_ReturnsFalse + - UnavailableSynthesisSession_State_Read_ReturnsCreated + - UnavailableSynthesisSession_StateChanged_SubscribeAndUnsubscribe_DoesNotThrow + + - id: Speech-Synthesis-UnavailableSynthesisSession-OperationsThrow + title: >- + The UnavailableSynthesisSession class shall throw + SpeechSynthesizerUnavailableException when SpeakAsync or SynthesizeAsync is + invoked with a non-null text argument, and ArgumentNullException for a null + text argument, while treating StopAsync as a safe no-op. + justification: | + A caller that ignores the IsAvailable flag and invokes an operational member + has made a programming error that must surface immediately rather than silently + speaking nothing; a null argument is a distinct, more fundamental programming + error that must be reported as such regardless of availability. + tests: + - UnavailableSynthesisSession_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException + - UnavailableSynthesisSession_SpeakAsync_NullText_ThrowsArgumentNullException + - UnavailableSynthesisSession_SynthesizeAsync_Always_ThrowsSpeechSynthesizerUnavailableException + - UnavailableSynthesisSession_SynthesizeAsync_NullText_ThrowsArgumentNullException + - UnavailableSynthesisSession_StopAsync_NoSessionInFlight_IsNoOp + + - id: Speech-Synthesis-UnavailableSynthesisSession-SafeDispose + title: >- + The UnavailableSynthesisSession class shall treat disposal as a safe no-op that + never throws and never invalidates the shared fallback. + justification: | + A host that wraps its session in a disposal scope must be able to run that same + code unchanged on a machine where synthesis is unavailable. + tests: + - UnavailableSynthesisSession_DisposeAsync_CalledTwice_DoesNotThrow diff --git a/docs/sysml2/model/speech-cli/conversation-command-subsystem.sysml b/docs/sysml2/model/speech-cli/conversation-command-subsystem.sysml index ca8ae8c..745efa3 100644 --- a/docs/sysml2/model/speech-cli/conversation-command-subsystem.sysml +++ b/docs/sysml2/model/speech-cli/conversation-command-subsystem.sysml @@ -4,9 +4,9 @@ package SpeechCli { * after all ten scaffolded subcommands were implemented - combining speak's * synthesis-then-play flow and recognize --mic's mic-listen flow into a * single invocation. Introduces no new ICliModelCatalog seam member: it - * reuses SynthesisCommandSubsystem's CreateSynthesizer and - * RecognitionCommandSubsystem's CreateRecognizer members exactly as speak - * and recognize already do individually. Introduces its own small + * reuses SynthesisCommandSubsystem's CreateSynthesizerEngineAsync and + * RecognitionCommandSubsystem's CreateRecognizerEngineAsync members exactly + * as speak and recognize already do individually. Introduces its own small * ICliCaptureDeviceSource seam over real capture-device resolution, * mirroring SynthesisCommandSubsystem's ICliPlaybackDeviceSource, letting * ask's mic-mode happy-path tests run deterministically without depending @@ -26,24 +26,30 @@ package SpeechCli { * resolves and validates both a synthesis-role --tts-model and a * recognition-role --stt-model, validates --tts-param/--stt-param values * via the reused ParameterBagParser, speaks the resolved text through a - * real playback device (Phase 1), then - once playback finishes - listens - * through a real capture device resolved via ICliCaptureDeviceSource - * (Phase 2) until the first final recognition result arrives, a reused - * SilenceTimeoutRecognizerSession times out, or Ctrl+C is pressed. Phase - * 2's recognizer (and its capture device) is now constructed concurrently - * with Phase 1's speak/playback wait, via a background task, rather than - * only once playback finishes, so the expensive model-load step overlaps - * with prompt playback and the turnaround gap before listening starts is - * minimized; only the model load is pre-warmed, never Start() (real + * real playback device via an ISpeechSynthesizerEngine/ISynthesisSession + * (Phase 1), then - once playback finishes - listens through a real + * capture device resolved via ICliCaptureDeviceSource using an + * ISpeechRecognizerEngine/IRecognitionSession wrapped in a reused + * SilenceTimeoutRecognizerSession (Phase 2) until the first final + * recognition result arrives, the silence/start timeout elapses, or Ctrl+C + * is pressed. Phase 2's recognizer engine and session (and its capture + * device) are now constructed concurrently with Phase 1's SpeakAsync wait, + * via a background Task.Run (PrewarmRecognizerAsync), rather than only once + * playback finishes, so the expensive model-load step overlaps with prompt + * playback and the turnaround gap before listening starts is minimized; + * only the model load is pre-warmed, never session.StartAsync (real * microphone capture), which still begins only once Phase 2 genuinely - * runs. A genuine Ctrl+C during Phase 2 is distinguished from a legitimate - * empty result (silence/start timeout) via the shared CancellationToken, and is - * reported the same way a Phase-1 cancellation already is (WriteError), - * skipping the output-writing step entirely rather than printing/writing - * empty text as a false success. Otherwise, the final recognized text is - * printed (or written to --output-text), and the synthesizer, playback - * device, recognizer, and silence-timeout session (when present) are - * disposed on every exit path. */ + * runs - if Phase 1 is canceled or fails first, the pre-warmed engine and + * session are disposed fire-and-forget via DisposePrewarmedRecognizerAsync + * instead of being awaited directly. A genuine Ctrl+C during Phase 2 is + * distinguished from a legitimate empty result (silence/start timeout) via + * the shared CancellationToken, and is reported the same way a Phase-1 + * cancellation already is (WriteError), skipping the output-writing step + * entirely rather than printing/writing empty text as a false success. + * Otherwise, the final recognized text is printed (or written to + * --output-text), and the synthesis session/engine, playback device, + * recognition session/engine, and capture device are all disposed (await + * using) on every exit path. */ comment sourceRef /* Source: src/DemaConsulting.Speech.Cli/Commands/ConversationCommandSubsystem/AskCommand.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Cli.Tests/Commands/ConversationCommandSubsystem/AskCommandTests.cs */ diff --git a/docs/sysml2/model/speech-cli/model-commands-subsystem.sysml b/docs/sysml2/model/speech-cli/model-commands-subsystem.sysml index 9ae3e32..5824ef2 100644 --- a/docs/sysml2/model/speech-cli/model-commands-subsystem.sysml +++ b/docs/sysml2/model/speech-cli/model-commands-subsystem.sysml @@ -22,7 +22,13 @@ package SpeechCli { part def ICliModelCatalog { doc /* CLI-owned seam for enumerating known models, downloading, uninstalling, and * cleaning up leftovers by id, so every command can be tested against controlled - * catalog data with no network access. */ + * catalog data with no network access. Also the single seam point extended by + * SynthesisCommandSubsystem/RecognitionCommandSubsystem with async engine-loading + * members (GetPreferredAudioFormat/CreateSynthesizerEngineAsync, + * GetAudioFormat/CreateRecognizerEngineAsync) returning the library's async + * ISpeechSynthesizerEngine/ISpeechRecognizerEngine, each later bound to a device + * via CreateSessionAsync rather than returning a device-bound synthesizer or + * recognizer directly. */ comment sourceRef /* Source: src/DemaConsulting.Speech.Cli/Commands/ModelCommandsSubsystem/ICliModelCatalog.cs */ } @@ -30,7 +36,11 @@ package SpeechCli { part def SpeechModelCatalogAdapter { doc /* Real catalog seam implementation, owning and delegating to a real * SpeechModelCatalog (and, for uninstall/cleanup, its Store) without exposing the - * library's internal-only constructor. */ + * library's internal-only constructor. Forwards GetPreferredAudioFormat/ + * CreateSynthesizerEngineAsync and GetAudioFormat/CreateRecognizerEngineAsync + * unchanged to the resolved model's real SpeechSynthesizerFactory/ + * SpeechRecognizerFactory, preserving each factory's own honestly-unavailable + * fallback contract rather than masking it. */ comment sourceRef /* Source: src/DemaConsulting.Speech.Cli/Commands/ModelCommandsSubsystem/SpeechModelCatalogAdapter.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Cli.Tests/Commands/ModelCommandsSubsystem/SpeechModelCatalogAdapterTests.cs */ diff --git a/docs/sysml2/model/speech-cli/recognition-command-subsystem.sysml b/docs/sysml2/model/speech-cli/recognition-command-subsystem.sysml index d15d1f4..270aff7 100644 --- a/docs/sysml2/model/speech-cli/recognition-command-subsystem.sysml +++ b/docs/sysml2/model/speech-cli/recognition-command-subsystem.sysml @@ -1,10 +1,12 @@ package SpeechCli { part def RecognitionCommandSubsystem { - doc /* The one speech-to-text subcommand (recognize) - the last of all 10 - * subcommands to be implemented - plus the SilenceTimeoutRecognizerSession - * idle-timeout utility, extending the ModelCommandsSubsystem's - * ICliModelCatalog seam with two further members (GetAudioFormat, - * CreateRecognizer) rather than introducing a second, competing seam. */ + doc /* The one speech-to-text subcommand (recognize) plus the + * SilenceTimeoutRecognizerSession idle-timeout decorator, extending the + * ModelCommandsSubsystem's ICliModelCatalog seam with two further members + * (GetAudioFormat, CreateRecognizerEngineAsync) rather than introducing a + * second, competing seam. Consumes the library's async Engine/Session + * RecognitionSubsystem surface (ISpeechRecognizerEngine/IRecognitionSession) + * rather than the prior synchronous, event-based ISpeechRecognizer. */ comment designRef /* Design: docs/design/speech-cli/recognition-command-subsystem.md */ comment verificationRef /* Verification: docs/verification/speech-cli/recognition-command-subsystem.md */ @@ -17,26 +19,34 @@ package SpeechCli { part def RecognizeCommand { doc /* Implements recognize: resolves exactly one of --input/--mic, resolves and * validates the requested recognition model, validates --stt-param values via - * the reused ParameterBagParser, drives file-input mode to completion via - * the capture device's own EndOfFileReached event (reentrant Stop()), wires - * mic-input mode to Ctrl+C and an optional SilenceTimeoutRecognizerSession, - * filters console output per --interim/--final-only, writes only final - * results to --output-text, and disposes the recognizer (and silence-timeout - * session, when present) on every exit path. */ + * the reused ParameterBagParser, loads an ISpeechRecognizerEngine via + * ICliModelCatalog.CreateRecognizerEngineAsync and binds an IRecognitionSession + * to the resolved capture device, drives file-input mode to completion via + * StartAsync's blocking file delivery followed by an explicit drain StopAsync, + * wires mic-input mode to Ctrl+C (session.StopAsync) and an always-constructed + * SilenceTimeoutRecognizerSession decorator around GetResultsAsync (file mode + * pumps the session directly, with no decorator needed), filters + * console output per --interim/--final-only, writes only final results to + * --output-text, and disposes the engine and session (await using) on every + * exit path. */ comment sourceRef /* Source: src/DemaConsulting.Speech.Cli/Commands/RecognitionCommandSubsystem/RecognizeCommand.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/RecognizeCommandTests.cs */ } part def SilenceTimeoutRecognizerSession { - doc /* Observes an ISpeechRecognizer's ResultReceived event and calls Stop(), - * then raises TimedOut, when no result (partial or final) arrives within a - * configured idle window. Enforces two sequential idle windows via a - * TimeProvider-based timer: a start-timeout grace period (defaulting to the - * silence-timeout value) armed once at construction, before any result has - * arrived, then re-armed with the silence-timeout value on every subsequent - * result from the first onward - deterministic and real-wall-clock-free for - * unit testing. */ + doc /* Stateless IAsyncEnumerable decorator wrapping an + * IRecognitionSession's GetResultsAsync: each iteration races the wrapped + * enumerator's MoveNextAsync against a TimeProvider-based Task.Delay via + * Task.WhenAny, calls the wrapped session's StopAsync and raises TimedOut when + * the delay wins before any result (partial or final) arrives, then continues + * draining so trailing results still surface. Enforces two sequential idle + * windows: a start-timeout grace period (defaulting to the silence-timeout + * value) armed once before any result has arrived, then re-armed with the + * silence-timeout value on every subsequent result from the first onward - no + * IDisposable, no locks, and deterministic/real-wall-clock-free for unit + * testing since cleanup is solely via the compiler-generated await using on + * the inner enumerator. */ comment sourceRef /* Source: src/DemaConsulting.Speech.Cli/Commands/RecognitionCommandSubsystem/SilenceTimeoutRecognizerSession.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/SilenceTimeoutRecognizerSessionTests.cs */ diff --git a/docs/sysml2/model/speech-cli/synthesis-command-subsystem.sysml b/docs/sysml2/model/speech-cli/synthesis-command-subsystem.sysml index b47fec8..e7ffa74 100644 --- a/docs/sysml2/model/speech-cli/synthesis-command-subsystem.sysml +++ b/docs/sysml2/model/speech-cli/synthesis-command-subsystem.sysml @@ -3,10 +3,10 @@ package SpeechCli { doc /* The one text-to-speech subcommand (speak), plus the ParameterBagParser * --tts-param validation utility, extending the ModelCommandsSubsystem's * ICliModelCatalog seam with two further members (GetPreferredAudioFormat, - * CreateSynthesizer) rather than introducing a second, competing seam, and - * its own small ICliPlaybackDeviceSource seam over real playback-device - * resolution, letting speak's device-dispatch logic be unit tested - * deterministically without depending on real PortAudio hardware. */ + * CreateSynthesizerEngineAsync) rather than introducing a second, competing + * seam, and its own small ICliPlaybackDeviceSource seam over real + * playback-device resolution, letting speak's device-dispatch logic be unit + * tested deterministically without depending on real PortAudio hardware. */ comment designRef /* Design: docs/design/speech-cli/synthesis-command-subsystem.md */ comment verificationRef /* Verification: docs/verification/speech-cli/synthesis-command-subsystem.md */ @@ -23,8 +23,11 @@ package SpeechCli { * exclusive), resolves and validates the requested synthesis model, validates * --tts-param values, strips audio tags when --no-tags is given, dispatches to * either a WAV file (--output-audio) or a real playback device - * (--playback-device/default), and disposes the synthesizer before the - * playback device. */ + * (--playback-device/default), loads an ISpeechSynthesizerEngine via + * ICliModelCatalog.CreateSynthesizerEngineAsync, binds an ISynthesisSession to + * the resolved device, awaits session.SpeakAsync, and disposes the session, + * then the engine, then the playback device (await using in declaration + * order, reversed) on every exit path. */ comment sourceRef /* Source: src/DemaConsulting.Speech.Cli/Commands/SynthesisCommandSubsystem/SpeakCommand.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/SpeakCommandTests.cs */ diff --git a/docs/sysml2/model/speech-demo/recognition-panel-subsystem.sysml b/docs/sysml2/model/speech-demo/recognition-panel-subsystem.sysml index d4a2e93..eca59bc 100644 --- a/docs/sysml2/model/speech-demo/recognition-panel-subsystem.sysml +++ b/docs/sysml2/model/speech-demo/recognition-panel-subsystem.sysml @@ -16,8 +16,11 @@ package SpeechDemo { part def IRecognizerSessionFactory { doc /* Demo-owned seam accepting the library's public ISpeechModel contract to compose - * a speech recognizer, so the panel can be tested with a plain fake model with no - * InternalsVisibleTo grant from the library. */ + * a speech recognizer engine, so the panel can be tested with a plain fake model + * with no InternalsVisibleTo grant from the library. Exposes a single LoadAsync + * member returning an ISpeechRecognizerEngine; per-run capture-device binding is + * done later, directly against the returned engine's CreateSessionAsync, so a host + * can load one engine per model and reuse it across many IRecognitionSession runs. */ comment sourceRef /* Source: src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/IRecognizerSessionFactory.cs */ comment designRef /* Design: docs/design/speech-demo/recognition-panel-subsystem.md (documented as a subsystem unit) */ @@ -28,8 +31,8 @@ package SpeechDemo { part def RecognizerSessionFactory { doc /* Real seam implementation: resolves the model's installed-files directory from * the shared SpeechModelStore, narrows the public model to IRecognitionModel, and - * forwards to SpeechRecognizerFactory.Create, returning the library's own - * UnavailableSpeechRecognizer for a model of the wrong role. */ + * forwards to SpeechRecognizerFactory.LoadAsync, returning the library's own + * UnavailableSpeechRecognizerEngine for a model of the wrong role. */ comment sourceRef /* Source: src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/RecognizerSessionFactory.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Demo.Tests/RecognitionPanelSubsystem/RecognizerSessionFactoryTests.cs */ @@ -40,15 +43,20 @@ package SpeechDemo { part def RecognitionPanelViewModel { doc /* Presentation state for the speech-to-text panel: installed recognition models, - * the Start/Stop streaming lifecycle, the progressive partial-then-final transcript, - * a CanChangeModel guard that disables model selection while Listening, and honest - * unavailable-state reporting. Caches at most one composed recognizer and reuses it - * across repeated Start/Stop cycles instead of recomposing it on every click, - * invalidating the cache - stopping an active session first, even while Listening - - * whenever the selected model changes, the selected capture device changes, or a - * shared device refresh replaces the underlying capture device. Implements - * IDisposable to release an active recognizer and its capture device - * deterministically. */ + * the async StartAsync/StopAsync streaming lifecycle driving an ISpeechRecognizerEngine + * and a single-use IRecognitionSession, the progressive partial-then-final transcript + * pumped from IRecognitionSession.GetResultsAsync, a CanChangeModel guard that + * disables model selection while Listening, and honest unavailable-state reporting. + * RecognitionStreamingState is derived from RecognitionSessionState via + * IRecognitionSession.StateChanged rather than ad hoc assignment at each call site. + * Caches at most one loaded engine (reused across repeated Start/Stop cycles for as + * long as the selected model is unchanged, instead of reloading the model on every + * click) and creates a fresh session from it on every Start, since a session is + * single-use; invalidates the engine - stopping and releasing an active session + * first, even while Listening - whenever the selected model changes, and invalidates + * only the session (keeping the engine) whenever the selected capture device changes + * or a shared device refresh replaces the underlying capture device. Implements + * IAsyncDisposable to release an active session and engine deterministically. */ comment sourceRef /* Source: src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/RecognitionPanelViewModel.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Demo.Tests/RecognitionPanelSubsystem/RecognitionPanelViewModelTests.cs */ diff --git a/docs/sysml2/model/speech-demo/synthesis-panel-subsystem.sysml b/docs/sysml2/model/speech-demo/synthesis-panel-subsystem.sysml index 10174f2..0f2c9be 100644 --- a/docs/sysml2/model/speech-demo/synthesis-panel-subsystem.sysml +++ b/docs/sysml2/model/speech-demo/synthesis-panel-subsystem.sysml @@ -17,9 +17,13 @@ package SpeechDemo { part def ISynthesizerSessionFactory { doc /* Demo-owned seam accepting the library's public ISpeechModel contract to compose - * a speech synthesizer, so the panel can be tested with a plain fake model with no - * InternalsVisibleTo grant from the library. Forwards an optional parameterValues - * bag (the settings panel's Settings.BuildValueBag()) unchanged. */ + * a speech synthesizer engine, so the panel can be tested with a plain fake model + * with no InternalsVisibleTo grant from the library. Exposes a single LoadAsync + * member returning an ISpeechSynthesizerEngine, forwarding an optional + * parameterValues bag (the settings panel's Settings.BuildValueBag()) unchanged; + * per-run playback-device binding is done later, directly against the returned + * engine's CreateSessionAsync, so a host can load one engine per model/parameter + * combination and reuse it across many ISynthesisSession runs. */ comment sourceRef /* Source: src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/ISynthesizerSessionFactory.cs */ comment designRef /* Design: docs/design/speech-demo/synthesis-panel-subsystem/synthesizer-session-factory.md (documented together with its sole implementation, SynthesizerSessionFactory) */ @@ -28,11 +32,11 @@ package SpeechDemo { } part def SynthesizerSessionFactory { - doc /* Real seam implementation: resolves the model's installed-files directory from - * the shared SpeechModelStore, narrows the public model to ISynthesisModel, and - * forwards to SpeechSynthesizerFactory.Create (including an optional parameterValues - * bag unchanged), returning the library's own UnavailableSpeechSynthesizer for a - * model of the wrong role. */ + doc /* Real seam implementation: narrows the public model to ISynthesisModel and + * forwards it, together with the shared SpeechModelStore and an optional + * parameterValues bag unchanged, to SpeechSynthesizerFactory.LoadAsync, returning + * the library's own UnavailableSpeechSynthesizerEngine for a model of the wrong + * role. */ comment sourceRef /* Source: src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/SynthesizerSessionFactory.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Demo.Tests/SynthesisPanelSubsystem/SynthesizerSessionFactoryTests.cs */ @@ -44,10 +48,18 @@ package SpeechDemo { part def SynthesisPanelViewModel { doc /* Presentation state for the text-to-speech panel: installed synthesis models, * the embedded settings panel for the selected model, example audio tag hints, the - * Play/Stop lifecycle with honest unavailable-state reporting, and a CanChangeModel - * guard that disables model/voice selection while Synthesizing or Playing. Forwards - * Settings.BuildValueBag() to the session factory on Play, so a selected voice - * genuinely changes synthesized output. */ + * async PlayAsync/StopAsync lifecycle driving an ISpeechSynthesizerEngine and an + * ISynthesisSession, and a CanChangeModel guard that disables model/voice selection + * while Synthesizing or Playing. SynthesisPlaybackState is derived from + * SynthesisSessionState via ISynthesisSession.StateChanged rather than ad hoc + * assignment at each call site. Caches at most one loaded engine (reloaded only when + * the selected model or the settings panel's built parameter values change) and at + * most one session created from it (recreated only when the engine was just + * reloaded or the selected playback device changes), reusing both across many Play + * calls instead of recomposing them on every click - fixing the bug where a prior + * design recreated the engine and session on every Play. Forwards + * Settings.BuildValueBag() to the session factory on an engine reload, so a selected + * voice genuinely changes synthesized output. */ comment sourceRef /* Source: src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/SynthesisPanelViewModel.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Demo.Tests/SynthesisPanelSubsystem/SynthesisPanelViewModelTests.cs */ diff --git a/docs/sysml2/model/speech/recognition-subsystem.sysml b/docs/sysml2/model/speech/recognition-subsystem.sysml index 9561b74..e980864 100644 --- a/docs/sysml2/model/speech/recognition-subsystem.sysml +++ b/docs/sysml2/model/speech/recognition-subsystem.sysml @@ -1,20 +1,31 @@ package Speech { part def RecognitionSubsystem { - doc /* Streaming speech-to-text: the public recognizer contract and its result/event - * types, the composition root that returns either a working recognizer or an honest - * unavailable fallback, the capture-to-engine pipeline with its audio-format - * converter, and the internal mockable speech-inference seam. */ + doc /* Streaming speech-to-text: a Layer 3 engine contract bound to one loaded + * recognition model, a Layer 5 session contract bound to one capture device, the + * composition root that returns either a working engine or an honest unavailable + * fallback, the capture-to-backend pipeline with its audio-format converter and + * backpressure buffer, the cooperative-cancel-then-abandon native-call policy, and + * the internal mockable speech-inference seam. */ comment designRef /* Design: docs/design/speech/recognition-subsystem.md */ comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem.md */ comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem.yaml */ - part iSpeechRecognizer : ISpeechRecognizer; + part iSpeechRecognizerEngine : ISpeechRecognizerEngine; + part iRecognitionSession : IRecognitionSession; + part recognitionSessionState : RecognitionSessionState; + part sessionStateChangedEventArgs : SessionStateChangedEventArgs; + part recognitionSessionFaultedException : RecognitionSessionFaultedException; + part recognitionEngineBusyException : RecognitionEngineBusyException; part speechRecognizerFactory : SpeechRecognizerFactory; - part sherpaOnnxSpeechRecognizer : SherpaOnnxSpeechRecognizer; - part recognitionEngine : RecognitionEngine; + part sherpaOnnxRecognitionSession : SherpaOnnxRecognitionSession; + part sherpaOnnxSpeechRecognizerEngine : SherpaOnnxSpeechRecognizerEngine; + part recognitionBackend : RecognitionBackend; + part recognitionResultBuffer : RecognitionResultBuffer; + part dedicatedWorker : DedicatedWorker; part audioFrameResampler : AudioFrameResampler; - part unavailableSpeechRecognizer : UnavailableSpeechRecognizer; + part unavailableSpeechRecognizerEngine : UnavailableSpeechRecognizerEngine; + part unavailableRecognitionSession : UnavailableRecognitionSession; part speechRecognizerUnavailableException : SpeechRecognizerUnavailableException; } } diff --git a/docs/sysml2/model/speech/recognition-subsystem/audio-frame-resampler.sysml b/docs/sysml2/model/speech/recognition-subsystem/audio-frame-resampler.sysml index 79e0eaa..a447316 100644 --- a/docs/sysml2/model/speech/recognition-subsystem/audio-frame-resampler.sysml +++ b/docs/sysml2/model/speech/recognition-subsystem/audio-frame-resampler.sysml @@ -8,8 +8,8 @@ package Speech { comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/AudioFrameResampler.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/AudioFrameResamplerTests.cs */ - comment designRef /* Design: docs/design/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md (documented with the pipeline it serves) */ - comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md (documented with the pipeline it serves) */ - comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.yaml */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/sherpa-onnx-recognition-session.md (documented with the pipeline it serves) */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/sherpa-onnx-recognition-session.md (documented with the pipeline it serves) */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/sherpa-onnx-recognition-session.yaml */ } } diff --git a/docs/sysml2/model/speech/recognition-subsystem/dedicated-worker.sysml b/docs/sysml2/model/speech/recognition-subsystem/dedicated-worker.sysml new file mode 100644 index 0000000..2d45d91 --- /dev/null +++ b/docs/sysml2/model/speech/recognition-subsystem/dedicated-worker.sysml @@ -0,0 +1,16 @@ +package Speech { + part def DedicatedWorker { + doc /* Internal long-running-task helper that runs one blocking delegate on its own + * dedicated thread (TaskCreationOptions.LongRunning) and applies a + * cooperative-cancel-then-abandon policy: requests cancellation first, then abandons + * the delegate after a bounded timeout (default 2 seconds) if it has not observed + * cancellation, reporting the abandonment through ISpeechDiagnostics at Warning level + * rather than blocking the caller forever on a stuck native call. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/DedicatedWorker.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/DedicatedWorkerTests.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/sherpa-onnx-recognition-session.md (documented with the pipeline it serves) */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/sherpa-onnx-recognition-session.md (documented with the pipeline it serves) */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/dedicated-worker.yaml */ + } +} diff --git a/docs/sysml2/model/speech/recognition-subsystem/i-recognition-session.sysml b/docs/sysml2/model/speech/recognition-subsystem/i-recognition-session.sysml new file mode 100644 index 0000000..bb49c8f --- /dev/null +++ b/docs/sysml2/model/speech/recognition-subsystem/i-recognition-session.sysml @@ -0,0 +1,17 @@ +package Speech { + part def IRecognitionSession { + doc /* Layer 5 contract for one streaming recognition session bound to one capture device: + * a forward-only RecognitionSessionState machine with a StateChanged event, + * StartAsync/StopAsync lifecycle control, and GetResultsAsync, an async-enumerable, + * single-consumer stream of ordered provisional and final SpeechRecognitionEvent + * results. Single-use: a session may not be restarted after Stopped. A session fault + * surfaces as RecognitionSessionFaultedException from the active GetResultsAsync + * enumeration. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionSession.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionSessionTests.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/i-recognition-session.md */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/i-recognition-session.md */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/i-recognition-session.yaml */ + } +} diff --git a/docs/sysml2/model/speech/recognition-subsystem/i-speech-recognizer-engine.sysml b/docs/sysml2/model/speech/recognition-subsystem/i-speech-recognizer-engine.sysml new file mode 100644 index 0000000..927114d --- /dev/null +++ b/docs/sysml2/model/speech/recognition-subsystem/i-speech-recognizer-engine.sysml @@ -0,0 +1,15 @@ +package Speech { + part def ISpeechRecognizerEngine { + doc /* Layer 3 contract for one loaded speech-inference engine bound to one recognition + * model: an availability flag plus CreateSessionAsync, which leases exclusive use of + * the engine to a new IRecognitionSession for one capture device. Exposes no + * inference-engine type. Never throws for an ordinary unavailable machine state; + * throws RecognitionEngineBusyException only when a session is already leased. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/ISpeechRecognizerEngine.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerEngineTests.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/i-speech-recognizer-engine.md */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/i-speech-recognizer-engine.md */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/i-speech-recognizer-engine.yaml */ + } +} diff --git a/docs/sysml2/model/speech/recognition-subsystem/i-speech-recognizer.sysml b/docs/sysml2/model/speech/recognition-subsystem/i-speech-recognizer.sysml deleted file mode 100644 index a4c6530..0000000 --- a/docs/sysml2/model/speech/recognition-subsystem/i-speech-recognizer.sysml +++ /dev/null @@ -1,11 +0,0 @@ -package Speech { - part def ISpeechRecognizer { - doc /* Public streaming speech-to-text contract (availability flag, start/stop, recognition-result event) plus its immutable SpeechRecognitionResult/SpeechRecognitionEvent payload types, exposing no inference-engine type. */ - - comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/ISpeechRecognizer.cs */ - comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerTests.cs */ - comment designRef /* Design: docs/design/speech/recognition-subsystem/i-speech-recognizer.md */ - comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/i-speech-recognizer.md */ - comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/i-speech-recognizer.yaml */ - } -} \ No newline at end of file diff --git a/docs/sysml2/model/speech/recognition-subsystem/recognition-engine.sysml b/docs/sysml2/model/speech/recognition-subsystem/recognition-backend.sysml similarity index 57% rename from docs/sysml2/model/speech/recognition-subsystem/recognition-engine.sysml rename to docs/sysml2/model/speech/recognition-subsystem/recognition-backend.sysml index 929a4fb..e57a2c0 100644 --- a/docs/sysml2/model/speech/recognition-subsystem/recognition-engine.sysml +++ b/docs/sysml2/model/speech/recognition-subsystem/recognition-backend.sysml @@ -1,13 +1,13 @@ package Speech { - part def RecognitionEngine { - doc /* Internal mockable seam over one loaded speech-inference engine, together with its + part def RecognitionBackend { + doc /* Internal mockable seam over one loaded speech-inference backend, together with its * factory and their real sherpa-onnx implementations, confining every inference * interop call so the recognition pipeline is testable with pure managed fakes. */ - comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionEngine.cs */ - comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerTests.cs */ - comment designRef /* Design: docs/design/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md (documented with the pipeline it serves) */ - comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md (documented with the pipeline it serves) */ - comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.yaml */ + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionBackend.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionEngineTests.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/sherpa-onnx-recognition-session.md (documented with the pipeline it serves) */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/sherpa-onnx-recognition-session.md (documented with the pipeline it serves) */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/sherpa-onnx-recognition-session.yaml */ } } diff --git a/docs/sysml2/model/speech/recognition-subsystem/recognition-engine-busy-exception.sysml b/docs/sysml2/model/speech/recognition-subsystem/recognition-engine-busy-exception.sysml new file mode 100644 index 0000000..39ba54b --- /dev/null +++ b/docs/sysml2/model/speech/recognition-subsystem/recognition-engine-busy-exception.sysml @@ -0,0 +1,13 @@ +package Speech { + part def RecognitionEngineBusyException { + doc /* Exception signaling that ISpeechRecognizerEngine.CreateSessionAsync was called while + * the engine's single lease is already held by another live IRecognitionSession. The + * engine fails fast (no queueing) rather than blocking the caller. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionEngineBusyException.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerEngineTests.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/i-speech-recognizer-engine.md (documented as a supporting fault type) */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/i-speech-recognizer-engine.md (documented as a supporting fault type) */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/recognition-session-lease.yaml */ + } +} diff --git a/docs/sysml2/model/speech/recognition-subsystem/recognition-result-buffer.sysml b/docs/sysml2/model/speech/recognition-subsystem/recognition-result-buffer.sysml new file mode 100644 index 0000000..7f45f45 --- /dev/null +++ b/docs/sysml2/model/speech/recognition-subsystem/recognition-result-buffer.sysml @@ -0,0 +1,15 @@ +package Speech { + part def RecognitionResultBuffer { + doc /* Internal two-tier backpressure buffer used by SherpaOnnxRecognitionSession between + * the pump thread and a possibly-slow GetResultsAsync consumer: one overwritable + * "latest provisional" slot plus a byte-capped (64 KiB) FIFO of final results, evicting + * the oldest final only as a last resort and reporting that eviction as a Warning + * diagnostic. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionResultBuffer.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionSessionTests.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/sherpa-onnx-recognition-session.md (documented with the pipeline it serves) */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/sherpa-onnx-recognition-session.md (documented with the pipeline it serves) */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/sherpa-onnx-recognition-session.yaml */ + } +} diff --git a/docs/sysml2/model/speech/recognition-subsystem/recognition-session-faulted-exception.sysml b/docs/sysml2/model/speech/recognition-subsystem/recognition-session-faulted-exception.sysml new file mode 100644 index 0000000..77d3af9 --- /dev/null +++ b/docs/sysml2/model/speech/recognition-subsystem/recognition-session-faulted-exception.sysml @@ -0,0 +1,13 @@ +package Speech { + part def RecognitionSessionFaultedException { + doc /* Exception surfaced from an active IRecognitionSession.GetResultsAsync enumeration + * when the session transitions to RecognitionSessionState.Faulted, wrapping the + * underlying cause (for example a capture device becoming unavailable mid-session). */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionSessionFaultedException.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionSessionTests.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/i-recognition-session.md (documented as a supporting fault type) */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/i-recognition-session.md (documented as a supporting fault type) */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/i-recognition-session.yaml */ + } +} diff --git a/docs/sysml2/model/speech/recognition-subsystem/recognition-session-state.sysml b/docs/sysml2/model/speech/recognition-subsystem/recognition-session-state.sysml new file mode 100644 index 0000000..0724a7d --- /dev/null +++ b/docs/sysml2/model/speech/recognition-subsystem/recognition-session-state.sysml @@ -0,0 +1,12 @@ +package Speech { + part def RecognitionSessionState { + doc /* Forward-only IRecognitionSession lifecycle enumeration: Created, Starting, Running, + * Stopping, Stopped, Disposing, Disposed, Faulted. Faulted is reachable from Starting, + * Running, or Stopping; no other backward or skipped transition is valid. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionSessionState.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/i-recognition-session.md (documented as a supporting value type) */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/i-recognition-session.md (documented as a supporting value type) */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/i-recognition-session.yaml */ + } +} diff --git a/docs/sysml2/model/speech/recognition-subsystem/session-state-changed-event-args.sysml b/docs/sysml2/model/speech/recognition-subsystem/session-state-changed-event-args.sysml new file mode 100644 index 0000000..3a3d997 --- /dev/null +++ b/docs/sysml2/model/speech/recognition-subsystem/session-state-changed-event-args.sysml @@ -0,0 +1,11 @@ +package Speech { + part def SessionStateChangedEventArgs { + doc /* Immutable record payload for IRecognitionSession.StateChanged, carrying the + * Previous and Current RecognitionSessionState values of one transition. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/SessionStateChangedEventArgs.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/i-recognition-session.md (documented as a supporting value type) */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/i-recognition-session.md (documented as a supporting value type) */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/i-recognition-session.yaml */ + } +} diff --git a/docs/sysml2/model/speech/recognition-subsystem/sherpa-onnx-recognition-session.sysml b/docs/sysml2/model/speech/recognition-subsystem/sherpa-onnx-recognition-session.sysml new file mode 100644 index 0000000..0b1ee31 --- /dev/null +++ b/docs/sysml2/model/speech/recognition-subsystem/sherpa-onnx-recognition-session.sysml @@ -0,0 +1,17 @@ +package Speech { + part def SherpaOnnxRecognitionSession { + doc /* Real streaming IRecognitionSession implementation: enqueues captured audio off the + * audio callback thread via a bounded drop-oldest channel, resamples it to the model's + * required format, drains it on a DedicatedWorker pump thread through the leased + * IRecognitionBackend, normalizes each result's text via the owning + * IRecognitionModel.NormalizeText, buffers ordered provisional and final results in a + * RecognitionResultBuffer, and streams them through GetResultsAsync. The backend stays + * owned and "hot" on the owning SherpaOnnxSpeechRecognizerEngine across sessions. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionSession.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionSessionTests.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/sherpa-onnx-recognition-session.md */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/sherpa-onnx-recognition-session.md */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/sherpa-onnx-recognition-session.yaml */ + } +} \ No newline at end of file diff --git a/docs/sysml2/model/speech/recognition-subsystem/sherpa-onnx-speech-recognizer-engine.sysml b/docs/sysml2/model/speech/recognition-subsystem/sherpa-onnx-speech-recognizer-engine.sysml new file mode 100644 index 0000000..c324afe --- /dev/null +++ b/docs/sysml2/model/speech/recognition-subsystem/sherpa-onnx-speech-recognizer-engine.sysml @@ -0,0 +1,15 @@ +package Speech { + part def SherpaOnnxSpeechRecognizerEngine { + doc /* Real ISpeechRecognizerEngine implementation: holds one loaded, "hot" IRecognitionBackend + * and a single-slot exclusivity lease (SemaphoreSlim(1,1), fail-fast, no queueing), and + * constructs a new SherpaOnnxRecognitionSession bound to one capture device per + * successful CreateSessionAsync call. Throws RecognitionEngineBusyException rather than + * blocking when the lease is already held. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxSpeechRecognizerEngine.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerEngineTests.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/sherpa-onnx-recognition-session.md (documented with the pipeline it serves) */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/sherpa-onnx-recognition-session.md (documented with the pipeline it serves) */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/recognition-session-lease.yaml */ + } +} diff --git a/docs/sysml2/model/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.sysml b/docs/sysml2/model/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.sysml deleted file mode 100644 index 74e35b0..0000000 --- a/docs/sysml2/model/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.sysml +++ /dev/null @@ -1,11 +0,0 @@ -package Speech { - part def SherpaOnnxSpeechRecognizer { - doc /* Real streaming recognition pipeline: enqueues captured audio off the audio callback thread, converts it to the model required format, feeds it to the recognition engine, normalizes each result's text via the owning IRecognitionModel.NormalizeText, and raises ordered provisional and final results. */ - - comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxSpeechRecognizer.cs */ - comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerTests.cs */ - comment designRef /* Design: docs/design/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md */ - comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md */ - comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.yaml */ - } -} \ No newline at end of file diff --git a/docs/sysml2/model/speech/recognition-subsystem/speech-recognizer-factory.sysml b/docs/sysml2/model/speech/recognition-subsystem/speech-recognizer-factory.sysml index ca2a11b..ca617c8 100644 --- a/docs/sysml2/model/speech/recognition-subsystem/speech-recognizer-factory.sysml +++ b/docs/sysml2/model/speech/recognition-subsystem/speech-recognizer-factory.sysml @@ -1,6 +1,12 @@ package Speech { part def SpeechRecognizerFactory { - doc /* Composition entry point that returns a working speech recognizer for an installed recognition model and available capture device, or an honest unavailable fallback, without ever throwing for an ordinary machine state. The preferred composition is to open the capture device with the model's declared AudioFormat first when the backend can honor that hint. */ + doc /* Composition entry point whose LoadAsync overloads return a working + * ISpeechRecognizerEngine for an installed recognition model, or an honest + * UnavailableSpeechRecognizerEngine fallback, without ever throwing for an ordinary + * machine state. Loading runs on a dedicated worker thread so the expensive native + * backend construction never blocks the calling thread. A capture device is bound + * later, per session, via ISpeechRecognizerEngine.CreateSessionAsync - not here - so + * one loaded engine can be reused across many devices or many sequential sessions. */ comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognizerFactory.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SpeechRecognizerFactoryTests.cs */ diff --git a/docs/sysml2/model/speech/recognition-subsystem/speech-recognizer-unavailable-exception.sysml b/docs/sysml2/model/speech/recognition-subsystem/speech-recognizer-unavailable-exception.sysml index 12c52a4..c938f1c 100644 --- a/docs/sysml2/model/speech/recognition-subsystem/speech-recognizer-unavailable-exception.sysml +++ b/docs/sysml2/model/speech/recognition-subsystem/speech-recognizer-unavailable-exception.sysml @@ -1,12 +1,13 @@ package Speech { part def SpeechRecognizerUnavailableException { doc /* Exception signaling that an operational member of an unavailable speech recognizer - * was invoked, or that a recognizer which claimed to be available failed on first use. */ + * engine or session was invoked, or that a session which claimed to be available + * failed on first use (for example the capture device going unavailable mid-session). */ comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognizerUnavailableException.cs */ - comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableSpeechRecognizerTests.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableSpeechRecognizerEngineTests.cs */ comment designRef /* Design: docs/design/speech/recognition-subsystem.md (documented as a supporting fallback type) */ comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem.md (documented as a supporting fallback type) */ - comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer.yaml */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer-engine.yaml */ } } diff --git a/docs/sysml2/model/speech/recognition-subsystem/unavailable-recognition-session.sysml b/docs/sysml2/model/speech/recognition-subsystem/unavailable-recognition-session.sysml new file mode 100644 index 0000000..a58be9d --- /dev/null +++ b/docs/sysml2/model/speech/recognition-subsystem/unavailable-recognition-session.sysml @@ -0,0 +1,14 @@ +package Speech { + part def UnavailableRecognitionSession { + doc /* Honest "unavailable" IRecognitionSession fallback returned by + * UnavailableSpeechRecognizerEngine.CreateSessionAsync: reports IsAvailable as false + * and State as Created, and throws SpeechRecognizerUnavailableException only when an + * operational member is invoked. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableRecognitionSession.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableRecognitionSessionTests.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/unavailable-recognition-session.md */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/unavailable-recognition-session.md */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/unavailable-recognition-session.yaml */ + } +} diff --git a/docs/sysml2/model/speech/recognition-subsystem/unavailable-speech-recognizer-engine.sysml b/docs/sysml2/model/speech/recognition-subsystem/unavailable-speech-recognizer-engine.sysml new file mode 100644 index 0000000..0191685 --- /dev/null +++ b/docs/sysml2/model/speech/recognition-subsystem/unavailable-speech-recognizer-engine.sysml @@ -0,0 +1,13 @@ +package Speech { + part def UnavailableSpeechRecognizerEngine { + doc /* Honest "unavailable" ISpeechRecognizerEngine fallback: reports IsAvailable as false + * and CreateSessionAsync returns UnavailableRecognitionSession.Instance rather than + * throwing for an ordinary unavailable machine state. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableSpeechRecognizerEngine.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableSpeechRecognizerEngineTests.cs */ + comment designRef /* Design: docs/design/speech/recognition-subsystem/unavailable-speech-recognizer-engine.md */ + comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/unavailable-speech-recognizer-engine.md */ + comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer-engine.yaml */ + } +} diff --git a/docs/sysml2/model/speech/recognition-subsystem/unavailable-speech-recognizer.sysml b/docs/sysml2/model/speech/recognition-subsystem/unavailable-speech-recognizer.sysml deleted file mode 100644 index d98a046..0000000 --- a/docs/sysml2/model/speech/recognition-subsystem/unavailable-speech-recognizer.sysml +++ /dev/null @@ -1,11 +0,0 @@ -package Speech { - part def UnavailableSpeechRecognizer { - doc /* Honest "unavailable" ISpeechRecognizer fallback: reports IsAvailable as false and throws SpeechRecognizerUnavailableException only when an operational member is invoked. */ - - comment sourceRef /* Source: src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableSpeechRecognizer.cs */ - comment testRef /* Test: test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableSpeechRecognizerTests.cs */ - comment designRef /* Design: docs/design/speech/recognition-subsystem/unavailable-speech-recognizer.md */ - comment verificationRef /* Verification: docs/verification/speech/recognition-subsystem/unavailable-speech-recognizer.md */ - comment reqRef /* Requirements: docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer.yaml */ - } -} \ No newline at end of file diff --git a/docs/sysml2/model/speech/synthesis-subsystem.sysml b/docs/sysml2/model/speech/synthesis-subsystem.sysml index ce4fd38..9c025b7 100644 --- a/docs/sysml2/model/speech/synthesis-subsystem.sysml +++ b/docs/sysml2/model/speech/synthesis-subsystem.sysml @@ -1,24 +1,34 @@ package Speech { part def SynthesisSubsystem { doc /* Text-to-speech: the closed, fixed Natural Language Audio Tag vocabulary and the - * model-independent Layer 1 parser (Sub-phase 4a), plus Layer 2 per-model rendering, - * sentence chunking, the public ISpeechSynthesizer streaming/playback pipeline with - * its composition root and honest unavailable fallback, and the real sherpa-onnx - * synthesis engine and playback-format converter (Sub-phase 4b). */ + * model-independent Layer 1 parser, plus Layer 2 per-model rendering, sentence + * chunking, the public Layer 3/Layer 5 Engine/Session text-to-speech contract + * (ISpeechSynthesizerEngine/ISynthesisSession) with its composition root and honest + * unavailable fallbacks at every layer, engine exclusivity leasing, a + * cancel-then-abandon dedicated worker, and the real sherpa-onnx synthesis backend + * and playback-format converter. */ comment designRef /* Design: docs/design/speech/synthesis-subsystem.md */ comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem.md */ comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem.yaml */ part audioTagParser : AudioTagParser; - part iSpeechSynthesizer : ISpeechSynthesizer; + part iSpeechSynthesizerEngine : ISpeechSynthesizerEngine; + part iSynthesisSession : ISynthesisSession; part speechSynthesizerFactory : SpeechSynthesizerFactory; - part sherpaOnnxSpeechSynthesizer : SherpaOnnxSpeechSynthesizer; - part synthesisEngine : SynthesisEngine; + part sherpaOnnxSpeechSynthesizerEngine : SherpaOnnxSpeechSynthesizerEngine; + part sherpaOnnxSynthesisSession : SherpaOnnxSynthesisSession; + part synthesisBackend : SynthesisBackend; part layer2Rendering : Layer2Rendering; part sentenceChunker : SentenceChunker; part playbackAudioResampler : PlaybackAudioResampler; - part unavailableSpeechSynthesizer : UnavailableSpeechSynthesizer; + part unavailableSpeechSynthesizerEngine : UnavailableSpeechSynthesizerEngine; + part unavailableSynthesisSession : UnavailableSynthesisSession; part speechSynthesizerUnavailableException : SpeechSynthesizerUnavailableException; + part synthesisSessionState : SynthesisSessionState; + part sessionStateChangedEventArgs : SessionStateChangedEventArgs; + part synthesisEngineBusyException : SynthesisEngineBusyException; + part synthesisSessionFaultedException : SynthesisSessionFaultedException; + part dedicatedWorker : DedicatedWorker; } } diff --git a/docs/sysml2/model/speech/synthesis-subsystem/dedicated-worker.sysml b/docs/sysml2/model/speech/synthesis-subsystem/dedicated-worker.sysml new file mode 100644 index 0000000..83b3caf --- /dev/null +++ b/docs/sysml2/model/speech/synthesis-subsystem/dedicated-worker.sysml @@ -0,0 +1,18 @@ +package Speech { + part def DedicatedWorker { + doc /* Internal utility that runs a delegate on a dedicated, long-running background + * thread, applying a cooperative-cancel-then-abandon policy for native calls that do + * not honor cancellation promptly: on cancellation the worker gets a bounded grace + * period to stop cooperatively, after which the awaited task completes as cancelled, + * a Warning diagnostic is reported, and the worker thread is detached. This is the + * synthesis subsystem's own copy, duplicated rather than shared with + * RecognitionSubsystem's identically-shaped internal utility per this library's + * subsystem test-dependency boundary. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/DedicatedWorker.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/DedicatedWorkerTests.cs */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md (documented with the pipeline it serves) */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md (documented with the pipeline it serves) */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/dedicated-worker.yaml */ + } +} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/i-speech-synthesizer-engine.sysml b/docs/sysml2/model/speech/synthesis-subsystem/i-speech-synthesizer-engine.sysml new file mode 100644 index 0000000..6cc4d2b --- /dev/null +++ b/docs/sysml2/model/speech/synthesis-subsystem/i-speech-synthesizer-engine.sysml @@ -0,0 +1,17 @@ +package Speech { + part def ISpeechSynthesizerEngine { + doc /* Layer 3: the public, loaded, expensive, native-backed text-to-speech model + * contract - availability flag, async CreateSessionAsync to bind a playback device + * into an ISynthesisSession, and one-shot SpeakAsync/SynthesizeAsync convenience + * overloads that create a session internally and dispose it - plus its adjacent, + * immutable SynthesizedSpeech audio-segment payload type, exposing no + * inference-engine type. Obtaining this engine is the expensive step; a session + * obtained from it is cheap and reusable across many calls. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/ISpeechSynthesizerEngine.cs, SynthesizedSpeech.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerEngineTests.cs */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/i-speech-synthesizer-engine.md */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/i-speech-synthesizer-engine.md */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/i-speech-synthesizer-engine.yaml */ + } +} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/i-speech-synthesizer.sysml b/docs/sysml2/model/speech/synthesis-subsystem/i-speech-synthesizer.sysml deleted file mode 100644 index 7b145fb..0000000 --- a/docs/sysml2/model/speech/synthesis-subsystem/i-speech-synthesizer.sysml +++ /dev/null @@ -1,13 +0,0 @@ -package Speech { - part def ISpeechSynthesizer { - doc /* Public streaming text-to-speech contract (availability flag, streaming/playing - * SpeakAsync, Stop), plus its immutable SynthesizedSpeech audio-segment payload - * type, exposing no inference-engine type. */ - - comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/ISpeechSynthesizer.cs */ - comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerTests.cs */ - comment designRef /* Design: docs/design/speech/synthesis-subsystem/i-speech-synthesizer.md */ - comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/i-speech-synthesizer.md */ - comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/i-speech-synthesizer.yaml */ - } -} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/i-synthesis-session.sysml b/docs/sysml2/model/speech/synthesis-subsystem/i-synthesis-session.sysml new file mode 100644 index 0000000..68e3ca4 --- /dev/null +++ b/docs/sysml2/model/speech/synthesis-subsystem/i-synthesis-session.sysml @@ -0,0 +1,17 @@ +package Speech { + part def ISynthesisSession { + doc /* Layer 5: a cheap text-to-speech session, bound to exactly one playback device + * instance for its entire life, obtained from + * ISpeechSynthesizerEngine.CreateSessionAsync. Exposes an availability flag, a + * SynthesisSessionState lifecycle state with a StateChanged event, non-overlapping + * SpeakAsync/SynthesizeAsync operations (a second call while one is in flight throws + * InvalidOperationException), and StopAsync to request cooperative cancellation of + * whichever operation is currently in flight. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisSession.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSynthesisSessionTests.cs */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/i-synthesis-session.md */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/i-synthesis-session.md */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/i-synthesis-session.yaml */ + } +} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/layer2-rendering.sysml b/docs/sysml2/model/speech/synthesis-subsystem/layer2-rendering.sysml index b1f52cb..044584f 100644 --- a/docs/sysml2/model/speech/synthesis-subsystem/layer2-rendering.sysml +++ b/docs/sysml2/model/speech/synthesis-subsystem/layer2-rendering.sysml @@ -11,8 +11,8 @@ package Speech { comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/IModelCapabilityProfile.cs, DefaultModelCapabilityProfile.cs, SpeechSegment.cs, SpeechPlan.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/DefaultModelCapabilityProfileTests.cs */ - comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md (documented with the pipeline it serves) */ - comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md (documented with the pipeline it serves) */ - comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.yaml */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md (documented with the pipeline it serves) */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md (documented with the pipeline it serves) */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.yaml */ } } diff --git a/docs/sysml2/model/speech/synthesis-subsystem/playback-audio-resampler.sysml b/docs/sysml2/model/speech/synthesis-subsystem/playback-audio-resampler.sysml index 1530aa6..b29b928 100644 --- a/docs/sysml2/model/speech/synthesis-subsystem/playback-audio-resampler.sysml +++ b/docs/sysml2/model/speech/synthesis-subsystem/playback-audio-resampler.sysml @@ -9,8 +9,8 @@ package Speech { comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/PlaybackAudioResampler.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/PlaybackAudioResamplerTests.cs */ - comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md (documented with the pipeline it serves) */ - comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md (documented with the pipeline it serves) */ - comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.yaml */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md (documented with the pipeline it serves) */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md (documented with the pipeline it serves) */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.yaml */ } } diff --git a/docs/sysml2/model/speech/synthesis-subsystem/sentence-chunker.sysml b/docs/sysml2/model/speech/synthesis-subsystem/sentence-chunker.sysml index 0e71c03..cf3f733 100644 --- a/docs/sysml2/model/speech/synthesis-subsystem/sentence-chunker.sysml +++ b/docs/sysml2/model/speech/synthesis-subsystem/sentence-chunker.sysml @@ -21,8 +21,8 @@ package Speech { comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/SentenceChunker.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SentenceChunkerTests.cs */ - comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md (documented with the pipeline it serves) */ - comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md (documented with the pipeline it serves) */ - comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.yaml */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md (documented with the pipeline it serves) */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md (documented with the pipeline it serves) */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.yaml */ } } diff --git a/docs/sysml2/model/speech/synthesis-subsystem/session-state-changed-event-args.sysml b/docs/sysml2/model/speech/synthesis-subsystem/session-state-changed-event-args.sysml new file mode 100644 index 0000000..3c7779f --- /dev/null +++ b/docs/sysml2/model/speech/synthesis-subsystem/session-state-changed-event-args.sysml @@ -0,0 +1,14 @@ +package Speech { + part def SessionStateChangedEventArgs { + doc /* Event arguments record carrying the Previous/Current SynthesisSessionState of an + * ISynthesisSession.StateChanged transition. Declared separately from + * RecognitionSubsystem's identically-shaped record (which carries + * RecognitionSessionState instead) rather than shared, matching this library's + * existing "no cross-subsystem public type" boundary. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/SessionStateChangedEventArgs.cs */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/i-synthesis-session.md (documented as a supporting value type) */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/i-synthesis-session.md (documented as a supporting value type) */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/i-synthesis-session.yaml */ + } +} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.sysml b/docs/sysml2/model/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.sysml new file mode 100644 index 0000000..7d6195b --- /dev/null +++ b/docs/sysml2/model/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.sysml @@ -0,0 +1,17 @@ +package Speech { + part def SherpaOnnxSpeechSynthesizerEngine { + doc /* Real ISpeechSynthesizerEngine implementation: holds one loaded SynthesisBackend, + * enforces single-session exclusivity with a fail-fast binary semaphore lease held + * for a leased session's entire life through DisposeAsync completion (a concurrent + * CreateSessionAsync call while the lease is held throws + * SynthesisEngineBusyException), and constructs SherpaOnnxSynthesisSession instances + * bound to a caller-supplied playback device. Disposing the engine disposes any + * still-active leased session first, then the owned backend. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerEngine.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerEngineTests.cs */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/synthesis-session-lease.yaml */ + } +} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.sysml b/docs/sysml2/model/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.sysml deleted file mode 100644 index e9f1118..0000000 --- a/docs/sysml2/model/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.sysml +++ /dev/null @@ -1,18 +0,0 @@ -package Speech { - part def SherpaOnnxSpeechSynthesizer { - doc /* Real streaming synthesis pipeline: normalizes and tag-parses input text, Layer 2 - * renders it into an ordered SpeechPlan, chunks and synthesizes each segment ahead - * of playback, and plays segments in order on the playback device while the next - * segment synthesizes. Resolves the speaker id passed to each segment's Generate - * call from the constructor-supplied parameterValues bag via - * ISynthesisModel.ResolveSpeakerId once per segment, replacing a previously - * hard-coded speakerId: 0, kept correctly independent of the per-segment - * ParameterOverrides speed/volume mechanism. */ - - comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSpeechSynthesizer.cs */ - comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerTests.cs */ - comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md */ - comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md */ - comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.yaml */ - } -} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.sysml b/docs/sysml2/model/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.sysml new file mode 100644 index 0000000..3691aca --- /dev/null +++ b/docs/sysml2/model/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.sysml @@ -0,0 +1,22 @@ +package Speech { + part def SherpaOnnxSynthesisSession { + doc /* Real ISynthesisSession implementation: normalizes and tag-parses input text, Layer + * 2 renders it into an ordered SpeechPlan, synthesizes each segment on a dedicated + * worker thread (bounding how long a non-cooperative native call is waited on before + * being abandoned), and - for SpeakAsync - plays the resulting audio through this + * session's bound playback device while waiting for genuine playback drain before + * stopping it. An explicit SynthesisSessionState state machine and an overlap guard + * (at most one SpeakAsync/SynthesizeAsync call in flight at a time) wrap this + * pipeline; StopAsync cancels an in-flight operation deterministically and a failed + * operation transitions the session to the terminal Faulted state. Resolves the + * speaker id passed to each segment's Generate call from the constructor-supplied + * parameterValues bag via ISynthesisModel.ResolveSpeakerId once per segment, + * independent of the per-segment ParameterOverrides speed/volume mechanism. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisSession.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSynthesisSessionTests.cs */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.yaml */ + } +} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/speech-synthesizer-factory.sysml b/docs/sysml2/model/speech/synthesis-subsystem/speech-synthesizer-factory.sysml index fc607ff..e2c89f4 100644 --- a/docs/sysml2/model/speech/synthesis-subsystem/speech-synthesizer-factory.sysml +++ b/docs/sysml2/model/speech/synthesis-subsystem/speech-synthesizer-factory.sysml @@ -1,12 +1,13 @@ package Speech { part def SpeechSynthesizerFactory { - doc /* Composition entry point that returns a working speech synthesizer for an - * installed synthesis model and available playback device, or an honest unavailable - * fallback, without ever throwing for an ordinary machine state. Forwards an - * optional, session-level parameterValues bag (for example a selected voice) - * unchanged to the constructed synthesizer. The preferred composition is to open the - * playback device with the model's best-effort PreferredAudioFormat first, while - * still relying on runtime resampling if the loaded engine reports a different rate. */ + doc /* Composition entry point that asynchronously loads an installed synthesis model + * into a working ISpeechSynthesizerEngine, or returns an honest + * UnavailableSpeechSynthesizerEngine fallback, without ever throwing for an ordinary + * machine state. No longer takes a playback device parameter - a device is bound + * later, per session, via ISpeechSynthesizerEngine.CreateSessionAsync, so the + * model-load step and the device-bind step can fail and be retried independently. + * Forwards an optional, engine-level parameterValues bag (for example a selected + * voice) unchanged to the constructed engine. */ comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/SpeechSynthesizerFactory.cs */ comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SpeechSynthesizerFactoryTests.cs */ diff --git a/docs/sysml2/model/speech/synthesis-subsystem/speech-synthesizer-unavailable-exception.sysml b/docs/sysml2/model/speech/synthesis-subsystem/speech-synthesizer-unavailable-exception.sysml index 755d7b7..dd2e223 100644 --- a/docs/sysml2/model/speech/synthesis-subsystem/speech-synthesizer-unavailable-exception.sysml +++ b/docs/sysml2/model/speech/synthesis-subsystem/speech-synthesizer-unavailable-exception.sysml @@ -1,13 +1,13 @@ package Speech { part def SpeechSynthesizerUnavailableException { - doc /* Exception signaling that an operational member of an unavailable speech - * synthesizer was invoked, or that a synthesizer which claimed to be available - * failed on first use. */ + doc /* Exception signaling that an operational member of an unavailable + * ISpeechSynthesizerEngine or ISynthesisSession was invoked, or that a session which + * claimed to be available faulted and was asked to operate again. */ comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/SpeechSynthesizerUnavailableException.cs */ - comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSpeechSynthesizerTests.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSpeechSynthesizerEngineTests.cs, UnavailableSynthesisSessionTests.cs */ comment designRef /* Design: docs/design/speech/synthesis-subsystem.md (documented as a supporting fallback type) */ comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem.md (documented as a supporting fallback type) */ - comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer.yaml */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.yaml, unavailable-synthesis-session.yaml */ } } diff --git a/docs/sysml2/model/speech/synthesis-subsystem/synthesis-backend.sysml b/docs/sysml2/model/speech/synthesis-subsystem/synthesis-backend.sysml new file mode 100644 index 0000000..3ce43bd --- /dev/null +++ b/docs/sysml2/model/speech/synthesis-subsystem/synthesis-backend.sysml @@ -0,0 +1,15 @@ +package Speech { + part def SynthesisBackend { + doc /* Internal mockable seam over one loaded speech-synthesis engine, together with its + * factory and their real sherpa-onnx implementations, confining every inference + * interop call so the synthesis pipeline is testable with pure managed fakes. Renamed + * from ISynthesisEngine/ISynthesisEngineFactory so the "engine" vocabulary is reserved + * for the public Layer 3 ISpeechSynthesizerEngine contract; members unchanged. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisBackend.cs, ISynthesisBackendFactory.cs, SherpaOnnxSynthesisEngine.cs, SherpaOnnxSynthesisEngineFactory.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSynthesisSessionTests.cs */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md (documented with the pipeline it serves) */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md (documented with the pipeline it serves) */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.yaml */ + } +} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/synthesis-engine-busy-exception.sysml b/docs/sysml2/model/speech/synthesis-subsystem/synthesis-engine-busy-exception.sysml new file mode 100644 index 0000000..34260ca --- /dev/null +++ b/docs/sysml2/model/speech/synthesis-subsystem/synthesis-engine-busy-exception.sysml @@ -0,0 +1,15 @@ +package Speech { + part def SynthesisEngineBusyException { + doc /* Exception thrown when ISpeechSynthesizerEngine.CreateSessionAsync is called while + * this engine's exclusivity lease is already held by another session. Fails fast + * rather than queueing or awaiting release, since waiting would make this call's + * latency depend on an unrelated session's teardown with no caller-visible way to + * bound that wait. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisEngineBusyException.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerEngineTests.cs */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md (documented as a supporting exception type) */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md (documented as a supporting exception type) */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/synthesis-session-lease.yaml */ + } +} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/synthesis-engine.sysml b/docs/sysml2/model/speech/synthesis-subsystem/synthesis-engine.sysml deleted file mode 100644 index 15aeb5f..0000000 --- a/docs/sysml2/model/speech/synthesis-subsystem/synthesis-engine.sysml +++ /dev/null @@ -1,13 +0,0 @@ -package Speech { - part def SynthesisEngine { - doc /* Internal mockable seam over one loaded speech-synthesis engine, together with its - * factory and their real sherpa-onnx implementations, confining every inference - * interop call so the synthesis pipeline is testable with pure managed fakes. */ - - comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisEngine.cs */ - comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerTests.cs */ - comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md (documented with the pipeline it serves) */ - comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md (documented with the pipeline it serves) */ - comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.yaml */ - } -} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/synthesis-session-faulted-exception.sysml b/docs/sysml2/model/speech/synthesis-subsystem/synthesis-session-faulted-exception.sysml new file mode 100644 index 0000000..e31fa3e --- /dev/null +++ b/docs/sysml2/model/speech/synthesis-subsystem/synthesis-session-faulted-exception.sysml @@ -0,0 +1,16 @@ +package Speech { + part def SynthesisSessionFaultedException { + doc /* Exception thrown when an operation is attempted on, or surfaced from, an + * ISynthesisSession that has transitioned to SynthesisSessionState.Faulted (an + * in-flight SpeakAsync/SynthesizeAsync operation failed for a reason other than its + * own requested cancellation, including a native call abandoned without ever being + * requested to stop). Carries the original fault as its inner exception; the session + * is terminal and must be disposed and replaced. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisSessionFaultedException.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSynthesisSessionTests.cs */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md (documented as a supporting exception type) */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md (documented as a supporting exception type) */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.yaml */ + } +} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/synthesis-session-state.sysml b/docs/sysml2/model/speech/synthesis-subsystem/synthesis-session-state.sysml new file mode 100644 index 0000000..900c621 --- /dev/null +++ b/docs/sysml2/model/speech/synthesis-subsystem/synthesis-session-state.sysml @@ -0,0 +1,15 @@ +package Speech { + part def SynthesisSessionState { + doc /* The lifecycle states an ISynthesisSession passes through: Created, Starting, + * Running, Stopping, Stopped, Disposing, Disposed, Faulted. Starting/Running/Stopping + * denote one discrete in-flight SpeakAsync/SynthesizeAsync operation rather than a + * continuous stream - unlike recognition's continuous capture window, a session + * returns to Stopped after each operation and is ready to accept another call. + * Faulted is terminal. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisSessionState.cs */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/i-synthesis-session.md (documented as a supporting value type) */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/i-synthesis-session.md (documented as a supporting value type) */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/i-synthesis-session.yaml */ + } +} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.sysml b/docs/sysml2/model/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.sysml new file mode 100644 index 0000000..82019c2 --- /dev/null +++ b/docs/sysml2/model/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.sysml @@ -0,0 +1,16 @@ +package Speech { + part def UnavailableSpeechSynthesizerEngine { + doc /* Honest "unavailable" ISpeechSynthesizerEngine fallback: reports IsAvailable as + * false; CreateSessionAsync always succeeds, returning the shared + * UnavailableSynthesisSession instance, since binding a device to an already + * unavailable engine is itself an ordinary (if useless) composition, not an error - + * only the returned session's operational members throw + * SpeechSynthesizerUnavailableException. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSpeechSynthesizerEngine.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSpeechSynthesizerEngineTests.cs */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.md */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.md */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.yaml */ + } +} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/unavailable-speech-synthesizer.sysml b/docs/sysml2/model/speech/synthesis-subsystem/unavailable-speech-synthesizer.sysml deleted file mode 100644 index d7d9056..0000000 --- a/docs/sysml2/model/speech/synthesis-subsystem/unavailable-speech-synthesizer.sysml +++ /dev/null @@ -1,13 +0,0 @@ -package Speech { - part def UnavailableSpeechSynthesizer { - doc /* Honest "unavailable" ISpeechSynthesizer fallback: reports IsAvailable as false and - * throws SpeechSynthesizerUnavailableException only when an operational member is - * invoked. */ - - comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSpeechSynthesizer.cs */ - comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSpeechSynthesizerTests.cs */ - comment designRef /* Design: docs/design/speech/synthesis-subsystem/unavailable-speech-synthesizer.md */ - comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/unavailable-speech-synthesizer.md */ - comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer.yaml */ - } -} diff --git a/docs/sysml2/model/speech/synthesis-subsystem/unavailable-synthesis-session.sysml b/docs/sysml2/model/speech/synthesis-subsystem/unavailable-synthesis-session.sysml new file mode 100644 index 0000000..e524b84 --- /dev/null +++ b/docs/sysml2/model/speech/synthesis-subsystem/unavailable-synthesis-session.sysml @@ -0,0 +1,14 @@ +package Speech { + part def UnavailableSynthesisSession { + doc /* Honest "unavailable" ISynthesisSession fallback: reports IsAvailable as false, + * always reports SynthesisSessionState.Created (it never transitions), and throws + * SpeechSynthesizerUnavailableException only when SpeakAsync or SynthesizeAsync is + * invoked; StopAsync and DisposeAsync are safe no-ops. */ + + comment sourceRef /* Source: src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSynthesisSession.cs */ + comment testRef /* Test: test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSynthesisSessionTests.cs */ + comment designRef /* Design: docs/design/speech/synthesis-subsystem/unavailable-synthesis-session.md */ + comment verificationRef /* Verification: docs/verification/speech/synthesis-subsystem/unavailable-synthesis-session.md */ + comment reqRef /* Requirements: docs/reqstream/speech/synthesis-subsystem/unavailable-synthesis-session.yaml */ + } +} diff --git a/docs/user_guide/introduction.md b/docs/user_guide/introduction.md index 26b840d..a66cdfe 100644 --- a/docs/user_guide/introduction.md +++ b/docs/user_guide/introduction.md @@ -184,17 +184,24 @@ if (device.IsAvailable) This release also ships the `RecognitionSubsystem` (`DemaConsulting.Speech.RecognitionSubsystem`), which turns captured audio into text: -- **`ISpeechRecognizer`**: the streaming speech-to-text contract — `IsAvailable`, `Start()`, - `Stop()`, a `ResultReceived` event, and `Dispose()`. No speech-engine type appears anywhere in - this contract, so a future engine change cannot break your code. +- **`ISpeechRecognizerEngine`**: the loaded, expensive, native-backed recognition model, with no + capture device bound yet — `IsAvailable` and `CreateSessionAsync(device, cancellationToken)`. + No speech-engine type appears anywhere in this contract, so a future engine change cannot break + your code. +- **`IRecognitionSession`**: the cheap, single-use streaming speech-to-text session bound to + exactly one capture device for its entire life — `IsAvailable`, `State`, a `StateChanged` + event, `StartAsync(cancellationToken)`, `StopAsync(cancellationToken)`, + `GetResultsAsync(cancellationToken)` (an `IAsyncEnumerable`), and + `DisposeAsync()`. - **`SpeechRecognitionResult`**: the recognized `Text` of the current utterance plus an `IsFinal` flag. `Text` is always the full utterance so far, never a fragment, so you can render it directly and simply replace it when the next result arrives. -- **`SpeechRecognizerFactory`**: the composition entry point. It returns a working recognizer - only when the model is installed, declares the recognition role, the capture device is - available, and the speech engine loads; otherwise it returns - **`UnavailableSpeechRecognizer`**, which reports `IsAvailable == false`. It never throws for - any of these ordinary machine states. +- **`SpeechRecognizerFactory`**: the composition entry point. `LoadAsync(...)` returns a working + engine only when the model is installed, declares the recognition role, and the speech engine + loads; otherwise it returns **`UnavailableSpeechRecognizerEngine`**, which reports + `IsAvailable == false` and whose `CreateSessionAsync` returns + **`UnavailableRecognitionSession`**. Neither layer throws for any of these ordinary machine + states. Typical recognition composition: @@ -223,20 +230,28 @@ var captureDevice = new AudioDeviceFactory().CreateCaptureDevice( AudioDeviceSelection.SystemDefault, model.AudioFormat); -// 5. Compose the recognizer and stream recognized text as it arrives. -using var recognizer = SpeechRecognizerFactory.Create(model, catalog, captureDevice); +// 5. Load the engine once, create a session bound to the capture device, and stream +// recognized text as it arrives. +await using var engine = await SpeechRecognizerFactory.LoadAsync(model, catalog); -if (recognizer.IsAvailable) +if (engine.IsAvailable) { - recognizer.ResultReceived += (_, args) => + await using var session = await engine.CreateSessionAsync(captureDevice); + + await session.StartAsync(); + + var resultsTask = Task.Run(async () => { - var status = args.Result.IsFinal ? "final" : "partial"; - Console.WriteLine($"{status}: {args.Result.Text}"); - }; + await foreach (var evt in session.GetResultsAsync()) + { + var status = evt.Result.IsFinal ? "final" : "partial"; + Console.WriteLine($"{status}: {evt.Result.Text}"); + } + }); - recognizer.Start(); // ... capture speech ... - recognizer.Stop(); + await session.StopAsync(); + await resultsTask; } ``` @@ -249,17 +264,26 @@ Points worth knowing: the target format and the recognizer often stays on its no-op equal-rate fast path. When resampling is still required, the downsampling path now applies anti-alias filtering before decimation. -- **Results arrive off the audio thread.** `ResultReceived` is raised from the recognizer's own - background decoding thread, never from the audio callback thread, so a handler may do moderate - work. Handlers are invoked serially, and an exception thrown by a handler is reported through - your diagnostics sink rather than propagated. -- **`Stop()` does not lose the tail of an utterance.** It drains audio already captured before - the call, so every result derived from it has been delivered by the time `Stop()` returns. -- **Dispose the recognizer.** A real recognizer holds a loaded engine; disposal releases it and - implies `Stop()`. Disposal is idempotent, and disposing the unavailable fallback is a safe - no-op. +- **Results are delivered through a dedicated worker thread, not the audio callback thread.** + `GetResultsAsync` streams events produced by the session's own dedicated decoding thread, which + pumps captured audio through the recognition backend independently of the audio device's own + callback thread. Two bounded buffers isolate a slow consumer from that pump: captured audio + waiting to be decoded queues in a bounded, drop-oldest buffer, and converted results queue in a + backpressure buffer that coalesces unread provisional results into a single "latest" slot while + keeping a byte-capped FIFO of final results (evicting only the oldest unread final, as a + last-resort safety valve, and reporting that through diagnostics). A single concurrent + `GetResultsAsync` enumeration is supported per session; a second overlapping call throws + `InvalidOperationException`. +- **`StopAsync` does not lose the tail of an utterance.** It drains audio already captured before + the call, so every result derived from it is enumerable via `GetResultsAsync` by the time + `StopAsync` returns. +- **Dispose the session (and eventually the engine).** A real session holds a capture-device + subscription and the engine's exclusivity lease; disposal releases both and implies `StopAsync`. + Disposal is idempotent, and disposing the unavailable fallback is a safe no-op. A session is + single-use: once `State` reaches `Stopped`, `StartAsync` throws `InvalidOperationException` + rather than restarting — create a new session via `CreateSessionAsync` for another run. - **`Text` is already restored, not raw engine output.** Before a result reaches your - `ResultReceived` handler, the recognizer calls the owning model's + `GetResultsAsync` consumer, the session calls the owning model's `IRecognitionModel.NormalizeText(text, isFinal)`. Most models simply pass text through unchanged (the interface's default), but `SherpaOnnxZipformerEnRecognitionModel` overrides it with `UppercaseTranscriptRestorer` to turn its raw shouted, unpunctuated output (for example @@ -274,7 +298,7 @@ Points worth knowing: casing of project-glossary terms (acronyms, product names) that general-purpose recognition cannot know about — is an anticipated, supported use of the delivered text, not an undocumented workaround. The ordering is guaranteed: the owning model's `NormalizeText` has - already run by the time `Text` reaches your `ResultReceived` handler, so your transformation + already run by the time `Text` reaches your `GetResultsAsync` consumer, so your transformation composes after the library's restoration rather than racing it. Attach that transformation to final (`isFinal: true`) results only — provisional results carry just the cheap pass and are still being revised, so running your own restoration on them would make the draft flicker @@ -335,11 +359,12 @@ narration rather than being dropped or rejected, so a reply is never worse than ## Synthesizing Speech -`SpeechSynthesizerFactory.Create(...)` composes an `ISpeechSynthesizer` over an installed -`ISynthesisModel` and a playback device, mirroring `SpeechRecognizerFactory`'s -nothing-throws-at-composition contract: it never throws for an ordinary machine state, and -instead returns a synthesizer reporting `IsAvailable == false` for a model that is not installed, -a machine with no speakers, or a missing speech-engine native runtime. +`SpeechSynthesizerFactory.LoadAsync(...)` composes an `ISpeechSynthesizerEngine` over an +installed `ISynthesisModel`, mirroring `SpeechRecognizerFactory`'s nothing-throws-at-composition +contract: it never throws for an ordinary machine state, and instead returns an engine reporting +`IsAvailable == false` for a model that is not installed, a machine with no speakers, or a +missing speech-engine native runtime. A playback device is bound later, per session (or per +one-shot call), not at `LoadAsync` time. ```csharp using DemaConsulting.Speech.AudioSubsystem; @@ -366,26 +391,26 @@ var playbackDevice = new AudioDeviceFactory().CreatePlaybackDevice( AudioDeviceSelection.SystemDefault, model.PreferredAudioFormat); -// 5. Compose the synthesizer and speak. -using var synthesizer = SpeechSynthesizerFactory.Create(model, catalog, playbackDevice); +// 5. Load the engine and speak a one-shot phrase through the engine-level convenience overload, +// which internally creates a session, speaks, and disposes the session again. +await using var engine = await SpeechSynthesizerFactory.LoadAsync(model, catalog); -if (synthesizer.IsAvailable) +if (engine.IsAvailable) { - await synthesizer.SpeakAsync("Welcome. [short pause] Let's get started!"); + await engine.SpeakAsync(playbackDevice, "Welcome. [short pause] Let's get started!"); } ``` For a model that declares a `ChoiceParameter` or `NumericParameter` for voice/speaker selection (such as `SherpaOnnxKokoroEnglishSynthesisModel`'s `voice` parameter, or `SherpaOnnxVitsLibriTtsEnglishSynthesisModel`'s numeric `speaker` parameter), pass a -`parameterValues` bag keyed by each declared parameter's `Id` to `Create(...)` to select a +`parameterValues` bag keyed by each declared parameter's `Id` to `LoadAsync(...)` to select a non-default value: ```csharp -using var synthesizer = SpeechSynthesizerFactory.Create( +await using var engine = await SpeechSynthesizerFactory.LoadAsync( model, catalog, - playbackDevice, parameterValues: new Dictionary { ["voice"] = "bm_george" }); ``` @@ -397,11 +422,11 @@ sink is wired up) so one settings dictionary stays reusable across different mod supplied value for a parameter the model *does* declare that fails that parameter's own validation (wrong CLR type, out of range, a fractional value for a whole-number-only parameter, or a string matching no declared `ChoiceParameterOption`) throws `ArgumentException` synchronously -from `Create()` naming the parameter, the model, and the reason the value is invalid - -`SpeechRecognizerFactory.Create`'s `parameterValues` argument follows the same rule. +from `LoadAsync(...)` naming the parameter, the model, and the reason the value is invalid - +`SpeechRecognizerFactory.LoadAsync`'s `parameterValues` argument follows the same rule. `model.PreferredAudioFormat` is likewise only a best-effort playback -hint: after construction, the synthesizer always treats the loaded engine's actual `SampleRate` -as authoritative and resamples whenever needed. This is a session-level choice: it is +hint: after the engine loads, every session created from it always treats the loaded engine's +actual `SampleRate` as authoritative and resamples whenever needed. This is a session-level choice: it is independent of, and does not disturb, the existing per-segment Natural Language Audio Tag speed/volume overrides described above, which continue to apply per rendered segment regardless of which voice is selected. The @@ -457,11 +482,11 @@ instead - consistent with the "never worse than plain narration" guarantee. The rendered result is an ordered `SpeechPlan` of `SpeechSegment`s, which a sentence/clause-sized chunker further splits so that synthesis and playback can pipeline: an earlier chunk plays on the playback device while a later chunk is still being synthesized, rather than waiting for an entire -utterance's inference to finish before any sound is heard. Calling `Stop()` cancels an in-flight -`SpeakAsync` call deterministically (for example, in response to a user interruption) and is a -safe no-op when nothing is speaking. A synthesis-engine fault or a playback device that drops -mid-utterance fails the awaited `SpeakAsync` task honestly rather than hanging or crashing the -process. +utterance's inference to finish before any sound is heard. Calling `StopAsync()` cancels an +in-flight `SpeakAsync` call deterministically (for example, in response to a user interruption) +and is a safe no-op when nothing is speaking. A synthesis-engine fault or a playback device that +drops mid-utterance fails the awaited `SpeakAsync` task honestly rather than hanging or crashing +the process. This release ships two production synthesis models (each also reports its license programmatically via `ISpeechModel.LicenseName`/`LicenseUrl`, without requiring you to parse @@ -473,8 +498,9 @@ this prose or `DisplayName`): (`http://www.openslr.org/141/`) and the Piper text-to-speech project is required if you redistribute the model or audio generated by it, but downstream relicensing under different terms is not otherwise restricted, unlike a ShareAlike/CC BY-SA license). All 904 speakers are - selectable through `ISpeechSynthesizer` by plain numeric index (`0`-`903`) via this model's - declared `NumericParameter` named `speaker`: LibriTTS-R's speaker embeddings have no published + selectable via `SpeechSynthesizerFactory.LoadAsync`'s `parameterValues` bag by plain numeric + index (`0`-`903`) through this model's declared `NumericParameter` named `speaker`: LibriTTS-R's + speaker embeddings have no published human-readable name mapping, so speakers are identified only by their numeric id, unlike the named voices below. - **`SherpaOnnxKokoroEnglishSynthesisModel`** (`kokoro-int8-en-v0_19`, an English-only, @@ -495,34 +521,57 @@ official GitHub Releases URL only when you explicitly request it. ## Hot TTS/STT: Reusing an Instance Across Turns -`SpeechRecognizerFactory.Create(...)`/`SpeechSynthesizerFactory.Create(...)` are the expensive -step in either direction: on success, each loads a model into native memory. `Start()`/`Stop()` -on an already-created `ISpeechRecognizer`, and a `SpeakAsync`/`SynthesizeStreamAsync`/ -`PlayStreamAsync` session on an already-created `ISpeechSynthesizer`, are comparatively cheap and -fully repeatable on the same instance - neither reloads the model. For a low-latency, multi-turn -scenario such as a voice conversation, create each instance **once** and reuse it across many -turns, rather than disposing and recreating it per turn: +`SpeechRecognizerFactory.LoadAsync(...)`/`SpeechSynthesizerFactory.LoadAsync(...)` are the +expensive step in either direction: on success, each loads a model into native memory. Beneath +that, there is a three-way cost split: + +- **`LoadAsync`** (engine): expensive - do this once per model/parameter combination. +- **`CreateSessionAsync`** (engine → session): binds exactly one device for the session's entire + life. Comparatively cheap, but exclusive - an engine leases its single `IRecognitionSession`/ + `ISynthesisSession` to only one live session at a time, and a concurrent `CreateSessionAsync` + call while a lease is held fails fast with `RecognitionEngineBusyException`/ + `SynthesisEngineBusyException` rather than queueing. +- **`StartAsync`/`StopAsync`/`GetResultsAsync`** (recognition) and **`SpeakAsync`/ + `SynthesizeAsync`** (synthesis): per-turn, cheap, and fully repeatable - neither reloads the + model nor rebinds the device. + +For a low-latency, multi-turn scenario such as a voice conversation, load each engine **once** +and reuse it across many turns, rather than disposing and recreating it per turn. A synthesis +session may itself be reused across many `SpeakAsync` calls (it cycles back to `Starting` on each +new call). A recognition session is single-use - once it reaches `Stopped`, `StartAsync` throws +`InvalidOperationException` - so reuse the *engine* across turns and create a fresh session per +turn instead: ```csharp -// Synthesis: create once, speak many times. -using var synthesizer = SpeechSynthesizerFactory.Create(model, catalog, playbackDevice); -await synthesizer.SpeakAsync("First turn."); -await synthesizer.SpeakAsync("Second turn - the model was never reloaded."); - -// Recognition: create once, Start()/Stop() many times. -using var recognizer = SpeechRecognizerFactory.Create(model, catalog, captureDevice); -recognizer.Start(); -// ... wait for this turn's result via ResultReceived, then: -recognizer.Stop(); -recognizer.Start(); // Next turn - again, no reload. -// ... wait for this turn's result, then: -recognizer.Stop(); +// Synthesis: load the engine once, create the session once, speak many times. +await using var synthesizerEngine = await SpeechSynthesizerFactory.LoadAsync(model, catalog); +await using var synthesisSession = await synthesizerEngine.CreateSessionAsync(playbackDevice); +await synthesisSession.SpeakAsync("First turn."); +await synthesisSession.SpeakAsync("Second turn - neither the model nor the device were reloaded."); + +// Recognition: load the engine once, create a fresh session per turn (sessions are single-use). +await using var recognizerEngine = await SpeechRecognizerFactory.LoadAsync(model, catalog); + +await using (var turn1 = await recognizerEngine.CreateSessionAsync(captureDevice)) +{ + await turn1.StartAsync(); + // ... wait for this turn's result via GetResultsAsync, then: + await turn1.StopAsync(); +} + +await using (var turn2 = await recognizerEngine.CreateSessionAsync(captureDevice)) +{ + // Creating turn2 only succeeds once turn1 has fully disposed and released the engine's lease. + await turn2.StartAsync(); + // ... wait for this turn's result, then: + await turn2.StopAsync(); +} ``` -Only dispose and recreate an instance when you need to change its model, device, or parameter -values - not simply to begin a new turn. `speech-cli ask` applies the same principle at the +Only dispose and reload an engine when you need to change its model or parameter values - not +simply to begin a new turn. `speech-cli ask` applies the same principle at the process level: because a single CLI invocation only ever runs one turn, there is no instance to -reuse across turns, so instead `ask` pre-warms (constructs, and so loads) its STT recognizer +reuse across turns, so instead `ask` pre-warms (constructs, and so loads) its STT engine concurrently with speaking the prompt, rather than only afterward, removing the same avoidable model-load latency a long-lived host would instead avoid by reusing one instance across many turns. See the "SpeechCli" section's worked examples below for the exact command. @@ -542,7 +591,7 @@ using DemaConsulting.Speech.SynthesisSubsystem; // 1. Compose the per-user model store/catalog and the audio devices. None of this throws for an // ordinary machine state - a missing microphone, missing speakers, or a missing native -// runtime all degrade to an honest "unavailable" device/recognizer/synthesizer instead. +// runtime all degrade to an honest "unavailable" device/engine/session instead. using var catalog = new SpeechModelCatalog(); var audioFactory = new AudioDeviceFactory(); @@ -567,41 +616,50 @@ var playbackDevice = audioFactory.CreatePlaybackDevice( AudioDeviceSelection.SystemDefault, synthesisModel.PreferredAudioFormat); -// 4. Compose the recognizer and synthesizer over the resolved models and devices. -using var recognizer = SpeechRecognizerFactory.Create( +// 4. Load the recognizer and synthesizer engines over the resolved models, then create a +// session bound to each resolved device. The recognition session is single-use (one session +// per "turn"); the synthesis session is reused for every reply. +await using var recognizerEngine = await SpeechRecognizerFactory.LoadAsync( recognitionModel, - catalog, - captureDevice); + catalog); -using var synthesizer = SpeechSynthesizerFactory.Create( +await using var synthesizerEngine = await SpeechSynthesizerFactory.LoadAsync( synthesisModel, - catalog, - playbackDevice); + catalog); -if (!recognizer.IsAvailable || !synthesizer.IsAvailable) +if (!recognizerEngine.IsAvailable || !synthesizerEngine.IsAvailable) { Console.WriteLine("No microphone/speakers (or the models failed to load) - exiting."); return; } +await using var recognitionSession = await recognizerEngine.CreateSessionAsync(captureDevice); +await using var synthesisSession = await synthesizerEngine.CreateSessionAsync(playbackDevice); + // 5. Speak a greeting, then listen and echo back each final result until Enter is pressed. -await synthesizer.SpeakAsync("Hello! [short pause] Say something and I will repeat it back."); +await synthesisSession.SpeakAsync("Hello! [short pause] Say something and I will repeat it back."); + +using var cts = new CancellationTokenSource(); -recognizer.ResultReceived += async (_, args) => +var resultsTask = Task.Run(async () => { - if (!args.Result.IsFinal) + await foreach (var evt in recognitionSession.GetResultsAsync(cts.Token)) { - return; - } + if (!evt.Result.IsFinal) + { + continue; + } - Console.WriteLine($"You said: {args.Result.Text}"); - await synthesizer.SpeakAsync($"You said: {args.Result.Text}"); -}; + Console.WriteLine($"You said: {evt.Result.Text}"); + await synthesisSession.SpeakAsync($"You said: {evt.Result.Text}"); + } +}); -recognizer.Start(); +await recognitionSession.StartAsync(); Console.WriteLine("Listening - press Enter to stop."); Console.ReadLine(); -recognizer.Stop(); +await recognitionSession.StopAsync(); +cts.Cancel(); ``` This example deliberately keeps error handling minimal for readability; a production application @@ -633,16 +691,25 @@ The application opens with four panels, one of which embeds a fifth: download action that reports progress and reports failure with an explanation. - **Text-to-Speech**: lists installed synthesis models, offers a text box with inline hints showing a few of the library's Natural Language Audio Tags (such as `[whispers]`, - `[short pause]`, and `[excited]`), and Play/Stop controls that compose an `ISpeechSynthesizer` - through a demo-owned seam over `SpeechSynthesizerFactory`. Playback status reflects - synthesizing, playing, idle, or an honest error; the panel explains itself if no synthesis - model is installed or no playback device is available. The panel refreshes itself + `[short pause]`, and `[excited]`), and Play/Stop controls that compose an + `ISpeechSynthesizerEngine`/`ISynthesisSession` through a demo-owned seam over + `SpeechSynthesizerFactory`. The panel caches its engine and session across Play calls - + reloading the engine only when the selected model or its parameter values change, and + recreating the session only when the engine was just reloaded or the playback device changes - + rather than recreating a synthesizer on every Play. Playback status is derived from the + session's own `State`/`StateChanged` (synthesizing, playing, idle, or an honest error) rather + than tracked ad hoc; the panel explains itself if no synthesis model is installed or no + playback device is available. The panel refreshes itself automatically the moment a synthesis model finishes downloading from the Model Catalog panel, with no manual click or application restart needed. The model picker and its Model Settings controls disable while audio is synthesizing or playing, so a voice/speaker cannot be changed mid-playback. - **Speech-to-Text**: offers Start/Stop streaming transcription that composes an - `ISpeechRecognizer` through a demo-owned seam over `SpeechRecognizerFactory`. Committed final + `ISpeechRecognizerEngine`/`IRecognitionSession` through a demo-owned seam over + `SpeechRecognizerFactory`. Start loads/reuses the cached engine, creates a fresh session (a + recognition session is single-use), and pumps `GetResultsAsync` on a background task; Stop + requests `StopAsync` on the active session. Listening state is derived from the session's own + `State`/`StateChanged` rather than tracked ad hoc. Committed final results accumulate in order while a trailing partial line updates live as the recognizer refines it. The panel explains itself if no recognition model is installed or no capture device is available. The panel refreshes itself automatically the moment a recognition model @@ -656,8 +723,9 @@ The application opens with four panels, one of which embeds a fifth: entirely by the library's `ISpeechModelParameter` concrete type. For the Text-to-Speech panel, the current value bag (including a selected voice, for a model such as `SherpaOnnxKokoroEnglishSynthesisModel` that declares one) is genuinely forwarded to - `SpeechSynthesizerFactory.Create(...)` on Play - selecting a different voice in the dropdown - audibly changes the synthesized speech. + `SpeechSynthesizerFactory.LoadAsync(...)` whenever it changes (triggering a cached-engine + reload on the next Play) - selecting a different voice in the dropdown audibly changes the + synthesized speech. ### Diagnostics: Raw Capture Recording (Temporary) diff --git a/docs/verification/definition.yaml b/docs/verification/definition.yaml index 2cba090..79585ab 100644 --- a/docs/verification/definition.yaml +++ b/docs/verification/definition.yaml @@ -41,10 +41,21 @@ input-files: - docs/verification/speech/model-management-subsystem/speech-model-descriptor.md - docs/verification/speech/model-management-subsystem/speech-model-catalog.md - docs/verification/speech/recognition-subsystem.md - - docs/verification/speech/recognition-subsystem/i-speech-recognizer.md + - docs/verification/speech/recognition-subsystem/i-speech-recognizer-engine.md + - docs/verification/speech/recognition-subsystem/i-recognition-session.md - docs/verification/speech/recognition-subsystem/speech-recognizer-factory.md - - docs/verification/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md - - docs/verification/speech/recognition-subsystem/unavailable-speech-recognizer.md + - docs/verification/speech/recognition-subsystem/sherpa-onnx-recognition-session.md + - docs/verification/speech/recognition-subsystem/unavailable-speech-recognizer-engine.md + - docs/verification/speech/recognition-subsystem/unavailable-recognition-session.md + - docs/verification/speech/synthesis-subsystem.md + - docs/verification/speech/synthesis-subsystem/i-speech-synthesizer-engine.md + - docs/verification/speech/synthesis-subsystem/i-synthesis-session.md + - docs/verification/speech/synthesis-subsystem/speech-synthesizer-factory.md + - docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md + - docs/verification/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md + - docs/verification/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.md + - docs/verification/speech/synthesis-subsystem/unavailable-synthesis-session.md + - docs/verification/speech/synthesis-subsystem/audio-tag-parser.md - docs/verification/speech-demo.md - docs/verification/speech-demo/shell-subsystem.md - docs/verification/speech-demo/device-selection-subsystem.md diff --git a/docs/verification/ots/sherpa-onnx.md b/docs/verification/ots/sherpa-onnx.md index 9b4bf55..1d838c8 100644 --- a/docs/verification/ots/sherpa-onnx.md +++ b/docs/verification/ots/sherpa-onnx.md @@ -49,7 +49,7 @@ declaration the streaming API requires is available to the recognition pipeline. **Requirement coverage**: `Speech-OTS-SherpaOnnx-ManagedStreamingApi`. -#### SpeechRecognizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealRecognizer +#### SpeechRecognizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine **Scenario**: Recognition is composed for an installed model and an available capture device. @@ -59,12 +59,12 @@ pattern works end to end. **Requirement coverage**: `Speech-OTS-SherpaOnnx-ModelOwnedConfiguration`. -#### SpeechRecognizerFactory_Create_EngineLoadFails_ReturnsUnavailableRecognizerAndDoesNotThrow +#### SpeechRecognizerFactory_LoadAsync_EngineLoadFails_ReturnsUnavailableEngineAndDoesNotFaultTask **Scenario**: Loading the speech-inference engine fails, as it would on a machine whose native runtime is absent. -**Expected**: Composition returns the honest unavailable recognizer and reports the reason, +**Expected**: Composition returns the honest unavailable engine and reports the reason, without throwing, proving the native-runtime boundary degrades in the required honest way. **Requirement coverage**: `Speech-OTS-SherpaOnnx-ModelOwnedConfiguration`. @@ -75,5 +75,5 @@ without throwing, proving the native-runtime boundary degrades in the required h `IRecognitionModel_CreateEngineConfig_InstalledDirectory_ResolvesPathsAndSampleRate`, `IRecognitionModel_AudioFormat_DeclaredByModel_IsExposed` - **`Speech-OTS-SherpaOnnx-ModelOwnedConfiguration`**: - `SpeechRecognizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealRecognizer`, - `SpeechRecognizerFactory_Create_EngineLoadFails_ReturnsUnavailableRecognizerAndDoesNotThrow` + `SpeechRecognizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine`, + `SpeechRecognizerFactory_LoadAsync_EngineLoadFails_ReturnsUnavailableEngineAndDoesNotFaultTask` diff --git a/docs/verification/speech-cli.md b/docs/verification/speech-cli.md index 3f73ad5..9437b44 100644 --- a/docs/verification/speech-cli.md +++ b/docs/verification/speech-cli.md @@ -16,8 +16,10 @@ implemented in Pass 4 (`list-devices`, `devices test`, `doctor`, verified in det _SpeechCli DeviceCommandsSubsystem Verification_), the one text-to-speech subcommand implemented in Pass 5 (`speak`, verified in detail in _SpeechCli SynthesisCommandSubsystem Verification_), and the one speech-to-text subcommand implemented in this pass (`recognize`, verified in detail -in _SpeechCli RecognitionCommandSubsystem Verification_) - the last of all ten subcommands to be -implemented; no subcommand handler remains a `NotImplementedException` stub. +in _SpeechCli RecognitionCommandSubsystem Verification_) - the tenth of what are now eleven +recognized subcommands to be implemented (the eleventh, `ask`, was added afterward and is +verified separately in _SpeechCli ConversationCommandSubsystem Verification_); no subcommand +handler remains a `NotImplementedException` stub. Automated coverage **does not** extend to real audio hardware or to a real, downloaded speech model: the `--validate` self-test's audio and model-store checks are composed exactly as the @@ -64,7 +66,7 @@ as a child process. `SpeechCli_NoArguments_Invoked_DisplaysBannerAndUsage` Verifies that `-h`/`-?`/`--help`, and running with no arguments at all, print the banner and usage -text listing every one of the ten recognized subcommands, and exit `0`. +text listing every one of the eleven recognized subcommands, and exit `0`. ### Subcommand Dispatch @@ -75,9 +77,9 @@ text listing every one of the ten recognized subcommands, and exit `0`. Verifies that `list-models` (representative of the five model-management subcommands), `list-devices`/`doctor` (representative of the three device-related subcommands), `speak`, and -`recognize` no longer throw `NotImplementedException` - `recognize` being the last of all ten -subcommands to reach that state - and that an unrecognized subcommand name is rejected cleanly -with a non-zero exit code rather than a stack trace. See _SpeechCli ModelCommandsSubsystem +`recognize` no longer throw `NotImplementedException` - `recognize` being the tenth of what are +now eleven subcommands to reach that state - and that an unrecognized subcommand name is rejected +cleanly with a non-zero exit code rather than a stack trace. See _SpeechCli ModelCommandsSubsystem Verification_, _SpeechCli DeviceCommandsSubsystem Verification_, _SpeechCli SynthesisCommandSubsystem Verification_, and _SpeechCli RecognitionCommandSubsystem Verification_ for the implemented subcommands' own detailed test scenarios. diff --git a/docs/verification/speech-cli/conversation-command-subsystem.md b/docs/verification/speech-cli/conversation-command-subsystem.md index f7fc6ac..1ab2e0f 100644 --- a/docs/verification/speech-cli/conversation-command-subsystem.md +++ b/docs/verification/speech-cli/conversation-command-subsystem.md @@ -4,48 +4,57 @@ The ConversationCommandSubsystem is verified entirely through deterministic unit tests against the same test doubles the synthesis and recognition passes already established: no new fake test -double is needed for the synthesizer/recognizer/catalog. `FakeCliModelCatalog`'s existing -`CreateSynthesizerOverride`/`CreateRecognizerOverride` delegates resolve `AskCommand`'s two -models; `FakeSpeechSynthesizer` (recording `SpeakAsync`/`Dispose` call counts, with a settable -exception to simulate a canceled Phase 1) and `FakeSpeechRecognizer` (recording `Start`/`Stop`/ -`Dispose` call counts, with an `OnStart` callback that raises a result synchronously before -`Start()` returns) drive both phases. Device resolution reuses `FakePlaybackDeviceSource` -(unmodified, from `SynthesisCommandSubsystem`'s own tests) for Phase 1, and this pass's own new +double is needed for the engine/session/catalog types themselves. `FakeCliModelCatalog`'s existing +`CreateSynthesizerEngineOverride`/`CreateRecognizerEngineOverride` delegates resolve `AskCommand`'s +two engines; `FakeSynthesisSession`/`FakeSpeechSynthesizerEngine` (recording `SpeakAsync`/ +`DisposeAsync` call counts, with a settable exception to simulate a canceled Phase 1) and +`FakeRecognitionSession`/`FakeSpeechRecognizerEngine` (recording `StartAsync`/`StopAsync`/ +`DisposeAsync` call counts, driving `GetResultsAsync` from an in-memory result sequence) drive both +phases. Device resolution reuses `FakePlaybackDeviceSource` (unmodified, from +`SynthesisCommandSubsystem`'s own tests) for Phase 1, and this pass's own new `FakeCaptureDeviceSource`/`FakeAudioCaptureDevice` (mirroring `FakePlaybackDeviceSource`/ `FakeAudioPlaybackDevice` exactly) for Phase 2 - never a real `AudioDeviceFactory` - so every `AskCommandTests` scenario that expects a successfully resolved capture device is deterministic on any machine, including headless CI runners with no real PortAudio hardware at all. No real model catalog, network access, or audio hardware is required by any `AskCommandTests` scenario. -Because `FakeSpeechRecognizer.Start()` and `FakeSpeechSynthesizer.SpeakAsync()` both invoke their -configured callbacks/results synchronously before returning, a fully deterministic, single-threaded -test can simulate a complete speak-then-listen turn - including a genuine final-result-driven -listen-phase stop - with no real threading or wall-clock delay. The one scenario that does need a -real elapsed delay (`AskCommand_Run_SilenceTimeoutWithNoFinalResult_EndsTurnWithEmptyText`) uses a -tiny (`0.05`s) real `--silence-timeout` value so the reused, unmodified -`SilenceTimeoutRecognizerSession`'s real `TimeProvider.System`-backed timer genuinely fires, -mirroring the same technique `RecognizeCommandTests` uses for its own silence-timeout scenarios. +Because `FakeRecognitionSession.GetResultsAsync()` and `FakeSynthesisSession.SpeakAsync()` both +yield/complete deterministically from in-memory configuration, a fully deterministic, +single-threaded test can simulate a complete speak-then-listen turn - including a genuine +final-result-driven listen-phase stop - with no real threading or wall-clock delay. The one +scenario that does need a real elapsed delay +(`AskCommand_Run_SilenceTimeoutWithNoFinalResult_EndsTurnWithEmptyText`) uses a tiny (`0.05`s) real +`--silence-timeout` value so the reused, unmodified `SilenceTimeoutRecognizerSession`'s real +`Task.Delay`-backed idle-timeout race genuinely fires, mirroring the same technique +`SilenceTimeoutRecognizerSessionTests`/`RecognizeCommandTests` use for their own silence-timeout +scenarios. A genuine `Ctrl+C` landing during Phase 2 cannot be simulated by raising a real -`Console.CancelKeyPress` event from a test (there is no supported way to do so), so the three +`Console.CancelKeyPress` event from a test (there is no supported way to do so), so the `AskCommand_RunAsync_*` cancellation scenarios drive `AskCommand.RunAsync` directly - the same internal entry point the public `Run` overload's real `Console.CancelKeyPress` handler calls - supplying their own `CancellationTokenSource`/`ManualResetEventSlim` pair and canceling/signaling -them from within a fake recognizer's `Start()` callback (or before `RunAsync` is even invoked, for -the early-return edge case) to reproduce the exact interleaving the real handler produces. +them from within a fake session's result sequence (or before `RunAsync` is even invoked, for the +early-return edge case) to reproduce the exact interleaving the real handler produces. +`AskCommand_Run_PrewarmsRecognizerConcurrentlyWithPlayback_CreatesRecognizerBeforePlaybackCompletes` +proves the recognizer engine/session pre-warm genuinely overlaps Phase 1's wait, using +`FakeSynthesisSession.SpeakAsyncAwaiter` to hold Phase 1's `SpeakAsync` in flight until +`FakeCliModelCatalog.CreateRecognizerEngineOverride` signals a `ManualResetEventSlim`, with a +bounded wait so a future regression that reorders pre-warming back to strictly after playback +fails the test explicitly rather than deadlocking. ### Test Environment - **Framework**: xUnit v3 running under the .NET SDK - **Execution**: `dotnet test` invoked by `build.ps1` and the CI pipeline -- **Isolation**: Every test constructs its own fake catalog/synthesizer/recognizer/device probes; +- **Isolation**: Every test constructs its own fake catalog/engine/session/device probes; `--output-text` tests write to a uniquely named temporary file, deleted afterward -- **Test doubles**: `FakeCliModelCatalog`, `FakeSpeechSynthesizer`, `FakeSpeechRecognizer`, - `FakePlaybackDeviceSource`, `FakeAudioPlaybackDevice` (now tracking a `DisposeCallCount`, - mirroring `FakeSpeechRecognizer.DisposeCallCount`, so a test can prove the playback device - resolved by `SpeakPromptAsync` is disposed exactly once), `FakeAudioPlaybackDeviceProbe` - (reused, mostly unmodified, from `SynthesisCommandSubsystem`'s own tests), and this pass's own - new `FakeCaptureDeviceSource`/`FakeAudioCaptureDevice`/`FakeAudioCaptureDeviceProbe` (mirroring +- **Test doubles**: `FakeCliModelCatalog`, `FakeSynthesisSession`/`FakeSpeechSynthesizerEngine`, + `FakeRecognitionSession`/`FakeSpeechRecognizerEngine`, `FakePlaybackDeviceSource`, + `FakeAudioPlaybackDevice` (tracking a `DisposeCallCount`, so a test can prove the playback device + resolved by `SpeakPromptAsync` is disposed exactly once), `FakeAudioPlaybackDeviceProbe` (reused, + mostly unmodified, from `SynthesisCommandSubsystem`'s own tests), and this pass's own new + `FakeCaptureDeviceSource`/`FakeAudioCaptureDevice`/`FakeAudioCaptureDeviceProbe` (mirroring `FakePlaybackDeviceSource`/`FakeAudioPlaybackDevice` exactly), so no `AskCommandTests` scenario ever depends on a real `AudioDeviceFactory` @@ -80,8 +89,9 @@ flag on `ask`; `--text` parses correctly; supplying both `--text` and `--file` t model id, and a not-yet-downloaded model id are each rejected with a distinct, actionable `ArgumentException` message (naming `--tts-model` or `--stt-model` as appropriate, suggesting `list-models`, `list-models --role tts`/`--role stt`, and `download ` respectively); a -successful run speaks the resolved text through the resolved TTS model, then listens through the -resolved STT model, then prints the final recognized text. +successful run awaits `CreateSynthesizerEngineAsync`/`CreateSessionAsync`/`SpeakAsync` through the +resolved TTS model, then `CreateRecognizerEngineAsync`/`CreateSessionAsync`/`StartAsync`/ +`GetResultsAsync` through the resolved STT model, then prints the final recognized text. **Requirement coverage**: `SpeechCli-ConversationCommands-Ask`. @@ -93,9 +103,9 @@ resolved STT model, then prints the final recognized text. **Scenario/Expected**: Repeated `--tts-param`/`--stt-param` flags each accumulate independently, in the order given, and are forwarded, fully resolved via the shared `ParameterBagParser`, to -`CreateSynthesizer`'s and `CreateRecognizer`'s own `parameterValues` argument respectively; an -invalid value for a declared TTS parameter is rejected before any synthesis or recognition is -attempted. Full `ParameterBagParser` scenario coverage already lives in +`CreateSynthesizerEngineAsync`'s and `CreateRecognizerEngineAsync`'s own `parameterValues` argument +respectively; an invalid value for a declared TTS parameter is rejected before any engine is +loaded. Full `ParameterBagParser` scenario coverage already lives in `SpeechCli-SynthesisCommands-ParamValidation`'s own tests and is not duplicated here. **Requirement coverage**: `SpeechCli-ConversationCommands-ParamValidation`. @@ -104,14 +114,16 @@ attempted. Full `ParameterBagParser` scenario coverage already lives in **Tests**: `AskCommand_ParseArguments_TimeoutFlags_ParseAsPositiveDoubles`, `AskCommand_ParseArguments_NonPositiveSilenceTimeout_ThrowsArgumentException`, -`AskCommand_Run_SilenceTimeoutWithNoFinalResult_EndsTurnWithEmptyText` +`AskCommand_Run_SilenceTimeoutWithNoFinalResult_EndsTurnWithEmptyText`, +`AskCommand_Run_NoTimeoutFlagsGiven_StillEndsTurnViaDefaultSilenceTimeout` **Scenario/Expected**: `--start-timeout`/`--silence-timeout` each parse a positive number of seconds, rejecting a non-positive value; the listen phase ends as soon as the first final recognition result arrives (verified as part of the success-path scenario above); when no final result ever arrives, a `--silence-timeout` firing (via the reused, unmodified `SilenceTimeoutRecognizerSession`) ends the turn with an empty recognized-text result rather than -blocking forever. +blocking forever; omitting both `--start-timeout` and `--silence-timeout` still ends the turn via +the built-in default silence timeout rather than listening indefinitely. **Requirement coverage**: `SpeechCli-ConversationCommands-ListenTermination`. @@ -134,8 +146,8 @@ stdout. `AskCommand_Run_NoCaptureDeviceAvailable_ThrowsInvalidOperationException` **Scenario/Expected**: Requesting an unrecognized `--playback-device`/`--capture-device` name -throws `ArgumentException` before any synthesizer/recognizer is created; an unavailable resolved -real playback or capture device throws `InvalidOperationException`. +throws `ArgumentException` before any engine is loaded; an unavailable resolved real playback or +capture device throws `InvalidOperationException`. **Requirement coverage**: `SpeechCli-ConversationCommands-DeviceDispatch`. @@ -144,6 +156,7 @@ real playback or capture device throws `InvalidOperationException`. **Tests**: `AskCommand_Run_CanceledDuringSpeak_SkipsListenPhase`, `AskCommand_Run_ExceptionDuringSpeak_DisposesPlaybackDeviceAndRethrows`, `AskCommand_Run_Success_SpeaksThenListensAndPrintsFinalResult`, +`AskCommand_RunAsync_CanceledDuringSpeak_ReturnsPromptlyWithoutAwaitingInFlightPrewarm`, `AskCommand_RunAsync_CtrlCDuringListen_ReportsCanceledAndDoesNotPrintText`, `AskCommand_RunAsync_CtrlCBeforeListenStarts_ReportsCanceledAndDoesNotPrintText`, `AskCommand_RunAsync_SilenceTimeoutWithNoCtrlC_ReportsSuccessNotCanceled`, @@ -151,30 +164,31 @@ real playback or capture device throws `InvalidOperationException`. `AskCommand_RunAsync_CtrlCImmediatelyBeforeFileWrite_ReportsCanceledAndDoesNotWriteFile` **Scenario/Expected**: A cancellation raised during Phase 1 (speak) is reported and causes Phase 2 -(listen) to be skipped entirely - no recognizer is ever *started*, the concurrently pre-warmed -recognizer (whether already constructed or still in flight) is disposed rather than leaked, and -the command returns cleanly with no partial recognized-text or playback state leaked. A genuine -`Ctrl+C` landing during Phase 2 - whether mid-listen or in the narrow window before Phase 2 even -starts a recognizer - is reported via `context.WriteError` with a non-zero exit code, and no -recognized text (even if some was already captured) is ever printed or written to -`--output-text`/stdout. A `Ctrl+C` landing in the narrow window immediately after `Listen()` -returns - whether or not `--output-text` is configured - is likewise reported as a cancellation -and never writes the output file or prints "Recognized text written", proving `RunAsync`'s -output-writing step (which now passes the real `cancellationToken` into +(listen) to be skipped entirely - no session is ever *started*, the concurrently pre-warmed +engine/session (whether already constructed or still in flight) is disposed rather than leaked via +the fire-and-forget `DisposePrewarmedRecognizerAsync` path, and `RunAsync` returns promptly without +awaiting an in-flight pre-warm, proving the fast-failure/`Ctrl+C`-responsiveness contract the +pre-warm design depends on. A genuine `Ctrl+C` landing during Phase 2 - whether mid-listen or in +the narrow window before Phase 2 even starts a session - is reported via `context.WriteError` with +a non-zero exit code, and no recognized text (even if some was already captured) is ever printed or +written to `--output-text`/stdout. A `Ctrl+C` landing in the narrow window immediately after +`Listen()` returns - whether or not `--output-text` is configured - is likewise reported as a +cancellation and never writes the output file or prints "Recognized text written", proving +`RunAsync`'s output-writing step (which passes the real `cancellationToken` into `File.WriteAllTextAsync` wrapped in a `try`/`catch` mirroring every other cancellation exit point in `RunAsync`) never falsely reports success once cancellation has been observed. Conversely, a legitimate empty result from a `--silence-timeout`/`--start-timeout` firing with no `Ctrl+C` involved is still reported as a normal, zero-exit-code success, proving the two outcomes are correctly distinguished rather than both being silently treated as success. -The resolved playback device's disposal - `SpeakPromptAsync`'s own `finally`-guaranteed -`(playbackDevice as IDisposable)?.Dispose()` - is verified as disposed exactly once across all -three of Phase 1's possible exits: the ordinary success path -(`AskCommand_Run_Success_SpeaksThenListensAndPrintsFinalResult`), a canceled Phase 1 -(`AskCommand_Run_CanceledDuringSpeak_SkipsListenPhase`), and a non-cancellation exception thrown -from Phase 1's `SpeakAsync` (`AskCommand_Run_ExceptionDuringSpeak_DisposesPlaybackDeviceAndRethrows`), -using `FakeAudioPlaybackDevice.DisposeCallCount` - proving playback-device disposal is not only -correct on the ordinary path but also never skipped when playback is canceled or faults. +The resolved playback device's disposal - `SpeakPromptAsync`'s own `using var playbackDeviceLease` +cast - is verified as disposed exactly once across all three of Phase 1's possible exits: the +ordinary success path (`AskCommand_Run_Success_SpeaksThenListensAndPrintsFinalResult`), a canceled +Phase 1 (`AskCommand_Run_CanceledDuringSpeak_SkipsListenPhase`), and a non-cancellation exception +thrown from Phase 1's `SpeakAsync` +(`AskCommand_Run_ExceptionDuringSpeak_DisposesPlaybackDeviceAndRethrows`), using +`FakeAudioPlaybackDevice.DisposeCallCount` - proving playback-device disposal is not only correct +on the ordinary path but also never skipped when playback is canceled or faults. **Requirement coverage**: `SpeechCli-ConversationCommands-CancellationAndDisposal`. @@ -184,21 +198,22 @@ correct on the ordinary path but also never skipped when playback is canceled or `AskCommand_Run_CanceledDuringSpeak_SkipsListenPhase`, `AskCommand_Run_NoCaptureDeviceAvailable_ThrowsInvalidOperationException` -**Scenario/Expected**: The recognizer is constructed concurrently with Phase 1's speak/playback -wait, not only after it finishes. The primary test proves this deterministically rather than with -a flaky timing assertion: `FakeSpeechSynthesizer.SpeakAsyncAwaiter` holds Phase 1's `SpeakAsync` -"in flight" until a `ManualResetEventSlim` set from inside `FakeCliModelCatalog`'s -`CreateRecognizerOverride` is signaled, with a bounded (five-second) wait. If a future regression -moves recognizer construction back to strictly after playback finishes, `SpeakAsync` would never -observe the signal being set (since `CreateRecognizer` would not run until *after* `SpeakAsync` -itself returns), and the bounded wait fails the test explicitly with a descriptive message -instead of either silently passing or deadlocking the test run indefinitely. The same scenario -also proves Phase 2 still proceeds correctly once Phase 1 completes (the recognizer starts, stops -on the first final result, and is disposed exactly once). The two supporting tests prove pre-warm -failure/cancellation is handled correctly rather than merely proving pre-warm construction itself: -a canceled Phase 1 disposes the concurrently pre-warmed recognizer rather than leaking it, and a -pre-warm failure (no capture device available) surfaces only *after* Phase 1 has already spoken -the prompt - proving Phase 1 is never skipped or reordered even when Phase 2's setup fails. +**Scenario/Expected**: The recognizer engine/session is constructed concurrently with Phase 1's +speak/playback wait, not only after it finishes. The primary test proves this deterministically +rather than with a flaky timing assertion: `FakeSynthesisSession.SpeakAsyncAwaiter` holds Phase 1's +`SpeakAsync` "in flight" until a `ManualResetEventSlim` set from inside `FakeCliModelCatalog`'s +`CreateRecognizerEngineOverride` is signaled, with a bounded (five-second) wait. If a future +regression moves recognizer engine construction back to strictly after playback finishes, +`SpeakAsync` would never observe the signal being set (since `CreateRecognizerEngineAsync` would +not run until *after* `SpeakAsync` itself returns), and the bounded wait fails the test explicitly +with a descriptive message instead of either silently passing or deadlocking the test run +indefinitely. The same scenario also proves Phase 2 still proceeds correctly once Phase 1 +completes (the session starts, stops on the first final result, and the engine/session are each +disposed exactly once). The two supporting tests prove pre-warm failure/cancellation is handled +correctly rather than merely proving pre-warm construction itself: a canceled Phase 1 disposes the +concurrently pre-warmed engine/session rather than leaking it, and a pre-warm failure (no capture +device available) surfaces only *after* Phase 1 has already spoken the prompt - proving Phase 1 is +never skipped or reordered even when Phase 2's setup fails. **Requirement coverage**: `SpeechCli-ConversationCommands-PrewarmRecognizer`. @@ -208,13 +223,20 @@ the prompt - proving Phase 1 is never skipped or reordered even when Phase 2's s `AskCommand_Run_NullCatalog_ThrowsArgumentNullException`, `AskCommand_Run_NullDeviceSource_ThrowsArgumentNullException`, `AskCommand_Run_NullCaptureSource_ThrowsArgumentNullException`, -`AskCommand_ParseArguments_NullArgs_ThrowsArgumentNullException` +`AskCommand_ParseArguments_NullArgs_ThrowsArgumentNullException`, +`AskCommand_RunAsync_NullContext_ThrowsArgumentNullException`, +`AskCommand_RunAsync_NullCatalog_ThrowsArgumentNullException`, +`AskCommand_RunAsync_NullDeviceSource_ThrowsArgumentNullException`, +`AskCommand_RunAsync_NullCaptureSource_ThrowsArgumentNullException`, +`AskCommand_RunAsync_NullStopSignal_ThrowsArgumentNullException`, +`AskCommand_RunAsync_NullOnRecognizerCreated_ThrowsArgumentNullException` **Scenario/Expected**: Every entry point rejects a `null` context, catalog, playback-device source, or capture-device source immediately with `ArgumentNullException`, proving `AskCommand` -never silently proceeds with a missing dependency. `ParseArguments` itself independently rejects -a `null` argument list with `ArgumentNullException`, proving the guard holds even when -`ParseArguments` is invoked directly rather than only through `Run`/`RunAsync`. +never silently proceeds with a missing dependency; the internal `RunAsync` overload additionally +rejects a `null` `stopSignal` or `onRecognizerCreated` callback. `ParseArguments` itself +independently rejects a `null` argument list with `ArgumentNullException`, proving the guard holds +even when `ParseArguments` is invoked directly rather than only through `Run`/`RunAsync`. **Requirement coverage**: `SpeechCli-ConversationCommands-NullGuards`. diff --git a/docs/verification/speech-cli/recognition-command-subsystem.md b/docs/verification/speech-cli/recognition-command-subsystem.md index d3fe4c2..0dab899 100644 --- a/docs/verification/speech-cli/recognition-command-subsystem.md +++ b/docs/verification/speech-cli/recognition-command-subsystem.md @@ -3,26 +3,28 @@ ### Verification Approach The RecognitionCommandSubsystem is verified through deterministic unit tests against a -hand-written `FakeSpeechRecognizer` (recording `Start`/`Stop`/`Dispose` call counts, with a -settable `IsAvailable` and an `OnStart` callback letting a test simulate a recognizer's own -synchronous `Start()`-drives-the-device contract) and `FakeCliModelCatalog`'s two new delegate -overrides (`GetAudioFormatOverride`/`CreateRecognizerOverride`), reused from +hand-written `FakeRecognitionSession`/`FakeSpeechRecognizerEngine` pair (recording +`StartAsync`/`StopAsync`/`DisposeAsync` call counts, with a settable `IsAvailable` and an +`OnStartAsync` callback letting a test simulate a session's own synchronous, blocking +`StartAsync`-drives-the-device file-mode contract) and `FakeCliModelCatalog`'s two new delegate +overrides (`GetAudioFormatOverride`/`CreateRecognizerEngineOverride`), reused from `ModelCommandsSubsystem`'s own tests. `SilenceTimeoutRecognizerSession` is verified entirely in isolation against a hand-written `FakeTimeProvider`/`FakeTimer` pair whose callback a test invokes directly, so every idle/reset/timeout scenario runs deterministically with no real wall-clock -delay. The two new `ICliModelCatalog` seam members (`GetAudioFormat`/`CreateRecognizer`) are -additionally verified against a **real** `SpeechModelCatalogAdapter`, proving the production `is -IRecognitionModel` cast genuinely throws for a real, compiled-in synthesis-role model, in -`SpeechModelCatalogAdapterTests.cs`. Out-of-process integration tests in `IntegrationTests.cs` -invoke the built tool as a child process for the model-resolution and input-source error paths -that matter most from an operator's perspective. - -The file-input EOF-driven-stop flow is verified against a **real** `WavFileAudioCaptureDevice` +delay. The two new `ICliModelCatalog` seam members (`GetAudioFormat`/ +`CreateRecognizerEngineAsync`) are additionally verified against a **real** +`SpeechModelCatalogAdapter`, proving the production `is IRecognitionModel` cast genuinely throws +for a real, compiled-in synthesis-role model, in `SpeechModelCatalogAdapterTests.cs`. +Out-of-process integration tests in `IntegrationTests.cs` invoke the built tool as a child process +for the model-resolution and input-source error paths that matter most from an operator's +perspective. + +The file-input, explicit-drain flow is verified against a **real** `WavFileAudioCaptureDevice` constructed over a minimal, self-generated, valid mono/16-bit-PCM WAV file (written directly by -the test, not a shared binary fixture), paired with `FakeSpeechRecognizer`'s `OnStart` callback -driving that real device's own `Start()` - proving the reentrant `Stop()`-from-`EndOfFileReached` -wiring genuinely round-trips through the real device rather than only being exercised against a -fully-faked recognizer/device pair. +the test, not a shared binary fixture), paired with `FakeRecognitionSession` wired to the real +`FakeSpeechRecognizerEngine.CreateSessionAsync` call - proving `RecognizeCommand`'s own +`StartAsync`-then-explicit-`StopAsync` sequencing for file mode genuinely drives a real device +rather than only being exercised against a fully-faked recognizer/device pair. This environment does have at least one real, downloaded speech-to-text model available (unlike the SynthesisCommandSubsystem pass's TTS gap), so `SpeechCli_RecognizeCommandWithRealSttModel_ @@ -39,14 +41,14 @@ adding a cross-project `.csproj` content-copy item. - **Framework**: xUnit v3 running under the .NET SDK - **Execution**: `dotnet test` invoked by `build.ps1` and the CI pipeline -- **Isolation**: Every unit test constructs its own fake catalog/recognizer/time provider/audio +- **Isolation**: Every unit test constructs its own fake catalog/engine/session/time provider/audio probes; `SpeechModelCatalogAdapterTests` uses a freshly created, uniquely named temporary directory as its model-store root, deleted afterward via `IDisposable` -- **Test doubles**: `FakeSpeechRecognizer` (hand-written `ISpeechRecognizer` fake), - `FakeTimeProvider`/`FakeTimer` (hand-written `TimeProvider`/`ITimer` fakes with a directly - invocable callback), `FakeCliModelCatalog` (extended with `GetAudioFormatOverride`/ - `CreateRecognizerOverride`), `FakeAudioCaptureDeviceProbe` (reused from - `DeviceCommandsSubsystem`'s own tests) +- **Test doubles**: `FakeRecognitionSession`/`FakeSpeechRecognizerEngine` (hand-written + `IRecognitionSession`/`ISpeechRecognizerEngine` fakes), `FakeTimeProvider`/`FakeTimer` + (hand-written `TimeProvider`/`ITimer` fakes with a directly invocable callback), + `FakeCliModelCatalog` (extended with `GetAudioFormatOverride`/`CreateRecognizerEngineOverride`), + `FakeAudioCaptureDeviceProbe` (reused from `DeviceCommandsSubsystem`'s own tests) ### Test Scenarios @@ -62,7 +64,7 @@ adding a cross-project `.csproj` content-copy item. `RecognizeCommand_Run_UnknownModelId_ThrowsArgumentException`, `RecognizeCommand_Run_WrongRoleModel_ThrowsArgumentException`, `RecognizeCommand_Run_NotDownloadedModel_ThrowsArgumentExceptionWithDownloadHint`, -`RecognizeCommand_Run_Success_DisposesRecognizerOnce`, +`RecognizeCommand_Run_Success_DisposesEngineAndSessionOnce`, `SpeechCli_RecognizeCommandWithoutModel_Invoked_ReturnsCleanError`, `SpeechCli_RecognizeCommandWithUnknownModel_Invoked_ReturnsCleanError`, `SpeechCli_RecognizeCommandWithWrongRoleModel_Invoked_ReturnsCleanError`, @@ -77,39 +79,40 @@ rejected with `ArgumentException`; `--input`/`--mic` each parse correctly; suppl up; an unknown model id, a wrong-role model id, and a not-yet-downloaded model id are each rejected with a distinct, actionable `ArgumentException` message (suggesting `list-models`, `list-models --role stt`, and `download ` respectively); a successful run disposes the -fake recognizer exactly once; every error path is proven both in-process and, cleanly and -non-zero-exit, against the built tool; a real, downloaded recognition model recognizing a real WAV -fixture produces non-empty text; `recognize` no longer throws `NotImplementedException` when -dispatched - the last of all 10 subcommands to reach that state. +fake engine and session exactly once each, in declaration order (session before engine); every +error path is proven both in-process and, cleanly and non-zero-exit, against the built tool; a +real, downloaded recognition model recognizing a real WAV fixture produces non-empty text; +`recognize` no longer throws `NotImplementedException` when dispatched. **Requirement coverage**: `SpeechCli-RecognitionCommands-Recognize`. #### `--stt-param` Validation Reuse **Tests**: `RecognizeCommand_ParseArguments_RepeatedParamFlags_AccumulatesInOrder`, -`RecognizeCommand_Run_ValidParam_ForwardsToCreateRecognizer`, +`RecognizeCommand_Run_ValidParam_ForwardsToCreateRecognizerEngine`, `RecognizeCommand_Run_InvalidParam_ThrowsArgumentException` **Scenario/Expected**: Repeated `--stt-param` flags accumulate in the order given and are forwarded, -fully resolved via the shared `ParameterBagParser`, to `CreateRecognizer`'s `parameterValues` -argument; an invalid value for a declared parameter is rejected before any recognizer is created. -Full `ParameterBagParser` scenario coverage (numeric range/integer checks, choice matching, -boolean parsing, unrecognized-key rejection) already lives in +fully resolved via the shared `ParameterBagParser`, to `CreateRecognizerEngineAsync`'s +`parameterValues` argument; an invalid value for a declared parameter is rejected before any +engine is loaded. Full `ParameterBagParser` scenario coverage (numeric range/integer checks, +choice matching, boolean parsing, unrecognized-key rejection) already lives in `SpeechCli-SynthesisCommands-ParamValidation`'s own tests and is not duplicated here. **Requirement coverage**: `SpeechCli-RecognitionCommands-ParamValidation`. -#### File-Input EOF-Driven Stop Flow +#### File-Input Explicit-Drain Flow -**Tests**: `RecognizeCommand_Run_FileInput_StartsAndStopsRecognizerViaEndOfFile`, -`RecognizeCommand_Run_FileInput_PassesWavFileCaptureDeviceToCreateRecognizer` +**Tests**: `RecognizeCommand_Run_FileInput_StartsAndStopsRecognizerViaExplicitDrain`, +`RecognizeCommand_Run_FileInput_PassesWavFileCaptureDeviceToCreateSession` **Scenario/Expected**: `--input ` constructs a real `WavFileAudioCaptureDevice` over -that path and passes it into `CreateRecognizer`; a fake recognizer whose `Start()` drives that -real device's own `Start()` to completion (simulating a real recognizer's documented contract) -results in `Start`/`Stop`/`Dispose` each being called exactly once, with `Stop()` triggered -reentrantly by the device's own `EndOfFileReached` event and no explicit wait added by the -command. +that path and passes it into `engine.CreateSessionAsync`; a fake session whose `StartAsync` +synchronously drives that real device's own delivery to completion (simulating a real session's +documented file-mode contract), followed by `RecognizeCommand`'s own explicit drain +`StopAsync` call, results in `StartAsync`/`StopAsync`/`DisposeAsync` each being called exactly +once, with no `SilenceTimeoutRecognizerSession` wrapper and no added wait by the command in file +mode. **Requirement coverage**: `SpeechCli-RecognitionCommands-FileInputEofDrivenStop`. @@ -118,25 +121,21 @@ command. **Tests**: `RecognizeCommand_ParseArguments_SilenceTimeoutFlag_ParsesSeconds`, `RecognizeCommand_ParseArguments_NonPositiveSilenceTimeout_ThrowsArgumentException`, `RecognizeCommand_ParseArguments_MalformedSilenceTimeout_ThrowsArgumentException`, -`SilenceTimeoutRecognizerSession_Construct_ArmsTimerWithGivenTimeout`, -`SilenceTimeoutRecognizerSession_PartialResultReceived_ResetsIdleTimer`, -`SilenceTimeoutRecognizerSession_FinalResultReceived_ResetsIdleTimer`, -`SilenceTimeoutRecognizerSession_IdleTimerFires_StopsRecognizerAndRaisesTimedOut`, -`SilenceTimeoutRecognizerSession_ResetThenFire_StopsOnlyOnActualFire`, -`SilenceTimeoutRecognizerSession_Dispose_UnsubscribesAndDisposesTimer`, -`SilenceTimeoutRecognizerSession_FireAfterDispose_DoesNotCallStopOrRaiseTimedOut`, -`SilenceTimeoutRecognizerSession_Construct_NullRecognizer_ThrowsArgumentNullException`, -`SilenceTimeoutRecognizerSession_Construct_NonPositiveTimeout_ThrowsArgumentOutOfRangeException`, -`SilenceTimeoutRecognizerSession_Construct_NullTimeProvider_UsesSystemTimeProvider` +`SilenceTimeoutRecognizerSession_Construct_NullSession_ThrowsArgumentNullException`, +`SilenceTimeoutRecognizerSession_Construct_NonPositiveIdleTimeout_ThrowsArgumentOutOfRangeException`, +`SilenceTimeoutRecognizerSession_Construct_NullTimeProvider_DoesNotThrow`, +`SilenceTimeoutRecognizerSession_GetResultsAsync_SecondResultReceived_StaysOnIdleTimeout`, +`SilenceTimeoutRecognizerSession_GetResultsAsync_TimeoutAfterResult_StopsSessionAndRaisesTimedOutOnce`, +`SilenceTimeoutRecognizerSession_GetResultsAsync_CancellationRequested_PropagatesOperationCanceledException` **Scenario/Expected**: `--silence-timeout` parses a positive number of seconds, rejecting zero, -negative, or malformed values; constructing a session arms its idle timer once with the given -timeout; a partial or final result re-arms the timer; the timer firing with no reset calls -`Stop()` exactly once and raises `TimedOut`; `Dispose()` unsubscribes and disposes the timer, -is idempotent, and a subsequent result never re-arms the disposed timer; a timer fire after -`Dispose()` never calls `Stop()` or raises `TimedOut` on the torn-down session; a `null` -recognizer or a non-positive timeout is rejected at construction; a `null` `TimeProvider` -defaults to `TimeProvider.System` without throwing. +negative, or malformed values; a `null` wrapped session or a non-positive idle timeout is rejected +at construction, while a `null` `TimeProvider` does not throw (defaulting to `TimeProvider.System`); +a second yielded result keeps re-arming with the idle timeout rather than reverting to the start +timeout; the idle timeout elapsing after a result has already been yielded calls +`_session.StopAsync` exactly once and raises `TimedOut` exactly once; a canceled +`CancellationToken` passed to `GetResultsAsync` propagates `OperationCanceledException` rather +than being silently swallowed. **Requirement coverage**: `SpeechCli-RecognitionCommands-SilenceTimeout`. @@ -146,23 +145,21 @@ defaults to `TimeProvider.System` without throwing. `RecognizeCommand_ParseArguments_StartTimeoutOmitted_DefaultsToNull`, `RecognizeCommand_ParseArguments_NonPositiveStartTimeout_ThrowsArgumentException`, `RecognizeCommand_ParseArguments_MalformedStartTimeout_ThrowsArgumentException`, -`SilenceTimeoutRecognizerSession_Construct_StartTimeoutOmitted_ArmsTimerWithSilenceTimeout`, -`SilenceTimeoutRecognizerSession_Construct_StartTimeoutGiven_ArmsTimerWithStartTimeout`, -`SilenceTimeoutRecognizerSession_FirstResultReceived_ReArmsWithSilenceTimeoutNotStartTimeout`, -`SilenceTimeoutRecognizerSession_SecondResultReceived_StaysOnSilenceTimeout`, -`SilenceTimeoutRecognizerSession_IdleTimerFiresBeforeFirstResult_StopsRecognizerAndRaisesTimedOut`, -`SilenceTimeoutRecognizerSession_Construct_NonPositiveStartTimeout_ThrowsArgumentOutOfRangeException` +`SilenceTimeoutRecognizerSession_Construct_NonPositiveStartTimeout_ThrowsArgumentOutOfRangeException`, +`SilenceTimeoutRecognizerSession_GetResultsAsync_StartTimeoutOmitted_ArmsTimerWithIdleTimeout`, +`SilenceTimeoutRecognizerSession_GetResultsAsync_StartTimeoutGiven_ArmsTimerWithStartTimeout`, +`SilenceTimeoutRecognizerSession_GetResultsAsync_FirstResultReceived_ReArmsWithIdleTimeoutNotStartTimeout`, +`SilenceTimeoutRecognizerSession_GetResultsAsync_TimeoutBeforeAnyResult_StopsSessionAndRaisesTimedOutOnce` **Scenario/Expected**: `--start-timeout` parses a positive number of seconds, rejecting zero, negative, or malformed values; parsing alone leaves `StartTimeoutSeconds` `null` when the flag is -omitted (no default is applied at parse time); constructing a session with `startTimeout` omitted -arms its idle timer with the same value as `idleTimeout` (today's exact behavior, preserved); -constructing a session with a distinct `startTimeout` value arms the timer with that value, not -`idleTimeout`; the first `ResultReceived` event (partial or final) re-arms the timer with -`idleTimeout`, not `startTimeout`; a second result stays re-armed with `idleTimeout`; the idle -timer firing before any result has arrived calls `Stop()` and raises `TimedOut`, proving -`startTimeout` genuinely governs the pre-first-result window rather than only being recorded; a -non-positive `startTimeout` is rejected at construction exactly like a non-positive `idleTimeout`. +omitted (no default is applied at parse time); a non-positive `startTimeout` is rejected at +construction exactly like a non-positive `idleTimeout`; omitting `startTimeout` arms the very +first race with `idleTimeout` (today's exact behavior, preserved); supplying a distinct +`startTimeout` arms the first race with that value instead; the first yielded result (partial or +final) re-arms subsequent races with `idleTimeout`, not `startTimeout`; the idle timeout elapsing +before any result has been yielded calls `_session.StopAsync` and raises `TimedOut`, proving +`startTimeout` genuinely governs the pre-first-result window rather than only being recorded. **Requirement coverage**: `SpeechCli-RecognitionCommands-StartTimeout`. @@ -172,6 +169,7 @@ non-positive `startTimeout` is rejected at construction exactly like a non-posit `RecognizeCommand_ParseArguments_OutputFlag_ParsesOutputPath`, `RecognizeCommand_Run_InterimAndFinalOnlyBothGiven_ThrowsArgumentException`, `RecognizeCommand_Run_DefaultVerbosity_PrintsBothInterimAndFinal`, +`RecognizeCommand_Run_ShrinkingInterimSequence_DoesNotLeaveStaleCharacters`, `RecognizeCommand_Run_FinalOnly_SuppressesInterimConsoleOutput`, `RecognizeCommand_Run_Interim_SuppressesFinalSettleConsoleOutput`, `RecognizeCommand_Run_Output_WritesOnlyFinalResultsOverwritingPriorContent` @@ -180,8 +178,10 @@ non-positive `startTimeout` is rejected at construction exactly like a non-posit `ArgumentException`; by default both an interim and a final result are printed to the console; `--final-only` suppresses interim console output while still printing final results; `--interim` suppresses the final "settle" console output while still printing interim results; -`--output-text ` writes only final results, one per line, to the file - never interim results - -overwriting any prior file content, regardless of the console verbosity flags in effect. +a hypothesis that later shrinks still pads over every stale trailing character from the longer +prior write before repositioning the cursor; `--output-text ` writes only final results, +one per line, to the file - never interim results - overwriting any prior file content, regardless +of the console verbosity flags in effect. **Requirement coverage**: `SpeechCli-RecognitionCommands-VerbosityAndOutput`. @@ -200,16 +200,15 @@ throws `InvalidOperationException` suggesting `--input` as an alternative. **Tests**: `SpeechModelCatalogAdapter_GetAudioFormat_SynthesisRoleModel_ThrowsArgumentException`, `SpeechModelCatalogAdapter_GetAudioFormat_NullDescriptor_ThrowsArgumentNullException`, -`SpeechModelCatalogAdapter_CreateRecognizer_SynthesisRoleModel_ThrowsArgumentException`, -`SpeechModelCatalogAdapter_CreateRecognizer_NullDescriptor_ThrowsArgumentNullException`, -`SpeechModelCatalogAdapter_CreateRecognizer_NullCaptureDevice_ThrowsArgumentNullException` +`SpeechModelCatalogAdapter_CreateRecognizerEngineAsync_SynthesisRoleModel_ThrowsArgumentException`, +`SpeechModelCatalogAdapter_CreateRecognizerEngineAsync_NullDescriptor_ThrowsArgumentNullException` **Scenario/Expected**: Against a real, temp-directory-rooted `SpeechModelCatalogAdapter`, calling either new seam member with a real, compiled-in synthesis-role model's descriptor throws `ArgumentException` naming the `descriptor` parameter, proving the adapter's `is IRecognitionModel` cast genuinely rejects a non-recognition model rather than only being exercised -through a fake; a `null` descriptor or capture device is rejected with `ArgumentNullException` -before the cast is even attempted. +through a fake; a `null` descriptor is rejected with `ArgumentNullException` before the cast is +even attempted. **Requirement coverage**: `SpeechCli-RecognitionCommands-CatalogSeamExtension`. @@ -231,7 +230,7 @@ missing dependency. - **`SpeechCli-RecognitionCommands-Recognize`**: see _RecognizeCommand — Input Source, Model Resolution, and Argument Parsing_ above - **`SpeechCli-RecognitionCommands-ParamValidation`**: see _`--stt-param` Validation Reuse_ above -- **`SpeechCli-RecognitionCommands-FileInputEofDrivenStop`**: see _File-Input EOF-Driven Stop +- **`SpeechCli-RecognitionCommands-FileInputEofDrivenStop`**: see _File-Input Explicit-Drain Flow_ above - **`SpeechCli-RecognitionCommands-SilenceTimeout`**: see _Silence Timeout (`SilenceTimeoutRecognizerSession`)_ above @@ -249,8 +248,8 @@ missing dependency. A RecognitionCommandSubsystem test run passes when: `recognize` correctly enforces input-source mutual exclusion and reports actionable model-resolution errors; `--stt-param` values are forwarded through the reused `ParameterBagParser`; file-input mode drives a real capture device to -completion and stops the recognizer reentrantly via `EndOfFileReached` with no added wait; -silence-timeout logic resets and fires deterministically against a fake time provider; +completion via `StartAsync` followed by an explicit drain `StopAsync`, with no added wait; +silence-timeout logic races and times out deterministically against a fake time provider; `--start-timeout` correctly arms the two-phase idle window (start-timeout before the first result, silence-timeout thereafter) deterministically against a fake time provider; `--interim`/`--final-only`/`--output-text` each filter/write exactly as specified; an unknown/ diff --git a/docs/verification/speech-cli/synthesis-command-subsystem.md b/docs/verification/speech-cli/synthesis-command-subsystem.md index 9ef3849..eae3f04 100644 --- a/docs/verification/speech-cli/synthesis-command-subsystem.md +++ b/docs/verification/speech-cli/synthesis-command-subsystem.md @@ -3,30 +3,31 @@ ### Verification Approach The SynthesisCommandSubsystem is verified through deterministic unit tests against a -hand-written `FakeSpeechSynthesizer` (recording `SpeakAsync` calls and `Stop`/`Dispose` call -counts, with settable `IsAvailable`/`SpeakAsyncException`) and `FakeCliModelCatalog`'s two new -delegate overrides (`GetPreferredAudioFormatOverride`/`CreateSynthesizerOverride`), reused from -`ModelCommandsSubsystem`'s own tests. Real-device dispatch is verified against a `FakePlaybackDeviceSource` -(a hand-written `ICliPlaybackDeviceSource` fake resolving purely from an in-memory device list, -touching no real PortAudio state at all) and a `FakeAudioPlaybackDevice`, so every `speak` -device-dispatch scenario is deterministic on every machine, headless or not - unlike constructing -a real `AudioDeviceFactory` with an injected probe, which still resolves to the honestly -unavailable fallback whenever `PortAudioEnvironment.Shared.IsInitialized` is `false`, regardless -of the injected probe (the exact bug this pass fixed; see `AudioDeviceFactoryPlaybackDeviceSourceTests` -below for the seam's own forwarding proof). `ParameterBagParser` is verified entirely in isolation -against hand-built `NumericParameter`/`ChoiceParameter`/`BooleanParameter` instances, with no -model catalog or synthesizer involved at all. The two new `ICliModelCatalog` seam members -(`GetPreferredAudioFormat`/`CreateSynthesizer`) are additionally verified against a **real** -`SpeechModelCatalogAdapter`, proving the production `is ISynthesisModel` cast genuinely throws for -a real, compiled-in recognition-role model, in -`SpeechModelCatalogAdapterTests.cs`. `AudioDeviceFactoryPlaybackDeviceSource` - the production -`ICliPlaybackDeviceSource` implementation - is separately verified against a **real**, composed -`AudioDeviceFactory` in `AudioDeviceFactoryPlaybackDeviceSourceTests.cs`, proving it is a pure -pass-through and does not alter `AudioDeviceFactory`'s own hardware-detection behavior in any way, -without asserting on whether real playback hardware happens to be present on the machine running -the test. Out-of-process integration tests in `IntegrationTests.cs` invoke the built tool as a -child process for the model-resolution and text-source error paths that matter most from an -operator's perspective. +hand-written `FakeSynthesisSession`/`FakeSpeechSynthesizerEngine` pair (recording `SpeakAsync` +calls and `DisposeAsync` call counts, with settable `IsAvailable`/`SpeakAsyncException`) and +`FakeCliModelCatalog`'s two new delegate overrides (`GetPreferredAudioFormatOverride`/ +`CreateSynthesizerEngineOverride`), reused from `ModelCommandsSubsystem`'s own tests. Real-device +dispatch is verified against a `FakePlaybackDeviceSource` (a hand-written `ICliPlaybackDeviceSource` +fake resolving purely from an in-memory device list, touching no real PortAudio state at all) and +a `FakeAudioPlaybackDevice`, so every `speak` device-dispatch scenario is deterministic on every +machine, headless or not - unlike constructing a real `AudioDeviceFactory` with an injected probe, +which still resolves to the honestly unavailable fallback whenever +`PortAudioEnvironment.Shared.IsInitialized` is `false`, regardless of the injected probe (the +exact bug this pass fixed; see `AudioDeviceFactoryPlaybackDeviceSourceTests` below for the seam's +own forwarding proof). `ParameterBagParser` is verified entirely in isolation against hand-built +`NumericParameter`/`ChoiceParameter`/`BooleanParameter` instances, with no model catalog or +synthesizer involved at all. The two new `ICliModelCatalog` seam members +(`GetPreferredAudioFormat`/`CreateSynthesizerEngineAsync`) are additionally verified against a +**real** `SpeechModelCatalogAdapter`, proving the production `is ISynthesisModel` cast genuinely +throws for a real, compiled-in recognition-role model, in `SpeechModelCatalogAdapterTests.cs`. +`AudioDeviceFactoryPlaybackDeviceSource` - the production `ICliPlaybackDeviceSource` +implementation - is separately verified against a **real**, composed `AudioDeviceFactory` in +`AudioDeviceFactoryPlaybackDeviceSourceTests.cs`, proving it is a pure pass-through and does not +alter `AudioDeviceFactory`'s own hardware-detection behavior in any way, without asserting on +whether real playback hardware happens to be present on the machine running the test. +Out-of-process integration tests in `IntegrationTests.cs` invoke the built tool as a child process +for the model-resolution and text-source error paths that matter most from an operator's +perspective. Automated tests do **not** exercise a real, downloaded synthesis model's actual `--output-audio` WAV-writing path end to end: CI has no cached TTS model available (downloading one requires @@ -41,15 +42,15 @@ documents for the same convention applied elsewhere). - **Framework**: xUnit v3 running under the .NET SDK - **Execution**: `dotnet test` invoked by `build.ps1` and the CI pipeline -- **Isolation**: Every unit test constructs its own fake catalog/synthesizer/audio probes; +- **Isolation**: Every unit test constructs its own fake catalog/engine/session/audio probes; `SpeechModelCatalogAdapterTests` uses a freshly created, uniquely named temporary directory as its model-store root, deleted afterward via `IDisposable` -- **Test doubles**: `FakeSpeechSynthesizer` (hand-written `ISpeechSynthesizer` fake), - `FakeCliModelCatalog` (extended with `GetPreferredAudioFormatOverride`/ - `CreateSynthesizerOverride`), `FakePlaybackDeviceSource` (hand-written `ICliPlaybackDeviceSource` - fake resolving purely from an in-memory device list), `FakeAudioPlaybackDevice` (hand-written - `IAudioPlaybackDevice` fake), `FakeAudioPlaybackDeviceProbe` (reused from - `DeviceCommandsSubsystem`'s own tests) +- **Test doubles**: `FakeSynthesisSession`/`FakeSpeechSynthesizerEngine` (hand-written + `ISynthesisSession`/`ISpeechSynthesizerEngine` fakes), `FakeCliModelCatalog` (extended with + `GetPreferredAudioFormatOverride`/`CreateSynthesizerEngineOverride`), `FakePlaybackDeviceSource` + (hand-written `ICliPlaybackDeviceSource` fake resolving purely from an in-memory device list), + `FakeAudioPlaybackDevice` (hand-written `IAudioPlaybackDevice` fake), + `FakeAudioPlaybackDeviceProbe` (reused from `DeviceCommandsSubsystem`'s own tests) ### Test Scenarios @@ -64,7 +65,7 @@ documents for the same convention applied elsewhere). `SpeakCommand_RunAsync_UnknownModelId_ThrowsArgumentException`, `SpeakCommand_RunAsync_WrongRoleModel_ThrowsArgumentException`, `SpeakCommand_RunAsync_NotDownloadedModel_ThrowsArgumentExceptionWithDownloadHint`, -`SpeakCommand_RunAsync_Success_DisposesSynthesizerOnce`, +`SpeakCommand_RunAsync_Success_DisposesEngineAndSessionOnce`, `SpeechCli_SpeakCommandWithoutModel_Invoked_ReturnsCleanError`, `SpeechCli_SpeakCommandWithUnknownModel_Invoked_ReturnsCleanError`, `SpeechCli_SpeakCommandWithWrongRoleModel_Invoked_ReturnsCleanError`, @@ -78,9 +79,9 @@ rejected with `ArgumentException`; `--text` parses its value; supplying both `-- even looked up; an unknown model id, a wrong-role model id, and a not-yet-downloaded model id are each rejected with a distinct, actionable `ArgumentException` message (suggesting `list-models`, `list-models --role tts`, and `download ` respectively); a successful run disposes the -fake synthesizer exactly once; every error path is proven both in-process and, cleanly and -non-zero-exit, against the built tool; `speak` no longer throws `NotImplementedException` when -dispatched. +fake engine and session exactly once each, in declaration order (session, then engine, then the +playback-device lease); every error path is proven both in-process and, cleanly and non-zero-exit, +against the built tool; `speak` no longer throws `NotImplementedException` when dispatched. **Requirement coverage**: `SpeechCli-SynthesisCommands-Speak`. @@ -102,7 +103,7 @@ dispatched. `ParameterBagParser_Resolve_UnrecognizedKey_ThrowsArgumentException`, `ParameterBagParser_Resolve_NoRawValues_ReturnsEmptyBag`, `SpeakCommand_ParseArguments_RepeatedParamFlags_AccumulatesInOrder`, -`SpeakCommand_RunAsync_ValidParam_ForwardsToCreateSynthesizer`, +`SpeakCommand_RunAsync_ValidParam_ForwardsToCreateSynthesizerEngine`, `SpeakCommand_RunAsync_InvalidParam_ThrowsArgumentException` **Scenario/Expected**: A well-formed `key=value` token splits correctly; a token with no `=` or @@ -113,8 +114,8 @@ value succeeds; a `ChoiceParameter` value matching a declared option resolves to a non-matching value throws; a `BooleanParameter` accepts `true`/`false` (case-insensitively) as a boxed `bool` and rejects any other value; a key not declared by the resolved model throws `ArgumentException`; repeated `--tts-param` flags accumulate in the order given and are forwarded, -fully resolved, to `CreateSynthesizer`'s `parameterValues` argument; an invalid value for any -declared parameter kind is rejected before synthesis is attempted. +fully resolved, to `CreateSynthesizerEngineAsync`'s `parameterValues` argument; an invalid value +for any declared parameter kind is rejected before an engine is loaded. **Requirement coverage**: `SpeechCli-SynthesisCommands-ParamValidation`. @@ -170,12 +171,13 @@ test; a `null` factory is rejected with `ArgumentNullException`. #### Cancellation and Disposal Ordering **Tests**: `SpeakCommand_RunAsync_Canceled_ReportsErrorCleanly`, -`SpeakCommand_RunAsync_Success_DisposesSynthesizerOnce` +`SpeakCommand_RunAsync_Success_DisposesEngineAndSessionOnce` -**Scenario/Expected**: A `FakeSpeechSynthesizer` configured to throw `OperationCanceledException` +**Scenario/Expected**: A `FakeSynthesisSession` configured to throw `OperationCanceledException` from `SpeakAsync` results in a clean, one-line cancellation message rather than an unhandled -exception propagating out of `RunAsync`; a successful run disposes the synthesizer exactly once, -proving the synthesizer-before-device disposal ordering never double-disposes or skips disposal. +exception propagating out of `RunAsync`; a successful run disposes the session, then the engine, +then the playback-device lease, proving the session-before-engine-before-device disposal ordering +never double-disposes or skips disposal. **Requirement coverage**: `SpeechCli-SynthesisCommands-CancellationAndDisposal`. @@ -183,17 +185,16 @@ proving the synthesizer-before-device disposal ordering never double-disposes or **Tests**: `SpeechModelCatalogAdapter_GetPreferredAudioFormat_RecognitionRoleModel_ThrowsArgumentException`, `SpeechModelCatalogAdapter_GetPreferredAudioFormat_NullDescriptor_ThrowsArgumentNullException`, -`SpeechModelCatalogAdapter_CreateSynthesizer_RecognitionRoleModel_ThrowsArgumentException`, -`SpeechModelCatalogAdapter_CreateSynthesizer_NullDescriptor_ThrowsArgumentNullException`, -`SpeechModelCatalogAdapter_CreateSynthesizer_NullPlaybackDevice_ThrowsArgumentNullException` +`SpeechModelCatalogAdapter_CreateSynthesizerEngineAsync_RecognitionRoleModel_ThrowsArgumentException`, +`SpeechModelCatalogAdapter_CreateSynthesizerEngineAsync_NullDescriptor_ThrowsArgumentNullException` **Scenario/Expected**: Against a real, temp-directory-rooted `SpeechModelCatalogAdapter`, calling either new seam member with a real, compiled-in recognition-role model's descriptor throws `ArgumentException` naming the `descriptor` parameter, proving the adapter's `is ISynthesisModel` cast genuinely rejects a non-synthesis model rather than only being exercised through a fake; a -`null` descriptor or playback device is rejected with `ArgumentNullException` before the cast is -even attempted. No automated test exercises a real, downloaded synthesis-role model against these -two members - see _Manual / Build-Time Verification_ below. +`null` descriptor is rejected with `ArgumentNullException` before the cast is even attempted. No +automated test exercises a real, downloaded synthesis-role model against these two members - see +_Manual / Build-Time Verification_ below. **Requirement coverage**: `SpeechCli-SynthesisCommands-CatalogSeamExtension`. @@ -234,9 +235,9 @@ exclusion and reports actionable model-resolution errors; `--tts-param` values a each declared parameter kind's own constraints, rejecting an unrecognized key; `--no-tags` strips recognized tags exactly, leaving other text byte-for-byte unchanged; `--output-audio` and real-device dispatch each construct the correct device kind and reject an unknown/unavailable device; a -cancellation is reported cleanly and the synthesizer is always disposed before the playback -device; the two new `ICliModelCatalog` seam members correctly reject a non-synthesis model against -a real catalog; and every entry point rejects a missing required dependency. +cancellation is reported cleanly and the session, engine, and playback device are always disposed +in that order; the two new `ICliModelCatalog` seam members correctly reject a non-synthesis model +against a real catalog; and every entry point rejects a missing required dependency. ## Manual / Build-Time Verification diff --git a/docs/verification/speech-demo/recognition-panel-subsystem.md b/docs/verification/speech-demo/recognition-panel-subsystem.md index ab9708e..4ec952d 100644 --- a/docs/verification/speech-demo/recognition-panel-subsystem.md +++ b/docs/verification/speech-demo/recognition-panel-subsystem.md @@ -3,20 +3,21 @@ ### Verification Approach The RecognitionPanelSubsystem is verified through deterministic unit tests in two layers. -`RecognitionPanelViewModel` is tested against NSubstitute fakes of `IModelCatalogService`, -`IAudioDeviceService`, and `IRecognizerSessionFactory`, which lets every Start/Stop lifecycle -transition, the partial-then-final transcript sequencing, and every unavailable-state path (no -model, no device, unavailable recognizer, `Start()` throwing) be produced on demand with no -downloaded model, no native runtime, and no real microphone. `RecognizerSessionFactory` is -tested directly for argument validation and the honest "wrong role" outcome. - -The "correct role composes a working recognizer" path inside `RecognizerSessionFactory` -delegates to the library's own `SpeechRecognizerFactory`, which requires an `IRecognitionModel` - -- an interface only the library's own assemblies can implement (see the design document's -remarks). That composition path is therefore outside this test project's reach and remains -covered by the library's own recognition-subsystem tests; the system-level integration test -additionally proves the panel composes over the real seam without a downloaded model. +`RecognitionPanelViewModel` is tested against NSubstitute/fake doubles of `IModelCatalogService`, +`IAudioDeviceService`, `IRecognizerSessionFactory`, `ISpeechRecognizerEngine`, and +`IRecognitionSession`, which lets every async Start/Stop lifecycle transition, state derivation +from `IRecognitionSession.StateChanged`, the partial-then-final transcript sequencing, the +two-tier engine/session cache invalidation, and every unavailable-state path (no model, no +device, unavailable engine, a busy/faulted session) be produced on demand with no downloaded +model, no native runtime, and no real microphone. `RecognizerSessionFactory` is tested directly +for argument validation and the honest "wrong role" outcome. + +The "correct role composes a working engine" path inside `RecognizerSessionFactory` delegates to +the library's own `SpeechRecognizerFactory`, which requires an `IRecognitionModel` - an interface +only the library's own assemblies can implement (see the design document's remarks). That +composition path is therefore outside this test project's reach and remains covered by the +library's own recognition-subsystem tests; the system-level integration test additionally proves +the panel composes over the real seam without a downloaded model. ### Test Environment @@ -24,8 +25,10 @@ additionally proves the panel composes over the real seam without a downloaded m - **Execution**: `dotnet test` invoked by `build.ps1` and the CI pipeline - **Isolation**: `RecognizerSessionFactoryTests` roots the library's model store in a fresh directory under the test output folder -- **Test doubles**: NSubstitute fakes of `IModelCatalogService`, `IAudioDeviceService`, - `IRecognizerSessionFactory`, `ISpeechRecognizer`, and `IAudioCaptureDevice`; `FakeSpeechModel` +- **Test doubles**: NSubstitute fakes of `IModelCatalogService`, `IAudioDeviceService`, and + `IRecognizerSessionFactory`; hand-written `FakeSpeechRecognizerEngine` and + `FakeRecognitionSession` (from `Fakes/`) standing in for `ISpeechRecognizerEngine` and + `IRecognitionSession`; `FakeSpeechModel` ### Test Scenarios @@ -82,35 +85,36 @@ installed. #### RecognitionPanelViewModel_Start_RecognizerUnavailable_ReportsErrorStateAndDisposes -**Scenario**: The session seam composes a recognizer that honestly reports itself unavailable. +**Scenario**: The session seam loads an engine that honestly reports itself unavailable. -**Expected**: `RecognizerUnavailableMessage` and `Error` state, and the unavailable recognizer is +**Expected**: `RecognizerUnavailableMessage` and `Error` state, and the unavailable engine is disposed. **Requirement coverage**: `SpeechDemo-Recognition-HonestUnavailableStates`. #### RecognitionPanelViewModel_Start_RecognizerStartThrows_ReportsErrorStateAndDisposes -**Scenario**: A composed recognizer's `Start()` throws. +**Scenario**: A created session's `StartAsync()` throws `SpeechRecognizerUnavailableException`. **Expected**: The exception is caught, `Error` state is reported with the exception's message, -and the recognizer is disposed rather than leaked. +and the session is disposed rather than leaked. **Requirement coverage**: `SpeechDemo-Recognition-HonestUnavailableStates`, `SpeechDemo-Recognition-ResourceLifetime`. #### RecognitionPanelViewModel_Start_SuccessfulSession_EntersListeningState -**Scenario**: A successful Start with a working recognizer. +**Scenario**: A successful Start with a working engine and session. -**Expected**: `State` becomes `Listening` and the recognizer's `Start()` is invoked exactly once. +**Expected**: `State` becomes `Listening` and the session's `StartAsync()` is invoked exactly +once. **Requirement coverage**: `SpeechDemo-Recognition-StartStopLifecycle`. #### RecognitionPanelViewModel_ResultReceived_PartialThenFinal_UpdatesTranscriptInOrder -**Scenario**: The recognizer raises a partial result followed by a final result for the same -utterance. +**Scenario**: The session's result stream yields a partial result followed by a final result for +the same utterance. **Expected**: The partial is shown as the trailing line while provisional, then replaced by the committed final line once the final result arrives; the final is preserved and a fresh partial @@ -128,12 +132,25 @@ line. **Requirement coverage**: `SpeechDemo-Recognition-TranscriptSequencing`. +#### RecognitionPanelViewModel_StateChanged_SessionTransitionsToFaulted_ReportsErrorState + +**Scenario**: A listening session itself reports an unrecoverable `Faulted` state (for example, +the bound capture device being lost mid-session) via `StateChanged`, not caused by this panel +calling Stop or Start. + +**Expected**: `State` becomes `Error` and `StatusMessage` becomes `SessionFaultedMessage`, +driven entirely by the `StateChanged` mapping rather than an ad hoc assignment. + +**Requirement coverage**: `SpeechDemo-Recognition-HonestUnavailableStates`. + #### RecognitionPanelViewModel_Stop_DuringListening_StopsWithoutDisposingSession **Scenario**: Stop is invoked while listening. -**Expected**: The recognizer's `Stop()` is invoked, but the recognizer is **not** disposed and -remains cached so a subsequent Start can reuse it; `State` returns to `Idle`. +**Expected**: The session's `StopAsync()` is invoked and the session is released (disposed as +single-use), but the cached engine itself is **not** disposed and remains cached so a subsequent +Start reuses it instead of reloading its model; `State` returns to `Idle` and `StatusMessage` +becomes `StoppedMessage`. **Requirement coverage**: `SpeechDemo-Recognition-StartStopLifecycle`. @@ -145,73 +162,85 @@ remains cached so a subsequent Start can reuse it; `State` returns to `Idle`. **Requirement coverage**: `SpeechDemo-Recognition-StartStopLifecycle`. -#### RecognitionPanelViewModel_StartStopStart_SameSelection_ReusesRecognizer +#### RecognitionPanelViewModel_Stop_CalledTwiceConcurrently_BothCompleteWithoutThrowing + +**Scenario**: `StopCommand` (`AllowConcurrentExecutions`) is invoked twice without awaiting the +first before starting the second. + +**Expected**: Both complete without throwing, rather than racing to double-dispose the same +session, and the panel settles at `Idle`. + +**Requirement coverage**: `SpeechDemo-Recognition-StartStopLifecycle`. + +#### RecognitionPanelViewModel_StartStopStart_SameSelection_ReusesEngine **Scenario**: Start, Stop, then Start again, with the same model and capture device selected throughout. -**Expected**: The session seam composes a recognizer exactly once; the second Start reuses the -cached instance rather than recomposing (and reloading the model) a second time. +**Expected**: The session seam loads an engine exactly once (`LoadAsync` is invoked once); a +fresh single-use session is created from the same cached engine for each of the two runs. **Requirement coverage**: `SpeechDemo-Recognition-StartStopLifecycle`. -#### RecognitionPanelViewModel_DeviceRefresh_InvalidatesCachedRecognizer +#### RecognitionPanelViewModel_DeviceRefresh_InvalidatesCachedSessionButNotEngine -**Scenario**: A recognizer is cached (idle) for the current capture device, then the shared +**Scenario**: A session is cached (idle) from a prior Start/Stop cycle, then the shared `DeviceSelectionViewModel.Refresh()` runs. -**Expected**: The cached recognizer is disposed, so a subsequent Start composes a fresh one -bound to the post-refresh device table rather than reusing one for a now-stale device. +**Expected**: The cached session is disposed, but the engine is never reloaded; a subsequent +Start creates a fresh session bound to the post-refresh device table from the same cached engine. **Requirement coverage**: `SpeechDemo-Recognition-StartStopLifecycle`. -#### RecognitionPanelViewModel_SelectedModelChanged_InvalidatesCachedRecognizer +#### RecognitionPanelViewModel_SelectedModelChanged_InvalidatesCachedEngine -**Scenario**: The selected recognition model is changed while idle, with a recognizer already -cached for the previously selected model. +**Scenario**: The selected recognition model is changed while idle, with an engine already +cached (idle) for the previously selected model. -**Expected**: The recognizer cached for the previous model is disposed, so a subsequent Start -composes one for the newly selected model instead of reusing a stale instance. +**Expected**: The engine cached for the previous model is disposed, so a subsequent Start loads +one for the newly selected model instead of reusing a stale instance. **Requirement coverage**: `SpeechDemo-Recognition-StartStopLifecycle`. -#### RecognitionPanelViewModel_SelectedModelChanged_WhileListening_StopsAndInvalidatesRecognizer +#### RecognitionPanelViewModel_SelectedModelChanged_WhileListening_StopsAndInvalidatesEngine **Scenario**: The selected recognition model is changed while a session is actively listening - bypassing the view's disabled model picker, since `SelectedModel`'s setter remains public and is not guarded at the model level. -**Expected**: The active recognizer is stopped, then disposed; `State` returns to `Idle` rather -than remaining stuck at `Listening` with no cached recognizer for a later Stop to find. +**Expected**: The active session is stopped, then the engine is disposed; `State` returns to +`Idle` rather than remaining stuck at `Listening` with no cached session for a later Stop to +find. **Requirement coverage**: `SpeechDemo-Recognition-StartStopLifecycle`. -#### RecognitionPanelViewModel_SelectedCaptureDeviceChanged_InvalidatesCachedRecognizer +#### RecognitionPanelViewModel_SelectedCaptureDeviceChanged_InvalidatesSessionButNotEngine **Scenario**: The shared `DeviceSelectionViewModel`'s selected capture device is changed while -idle, with a recognizer already cached for the previously selected device. +idle, with a session already cached (idle) bound to the previously selected device. -**Expected**: The recognizer cached for the previous device is disposed, so a subsequent Start -composes one bound to the newly selected device instead of reusing a stale instance. +**Expected**: The session cached for the previous device is disposed, but the engine is never +reloaded, so a subsequent Start builds a session bound to the newly selected device while reusing +the already-loaded engine. **Requirement coverage**: `SpeechDemo-Recognition-StartStopLifecycle`. -#### RecognitionPanelViewModel_SelectedCaptureDeviceChanged_WhileListening_StopsAndInvalidatesRecognizer +#### RecognitionPanelViewModel_SelectedCaptureDeviceChanged_WhileListening_StopsAndInvalidatesSession **Scenario**: The shared `DeviceSelectionViewModel`'s selected capture device is changed while a session is actively listening - the capture-device picker, unlike the model picker, is never disabled while listening, so this is reachable directly from the view. -**Expected**: The active recognizer is stopped, then disposed; `State` returns to `Idle` rather -than remaining stuck at `Listening` bound to an abandoned device. +**Expected**: The active session is stopped and disposed, but the engine persists; `State` +returns to `Idle` rather than remaining stuck at `Listening` bound to an abandoned device. **Requirement coverage**: `SpeechDemo-Recognition-StartStopLifecycle`. #### RecognitionPanelViewModel_Dispose_ReleasesActiveSessionWithoutThrowing -**Scenario**: The panel is disposed while a recognizer is active. +**Scenario**: The panel is disposed (`DisposeAsync`) while a session is active. -**Expected**: The recognizer is disposed exactly once and disposal does not throw. +**Expected**: The session and the cached engine are disposed without throwing. **Requirement coverage**: `SpeechDemo-Recognition-ResourceLifetime`. @@ -223,29 +252,21 @@ than remaining stuck at `Listening` bound to an abandoned device. **Requirement coverage**: `SpeechDemo-Recognition-SessionSeam`. -#### RecognizerSessionFactory_Create_NullModel_ThrowsArgumentNullException +#### RecognizerSessionFactory_LoadAsync_NullModel_ThrowsArgumentNullException -**Scenario**: `Create` is called with a missing model. +**Scenario**: `LoadAsync` is called with a missing model. **Expected**: `ArgumentNullException`. **Requirement coverage**: `SpeechDemo-Recognition-SessionSeam`. -#### RecognizerSessionFactory_Create_NullDevice_ThrowsArgumentNullException +#### RecognizerSessionFactory_LoadAsync_ModelNotRecognitionRole_ReturnsUnavailableEngine -**Scenario**: `Create` is called with a missing capture device. +**Scenario**: `LoadAsync` is called with a model that does not implement the library's +recognition role. -**Expected**: `ArgumentNullException`. - -**Requirement coverage**: `SpeechDemo-Recognition-SessionSeam`. - -#### RecognizerSessionFactory_Create_ModelNotRecognitionRole_ReturnsUnavailableRecognizer - -**Scenario**: `Create` is called with a model that does not implement the library's recognition -role. - -**Expected**: The library's own `UnavailableSpeechRecognizer.Instance`, exactly like a model that -is not installed, rather than an exception. +**Expected**: The library's own `UnavailableSpeechRecognizerEngine.Instance`, exactly like a +model that is not installed, rather than an exception. **Requirement coverage**: `SpeechDemo-Recognition-SessionSeam`. @@ -300,15 +321,25 @@ to pick a different model after a failed attempt. **Scenario**: The panel is actively listening, sharing a `DeviceSelectionViewModel` whose device service refuses `RefreshDevices()` with `AudioDeviceInUseException` only while a "still -listening" flag is true; the recognizer's `Stop()` flips that flag false. The shared +listening" flag is true; the session's `StopAsync()` flips that flag false. The shared `DeviceSelectionViewModel.Refresh()` is then invoked, as the "Refresh devices" button would. -**Expected**: The panel's registered pre-refresh hook calls `Stop()` on the recognizer before the -device refresh is attempted, so the refresh completes without throwing; the panel returns to +**Expected**: The panel's registered pre-refresh hook calls `StopAsync()` on the session before +the device refresh is attempted, so the refresh completes without throwing; the panel returns to `Idle` with `CanStop` false. **Requirement coverage**: `SpeechDemo-Recognition-StopsBeforeDeviceRefresh`. +#### RecognitionPanelViewModel_PreRefreshHook_WhileIdle_IsNoOpAndDeviceRefreshSucceeds + +**Scenario**: The registered pre-refresh hook runs while no listening session was ever started +(no engine, no session created). + +**Expected**: The refresh completes without throwing, no session is created, and the panel +remains `Idle`. + +**Requirement coverage**: `SpeechDemo-Recognition-StopsBeforeDeviceRefresh`. + ### Requirements Coverage - **`SpeechDemo-Recognition-ModelSelection`**: @@ -318,12 +349,13 @@ device refresh is attempted, so the refresh completes without throwing; the pane `RecognitionPanelViewModel_Start_SuccessfulSession_EntersListeningState`, `RecognitionPanelViewModel_Stop_DuringListening_StopsWithoutDisposingSession`, `RecognitionPanelViewModel_Stop_NothingListening_IsSafeNoOp`, - `RecognitionPanelViewModel_StartStopStart_SameSelection_ReusesRecognizer`, - `RecognitionPanelViewModel_DeviceRefresh_InvalidatesCachedRecognizer`, - `RecognitionPanelViewModel_SelectedModelChanged_InvalidatesCachedRecognizer`, - `RecognitionPanelViewModel_SelectedModelChanged_WhileListening_StopsAndInvalidatesRecognizer`, - `RecognitionPanelViewModel_SelectedCaptureDeviceChanged_InvalidatesCachedRecognizer`, - `RecognitionPanelViewModel_SelectedCaptureDeviceChanged_WhileListening_StopsAndInvalidatesRecognizer` + `RecognitionPanelViewModel_Stop_CalledTwiceConcurrently_BothCompleteWithoutThrowing`, + `RecognitionPanelViewModel_StartStopStart_SameSelection_ReusesEngine`, + `RecognitionPanelViewModel_DeviceRefresh_InvalidatesCachedSessionButNotEngine`, + `RecognitionPanelViewModel_SelectedModelChanged_InvalidatesCachedEngine`, + `RecognitionPanelViewModel_SelectedModelChanged_WhileListening_StopsAndInvalidatesEngine`, + `RecognitionPanelViewModel_SelectedCaptureDeviceChanged_InvalidatesSessionButNotEngine`, + `RecognitionPanelViewModel_SelectedCaptureDeviceChanged_WhileListening_StopsAndInvalidatesSession` - **`SpeechDemo-Recognition-TranscriptSequencing`**: `RecognitionPanelViewModel_ResultReceived_PartialThenFinal_UpdatesTranscriptInOrder`, `RecognitionPanelViewModel_BuildTranscriptText_FinalsAndPartial_RendersInOrder` @@ -331,7 +363,8 @@ device refresh is attempted, so the refresh completes without throwing; the pane `RecognitionPanelViewModel_Start_NoModelSelected_ReportsErrorState`, `RecognitionPanelViewModel_Start_NoCaptureDevice_ReportsErrorState`, `RecognitionPanelViewModel_Start_RecognizerUnavailable_ReportsErrorStateAndDisposes`, - `RecognitionPanelViewModel_Start_RecognizerStartThrows_ReportsErrorStateAndDisposes` + `RecognitionPanelViewModel_Start_RecognizerStartThrows_ReportsErrorStateAndDisposes`, + `RecognitionPanelViewModel_StateChanged_SessionTransitionsToFaulted_ReportsErrorState` - **`SpeechDemo-Recognition-HonestEmptyCatalogState`**: `RecognitionPanelViewModel_Constructor_NoInstalledRecognitionModel_ReportsHonestEmptyState` - **`SpeechDemo-Recognition-ResourceLifetime`**: @@ -340,9 +373,8 @@ device refresh is attempted, so the refresh completes without throwing; the pane - **`SpeechDemo-Recognition-SessionSeam`**: `RecognitionPanelViewModel_Constructor_NullDependency_ThrowsArgumentNullException`, `RecognizerSessionFactory_Constructor_NullStore_ThrowsArgumentNullException`, - `RecognizerSessionFactory_Create_NullModel_ThrowsArgumentNullException`, - `RecognizerSessionFactory_Create_NullDevice_ThrowsArgumentNullException`, - `RecognizerSessionFactory_Create_ModelNotRecognitionRole_ReturnsUnavailableRecognizer` + `RecognizerSessionFactory_LoadAsync_NullModel_ThrowsArgumentNullException`, + `RecognizerSessionFactory_LoadAsync_ModelNotRecognitionRole_ReturnsUnavailableEngine` - **`SpeechDemo-Recognition-AutoRefreshOnInstall`**: `RecognitionPanelViewModel_ModelInstalled_MatchingRole_TriggersRefresh`, `RecognitionPanelViewModel_ModelInstalled_NonMatchingRole_DoesNotTriggerRefresh`, @@ -351,24 +383,29 @@ device refresh is attempted, so the refresh completes without throwing; the pane `RecognitionPanelViewModel_CanChangeModel_TogglesAcrossStateTransitions`, `RecognitionPanelViewModel_CanChangeModel_ErrorState_IsTrue` - **`SpeechDemo-Recognition-StopsBeforeDeviceRefresh`**: - `RecognitionPanelViewModel_PreRefreshHook_WhileListening_StopsSessionBeforeDeviceRefreshSucceeds` + `RecognitionPanelViewModel_PreRefreshHook_WhileListening_StopsSessionBeforeDeviceRefreshSucceeds`, + `RecognitionPanelViewModel_PreRefreshHook_WhileIdle_IsNoOpAndDeviceRefreshSucceeds` ### Acceptance Criteria A RecognitionPanelSubsystem test run passes when: only installed recognition models are offered and the selection survives a refresh; a successful Start enters the listening state and invokes -the recognizer exactly once; partial results replace the trailing transcript line while finals -commit permanently in order; Stop stops an active recognizer without disposing it, retaining it -cached for reuse, while Dispose releases an active recognizer exactly once; both are safe no-ops -with nothing active and neither throws; every unavailable state - no -model, no device, an unavailable recognizer, a throwing `Start()` - is reported honestly with the -recognizer released rather than leaked; a cached recognizer is reused across repeated Start/Stop -cycles for the same model and device, but is invalidated - stopping an active session first if -one is in progress, never leaving the panel stuck in `Listening` - whenever the selected model, -the selected capture device, or a pending device refresh requires a different recognizer; the -session seam validates its arguments and reports a -role mismatch the same honest way the library reports an uninstalled model; a matching-role -`ModelInstalled` event triggers an automatic refresh while a non-matching-role event does not; -`Dispose()` unsubscribes from `ModelInstalled` so a later event is never applied; and the panel's -registered pre-refresh hook stops an actively listening session before a shared device refresh is -attempted, so the refresh succeeds deterministically. +the session's `StartAsync()` exactly once; partial results replace the trailing transcript line +while finals commit permanently in order; Stop stops an active session without disposing the +cached engine, releasing only the single-use session for reuse of the engine, while two +concurrent Stop calls both complete without racing to double-dispose, and Dispose releases an +active session and the cached engine exactly once; both Stop and Dispose are safe no-ops with +nothing active and neither throws; every unavailable state - no model, no device, an unavailable +engine, a session `StartAsync()` throwing, or a session transitioning to `Faulted` mid-stream via +`StateChanged` - is reported honestly with resources released rather than leaked; a cached engine +is reused across repeated Start/Stop cycles for the same model, with a fresh single-use session +created each time, but the engine is invalidated - stopping an active session first if one is in +progress, never leaving the panel stuck in `Listening` - whenever the selected model requires a +different engine, while only the session (not the engine) is invalidated whenever the selected +capture device changes or a device refresh is pending; the session seam validates its arguments +and reports a role mismatch the same honest way the library reports an uninstalled model; a +matching-role `ModelInstalled` event triggers an automatic refresh while a non-matching-role +event does not; `DisposeAsync()` unsubscribes from `ModelInstalled` so a later event is never +applied; and the panel's registered pre-refresh hook stops an actively listening session before a +shared device refresh is attempted (and is a safe no-op while idle), so the refresh succeeds +deterministically. diff --git a/docs/verification/speech-demo/synthesis-panel-subsystem.md b/docs/verification/speech-demo/synthesis-panel-subsystem.md index 0b93f5c..ae283bd 100644 --- a/docs/verification/speech-demo/synthesis-panel-subsystem.md +++ b/docs/verification/speech-demo/synthesis-panel-subsystem.md @@ -3,19 +3,21 @@ ### Verification Approach The SynthesisPanelSubsystem is verified through deterministic unit tests in two layers. -`SynthesisPanelViewModel` is tested against NSubstitute fakes of `IModelCatalogService`, -`IAudioDeviceService`, and `ISynthesizerSessionFactory`, which lets every Play/Stop lifecycle -transition and every unavailable-state path (no model, no device, unavailable synthesizer) be -produced on demand with no downloaded model, no native runtime, and no real speakers. -`SynthesizerSessionFactory` is tested directly for argument validation and the honest "wrong -role" outcome. - -The "correct role composes a working synthesizer" path inside `SynthesizerSessionFactory` -delegates to the library's own `SpeechSynthesizerFactory`, which requires an `ISynthesisModel` - -an interface only the library's own assemblies can implement (see the design document's -remarks). That composition path is therefore outside this test project's reach and remains -covered by the library's own synthesis-subsystem tests; the system-level integration test -additionally proves the panel composes over the real seam without a downloaded model. +`SynthesisPanelViewModel` is tested against NSubstitute/fake doubles of `IModelCatalogService`, +`IAudioDeviceService`, `ISynthesizerSessionFactory`, `ISpeechSynthesizerEngine`, and +`ISynthesisSession`, which lets every async Play/Stop lifecycle transition, state derivation from +`ISynthesisSession.StateChanged`, the two-tier engine/session lazy-reload cache (the central +bugfix this redesign exists for), and every unavailable-state path (no model, no device, +unavailable engine) be produced on demand with no downloaded model, no native runtime, and no +real speakers. `SynthesizerSessionFactory` is tested directly for argument validation and the +honest "wrong role" outcome. + +The "correct role composes a working engine" path inside `SynthesizerSessionFactory` delegates to +the library's own `SpeechSynthesizerFactory`, which requires an `ISynthesisModel` - an interface +only the library's own assemblies can implement (see the design document's remarks). That +composition path is therefore outside this test project's reach and remains covered by the +library's own synthesis-subsystem tests; the system-level integration test additionally proves +the panel composes over the real seam without a downloaded model. ### Test Environment @@ -23,8 +25,10 @@ additionally proves the panel composes over the real seam without a downloaded m - **Execution**: `dotnet test` invoked by `build.ps1` and the CI pipeline - **Isolation**: `SynthesizerSessionFactoryTests` roots the library's model store in a fresh directory under the test output folder -- **Test doubles**: NSubstitute fakes of `IModelCatalogService`, `IAudioDeviceService`, - `ISynthesizerSessionFactory`, `ISpeechSynthesizer`, and `IAudioPlaybackDevice`; `FakeSpeechModel` +- **Test doubles**: NSubstitute fakes of `IModelCatalogService`, `IAudioDeviceService`, and + `ISynthesizerSessionFactory`; hand-written `FakeSpeechSynthesizerEngine` and + `FakeSynthesisSession` (from `Fakes/`) standing in for `ISpeechSynthesizerEngine` and + `ISynthesisSession`; `FakeSpeechModel` ### Test Scenarios @@ -80,31 +84,118 @@ model, and an installed recognition model. #### SynthesisPanelViewModel_Play_SynthesizerUnavailable_ReportsErrorStateAndDisposes -**Scenario**: The session seam composes a synthesizer that honestly reports itself unavailable. +**Scenario**: The session seam loads an engine that honestly reports itself unavailable. -**Expected**: `SynthesizerUnavailableMessage` and `Error` state, and the unavailable synthesizer -is disposed. +**Expected**: `SynthesizerUnavailableMessage` and `Error` state, and the unavailable engine is +disposed. **Requirement coverage**: `SpeechDemo-Synthesis-HonestUnavailableStates`. +#### SynthesisPanelViewModel_Play_ModelDeclaresChoiceParameter_ForwardsValueBagToSessionFactory + +**Scenario**: The selected model declares a `voice` `ChoiceParameter` with a non-default current +selection, and Play is invoked. + +**Expected**: `ISynthesizerSessionFactory.LoadAsync` is called with a `parameterValues` bag whose +content exactly matches `Settings.BuildValueBag()` (one entry, `"voice"` -> the selected value), +proving the embedded settings panel's selection genuinely reaches the session factory rather +than being built and discarded. + +**Requirement coverage**: `SpeechDemo-Synthesis-VoiceSelectionForwarding`. + #### SynthesisPanelViewModel_Play_SuccessfulSession_TransitionsThroughLifecycleToIdle -**Scenario**: A full Play with a working synthesizer. +**Scenario**: A full Play with a working engine and session. **Expected**: `State` transitions through `Synthesizing` then `Playing` before settling on -`Idle`, and the synthesizer is disposed afterward. +`Idle`, driven by `ISynthesisSession.StateChanged`, and the exact text is forwarded to +`SpeakAsync`. **Requirement coverage**: `SpeechDemo-Synthesis-PlayLifecycle`. +#### SynthesisPanelViewModel_Play_SuccessfulSession_CanChangeModelTogglesAcrossLifecycle + +**Scenario**: A successful `PlayAsync` transitions `State` through `Synthesizing`, `Playing`, and +back to `Idle`. + +**Expected**: `CanChangeModel` is `true` while `Idle`, `false` while `Synthesizing`/`Playing`, and +`true` again once settled back at `Idle`. + +**Requirement coverage**: `SpeechDemo-Synthesis-ModelSwitchGuard`. + +#### SynthesisPanelViewModel_CanChangeModel_ErrorState_IsTrue + +**Scenario**: `PlayAsync` fails (no playback device available) and the panel enters `Error`. + +**Expected**: `CanChangeModel` is `true`, matching `CanPlay`'s own `Idle`-or-`Error` condition, so +a user can pick a different model after a failed attempt. + +**Requirement coverage**: `SpeechDemo-Synthesis-ModelSwitchGuard`. + #### SynthesisPanelViewModel_Stop_DuringPlayback_CancelsSessionAndReportsStopped -**Scenario**: Stop is invoked while Play is in flight. +**Scenario**: Stop is invoked while Play is in flight (an indefinitely long `SpeakAsync` awaiting +cancellation). -**Expected**: The session is canceled, `State` settles on `Idle`, and `StatusMessage` reports -`StoppedMessage`. +**Expected**: The session's `StopAsync()` is invoked, cancellation unwinds the task cleanly, +`State` settles on `Idle`, and `StatusMessage` reports `StoppedMessage`. **Requirement coverage**: `SpeechDemo-Synthesis-StopControl`. +#### SynthesisPanelViewModel_PreRefreshHook_WhilePlaying_StopsAndAwaitsExecutionTaskBeforeDeviceRefreshSucceeds + +**Scenario**: A `PlayAsync` execution is in flight (an indefinitely long `SpeakAsync` awaiting +cancellation), sharing the panel's own `DeviceSelectionViewModel`. The shared +`DeviceSelectionViewModel.Refresh()` is then invoked, as the "Refresh devices" button would. + +**Expected**: The panel's registered pre-refresh hook calls `StopAsync()` on the session and +awaits the in-flight `PlayCommand.ExecutionTask` to its actual completion (not merely its +cancellation request) before the device refresh proceeds, so the refresh completes without +throwing, the session is released, and the panel settles on `Idle`. + +**Requirement coverage**: `SpeechDemo-Synthesis-StopsBeforeDeviceRefresh`. + +#### SynthesisPanelViewModel_PreRefreshHook_WhileIdle_IsNoOpAndDeviceRefreshSucceeds + +**Scenario**: The registered pre-refresh hook runs while no Play session was ever started (no +engine, no session created). + +**Expected**: The refresh completes without throwing, no session is created, and the panel +remains `Idle`. + +**Requirement coverage**: `SpeechDemo-Synthesis-StopsBeforeDeviceRefresh`. + +#### SynthesisPanelViewModel_Play_CalledTwiceWithUnchangedModelAndParameters_ReusesSameSessionWithoutReload + +**Scenario**: Play is invoked twice in a row with the selected model and settings parameter +values unchanged between calls - the central bugfix this redesign exists for. + +**Expected**: The engine is loaded exactly once (`LoadAsync` invoked once), exactly one session +is created, and both Play calls speak through that same cached session rather than reloading the +model and recreating the session on every click. + +**Requirement coverage**: `SpeechDemo-Synthesis-EngineSessionReuse`. + +#### SynthesisPanelViewModel_Play_ParameterValueChanged_ReloadsEngineAndRecreatesSession + +**Scenario**: A declared settings parameter's value is changed between two Play calls. + +**Expected**: The stale cached engine is disposed and a fresh one is loaded for the second Play +(`LoadAsync` invoked twice), and - as a direct consequence of the engine reload - a fresh session +is created from the new engine rather than reusing the previous one. + +**Requirement coverage**: `SpeechDemo-Synthesis-EngineSessionReuse`. + +#### SynthesisPanelViewModel_Play_PlaybackDeviceChanged_RecreatesSessionButNotEngine + +**Scenario**: The shared `DeviceSelectionViewModel`'s selected playback device is changed between +two Play calls, with the same model and parameter values throughout. + +**Expected**: The engine is loaded only once (`LoadAsync` invoked once), but a second, distinct +session is created bound to the newly selected device, and the stale first session is disposed. + +**Requirement coverage**: `SpeechDemo-Synthesis-EngineSessionReuse`. + #### SynthesisPanelViewModel_ExampleTagHints_ContainsExpectedTags **Scenario**: `ExampleTagHints` is read. @@ -122,54 +213,34 @@ is disposed. **Requirement coverage**: `SpeechDemo-Synthesis-SessionSeam`. -#### SynthesizerSessionFactory_Create_NullModel_ThrowsArgumentNullException +#### SynthesizerSessionFactory_LoadAsync_NullModel_ThrowsArgumentNullException -**Scenario**: `Create` is called with a missing model. +**Scenario**: `LoadAsync` is called with a missing model. **Expected**: `ArgumentNullException`. **Requirement coverage**: `SpeechDemo-Synthesis-SessionSeam`. -#### SynthesizerSessionFactory_Create_NullDevice_ThrowsArgumentNullException - -**Scenario**: `Create` is called with a missing playback device. +#### SynthesizerSessionFactory_LoadAsync_ModelNotSynthesisRole_ReturnsUnavailableEngine -**Expected**: `ArgumentNullException`. - -**Requirement coverage**: `SpeechDemo-Synthesis-SessionSeam`. - -#### SynthesizerSessionFactory_Create_ModelNotSynthesisRole_ReturnsUnavailableSynthesizer - -**Scenario**: `Create` is called with a model that does not implement the library's synthesis +**Scenario**: `LoadAsync` is called with a model that does not implement the library's synthesis role. -**Expected**: The library's own `UnavailableSpeechSynthesizer.Instance`, exactly like a model -that is not installed, rather than an exception. +**Expected**: The library's own `UnavailableSpeechSynthesizerEngine.Instance`, exactly like a +model that is not installed, rather than an exception. **Requirement coverage**: `SpeechDemo-Synthesis-SessionSeam`. -#### SynthesizerSessionFactory_Create_ModelNotSynthesisRoleWithParameterValues_ReturnsUnavailableSynthesizer +#### SynthesizerSessionFactory_LoadAsync_ModelNotSynthesisRoleWithParameterValues_ReturnsUnavailableEngine -**Scenario**: `Create` is called with a non-null `parameterValues` bag alongside a model that +**Scenario**: `LoadAsync` is called with a non-null `parameterValues` bag alongside a model that does not implement the library's synthesis role. -**Expected**: The library's own `UnavailableSpeechSynthesizer.Instance`, proving the new +**Expected**: The library's own `UnavailableSpeechSynthesizerEngine.Instance`, proving the `parameterValues` argument does not disturb the existing wrong-role fallback. **Requirement coverage**: `SpeechDemo-Synthesis-VoiceSelectionForwarding`. -#### SynthesisPanelViewModel_Play_ModelDeclaresChoiceParameter_ForwardsValueBagToSessionFactory - -**Scenario**: The selected model declares a `voice` `ChoiceParameter` with a non-default current -selection, and Play is invoked. - -**Expected**: `ISynthesizerSessionFactory.Create` is called with a `parameterValues` bag whose -content exactly matches `Settings.BuildValueBag()` (one entry, `"voice"` -> the selected value), -proving the embedded settings panel's selection genuinely reaches the session factory rather -than being built and discarded. - -**Requirement coverage**: `SpeechDemo-Synthesis-VoiceSelectionForwarding`. - #### SynthesisPanelViewModel_ModelInstalled_MatchingRole_TriggersRefresh **Scenario**: `IModelCatalogService.ModelInstalled` is raised for a synthesis-role model after @@ -188,58 +259,26 @@ construction. **Requirement coverage**: `SpeechDemo-Synthesis-AutoRefreshOnInstall`. -#### SynthesisPanelViewModel_Dispose_UnsubscribesFromModelInstalled_NoRefreshAfterDispose +#### SynthesisPanelViewModel_DisposeAsync_UnsubscribesFromModelInstalled_NoRefreshAfterDispose -**Scenario**: The panel is disposed, then `IModelCatalogService.ModelInstalled` is raised for a -matching-role model. +**Scenario**: The panel is disposed (`DisposeAsync`), then `IModelCatalogService.ModelInstalled` +is raised for a matching-role model. **Expected**: No exception, and the panel does not pick up the later install since it had already unsubscribed. **Requirement coverage**: `SpeechDemo-Synthesis-ResourceLifetime`. -#### SynthesisPanelViewModel_Dispose_NoActiveSynthesizer_IsSafeAndIdempotent +#### SynthesisPanelViewModel_DisposeAsync_NoActiveSession_IsSafeAndIdempotent -**Scenario**: The panel is disposed twice with no active synthesizer. +**Scenario**: The panel is disposed twice with no active session. **Expected**: No exception either time - this is a new capability on this class (it did not -previously implement `IDisposable`), so unlike `RecognitionPanelViewModel` it has no prior -coverage to rely on. +previously implement any disposable contract), so unlike `RecognitionPanelViewModel` it has no +prior coverage to rely on. **Requirement coverage**: `SpeechDemo-Synthesis-ResourceLifetime`. -#### SynthesisPanelViewModel_Play_SuccessfulSession_CanChangeModelTogglesAcrossLifecycle - -**Scenario**: A successful `PlayAsync` transitions `State` through `Synthesizing`, `Playing`, and -back to `Idle`. - -**Expected**: `CanChangeModel` is `true` while `Idle`, `false` while `Synthesizing`/`Playing`, and -`true` again once settled back at `Idle`. - -**Requirement coverage**: `SpeechDemo-Synthesis-ModelSwitchGuard`. - -#### SynthesisPanelViewModel_CanChangeModel_ErrorState_IsTrue - -**Scenario**: `PlayAsync` fails (no playback device available) and the panel enters `Error`. - -**Expected**: `CanChangeModel` is `true`, matching `CanPlay`'s own `Idle`-or-`Error` condition, so -a user can pick a different model after a failed attempt. - -**Requirement coverage**: `SpeechDemo-Synthesis-ModelSwitchGuard`. - -#### SynthesisPanelViewModel_PreRefreshHook_WhilePlaying_StopsAndAwaitsExecutionTaskBeforeDeviceRefreshSucceeds - -**Scenario**: A `PlayAsync` execution is in flight (an indefinitely long `SpeakAsync` awaiting -cancellation), sharing the panel's own `DeviceSelectionViewModel`. The shared -`DeviceSelectionViewModel.Refresh()` is then invoked, as the "Refresh devices" button would. - -**Expected**: The panel's registered pre-refresh hook calls `Stop()` on the synthesizer and awaits -the in-flight `PlayCommand.ExecutionTask` to its actual completion (not merely its cancellation -request) before the device refresh proceeds, so the refresh completes without throwing and the -panel settles on `Idle`. - -**Requirement coverage**: `SpeechDemo-Synthesis-StopsBeforeDeviceRefresh`. - ### Requirements Coverage - **`SpeechDemo-Synthesis-ModelSelection`**: @@ -252,6 +291,10 @@ panel settles on `Idle`. `SynthesisPanelViewModel_Play_SuccessfulSession_TransitionsThroughLifecycleToIdle` - **`SpeechDemo-Synthesis-StopControl`**: `SynthesisPanelViewModel_Stop_DuringPlayback_CancelsSessionAndReportsStopped` +- **`SpeechDemo-Synthesis-EngineSessionReuse`**: + `SynthesisPanelViewModel_Play_CalledTwiceWithUnchangedModelAndParameters_ReusesSameSessionWithoutReload`, + `SynthesisPanelViewModel_Play_ParameterValueChanged_ReloadsEngineAndRecreatesSession`, + `SynthesisPanelViewModel_Play_PlaybackDeviceChanged_RecreatesSessionButNotEngine` - **`SpeechDemo-Synthesis-HonestUnavailableStates`**: `SynthesisPanelViewModel_Play_NoModelSelected_ReportsErrorState`, `SynthesisPanelViewModel_Play_NoPlaybackDevice_ReportsErrorState`, @@ -261,37 +304,42 @@ panel settles on `Idle`. - **`SpeechDemo-Synthesis-SessionSeam`**: `SynthesisPanelViewModel_Constructor_NullDependency_ThrowsArgumentNullException`, `SynthesizerSessionFactory_Constructor_NullStore_ThrowsArgumentNullException`, - `SynthesizerSessionFactory_Create_NullModel_ThrowsArgumentNullException`, - `SynthesizerSessionFactory_Create_NullDevice_ThrowsArgumentNullException`, - `SynthesizerSessionFactory_Create_ModelNotSynthesisRole_ReturnsUnavailableSynthesizer` + `SynthesizerSessionFactory_LoadAsync_NullModel_ThrowsArgumentNullException`, + `SynthesizerSessionFactory_LoadAsync_ModelNotSynthesisRole_ReturnsUnavailableEngine` - **`SpeechDemo-Synthesis-VoiceSelectionForwarding`**: `SynthesisPanelViewModel_Play_ModelDeclaresChoiceParameter_ForwardsValueBagToSessionFactory`, - `SynthesizerSessionFactory_Create_ModelNotSynthesisRoleWithParameterValues_ReturnsUnavailableSynthesizer` + `SynthesizerSessionFactory_LoadAsync_ModelNotSynthesisRoleWithParameterValues_ReturnsUnavailableEngine` - **`SpeechDemo-Synthesis-AutoRefreshOnInstall`**: `SynthesisPanelViewModel_ModelInstalled_MatchingRole_TriggersRefresh`, `SynthesisPanelViewModel_ModelInstalled_NonMatchingRole_DoesNotTriggerRefresh` - **`SpeechDemo-Synthesis-ResourceLifetime`**: - `SynthesisPanelViewModel_Dispose_UnsubscribesFromModelInstalled_NoRefreshAfterDispose`, - `SynthesisPanelViewModel_Dispose_NoActiveSynthesizer_IsSafeAndIdempotent` + `SynthesisPanelViewModel_DisposeAsync_UnsubscribesFromModelInstalled_NoRefreshAfterDispose`, + `SynthesisPanelViewModel_DisposeAsync_NoActiveSession_IsSafeAndIdempotent` - **`SpeechDemo-Synthesis-ModelSwitchGuard`**: `SynthesisPanelViewModel_Play_SuccessfulSession_CanChangeModelTogglesAcrossLifecycle`, `SynthesisPanelViewModel_CanChangeModel_ErrorState_IsTrue` - **`SpeechDemo-Synthesis-StopsBeforeDeviceRefresh`**: - `SynthesisPanelViewModel_PreRefreshHook_WhilePlaying_StopsAndAwaitsExecutionTaskBeforeDeviceRefreshSucceeds` + `SynthesisPanelViewModel_PreRefreshHook_WhilePlaying_StopsAndAwaitsExecutionTaskBeforeDeviceRefreshSucceeds`, + `SynthesisPanelViewModel_PreRefreshHook_WhileIdle_IsNoOpAndDeviceRefreshSucceeds` ### Acceptance Criteria A SynthesisPanelSubsystem test run passes when: only installed synthesis models are offered and the selection survives a refresh; the embedded settings panel always reflects the selected model; the example tag hints are drawn from the library's own vocabulary; a successful Play -transitions through every documented lifecycle state and releases its synthesizer; Stop -interrupts an in-flight Play deterministically; every unavailable state - no model, no device, -an unavailable synthesizer - is reported honestly rather than crashing; the session seam +transitions through every documented lifecycle state (driven by `ISynthesisSession.StateChanged`) +and releases its session; Stop interrupts an in-flight Play deterministically; calling Play +repeatedly with the same model/parameter/device selection reuses the same cached engine and +session rather than reloading the model and recreating the session on every click, while a +changed parameter value reloads the engine (and, as a consequence, recreates the session) and a +changed playback device alone recreates only the session; every unavailable state - no model, no +device, an unavailable engine - is reported honestly rather than crashing; the session seam validates its arguments and reports a role mismatch the same honest way the library reports an uninstalled model, regardless of whether a `parameterValues` bag is supplied; the settings panel's current value bag genuinely reaches the session factory when Play is invoked; a matching-role `ModelInstalled` event triggers an automatic refresh while a non-matching-role -event does not; `Dispose()` - now implemented on this class for the first time - is safe and -idempotent, and unsubscribes from `ModelInstalled` so a later event is never applied; and the -panel's registered pre-refresh hook stops an in-flight Play and genuinely awaits its completion -before a shared device refresh is attempted, so the refresh succeeds deterministically. +event does not; `DisposeAsync()` - now implemented on this class for the first time - is safe and +idempotent, releases the cached engine/session, and unsubscribes from `ModelInstalled` so a later +event is never applied; and the panel's registered pre-refresh hook stops an in-flight Play and +genuinely awaits its completion before a shared device refresh is attempted (and is a safe no-op +while idle), so the refresh succeeds deterministically. diff --git a/docs/verification/speech.md b/docs/verification/speech.md index e6fc2a9..fb56ec1 100644 --- a/docs/verification/speech.md +++ b/docs/verification/speech.md @@ -26,8 +26,8 @@ or synthesis engine instead of a real model, so CI never depends on a real, mult download or the native sherpa-onnx runtime. System tests reside in `SpeechTests.cs` within the `DemaConsulting.Speech.Tests` project, with the -Natural Language Audio Tag and streaming-synthesis scenarios additionally proven by -`AudioTagParserTests.cs` and `SherpaOnnxSpeechSynthesizerTests.cs` in the same project. +Natural Language Audio Tag scenarios additionally proven by `AudioTagParserTests.cs` in the same +project. ## Test Environment @@ -136,21 +136,16 @@ Verifies that a bracketed word outside the closed vocabulary is passed through a text rather than dropped or rejected, proving the "never worse than plain narration" guarantee for unrecognized or malformed bracket content. -### Unit: Plain Text Synthesis Yields an Audio Segment +### Unit: Streaming Synthesis Produces Played Audio -**Test**: `SynthesizeStreamAsync_PlainText_YieldsAudioSegment` - -Verifies that `SherpaOnnxSpeechSynthesizer.SynthesizeStreamAsync` yields at least one audio segment -for plain text with no audio tags, proving the Layer 2 rendering and chunked synthesis pipeline -produce audio for the simplest input. - -### Unit: Ordered Segments Start, Write in Order, and Stop Playback - -**Test**: `PlayStreamAsync_OrderedSegments_StartsWritesInOrderAndStops` +**Test**: `Speech_SystemIntegration_StreamingSynthesis_TextProducesPlayedAudio` -Verifies that `SherpaOnnxSpeechSynthesizer.PlayStreamAsync` starts the playback device, writes a -mono 16 kHz silence segment followed by an audio segment to the device in order, and stops the -device once playback completes. +Verifies that composing `SpeechSynthesizerFactory.LoadAsync` with an installed test model and a +fake synthesis engine, then creating a session via `ISpeechSynthesizerEngine.CreateSessionAsync` +and speaking plain text via `ISynthesisSession.SpeakAsync`, starts the playback device, writes at +least one block of audio to it, and stops the device once playback completes - proving the +Layer 2 rendering and chunked streaming-synthesis pipeline produce and play audio for the +simplest input end to end through the current async Engine/Session API. ## Acceptance Criteria diff --git a/docs/verification/speech/model-management-subsystem.md b/docs/verification/speech/model-management-subsystem.md index 7bf2c81..7a18e4e 100644 --- a/docs/verification/speech/model-management-subsystem.md +++ b/docs/verification/speech/model-management-subsystem.md @@ -46,4 +46,8 @@ throws `ArgumentException` for a value invalid for a parameter a model declares ignoring (with only an `Info` diagnostic) a supplied key naming a parameter the model does not declare; and `SpeechModelCatalog` correctly reports `NotDownloaded`, `Downloading`, `Downloaded`, and `FailedOrCorrupt` for an injected fake model as a download is requested, in progress, -completes, or fails, using only injected fakes since this pass ships zero real model classes. +completes, or fails, using injected fakes for deterministic state-resolution coverage, with the +same state-resolution behavior additionally proven indirectly through the catalog's four real, +shipped model classes (`SherpaOnnxZipformerEnRecognitionModel`, +`SherpaOnnxNemotronStreamingEnRecognitionModel`, `SherpaOnnxVitsLibriTtsEnglishSynthesisModel`, +`SherpaOnnxKokoroEnglishSynthesisModel`) in their own unit-level test suites. diff --git a/docs/verification/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.md b/docs/verification/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.md index 61fb367..71a0140 100644 --- a/docs/verification/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.md +++ b/docs/verification/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.md @@ -41,7 +41,7 @@ and removes the archive file afterward; `ResolveSpeakerId` resolves every one of voices to its confirmed speaker id, and falls back to the default voice's id for an unknown value, a missing key, or a `null` bag; and - proven once, empirically, outside the automated suite - selecting two different voices against the same input sentence through the real -`SpeechSynthesizerFactory.Create` genuinely produces two different, non-silent audio outputs. +`SpeechSynthesizerFactory.LoadAsync` genuinely produces two different, non-silent audio outputs. #### Test Scenarios @@ -119,8 +119,8 @@ convention of leaving no scratch artifacts in the working tree). byte-identical, proving that selecting a different voice genuinely changes the synthesized output rather than the configuration field being silently ignored - **Full pipeline confirmed**: the same distinction was independently reproduced end-to-end - through the real `SpeechSynthesizerFactory.Create` with a `parameterValues` bag selecting each - voice by name, `SherpaOnnxSpeechSynthesizer.GenerateSegment` calling + through the real `SpeechSynthesizerFactory.LoadAsync` with a `parameterValues` bag selecting each + voice by name, `SherpaOnnxSynthesisSession.GenerateSegmentAsync` calling `ISynthesisModel.ResolveSpeakerId` once per segment instead of a hard-coded `speakerId: 0`, and producing the same two distinct, non-silent outputs described above - not merely the isolated engine call diff --git a/docs/verification/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.md b/docs/verification/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.md index dadc96d..fee3626 100644 --- a/docs/verification/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.md +++ b/docs/verification/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.md @@ -43,7 +43,7 @@ and removes the archive file afterward; `ResolveSpeakerId` resolves a valid nume boxed `double`, and falls back to the default speaker id (`0`) for an out-of-range value, a non-numeric value, a missing key, or a `null` bag; and - proven once, empirically, outside the automated suite - selecting two different speaker ids against the same input sentence through -the real `SpeechSynthesizerFactory.Create` genuinely produces two different, non-silent audio +the real `SpeechSynthesizerFactory.LoadAsync` genuinely produces two different, non-silent audio outputs. #### Test Scenarios @@ -118,7 +118,7 @@ convention of leaving no scratch artifacts in the working tree). `10dc268f3e371696d721486123e2705a9fc1faa113491979fde4d88dba1f1b1c` - **License**: CC BY 4.0, read directly from the archive's own extracted `MODEL_CARD` file - **Distinct-speaker proof**: the same English sentence ("The quick brown fox jumps over the lazy - dog.") was synthesized twice through the real `SpeechSynthesizerFactory.Create` with a + dog.") was synthesized twice through the real `SpeechSynthesizerFactory.LoadAsync` with a `parameterValues` bag selecting two different numeric speaker ids via the new `speaker` parameter: - Speaker id 0 (default): 47,104 samples generated, RMS amplitude ≈ 0.093208 - non-silent @@ -127,8 +127,8 @@ convention of leaving no scratch artifacts in the working tree). byte-identical, proving that selecting a different speaker genuinely changes the synthesized output rather than the configuration field being silently ignored - **Full pipeline confirmed**: this distinction was produced end-to-end through the real - `SpeechSynthesizerFactory.Create` with a `parameterValues` bag selecting each speaker by - numeric index, `SherpaOnnxSpeechSynthesizer.GenerateSegment` calling + `SpeechSynthesizerFactory.LoadAsync` with a `parameterValues` bag selecting each speaker by + numeric index, `SherpaOnnxSynthesisSession.GenerateSegmentAsync` calling `ISynthesisModel.ResolveSpeakerId` once per segment instead of a hard-coded `speakerId: 0`, and producing the two distinct, non-silent outputs described above - not merely an isolated engine call diff --git a/docs/verification/speech/recognition-subsystem.md b/docs/verification/speech/recognition-subsystem.md index e5ff2be..487d249 100644 --- a/docs/verification/speech/recognition-subsystem.md +++ b/docs/verification/speech/recognition-subsystem.md @@ -3,21 +3,25 @@ ### Verification Approach The RecognitionSubsystem is verified entirely through deterministic unit tests that substitute a -fake recognition engine behind the subsystem's internal engine seam and an NSubstitute -`IAudioCaptureDevice` in place of real hardware. This makes composition decisions, audio-format -conversion, the two-thread streaming pipeline, result ordering, fault containment, and honest +fake `IRecognitionBackend`/`IRecognitionBackendFactory` pair behind the subsystem's internal +backend seam and an NSubstitute `IAudioCaptureDevice` in place of real hardware. This makes +composition decisions, audio-format conversion, the two-thread streaming pipeline, engine +exclusivity, session lifecycle, result ordering/backpressure, fault containment, and honest degradation fully testable without a downloaded speech model, a microphone, or the platform-specific native speech-inference runtime. -Determinism is structural rather than timing-based: the recognizer's `Stop()` completes its -internal queue and joins its background consumer, so every result derived from a frame raised -before the call has been delivered by the time it returns. No test polls, sleeps, or waits on a -timeout. +Determinism is structural rather than timing-based: `StopAsync` completes its internal queue and +awaits its pump task, so every result derived from a frame raised before the call has been +delivered or accounted for by the time it returns; cancellation/abandon-timeout behavior is +verified with an injectable abandon-timeout override rather than real multi-second waits. Automated coverage **does not** include recognizing real speech. Proving that real audio from a real microphone produces correct text through a real model requires both a downloaded production model (which this phase deliberately does not ship) and audio hardware, so it remains a -manual/local verification activity. +manual/local verification activity. The bookkeeping and accuracy of the internal, native-backed +`SherpaOnnxRecognitionEngine` against real installed models (when present) is covered separately +in `SherpaOnnxRecognitionEngineTests`/`SherpaOnnxRecognitionEngineAccuracyTests`, unaffected by +this phase's Engine/Session split since neither class changed. ### Test Environment @@ -25,7 +29,7 @@ manual/local verification activity. - **Execution**: `dotnet test` invoked by `build.ps1` and the CI pipeline - **Dependencies**: No external services, no downloaded model, no native speech-inference runtime, and no physical audio hardware -- **Test doubles**: A fake `IRecognitionEngine`/`IRecognitionEngineFactory` pair, NSubstitute +- **Test doubles**: A fake `IRecognitionBackend`/`IRecognitionBackendFactory` pair, NSubstitute capture devices and diagnostics sinks, a fake recognition model, and a parameter-capturing fake recognition model used only to prove parameter-values pass-through - **Isolation**: Composition tests create and delete their own scratch installed-model directory @@ -34,9 +38,11 @@ manual/local verification activity. A RecognitionSubsystem test run passes when: -- Composition returns a real recognizer only when the model is installed, declares the - recognition role, the capture device is available, and the engine loads -- Every other composition outcome returns the honest unavailable recognizer without throwing +- Composition returns a real engine only when the model is installed, declares the recognition + role, and the backend loads; `LoadAsync`'s returned task never faults for an ordinary + unavailable machine state +- Every other composition outcome returns the honest unavailable engine without faulting the + returned task - An optional `parameterValues` bag supplied by the caller reaches the recognition model's own engine-configuration logic unchanged, and does not change behavior for a model that declares no parameters @@ -44,144 +50,169 @@ A RecognitionSubsystem test run passes when: silently ignored (with only an `Info` diagnostic reported) and composition still succeeds; a supplied value for a parameter the model *does* declare that fails that parameter's own validation (wrong CLR type, out-of-range or non-integral for a `NumericParameter`, an invalid - option for a `ChoiceParameter`, a non-`bool` for a `BooleanParameter`) throws `ArgumentException` - synchronously from `Create()`, before any installed/role/device/engine check runs + option for a `ChoiceParameter`, a non-`bool` for a `BooleanParameter`) faults the returned task + with `ArgumentException`, before any installed/role/backend-load check runs +- `CreateSessionAsync` returns exactly one live session per engine at a time, throwing + `RecognitionEngineBusyException` for a concurrent attempt, and permits a new session once the + prior one is fully disposed - Captured audio is downmixed and resampled to the model's declared `AudioFormat`, with above-target-Nyquist energy attenuated before downsampling decimation -- Every recognition result is delivered, in order, with its provisional/final flag preserved -- Start/stop/dispose behave idempotently, drain queued audio, and release engine resources -- `Stop()`/`Dispose()` flush trailing audio the engine had accepted but not yet decoded - the tail - of an utterance released with no trailing silence - as one last final result before resetting the - engine for the next session, so no accepted audio is silently lost -- Engine faults and throwing host handlers are reported and contained rather than propagated, - including a fault in the trailing-audio flush itself -- The unavailable recognizer stays honest and safe to hold, subscribe to, and dispose +- Every recognition result is delivered through `GetResultsAsync`, in order, with its + provisional/final flag preserved, subject to the documented backpressure policy (coalesced + provisionals, byte-capped finals) +- The session's `RecognitionSessionState` machine only ever makes forward-only, documented + transitions, raising `StateChanged` for each one +- `StartAsync`/`StopAsync`/`DisposeAsync` behave idempotently, drain queued audio, and release + backend/lease resources +- `StopAsync`/`DisposeAsync` flush trailing audio the backend had accepted but not yet decoded - + the tail of an utterance released with no trailing silence - as one last final result before + resetting the backend for the next session, so no accepted audio is silently lost +- Backend faults, a lost capture device, and throwing host handlers are reported/contained rather + than propagated uncontrolled, and surface to an active `GetResultsAsync` consumer as + `RecognitionSessionFaultedException` +- A non-cooperative native call is abandoned after its configured timeout rather than blocking a + caller forever, with the abandonment reported through diagnostics +- The unavailable engine and unavailable session both stay honest and safe to hold, subscribe to, + and dispose - The automated verification boundary remains honest about the absence of real-speech coverage ### Test Scenarios -#### Composition: Real Recognizer for an Installed Model and Available Device +#### Composition: Real Engine for an Installed Model -**Tests**: `SpeechRecognizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealRecognizer`, -`SpeechRecognizerFactory_Create_WithStoreModelInstalledAndDeviceAvailable_ReturnsRealRecognizer`, -`SpeechRecognizerFactory_Create_WithCatalogModelInstalledAndDeviceAvailable_ReturnsRealRecognizer` +**Tests**: `SpeechRecognizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine`, +`SpeechRecognizerFactory_LoadAsync_WithStoreModelInstalled_ReturnsRealEngine`, +`SpeechRecognizerFactory_LoadAsync_WithCatalogModelInstalled_ReturnsRealEngine` -Verifies that an installed recognition model plus an available capture device composes a real -recognizer wired to the injected engine factory, with the installed-model directory passed -through unchanged, whether that directory is supplied directly as a `string`, resolved from a +Verifies that an installed recognition model composes a real `SherpaOnnxSpeechRecognizerEngine` +wired to the injected backend factory, with the installed-model directory passed through +unchanged, whether that directory is supplied directly as a `string`, resolved from a `SpeechModelStore`, or resolved from a `SpeechModelCatalog`'s own store. #### Composition: Honest Fallback for Every Unavailable State -**Tests**: `SpeechRecognizerFactory_Create_ModelNotInstalled_ReturnsUnavailableRecognizer`, -`SpeechRecognizerFactory_Create_CaptureDeviceUnavailable_ReturnsUnavailableRecognizer`, -`SpeechRecognizerFactory_Create_ModelRoleIsNotRecognition_ReturnsUnavailableRecognizer`, -`SpeechRecognizerFactory_Create_EngineLoadFails_ReturnsUnavailableRecognizerAndDoesNotThrow`, -`SpeechRecognizerFactory_Create_WithStoreModelNotInstalled_ReturnsUnavailableRecognizer`, -`SpeechRecognizerFactory_Create_WithCatalogModelNotInstalled_ReturnsUnavailableRecognizer` +**Tests**: `SpeechRecognizerFactory_LoadAsync_ModelNotInstalled_ReturnsUnavailableEngine`, +`SpeechRecognizerFactory_LoadAsync_ModelRoleIsNotRecognition_ReturnsUnavailableEngine`, +`SpeechRecognizerFactory_LoadAsync_EngineLoadFails_ReturnsUnavailableEngineAndDoesNotFaultTask`, +`SpeechRecognizerFactory_LoadAsync_WithStoreModelNotInstalled_ReturnsUnavailableEngine`, +`SpeechRecognizerFactory_LoadAsync_WithCatalogModelNotInstalled_ReturnsUnavailableEngine` -Verifies that a missing model, an unavailable device, a wrong-role model, and a failed engine -load all degrade to the shared unavailable recognizer without throwing, and that no engine is -loaded when an earlier check already failed. +Verifies that a missing model, a wrong-role model, and a failed backend load all degrade to the +shared unavailable engine without faulting the returned task, and that no backend is loaded when +an earlier check already failed. -#### Composition: Null Arguments Are Programming Errors +#### Composition: Null Arguments and Cancellation Are Programming/Caller Errors -**Tests**: `SpeechRecognizerFactory_Create_NullModel_ThrowsArgumentNullException`, -`SpeechRecognizerFactory_Create_NullCaptureDevice_ThrowsArgumentNullException`, -`SpeechRecognizerFactory_Create_WithStoreNullModel_ThrowsArgumentNullException`, -`SpeechRecognizerFactory_Create_WithStoreNullStore_ThrowsArgumentNullException`, -`SpeechRecognizerFactory_Create_WithStoreNullCaptureDevice_ThrowsArgumentNullException`, -`SpeechRecognizerFactory_Create_WithCatalogNullModel_ThrowsArgumentNullException`, -`SpeechRecognizerFactory_Create_WithCatalogNullCatalog_ThrowsArgumentNullException`, -`SpeechRecognizerFactory_Create_WithCatalogNullCaptureDevice_ThrowsArgumentNullException` +**Tests**: `SpeechRecognizerFactory_LoadAsync_NullModel_FaultsWithArgumentNullException`, +`SpeechRecognizerFactory_LoadAsync_WithStoreNullModel_FaultsWithArgumentNullException`, +`SpeechRecognizerFactory_LoadAsync_WithStoreNullStore_FaultsWithArgumentNullException`, +`SpeechRecognizerFactory_LoadAsync_WithCatalogNullModel_FaultsWithArgumentNullException`, +`SpeechRecognizerFactory_LoadAsync_WithCatalogNullCatalog_FaultsWithArgumentNullException`, +`SpeechRecognizerFactory_LoadAsync_CancelledToken_FaultsWithOperationCanceledException` -Verifies that a null model, capture device, store, or catalog throws, distinguishing a -programming error from an ordinary machine state. +Verifies that a null model, store, or catalog faults the returned task with +`ArgumentNullException`, and that a cancellation token already cancelled before loading completes +faults it with `OperationCanceledException`, distinguishing both from an ordinary machine state. #### Composition: Parameter Value Bag Forwarding -**Tests**: `SpeechRecognizerFactory_Create_ParameterValuesSupplied_ReachesModelCreateEngineConfig`, -`SpeechRecognizerFactory_Create_WithStoreParameterValuesSupplied_ReachesModelCreateEngineConfig`, -`SpeechRecognizerFactory_Create_WithCatalogParameterValuesSupplied_ReachesModelCreateEngineConfig` +**Tests**: `SpeechRecognizerFactory_LoadAsync_ParameterValuesSupplied_ReachesModelCreateEngineConfig`, +`SpeechRecognizerFactory_LoadAsync_WithStoreParameterValuesSupplied_ReachesModelCreateEngineConfig`, +`SpeechRecognizerFactory_LoadAsync_WithCatalogParameterValuesSupplied_ReachesModelCreateEngineConfig` Verifies that an optional `parameterValues` bag supplied by the caller (for example, a selected recognition language built from a declared `ChoiceParameter`) genuinely reaches a model's own two-argument `IRecognitionModel.CreateEngineConfig` override rather than merely reaching the -engine factory. +backend factory. #### Composition: Parameter Value Validation -**Tests**: `SpeechRecognizerFactory_Create_UnrecognizedParameterId_ComposesAndReportsInfo`, -`SpeechRecognizerFactory_Create_RecognizedNumericParameterOutOfRange_Throws`, -`SpeechRecognizerFactory_Create_RecognizedNumericParameterWrongType_Throws`, -`SpeechRecognizerFactory_Create_RecognizedChoiceParameterInvalidOption_Throws`, -`SpeechRecognizerFactory_Create_RecognizedBooleanParameterWrongType_Throws` +**Tests**: `SpeechRecognizerFactory_LoadAsync_UnrecognizedParameterId_ComposesAndReportsInfo`, +`SpeechRecognizerFactory_LoadAsync_RecognizedNumericParameterOutOfRange_FaultsWithArgumentException`, +`SpeechRecognizerFactory_LoadAsync_RecognizedNumericParameterWrongType_FaultsWithArgumentException`, +`SpeechRecognizerFactory_LoadAsync_RecognizedChoiceParameterInvalidOption_FaultsWithArgumentException`, +`SpeechRecognizerFactory_LoadAsync_RecognizedBooleanParameterWrongType_FaultsWithArgumentException` Verifies the deliberate, breaking-change split introduced for this behavior: a supplied -`parameterValues` key naming a parameter the model does not declare still composes a real -recognizer and reports only an `Info` diagnostic, never throwing (preserving cross-model -compatibility); a supplied value for a parameter the model *does* declare, but that is invalid -for it, throws `ArgumentException` synchronously from `Create()` - before any -installed/role/device/engine check runs - naming the parameter id, the model id, and the specific -reason the value is invalid. This replaces this library's earlier behavior of silently -substituting a default for such a value. - -#### Pipeline: Capture Format Conversion - -**Tests**: `SherpaOnnxSpeechRecognizer_FrameCaptured_StereoAtHigherRate_FeedsResampledMonoToEngine`, -`SherpaOnnxSpeechRecognizer_Constructor_DeviceReportsUnusableFormat_FallsBackToPassThrough`, -`SherpaOnnxSpeechRecognizer_FrameCaptured_EmptyBlock_IsIgnored` - -Verifies that a stereo block at a higher rate reaches the engine as mono at the model's declared -rate, that a device reporting an unusable format degrades to pass-through with a warning rather -than throwing, and that an empty block is ignored. - -#### Pipeline: Result Delivery and Ordering - -**Tests**: `SherpaOnnxSpeechRecognizer_FrameCaptured_EngineDecodesResults_RaisesResultReceivedInOrder` - -Verifies that every result the engine decodes is raised in order with its provisional/final flag -preserved. - -#### Pipeline: Text Normalization - -**Tests**: `SherpaOnnxSpeechRecognizer_Constructor_NullModel_ThrowsArgumentNullException`, -`SherpaOnnxSpeechRecognizer_FrameCaptured_FinalResult_AppliesModelNormalizeTextWithIsFinalTrue`, -`SherpaOnnxSpeechRecognizer_FrameCaptured_ProvisionalResult_AppliesModelNormalizeTextWithIsFinalFalse` - -Verifies that a null `model` constructor argument throws `ArgumentNullException`, and that every -raised result's text has passed through the owning model's `IRecognitionModel.NormalizeText`, -called with `isFinal: true` for final results and `isFinal: false` for provisional results - the -model's returned text, not the engine's raw text, is what `ResultReceived` carries. - -#### Pipeline: Lifecycle and Draining - -**Tests**: `SherpaOnnxSpeechRecognizer_Start_Always_SubscribesAndStartsCaptureDevice`, -`SherpaOnnxSpeechRecognizer_Start_AlreadyRunning_IsNoOp`, -`SherpaOnnxSpeechRecognizer_Stop_WhileRunning_UnsubscribesAndStopsCaptureDevice`, -`SherpaOnnxSpeechRecognizer_Stop_NotRunning_IsNoOp`, -`SherpaOnnxSpeechRecognizer_Stop_EngineHasFlushableTrailingAudio_RaisesFlushedFinalResult`, -`SherpaOnnxSpeechRecognizer_Dispose_EngineHasFlushableTrailingAudio_RaisesFlushedFinalResult`, -`SherpaOnnxSpeechRecognizer_Stop_EngineHasNothingToFlush_RaisesNoExtraResult`, -`SherpaOnnxSpeechRecognizer_Dispose_CalledTwice_StopsAndDisposesEngineOnce`, -`SherpaOnnxSpeechRecognizer_Start_AfterDispose_ThrowsObjectDisposedException` - -Verifies that starting subscribes and starts capture, repeated starts and idle stops are no-ops, -stopping unsubscribes and stops the device so post-stop frames never reach the engine, disposal -stops once and releases the engine once, and starting after disposal is rejected. - -#### Pipeline: Fault Containment - -**Tests**: `SherpaOnnxSpeechRecognizer_FrameCaptured_EngineThrows_ReportsFaultAndKeepsRunning`, -`SherpaOnnxSpeechRecognizer_ResultReceived_HandlerThrows_ReportsFaultAndDoesNotRethrow`, -`SherpaOnnxSpeechRecognizer_Start_CaptureDeviceFails_ThrowsSpeechRecognizerUnavailableException`, -`SherpaOnnxSpeechRecognizer_Stop_EngineFlushFails_CompletesResetsEngineAndReportsFault` - -Verifies that engine faults and throwing host handlers are reported through the diagnostics sink -and never escape into the capture path, while a capture device that fails on first use surfaces -the documented recognizer exception with the device's failure as its inner exception. A fault in -the trailing-audio flush itself is contained the same way: reported, not thrown, with teardown and -the subsequent engine reset both still completing. +`parameterValues` key naming a parameter the model does not declare still composes a real engine +and reports only an `Info` diagnostic, never faulting the task (preserving cross-model +compatibility); a supplied value for a parameter the model *does* declare, but that is invalid for +it, faults the returned task with `ArgumentException` - before any installed/role/backend-load +check runs - naming the parameter id, the model id, and the specific reason the value is invalid. + +#### Engine Exclusivity and Lease Behavior + +**Tests**: `SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_NoActiveSession_ReturnsSession`, +`SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_SessionAlreadyLeased_ThrowsRecognitionEngineBusyException`, +`SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_PriorSessionDisposing_ThrowsRecognitionEngineBusyException`, +`SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_AfterPriorSessionFullyDisposed_ReturnsNewSession`, +`SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException`, +`SherpaOnnxSpeechRecognizerEngine_DisposeAsync_WithActiveSession_DisposesSessionFirst` + +Verifies that the engine's single lease permits exactly one live session at a time, fails fast +(no queueing) with `RecognitionEngineBusyException` for a concurrent attempt - including while the +prior session is still mid-`DisposeAsync` - and is released only once that prior session has fully +completed disposal, after which a new session can be created. + +#### Session Lifecycle: State Machine Transitions + +**Tests**: `SherpaOnnxRecognitionSession_StartAsync_FromCreated_TransitionsToRunning`, +`SherpaOnnxRecognitionSession_StartAsync_FromStopped_ThrowsInvalidOperationException`, +`SherpaOnnxRecognitionSession_StateChanged_EmitsEveryTransitionInOrder`, +`SherpaOnnxRecognitionSession_DeviceLostMidSession_TransitionsToFaulted` + +Verifies the forward-only `RecognitionSessionState` machine: `StartAsync` transitions `Created -> +Starting -> Running`, a session is single-use (starting again after `Stopped` throws +`InvalidOperationException`), `StateChanged` raises every transition in order, and a capture +device going unavailable mid-session transitions the session to `Faulted`. + +#### Session Lifecycle: Stop, Dispose, and Draining + +**Tests**: `SherpaOnnxRecognitionSession_StopAsync_FlushesTrailingResultsBeforeCompleting`, +`SherpaOnnxRecognitionSession_StopAsync_CalledConcurrentlyTwice_BothCompleteOnceStopped`, +`SherpaOnnxRecognitionSession_StopAsync_BackendResetFails_CompletesAndReportsFault` + +Verifies that `StopAsync` flushes trailing audio the backend had accepted but not yet decoded as +one last final result before completing, that two concurrent `StopAsync` callers both complete +only once the session has actually stopped, and that a backend reset failure during `StopAsync` is +reported rather than thrown while teardown still completes. + +#### Session Pipeline: Capture Format Conversion and Text Normalization + +**Tests**: `SherpaOnnxRecognitionSession_FrameCaptured_StereoAtModelRate_FeedsDownmixedMonoToBackend`, +`SherpaOnnxRecognitionSession_FrameCaptured_FinalResult_AppliesModelNormalizeTextWithIsFinalTrue` + +Verifies that a stereo block reaches the backend as downmixed mono at the model's declared rate, +and that a final result's text has passed through the owning model's +`IRecognitionModel.NormalizeText(text, isFinal: true)` before being buffered. + +#### Session Pipeline: Result Delivery and Backpressure + +**Tests**: `SherpaOnnxRecognitionSession_GetResultsAsync_CalledConcurrently_ThrowsInvalidOperationException`, +`SherpaOnnxRecognitionSession_GetResultsAsync_CancelledToken_EndsEnumerationWithoutStoppingSession`, +`SherpaOnnxRecognitionSession_GetResultsAsync_SessionFaulted_ThrowsRecognitionSessionFaultedException`, +`SherpaOnnxRecognitionSession_GetResultsAsync_SlowConsumer_CoalescesProvisionalResults`, +`SherpaOnnxRecognitionSession_GetResultsAsync_SlowConsumer_NeverDropsFinalResultsUnderByteCap` + +Verifies that `GetResultsAsync` is single-consumer (a concurrent second enumeration throws +`InvalidOperationException`), that cancelling the consumer's token ends its enumeration without +stopping the session itself, that a faulted session surfaces +`RecognitionSessionFaultedException` from the active enumeration, that a slow consumer only ever +sees the latest coalesced provisional rather than a queue of stale ones, and that final results +are never dropped while the byte cap is not exceeded. + +#### Cooperative-Cancel-Then-Abandon Policy + +**Tests**: `DedicatedWorker_Run_CooperativeCancellation_CompletesPromptly`, +`DedicatedWorker_Run_NonCooperativeDelegate_AbandonsAfterTimeoutAndReportsDiagnostics`, +`DedicatedWorker_Run_UsesLongRunningTaskCreationOption`, +`SherpaOnnxRecognitionSession_NativeCallExceedsAbandonTimeout_TaskCompletesAndDiagnosticsReportsWarning` + +Verifies that a delegate which observes cancellation promptly completes its task immediately, that +a delegate which does not observe cancellation is abandoned after the configured timeout with a +`Warning` diagnostic rather than blocking the caller forever, that the worker always runs its +delegate with `TaskCreationOptions.LongRunning`, and that this same abandon behavior is exercised +end-to-end through a session's pump thread via an injectable abandon timeout. #### Audio Conversion: Downmix, Rate Conversion, and Boundaries @@ -213,14 +244,21 @@ intermediate downmix, and rejection of non-positive rates and channel counts. #### Unavailable Fallback: Honest Degradation -**Tests**: `UnavailableSpeechRecognizer_IsAvailable_Read_ReturnsFalse`, -`UnavailableSpeechRecognizer_Start_Always_ThrowsSpeechRecognizerUnavailableException`, -`UnavailableSpeechRecognizer_Stop_Always_ThrowsSpeechRecognizerUnavailableException`, -`UnavailableSpeechRecognizer_SubscriptionAndDispose_Always_AreSafeNoOps`, +**Tests**: `UnavailableSpeechRecognizerEngine_IsAvailable_Read_ReturnsFalse`, +`UnavailableSpeechRecognizerEngine_CreateSessionAsync_Always_ReturnsUnavailableSession`, +`UnavailableSpeechRecognizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException`, +`UnavailableSpeechRecognizerEngine_DisposeAsync_CalledTwice_DoesNotThrow`, +`UnavailableRecognitionSession_IsAvailable_Read_ReturnsFalse`, +`UnavailableRecognitionSession_StartAsync_Always_ThrowsSpeechRecognizerUnavailableException`, +`UnavailableRecognitionSession_StopAsync_Always_IsSafeNoOp`, +`UnavailableRecognitionSession_GetResultsAsync_Always_ThrowsSpeechRecognizerUnavailableException`, +`UnavailableRecognitionSession_DisposeAsync_CalledTwice_DoesNotThrow`, `SpeechRecognizerUnavailableException_Constructor_WithMessage_ExposesMessage`, `SpeechRecognizerUnavailableException_Constructor_WithInnerException_ExposesBoth`, `SpeechRecognizerUnavailableException_Constructor_Default_HasNonEmptyMessage` -Verifies that the shared fallback stays honest and safe to hold, subscribe to, and dispose, that -operational misuse throws the documented exception, and that the exception conforms to the -standard three-constructor pattern. +Verifies that both shared fallbacks stay honest and safe to hold, subscribe to, and dispose +repeatedly, that `UnavailableSpeechRecognizerEngine.CreateSessionAsync` always returns the shared +unavailable session rather than throwing, that operational misuse of the unavailable session +throws the documented exception, and that the exception type conforms to the standard +three-constructor pattern. diff --git a/docs/verification/speech/recognition-subsystem/i-recognition-session.md b/docs/verification/speech/recognition-subsystem/i-recognition-session.md new file mode 100644 index 0000000..c85f1df --- /dev/null +++ b/docs/verification/speech/recognition-subsystem/i-recognition-session.md @@ -0,0 +1,32 @@ +### IRecognitionSession + +#### Verification Approach + +`IRecognitionSession` is verified indirectly through both shipped implementations: +`UnavailableRecognitionSession` proves the honest-unavailable path, and +`SherpaOnnxRecognitionSession` proves state-machine transitions, lifecycle behavior, and delivery +of provisional and final results. The result and event value types are verified through the +results observed at `GetResultsAsync`, since they carry no behavior of their own. + +#### Test Environment + +- **Framework**: xUnit v3 running under the .NET SDK +- No additional setup beyond the standard test runner; no model, microphone, or native + speech-inference runtime is required. + +#### Acceptance Criteria + +The contract is considered verified when the unavailable implementation reports `false` +availability, reports `State` as `Created`, throws on operational misuse, and treats subscription +and disposal as safe no-ops, and when the real implementation only makes forward-only, documented +`RecognitionSessionState` transitions, raises `StateChanged` for each one, starts and stops +capture, delivers ordered provisional and final results carrying the full recognized text - +including finalizing and delivering trailing audio accepted but not yet decoded before +`StopAsync()` returns rather than losing it - releases its engine lease on disposal, and is +single-use and single-consumer by contract. + +#### Test Scenarios + +See the RecognitionSubsystem-level scenarios "Session Lifecycle: State Machine Transitions", +"Session Lifecycle: Stop, Dispose, and Draining", "Session Pipeline: Result Delivery and +Backpressure", and "Unavailable Fallback: Honest Degradation". diff --git a/docs/verification/speech/recognition-subsystem/i-speech-recognizer-engine.md b/docs/verification/speech/recognition-subsystem/i-speech-recognizer-engine.md new file mode 100644 index 0000000..5ebe93d --- /dev/null +++ b/docs/verification/speech/recognition-subsystem/i-speech-recognizer-engine.md @@ -0,0 +1,27 @@ +### ISpeechRecognizerEngine + +#### Verification Approach + +`ISpeechRecognizerEngine` is verified indirectly through both shipped implementations: +`UnavailableSpeechRecognizerEngine` proves the honest-unavailable path, and +`SherpaOnnxSpeechRecognizerEngine` proves availability reporting and the engine's single-lease +exclusivity behavior. + +#### Test Environment + +- **Framework**: xUnit v3 running under the .NET SDK +- No additional setup beyond the standard test runner; no model, microphone, or native + speech-inference runtime is required. + +#### Acceptance Criteria + +The contract is considered verified when the unavailable implementation reports `false` +availability and always returns the shared unavailable session, and when the real implementation +leases its backend to exactly one live session at a time, throwing `RecognitionEngineBusyException` +for a concurrent `CreateSessionAsync` attempt and permitting a new session once the prior one is +fully disposed. + +#### Test Scenarios + +See the RecognitionSubsystem-level scenarios "Engine Exclusivity and Lease Behavior" and +"Unavailable Fallback: Honest Degradation". diff --git a/docs/verification/speech/recognition-subsystem/i-speech-recognizer.md b/docs/verification/speech/recognition-subsystem/i-speech-recognizer.md deleted file mode 100644 index 574f2a3..0000000 --- a/docs/verification/speech/recognition-subsystem/i-speech-recognizer.md +++ /dev/null @@ -1,32 +0,0 @@ -### ISpeechRecognizer - -#### Verification Approach - -`ISpeechRecognizer` is verified indirectly through both shipped implementations: -`UnavailableSpeechRecognizer` proves the honest-unavailable path, and -`SherpaOnnxSpeechRecognizer` proves availability reporting, lifecycle behavior, and delivery of -provisional and final results. The result and event value types are verified through the results -observed at the contract's event, since they carry no behavior of their own. - -#### Test Environment - -- **Framework**: xUnit v3 running under the .NET SDK -- No additional setup beyond the standard test runner; no model, microphone, or native - speech-inference runtime is required. - -#### Acceptance Criteria - -The contract is considered verified when the unavailable implementation reports `false` -availability, throws on operational misuse, and treats subscription and disposal as safe no-ops, -and when the real implementation starts and stops capture, delivers ordered provisional and final -results carrying the full recognized text - including finalizing and delivering trailing audio -accepted but not yet decoded before `Stop()` returns rather than losing it - releases its engine -on disposal, and supports many independent Start/Stop cycles on the same instance without needing -to be reconstructed. - -#### Test Scenarios - -See the RecognitionSubsystem-level scenarios "Pipeline: Result Delivery and Ordering", "Pipeline: -Lifecycle and Draining", and "Unavailable Fallback: Honest Degradation". The "construct once, -reuse across many turns" contract is verified directly by -`SherpaOnnxSpeechRecognizer_MultipleStartStopCycles_ReusesSameInstanceWithoutReconstruction`. diff --git a/docs/verification/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md b/docs/verification/speech/recognition-subsystem/sherpa-onnx-recognition-session.md similarity index 65% rename from docs/verification/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md rename to docs/verification/speech/recognition-subsystem/sherpa-onnx-recognition-session.md index ef9c12f..38314c5 100644 --- a/docs/verification/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.md +++ b/docs/verification/speech/recognition-subsystem/sherpa-onnx-recognition-session.md @@ -1,21 +1,24 @@ -### SherpaOnnxSpeechRecognizer +### SherpaOnnxRecognitionSession #### Verification Approach -The streaming recognizer, the recognition-engine seam, and the audio-format converter are -verified together, since the recognizer's whole purpose is to move audio from the converter into -the engine on the right thread. +The streaming session, its companion Layer 3 engine, the recognition-backend seam, and the +audio-format converter are verified together, since the session's whole purpose is to move audio +from the converter into the backend on the right thread. -The engine seam is replaced by a deterministic fake that records the mono samples it was fed and +The backend seam is replaced by a deterministic fake that records the mono samples it was fed and yields a scripted sequence of results, so no downloaded model and no platform-specific native speech-inference binary is ever needed. The capture device is an NSubstitute `IAudioCaptureDevice` whose `FrameCaptured` event the test raises directly, standing in for a real audio callback. Diagnostics are an NSubstitute sink, so fault containment is verified by observing what was reported rather than by asserting an absence of exceptions alone. -Threading is verified without timing assumptions. The recognizer's `Stop()` completes its -internal queue and joins its background consumer, so a test raises a frame, calls `Stop()`, and -then asserts on the fully drained result - no sleeps, polls, or timeouts appear anywhere. +Threading is verified without timing assumptions. `StopAsync` completes its internal queue and +awaits its pump task, so a test raises a frame, calls `StopAsync`, and then asserts on the fully +drained result - no sleeps, polls, or timeouts appear anywhere for ordinary behavior. The +cooperative-cancel-then-abandon policy is the one place a bounded wait is genuinely under test, +and it is verified deterministically with an injectable abandon-timeout override rather than the +real multi-second default. The converter is verified separately as a pure function over plain float arrays, with a tolerance-based float comparer so assertions describe signal content rather than exact binary @@ -157,15 +160,13 @@ installed streaming Zipformer model the same kind of abandoned, no-trailing-sile then calling `TryFlush()` instead of `Reset()` directly, produces a final result whose text contains every word of the abandoned line - proving the native `OnlineStream.InputFinished` mechanism genuinely recovers audio that would otherwise be silently discarded, not merely that -nothing crashes. `SherpaOnnxSpeechRecognizer_Stop_EngineHasFlushableTrailingAudio_RaisesFlushedFinalResult` -(`SherpaOnnxSpeechRecognizerTests`, against the fake engine) separately verifies the pipeline -wiring: `Stop()` raises the engine's flushed result through `ResultReceived` before resetting, -`SherpaOnnxSpeechRecognizer_Dispose_EngineHasFlushableTrailingAudio_RaisesFlushedFinalResult` -verifies `Dispose()` shares the same flush wiring rather than skipping it because it is terminal, -`SherpaOnnxSpeechRecognizer_Stop_EngineHasNothingToFlush_RaisesNoExtraResult` verifies a quiet -session raises nothing extra, and -`SherpaOnnxSpeechRecognizer_Stop_EngineFlushFails_CompletesResetsEngineAndReportsFault` verifies a -faulting flush is contained the same way a faulting reset already was - reported, not thrown, +nothing crashes. +`SherpaOnnxRecognitionSession_StopAsync_FlushesTrailingResultsBeforeCompleting` +(`SherpaOnnxRecognitionSessionTests`, against the fake backend) separately verifies the pipeline +wiring: `StopAsync()` buffers the backend's flushed result for `GetResultsAsync` before resetting +the backend, and +`SherpaOnnxRecognitionSession_StopAsync_BackendResetFails_CompletesAndReportsFault` verifies a +faulting reset is contained the same way a faulting flush already is - reported, not thrown, with teardown and the subsequent engine reset both still completing. #### Test Environment @@ -173,71 +174,68 @@ with teardown and the subsequent engine reset both still completing. - **Framework**: xUnit v3 running under the .NET SDK - **Execution**: `dotnet test` invoked by `build.ps1` and the CI pipeline - **Dependencies**: No external services, no downloaded model, no native speech-inference runtime, - and no physical audio hardware for `SherpaOnnxSpeechRecognizerTests`; the real, already-installed + and no physical audio hardware for `SherpaOnnxRecognitionSessionTests`/ + `SherpaOnnxSpeechRecognizerEngineTests`/`DedicatedWorkerTests`; the real, already-installed streaming Zipformer model and the real native runtime for `SherpaOnnxRecognitionEngineTests` (self-skipping when the model is not installed) -- **Test doubles**: Fake `IRecognitionEngine`/`IRecognitionEngineFactory` implementations plus - NSubstitute capture devices and diagnostics sinks for `SherpaOnnxSpeechRecognizerTests` +- **Test doubles**: Fake `IRecognitionBackend`/`IRecognitionBackendFactory` implementations plus + NSubstitute capture devices and diagnostics sinks for `SherpaOnnxRecognitionSessionTests` and + `SherpaOnnxSpeechRecognizerEngineTests` #### Acceptance Criteria -The units are considered verified when captured audio reaches the engine downmixed and resampled -to the model's declared `AudioFormat`, with above-target-Nyquist energy attenuated before any -downsampling decimation; every decoded result is raised in order with its provisional/final flag -intact, start/stop/dispose are idempotent and drain queued audio before returning, engine and -handler faults are reported without escaping, a capture-start failure surfaces the documented -exception, and the converter produces the documented output for identity, upsampling, +The units are considered verified when captured audio reaches the backend downmixed and +resampled to the model's declared `AudioFormat`, with above-target-Nyquist energy attenuated +before any downsampling decimation; every decoded result is delivered through `GetResultsAsync` in +order with its provisional/final flag intact; `StartAsync`/`StopAsync`/`DisposeAsync` are +idempotent and drain queued audio before returning; the session's state machine only ever makes +forward-only, documented transitions; backend and handler faults are reported without escaping +uncontrolled and surface to an active `GetResultsAsync` consumer as +`RecognitionSessionFaultedException`; a capture-start or mid-session device failure surfaces the +documented exception; the engine's single lease permits exactly one live session at a time; a +non-cooperative native call is abandoned after its configured timeout with a `Warning` +diagnostic; and the converter produces the documented output for identity, upsampling, downsampling, multi-channel, and boundary inputs while rejecting non-positive rates and channel -counts. `Stop()`/`Dispose()` flush the engine's trailing audio - recovering, as one last final -result, even the tail of an utterance a streaming engine could not otherwise decode without -audio it will now never receive (for example a push-to-talk release with no trailing silence) - -before resetting the engine, so no accepted audio is silently lost; a fault in either that flush -or the subsequent reset during `Stop()`/`Dispose()` is reported without escaping and teardown -still completes, in which case the trailing flush and/or the engine's clean state cannot be -guaranteed, and in rare cases restart may not fully recover the ability to decode. A later -`Start()` is still permitted after such a fault during `Stop()`, while `Dispose()` remains -terminal regardless of whether its flush or reset succeeds. For the -post-endpoint warm-up-replay feature specifically: a disabled (`0`) window allocates no buffer and -changes no observable behavior; an enabled window allocates a buffer sized to the configured -duration that accumulates and trims fed samples; a real endpoint that occurs after genuine -recognized text has been produced since the last reset consumes the buffer via a replay that -never raises more than one final `SpeechRecognitionResult` per utterance; a real endpoint that -fires on pure/near-silence before any genuine text has been recognized since the last reset never -replays - the buffer is dropped and the grace period never arms; and the post-replay grace period -suppresses a same-tick spurious re-trigger. For real-speech transcription accuracy specifically: -the real, installed Zipformer and Nemotron models each transcribe the real "Crossing the Bar" -recording with a Word Error Rate, against its ground-truth text, that does not exceed the -documented 20% tolerance. For the session-end `Reset()` fix specifically: audio accepted but not -yet decoded before a session-end reset never surfaces as text once only silence follows the -reset. For the `TryFlush()` recovery fix specifically: audio accepted but not yet decoded is -recovered as a final result's text when flushed before that reset, rather than only proving it is -not lost. +counts. `StopAsync`/`DisposeAsync` flush the backend's trailing audio - recovering, as one last +final result, even the tail of an utterance a streaming backend could not otherwise decode +without audio it will now never receive (for example a push-to-talk release with no trailing +silence) - before resetting the backend, so no accepted audio is silently lost; a fault in either +that flush or the subsequent reset during `StopAsync`/`DisposeAsync` is reported without escaping +and teardown still completes, in which case the trailing flush and/or the backend's clean state +cannot be guaranteed, and in rare cases restart (via a new session from the same engine) may not +fully recover the ability to decode. For the post-endpoint warm-up-replay feature specifically: a +disabled (`0`) window allocates no buffer and changes no observable behavior; an enabled window +allocates a buffer sized to the configured duration that accumulates and trims fed samples; a real +endpoint that occurs after genuine recognized text has been produced since the last reset consumes +the buffer via a replay that never raises more than one final `SpeechRecognitionResult` per +utterance; a real endpoint that fires on pure/near-silence before any genuine text has been +recognized since the last reset never replays - the buffer is dropped and the grace period never +arms; and the post-replay grace period suppresses a same-tick spurious re-trigger. For real-speech +transcription accuracy specifically: the real, installed Zipformer and Nemotron models each +transcribe the real "Crossing the Bar" recording with a Word Error Rate, against its ground-truth +text, that does not exceed the documented 20% tolerance. For the session-end `Reset()` fix +specifically: audio accepted but not yet decoded before a session-end reset never surfaces as text +once only silence follows the reset. For the `TryFlush()` recovery fix specifically: audio +accepted but not yet decoded is recovered as a final result's text when flushed before that reset, +rather than only proving it is not lost. #### Test Scenarios -See the RecognitionSubsystem-level scenarios "Pipeline: Capture Format Conversion", "Pipeline: -Result Delivery and Ordering", "Pipeline: Text Normalization", "Pipeline: Lifecycle and -Draining", "Pipeline: Fault Containment", and "Audio Conversion: Downmix, Rate Conversion, and -Boundaries", plus `SherpaOnnxRecognitionEngineTests`'s "Post-Endpoint Warm-Up Replay" scenarios -covering the disabled no-op path, buffer sizing/accumulation/trimming, replay-on-endpoint, -suppressed replay output, the post-replay endpoint grace period, a silence-only endpoint never -arming replay (`SilenceOnlyEndpoint_NeverArmsReplay`), and the replay-eligibility gate itself -exercised deterministically via direct reflection assertions on `_hasRecognizedTextSinceReset` -(`ReplayEligibility_GatedOnHasRecognizedTextSinceReset`), independent of whether the real model -happens to transcribe a synthesized tone as non-empty text. `SherpaOnnxRecognitionEngineAccuracyTests` -adds the "Real-Speech Transcription Accuracy" scenario: transcribing the real "Crossing the Bar" -recording end to end through the real, installed Zipformer and Nemotron models and asserting the -resulting Word Error Rate against the poem's ground-truth text stays within tolerance; the -"Session-End Reset Buffered-Audio Regression" scenario -(`Reset_AbandonedUtteranceWithNoTrailingSilence_DoesNotBleedIntoNextSession`) described above; and -the "Trailing-Audio Flush Recovery" scenario +See the RecognitionSubsystem-level scenarios "Engine Exclusivity and Lease Behavior", "Session +Lifecycle: State Machine Transitions", "Session Lifecycle: Stop, Dispose, and Draining", "Session +Pipeline: Capture Format Conversion and Text Normalization", "Session Pipeline: Result Delivery +and Backpressure", "Cooperative-Cancel-Then-Abandon Policy", and "Audio Conversion: Downmix, Rate +Conversion, and Boundaries", plus `SherpaOnnxRecognitionEngineTests`'s "Post-Endpoint Warm-Up +Replay" scenarios covering the disabled no-op path, buffer sizing/accumulation/trimming, +replay-on-endpoint, suppressed replay output, the post-replay endpoint grace period, a +silence-only endpoint never arming replay (`SilenceOnlyEndpoint_NeverArmsReplay`), and the +replay-eligibility gate itself exercised deterministically via direct reflection assertions on +`_hasRecognizedTextSinceReset` (`ReplayEligibility_GatedOnHasRecognizedTextSinceReset`), +independent of whether the real model happens to transcribe a synthesized tone as non-empty text. +`SherpaOnnxRecognitionEngineAccuracyTests` adds the "Real-Speech Transcription Accuracy" scenario: +transcribing the real "Crossing the Bar" recording end to end through the real, installed +Zipformer and Nemotron models and asserting the resulting Word Error Rate against the poem's +ground-truth text stays within tolerance; the "Session-End Reset Buffered-Audio Regression" +scenario (`Reset_AbandonedUtteranceWithNoTrailingSilence_DoesNotBleedIntoNextSession`) described +above; and the "Trailing-Audio Flush Recovery" scenario (`TryFlush_AbandonedUtteranceWithNoTrailingSilence_RecoversTrailingWords`) described above. -"Pipeline: Fault Containment" also covers -`SherpaOnnxSpeechRecognizer_Stop_EngineResetFails_CompletesReportsFaultAndPermitsRestart` (a -faulting engine `Reset()` during teardown is reported, `Stop()` still completes, and a later -`Start()` is still permitted) and, for the flush, -`SherpaOnnxSpeechRecognizer_Stop_EngineFlushFails_CompletesResetsEngineAndReportsFault`. -"Pipeline: Lifecycle and Draining" also covers -`SherpaOnnxSpeechRecognizer_Stop_EngineHasFlushableTrailingAudio_RaisesFlushedFinalResult`, -`SherpaOnnxSpeechRecognizer_Dispose_EngineHasFlushableTrailingAudio_RaisesFlushedFinalResult`, and -`SherpaOnnxSpeechRecognizer_Stop_EngineHasNothingToFlush_RaisesNoExtraResult`. diff --git a/docs/verification/speech/recognition-subsystem/speech-recognizer-factory.md b/docs/verification/speech/recognition-subsystem/speech-recognizer-factory.md index 5c926a2..450a76d 100644 --- a/docs/verification/speech/recognition-subsystem/speech-recognizer-factory.md +++ b/docs/verification/speech/recognition-subsystem/speech-recognizer-factory.md @@ -3,7 +3,7 @@ #### Verification Approach `SpeechRecognizerFactory` is verified through unit tests that inject a fake -`IRecognitionEngineFactory`, so every composition branch can be exercised without a downloaded +`IRecognitionBackendFactory`, so every composition branch can be exercised without a downloaded model or a native speech-inference runtime. Model installation is simulated with a real scratch directory created and removed per test instance, so the "is it installed?" check is exercised against the file system rather than a mock. The store-based overload is verified against a real @@ -12,9 +12,10 @@ against the file system rather than a mock. The store-based overload is verified through the store rather than hard-coded, and the catalog-based overload is verified against a `SpeechModelCatalog` wrapping that same real store (via the internal test constructor with an empty known-model list), proving `catalog.Store` genuinely resolves through to the same store -rather than a second, disconnected one. Capture-device availability is supplied by an -NSubstitute `IAudioCaptureDevice` and by the shared unavailable capture device. Diagnostics are -verified with an NSubstitute sink where the reported reason matters. +rather than a second, disconnected one. Diagnostics are verified with an NSubstitute sink where +the reported reason matters. Because `LoadAsync` is now `async`, every outcome is observed by +awaiting the returned task and, for the fault cases, asserting the specific exception type the +task faults with, rather than a synchronous throw. #### Test Environment @@ -25,29 +26,25 @@ verified with an NSubstitute sink where the reported reason matters. #### Acceptance Criteria -Composition is considered verified when a real recognizer is returned only for the fully -available case, every unavailable state returns the shared fallback without throwing, no engine -is loaded once an earlier check has failed, an engine load failure is caught and reported, null -arguments throw, an optional `parameterValues` bag supplied by the caller reaches the recognition +Composition is considered verified when a real engine is returned only for the fully available +case, every unavailable state returns the shared fallback without faulting the returned task, no +backend is loaded once an earlier check has failed, a backend load failure is caught and +reported, null arguments and a pre-cancelled token fault the returned task with the documented +exception type, an optional `parameterValues` bag supplied by the caller reaches the recognition model's own engine-configuration logic unchanged, an unrecognized `parameterValues` key composes successfully with only an `Info` diagnostic reported, and an invalid value for a parameter the -model does declare throws `ArgumentException` synchronously from `Create()`. +model does declare faults the returned task with `ArgumentException`. #### Test Scenarios -See the RecognitionSubsystem-level scenarios "Composition: Real Recognizer for an Installed Model -and Available Device", "Composition: Honest Fallback for Every Unavailable State", "Composition: -Null Arguments Are Programming Errors", "Composition: Parameter Value Bag Forwarding", and +See the RecognitionSubsystem-level scenarios "Composition: Real Engine for an Installed Model", +"Composition: Honest Fallback for Every Unavailable State", "Composition: Null Arguments and +Cancellation Are Programming/Caller Errors", "Composition: Parameter Value Bag Forwarding", and "Composition: Parameter Value Validation". #### Reuse and Concurrent Pre-Warming -This factory's "construct once, reuse across turns" and "safe to call `Create` concurrently from -a background task" guidance (see the design doc) is a documentation contract about the factory's -own statelessness, not independently testable behavior of `Create` itself: the factory holds no -state to race on. It is exercised end-to-end by -`DemaConsulting.Speech.Cli.Tests.Commands.ConversationCommandSubsystem.AskCommandTests -.AskCommand_Run_PrewarmsRecognizerConcurrentlyWithPlayback_CreatesRecognizerBeforePlaybackCompletes`, -which proves a host (the CLI's `ask` command) genuinely calls `Create` from a background task -while other work (Phase 1's synthesis/playback) proceeds concurrently, and that the resulting -recognizer is adopted correctly once both complete. +This factory's "load once, reuse across many sessions" and "safe to call `LoadAsync` concurrently +from a background task" guidance (see the design doc) is a documentation contract about the +factory's own statelessness, not independently testable behavior of `LoadAsync` itself: the +factory holds no state to race on. diff --git a/docs/verification/speech/recognition-subsystem/unavailable-recognition-session.md b/docs/verification/speech/recognition-subsystem/unavailable-recognition-session.md new file mode 100644 index 0000000..c47a4ef --- /dev/null +++ b/docs/verification/speech/recognition-subsystem/unavailable-recognition-session.md @@ -0,0 +1,25 @@ +### UnavailableRecognitionSession + +#### Verification Approach + +`UnavailableRecognitionSession` is verified directly against its shared instance; it has no +dependencies to mock. `SpeechRecognizerUnavailableException` is verified through its three +constructors. The safe-no-op test deliberately disposes the shared instance twice and then +re-reads it, proving that a host wrapping its session in a disposal scope cannot invalidate the +fallback for anyone else. + +#### Test Environment + +- **Framework**: xUnit v3 running under the .NET SDK +- No additional setup beyond the standard test runner. + +#### Acceptance Criteria + +The unit is considered verified when the shared instance reports `false` availability and +`Created` state, throws `SpeechRecognizerUnavailableException` from `StartAsync` and the first +`MoveNextAsync` of `GetResultsAsync`, treats `StopAsync` and repeated disposal as safe no-ops, and +the exception type exposes the message and inner exception supplied to each of its constructors. + +#### Test Scenarios + +See the RecognitionSubsystem-level scenario "Unavailable Fallback: Honest Degradation". diff --git a/docs/verification/speech/recognition-subsystem/unavailable-speech-recognizer-engine.md b/docs/verification/speech/recognition-subsystem/unavailable-speech-recognizer-engine.md new file mode 100644 index 0000000..b250620 --- /dev/null +++ b/docs/verification/speech/recognition-subsystem/unavailable-speech-recognizer-engine.md @@ -0,0 +1,22 @@ +### UnavailableSpeechRecognizerEngine + +#### Verification Approach + +`UnavailableSpeechRecognizerEngine` is verified directly against its shared instance; it has no +dependencies to mock. + +#### Test Environment + +- **Framework**: xUnit v3 running under the .NET SDK +- No additional setup beyond the standard test runner. + +#### Acceptance Criteria + +The unit is considered verified when the shared instance reports `false` availability, always +returns the shared `UnavailableRecognitionSession.Instance` from `CreateSessionAsync` regardless +of the supplied device, rejects a null device with `ArgumentNullException`, and treats repeated +disposal as a safe no-op. + +#### Test Scenarios + +See the RecognitionSubsystem-level scenario "Unavailable Fallback: Honest Degradation". diff --git a/docs/verification/speech/recognition-subsystem/unavailable-speech-recognizer.md b/docs/verification/speech/recognition-subsystem/unavailable-speech-recognizer.md deleted file mode 100644 index f1ae217..0000000 --- a/docs/verification/speech/recognition-subsystem/unavailable-speech-recognizer.md +++ /dev/null @@ -1,25 +0,0 @@ -### UnavailableSpeechRecognizer - -#### Verification Approach - -`UnavailableSpeechRecognizer` is verified directly against its shared instance; it has no -dependencies to mock. `SpeechRecognizerUnavailableException` is verified through its three -constructors. The safe-no-op test deliberately disposes the shared instance twice and then -re-reads it, proving that a host wrapping its recognizer in a disposal scope cannot invalidate -the fallback for anyone else. - -#### Test Environment - -- **Framework**: xUnit v3 running under the .NET SDK -- No additional setup beyond the standard test runner. - -#### Acceptance Criteria - -The unit is considered verified when the shared instance reports `false` availability, throws -`SpeechRecognizerUnavailableException` from both operational members, treats result-event -subscription and repeated disposal as no-ops, and the exception type exposes the message and -inner exception supplied to each of its constructors. - -#### Test Scenarios - -See the RecognitionSubsystem-level scenario "Unavailable Fallback: Honest Degradation". diff --git a/docs/verification/speech/synthesis-subsystem.md b/docs/verification/speech/synthesis-subsystem.md index a249b2d..1b886c9 100644 --- a/docs/verification/speech/synthesis-subsystem.md +++ b/docs/verification/speech/synthesis-subsystem.md @@ -2,24 +2,42 @@ ### Verification Approach -The SynthesisSubsystem's Sub-phase 4a scope - the closed Natural Language Audio Tag vocabulary -and the Layer 1 parser - is verified entirely through deterministic, pure unit tests over plain -strings. Neither `AudioTagCatalog` nor `AudioTagParser` depends on a model, an inference engine, -an audio device, or any native runtime, so no test double is required at this stage: every test -calls the production catalog/parser directly and asserts on their return values. - -Sub-phase 4b, described below, is verified the same way the RecognitionSubsystem is: a fake -`ISynthesisEngine`/`ISynthesisEngineFactory` pair and an NSubstitute `IAudioPlaybackDevice` stand -in for the native sherpa-onnx runtime and real speakers, making Layer 2 rendering, chunking, -pipelined synthesize-while-play behavior, cancellation, fault containment, and honest degradation -fully testable without a downloaded speech model or physical audio hardware. Determinism is -structural, not timing-based: the bounded channel between the producer and the caller's -enumeration guarantees ordering without polling or sleeping. +The SynthesisSubsystem's Layer 1 scope - the closed Natural Language Audio Tag vocabulary and the +Layer 1 parser - is verified entirely through deterministic, pure unit tests over plain strings. +Neither `AudioTagCatalog` nor `AudioTagParser` depends on a model, an inference engine, an audio +device, or any native runtime, so no test double is required at this stage: every test calls the +production catalog/parser directly and asserts on their return values. This remains unchanged by +the Engine/Session redesign. + +The Layer 3/Layer 5 Engine/Session API is verified the same way the RecognitionSubsystem's own +Engine/Session redesign is: a fake `ISynthesisBackend`/`ISynthesisBackendFactory` pair and an +NSubstitute `IAudioPlaybackDevice` stand in for the native sherpa-onnx runtime and real speakers, +making Layer 2 rendering, chunking, per-operation synthesize-then-play behavior, engine +exclusivity leasing, the session state machine, the overlap rule, cancellation, fault +containment, and honest degradation fully testable without a downloaded speech model or physical +audio hardware. Determinism is structural, not timing-based: a deterministic fake backend plus +`SemaphoreSlim`-based test doubles let a test assert ordering, in-flight state, and fault +transitions without polling or sleeping. + +The lease/exclusivity behavior is verified by driving `SherpaOnnxSpeechSynthesizerEngine`'s real +`CreateSessionAsync`/`DisposeAsync` logic against a fake backend, asserting that a second +concurrent session request fails fast with `SynthesisEngineBusyException` rather than hanging or +queueing, and that disposing a leased session (or the engine itself, with a session still active) +releases the lease so a subsequent request succeeds. The overlap rule and the full +`SynthesisSessionState` transition sequence are verified against the real +`SherpaOnnxSynthesisSession`, asserting the documented `Created → Starting → Running → Stopping → +Stopped` sequence for one successful operation (and the `Faulted` terminal transition for a +failing one) via the `StateChanged` event, and that a second call made while an operation is +already in flight on the same session throws `InvalidOperationException` immediately. The +`SynthesizeAsync` full-fidelity segment list is verified by asserting its returned +`IReadOnlyList` includes a pure-silence segment for a rendered pause tag, not +only the segments containing synthesized speech. Automated coverage **does not** include synthesizing real, intelligible speech. Proving that real text produces correct audible speech through a real model on real speakers requires a downloaded -production model (which this phase deliberately does not ship) and audio hardware, so it remains -a manual/local verification activity, mirroring the RecognitionSubsystem's identical boundary. +production model (which this library deliberately does not ship by default) and audio hardware, +so it remains a manual/local verification activity, mirroring the RecognitionSubsystem's +identical boundary. ### Test Environment @@ -27,12 +45,12 @@ a manual/local verification activity, mirroring the RecognitionSubsystem's ident - **Execution**: `dotnet test` invoked by `build.ps1` and the CI pipeline - **Dependencies**: None - no external services, no downloaded model, no native runtime, no audio hardware -- **Test doubles**: None for the Sub-phase 4a vocabulary/parser tests. For Sub-phase 4b: a fake - `ISynthesisEngine`/`ISynthesisEngineFactory` pair, NSubstitute playback devices and diagnostics +- **Test doubles**: None for the vocabulary/parser tests. For the Engine/Session pipeline: a fake + `ISynthesisBackend`/`ISynthesisBackendFactory` pair, NSubstitute playback devices (including a + semaphore-signaling `PendingSampleCount` stub for the genuine-drain-wait test) and diagnostics sinks, and a fake synthesis model -- **Isolation**: Every Sub-phase 4a test is a pure function call with no shared or persisted - state; Sub-phase 4b composition tests create and delete their own scratch installed-model - directory +- **Isolation**: Every vocabulary/parser test is a pure function call with no shared or persisted + state; composition tests create and delete their own scratch installed-model directory ### Acceptance Criteria @@ -55,34 +73,46 @@ A SynthesisSubsystem test run passes when: every non-pause tag; plain text is split into chunk-sized segments - `SentenceChunker.Chunk` splits on primary sentence-ending punctuation first, then unconditionally splits every resulting sentence-level piece further on secondary clause - punctuation (commas, semicolons, colons) regardless of length, except where that punctuation - is flanked by a digit on both sides (a time such as `12:30` or a thousands separator such as - `1,000`); falls back to a whitespace-budget split only when a piece is still over budget; - merges a degenerate, word-less punctuation piece onto the preceding chunk (or drops it when - there is no preceding chunk); never splits a single word; and rejects a non-positive budget or - null text. `ChunkWithMetadata` additionally flags, per chunk, whether the chunk's text ends in - a genuine ellipsis -- Composition returns a real synthesizer only when the model is installed, declares the - synthesis role, the playback device is available, and the engine loads; every other outcome - returns the honest unavailable synthesizer without throwing + punctuation, falls back to a whitespace-budget split only when still over budget, merges + degenerate word-less pieces onto the preceding chunk, never splits a single word, and rejects a + non-positive budget or null text. `ChunkWithMetadata` additionally flags, per chunk, whether the + chunk's text ends in a genuine ellipsis +- `SpeechSynthesizerFactory.LoadAsync` returns a real engine only when the model is installed, + declares the synthesis role, and the backend loads; every other outcome returns the honest + unavailable engine without throwing - A supplied `parameterValues` key naming a parameter not declared by the requested model is silently ignored (with only an `Info` diagnostic reported) and composition still succeeds; a supplied value for a parameter the model *does* declare that fails that parameter's own - validation (wrong CLR type, out-of-range or non-integral for a `NumericParameter`, an invalid - option for a `ChoiceParameter`, a non-`bool` for a `BooleanParameter`) throws - `ArgumentException` synchronously from `Create()`, before any installed/role/device/engine - check runs -- Text flows through chunking, rendering, and the engine to yield ordered audio segments, played - in order with correct pre/post silence, while a later chunk synthesizes during an earlier - chunk's playback -- A long, multi-sentence input still yields segments in the correct order at the increased - look-ahead capacity of 8 pending segments, and the single background producer never calls - `ISynthesisEngine.Generate` concurrently -- `Stop()` cancels an in-flight session deterministically and is a safe no-op when idle; engine - faults and an unavailable playback device fail the caller's task honestly rather than hanging -- `PlaybackAudioResampler` resamples and upmixes engine-rate mono audio to the playback device's + validation throws `ArgumentException` synchronously from `LoadAsync()`, before any + installed/role/backend-load check runs +- `ISpeechSynthesizerEngine.CreateSessionAsync` succeeds when no lease is held and fails fast with + `SynthesisEngineBusyException` when one is; disposing a leased session or the engine releases + the lease so a subsequent request succeeds again +- A session's `SpeakAsync`/`SynthesizeAsync` calls never overlap on the same instance - a second + call while one is in flight throws `InvalidOperationException` - and the session can be reused + across many sequential calls without reconstruction +- A session's `StateChanged` event reports the documented `Created → Starting → Running → + Stopping → Stopped` sequence for one successful operation, and transitions to the terminal + `Faulted` state (reported through diagnostics) for a failing one; every subsequent operation on + a faulted session throws `SynthesisSessionFaultedException` wrapping the original fault +- `SynthesizeAsync` returns the full, ordered `IReadOnlyList` segment list, + including a pure-silence segment for a rendered pause, without requiring a playback device +- Text flows through chunking, rendering, and the backend to yield ordered audio segments, played + in order with correct pre/post silence +- A long, multi-sentence input still yields segments in the correct order, and the backend is + never called concurrently +- `StopAsync` cancels an in-flight operation deterministically and is a safe no-op when idle; + backend faults and an unavailable playback device fail the caller's task honestly rather than + hanging +- A session-level `parameterValues` bag resolves to the correct speaker id via + `ISynthesisModel.ResolveSpeakerId` once per segment, coexisting correctly with an independent + per-segment Natural Language Audio Tag speed override in the same call +- `DedicatedWorker` completes promptly on cooperative cancellation, and abandons (reporting a + diagnostic and still returning control to its caller) a non-cooperative delegate after its + timeout, always running the delegate on a long-running task +- `PlaybackAudioResampler` resamples and upmixes backend-rate mono audio to the playback device's resolved format -- The unavailable synthesizer stays honest and safe to hold and dispose +- The unavailable engine and unavailable session both stay honest and safe to hold and dispose - The automated verification boundary remains honest about the absence of real-speech coverage ### Test Scenarios @@ -197,160 +227,214 @@ punctuation (commas, semicolons, colons) regardless of whether the piece is stil every clause becomes its own chunk; falls back to a whitespace budget split only when a piece is still over-length after both punctuation passes; never splits a single over-length word; and rejects a non-positive budget or null text. Also verifies that clause punctuation flanked by a -digit on both sides (a time such as `12:30` or a thousands separator such as `1,000`, including a -decimal point such as `0.5`) is never treated as a boundary, so numerals stay intact, while a -comma, semicolon, or colon still splits normally when only one side is a digit. Also verifies that a +digit on both sides is never treated as a boundary, so numerals stay intact, while a comma, +semicolon, or colon still splits normally when only one side is a digit. Also verifies that a decimal point followed immediately by a digit is kept attached to its numeral even without a -leading digit - a bare-fraction decimal such as `.5` or `$.99`, at start-of-text, after -whitespace, a sign, or a currency symbol - so it is not mistaken for a sentence-ending period and -dropped as a degenerate, word-less chunk (which would otherwise silence the "point" when the -number is spoken); a period directly preceded by a letter (e.g. `Wait.5 more.`) does not qualify -for this exception and still splits as an ordinary sentence boundary. Also verifies that a maximal -run of consecutive primary -sentence-ending characters (an ellipsis `...`, or mixed terminators such as `?!`/`!!`) is treated -as a single boundary and stays attached to the preceding sentence as one piece - rather than -producing degenerate single-punctuation-character chunks that cause audible synthesis glitches - -both mid-text and at the very end of the text, while a normal single-terminator sentence is -unaffected. Also verifies that a degenerate, word-less punctuation piece (produced by either -punctuation pass, e.g. a whitespace-spaced ellipsis or a lone comma) is merged onto the -immediately preceding non-empty chunk, or dropped entirely when it occurs at the very start of the -text with no preceding chunk to merge into. Finally, verifies that `ChunkWithMetadata` correctly -flags a chunk as ending in a genuine ellipsis (three or more consecutive `.` characters, adjacent -or whitespace-spaced) and not for fewer than three, and that `Chunk` produces exactly the same -chunk text as `ChunkWithMetadata`. - -#### Composition: Real Synthesizer for an Installed Model and Available Device - -**Tests**: `SpeechSynthesizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer`, -`SpeechSynthesizerFactory_Create_WithStoreModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer`, -`SpeechSynthesizerFactory_Create_WithCatalogModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer` - -Verifies that an installed synthesis model plus an available playback device composes a real -synthesizer wired to the injected engine factory, with the installed-model directory passed -through unchanged, whether that directory is supplied directly as a `string`, resolved from a -`SpeechModelStore`, or resolved from a `SpeechModelCatalog`'s own store. +leading digit, and that a maximal run of consecutive primary sentence-ending characters is treated +as a single boundary and stays attached to the preceding sentence as one piece. Also verifies that +a degenerate, word-less punctuation piece is merged onto the immediately preceding non-empty +chunk, or dropped entirely when it occurs at the very start of the text. Finally, verifies that +`ChunkWithMetadata` correctly flags a chunk as ending in a genuine ellipsis and not for fewer than +three dots, and that `Chunk` produces exactly the same chunk text as `ChunkWithMetadata`. + +#### Composition: Real Engine for an Installed Model + +**Tests**: `SpeechSynthesizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine`, +`SpeechSynthesizerFactory_LoadAsync_WithStoreModelInstalled_ReturnsRealEngine`, +`SpeechSynthesizerFactory_LoadAsync_WithCatalogModelInstalled_ReturnsRealEngine` + +Verifies that an installed synthesis model composes a real engine wired to the injected backend +factory, with the installed-model directory passed through unchanged, whether that directory is +supplied directly as a `string`, resolved from a `SpeechModelStore`, or resolved from a +`SpeechModelCatalog`'s own store. #### Composition: Honest Fallback for Every Unavailable State -**Tests**: `SpeechSynthesizerFactory_Create_ModelNotInstalled_ReturnsUnavailableSynthesizer`, -`SpeechSynthesizerFactory_Create_PlaybackDeviceUnavailable_ReturnsUnavailableSynthesizer`, -`SpeechSynthesizerFactory_Create_ModelRoleIsNotSynthesis_ReturnsUnavailableSynthesizer`, -`SpeechSynthesizerFactory_Create_EngineLoadFails_ReturnsUnavailableSynthesizerAndDoesNotThrow`, -`SpeechSynthesizerFactory_Create_WithStoreModelNotInstalled_ReturnsUnavailableSynthesizer`, -`SpeechSynthesizerFactory_Create_WithCatalogModelNotInstalled_ReturnsUnavailableSynthesizer` - -Verifies that a missing model, an unavailable device, a wrong-role model, and a failed engine -load all degrade to the shared unavailable synthesizer without throwing. - -#### Composition: Null Arguments Are Programming Errors - -**Tests**: `SpeechSynthesizerFactory_Create_NullModel_ThrowsArgumentNullException`, -`SpeechSynthesizerFactory_Create_NullPlaybackDevice_ThrowsArgumentNullException`, -`SpeechSynthesizerFactory_Create_WithStoreNullModel_ThrowsArgumentNullException`, -`SpeechSynthesizerFactory_Create_WithStoreNullStore_ThrowsArgumentNullException`, -`SpeechSynthesizerFactory_Create_WithStoreNullPlaybackDevice_ThrowsArgumentNullException`, -`SpeechSynthesizerFactory_Create_WithCatalogNullModel_ThrowsArgumentNullException`, -`SpeechSynthesizerFactory_Create_WithCatalogNullCatalog_ThrowsArgumentNullException`, -`SpeechSynthesizerFactory_Create_WithCatalogNullPlaybackDevice_ThrowsArgumentNullException` - -Verifies that a null model, playback device, store, or catalog throws, distinguishing a -programming error from an ordinary machine state. - -#### Composition: Voice Selection Value Bag Forwarding - -**Test**: `SpeechSynthesizerFactory_Create_ParameterValuesSupplied_ForwardedToSynthesizer` - -Verifies that an optional `parameterValues` bag supplied by the caller (for example, a chosen -voice built from a declared `ChoiceParameter`) is forwarded unchanged to the constructed -synthesizer. - -#### Composition: Parameter Value Validation - -**Tests**: `SpeechSynthesizerFactory_Create_UnrecognizedParameterId_ComposesAndReportsInfo`, -`SpeechSynthesizerFactory_Create_RecognizedNumericParameterOutOfRange_Throws`, -`SpeechSynthesizerFactory_Create_RecognizedNumericParameterWrongType_Throws`, -`SpeechSynthesizerFactory_Create_RecognizedIntegerParameterFractionalValue_Throws`, -`SpeechSynthesizerFactory_Create_RecognizedChoiceParameterInvalidOption_Throws` - -Verifies the deliberate, breaking-change split introduced for this behavior: a supplied -`parameterValues` key naming a parameter the model does not declare still composes a real -synthesizer and reports only an `Info` diagnostic, never throwing (preserving cross-model -compatibility); a supplied value for a parameter the model *does* declare, but that is invalid -for it, throws `ArgumentException` synchronously from `Create()` - before any -installed/role/device/engine check runs - naming the parameter id, the model id, and the specific -reason the value is invalid. This replaces this library's earlier behavior of silently -substituting a default the first time `ResolveSpeakerId` ran per segment, and does not change -`ResolveSpeakerId`'s or `ResolveOverrideRatios`'s own existing never-throw, per-segment runtime -contract. - -#### Pipeline: Chunked Synthesis and Ordered Playback - -**Tests**: `SynthesizeStreamAsync_PlainText_YieldsAudioSegment`, -`SynthesizeStreamAsync_TextWithPauseTag_YieldsSilenceSegmentWithNoEngineCall`, -`PlayStreamAsync_OrderedSegments_StartsWritesInOrderAndStops` - -Verifies that plain text yields a synthesized audio segment, that a pause tag yields a -pure-silence segment without invoking the engine, and that ordered segments are started, written -in order, and stopped on the playback device. - -#### Pipeline: Ordered, Strictly Sequential Production at the Increased Look-Ahead Capacity - -**Test**: `SynthesizeStreamAsync_LongMultiSentenceInput_ProducesOrderedSegmentsSequentially` - -Verifies the `PendingSegmentCapacity` tuning change from `5` to `8` (raised because unconditional -clause-punctuation splitting now yields more, smaller chunks per sentence): for a long, -12-sentence input - producing more chunks than fit in the buffer at once - the yielded segments -still arrive in the exact same order the sentences appear in the source text (cross-checked -against each chunk's expected sample count from `FakeSynthesisEngine`), and -`FakeSynthesisEngine`'s concurrency tracking (`MaxConcurrentGenerateCalls`) reports a maximum of -exactly `1`, proving the single background producer never calls `ISynthesisEngine.Generate` -concurrently. The test's fake playback device drains writes instantly, so it cannot itself -distinguish a capacity of `5` from `8`; it verifies ordering and sequential production hold at -scale, while the capacity value itself is simply the constant currently configured in -`SherpaOnnxSpeechSynthesizer`. - -#### Pipeline: Fault Containment - -**Tests**: `SynthesizeStreamAsync_EngineThrows_ReportsFaultAndPropagatesToCaller`, -`PlayStreamAsync_PlaybackDeviceWriteThrows_PropagatesAndStillStopsDevice`, -`PlayStreamAsync_PlaybackDeviceUnavailable_ThrowsRatherThanHanging` - -Verifies that an engine fault is reported and propagated to the caller rather than hanging, that -a playback write failure still stops the device before propagating, and that an unavailable -playback device fails promptly rather than hanging the pipeline. - -#### Pipeline: Genuine Playback Drain Before Stopping - -**Tests**: `PlayStreamAsync_PlaybackDeviceReportsPendingSamples_WaitsForDrainBeforeStopping` - -Verifies the fix for a bug where the TTS panel's status flashed from "Playing" back to "Idle" -almost instantly with no audible sound: `PlayStreamAsync` now polls the playback device's -`PendingSampleCount` and does not stop the device merely because every segment has been -enqueued. Uses an NSubstitute playback device whose `PendingSampleCount` getter signals a -semaphore on every read, so the test deterministically observes the wait has genuinely begun -(rather than racing a sleep) before asserting the awaited task has not completed and `Stop()` has -not been called; only after the test sets the reported pending count to zero does the task -complete and the device get stopped. - -#### Pipeline: Cancellation and Lifecycle - -**Tests**: `Stop_WhileSpeaking_CancelsInFlightSessionOnlyAfterInFlightGenerateReturns`, -`SynthesizeStreamAsync_CancelledMidGenerate_AwaitsProducerBeforeEnumerationCompletesAndDisposalIsSafe`, -`Stop_NoSessionInFlight_IsNoOp`, `Dispose_CalledTwice_DisposesEngineOnce`, -`SynthesizeStreamAsync_AfterDispose_ThrowsObjectDisposedException`, `IsAvailable_Always_ReturnsTrue` - -Verifies that `Stop()` cancels an in-flight session deterministically, is a safe no-op when idle, -disposal releases the engine exactly once even when called twice, operating after disposal is -rejected, and a real synthesizer always reports itself available. Also verifies the fix for a -confirmed `AccessViolationException` crash: `SynthesizeStreamAsync`/`SpeakAsync` never report -completion while the producer's in-flight native `Generate` call is still running, on either the -`Stop()`-driven or the directly-cancelled path, so a caller can never dispose the engine out from -under a still-executing call. A `BlockingSynthesisEngine` test double holds `Generate` open on two -`SemaphoreSlim`s until the test explicitly releases it, letting each test assert the outer task is -still incomplete immediately after cancellation and only completes (with `OperationCanceledException`) -once the in-flight call has genuinely returned; the second test additionally disposes the -synthesizer immediately afterward and asserts no exception, and that only the one expected -`Generate` call was ever made. +**Tests**: `SpeechSynthesizerFactory_LoadAsync_ModelNotInstalled_ReturnsUnavailableEngine`, +`SpeechSynthesizerFactory_LoadAsync_ModelRoleIsNotSynthesis_ReturnsUnavailableEngine`, +`SpeechSynthesizerFactory_LoadAsync_EngineLoadFails_ReturnsUnavailableEngineAndDoesNotThrow`, +`SpeechSynthesizerFactory_LoadAsync_WithStoreModelNotInstalled_ReturnsUnavailableEngine`, +`SpeechSynthesizerFactory_LoadAsync_WithCatalogModelNotInstalled_ReturnsUnavailableEngine` + +Verifies that a missing model, a wrong-role model, and a failed backend load all degrade to the +shared unavailable engine without throwing. + +#### Composition: Null Arguments and Cancellation Are Programming Errors + +**Tests**: `SpeechSynthesizerFactory_LoadAsync_NullModel_ThrowsArgumentNullException`, +`SpeechSynthesizerFactory_LoadAsync_CancelledToken_ThrowsOperationCanceledException`, +`SpeechSynthesizerFactory_LoadAsync_WithStoreNullModel_ThrowsArgumentNullException`, +`SpeechSynthesizerFactory_LoadAsync_WithStoreNullStore_ThrowsArgumentNullException`, +`SpeechSynthesizerFactory_LoadAsync_WithCatalogNullModel_ThrowsArgumentNullException`, +`SpeechSynthesizerFactory_LoadAsync_WithCatalogNullCatalog_ThrowsArgumentNullException` + +Verifies that a null model, store, or catalog throws, and that a cancelled `cancellationToken` +throws `OperationCanceledException`, distinguishing a programming error/explicit cancellation +request from an ordinary machine state. Unlike the former synchronous factory, no +`NullPlaybackDevice` variant exists for any overload: `LoadAsync` no longer takes a playback +device parameter at all. + +#### Composition: Parameter Value Forwarding and Validation + +**Tests**: `SpeechSynthesizerFactory_LoadAsync_ParameterValuesSupplied_ForwardedToEngine`, +`SpeechSynthesizerFactory_LoadAsync_UnrecognizedParameterId_ComposesAndReportsInfo`, +`SpeechSynthesizerFactory_LoadAsync_RecognizedNumericParameterOutOfRange_Throws`, +`SpeechSynthesizerFactory_LoadAsync_RecognizedNumericParameterWrongType_Throws`, +`SpeechSynthesizerFactory_LoadAsync_RecognizedIntegerParameterFractionalValue_Throws`, +`SpeechSynthesizerFactory_LoadAsync_RecognizedChoiceParameterInvalidOption_Throws` + +Verifies that an optional `parameterValues` bag supplied by the caller is forwarded unchanged to +the constructed engine; that a supplied `parameterValues` key naming a parameter the model does +not declare still composes a real engine and reports only an `Info` diagnostic, never throwing +(preserving cross-model compatibility); and that a supplied value for a parameter the model *does* +declare, but that is invalid for it, throws `ArgumentException` synchronously from `LoadAsync()` - +before any installed/role/backend-load check runs - naming the parameter id, the model id, and the +specific reason the value is invalid. + +#### Engine: Session Exclusivity and Lease Lifecycle + +**Tests**: `SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_NoLeaseHeld_ReturnsRealSession`, +`SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_LeaseAlreadyHeld_ThrowsSynthesisEngineBusyException`, +`SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_AfterPriorSessionDisposed_SucceedsAgain`, +`SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException`, +`SherpaOnnxSpeechSynthesizerEngine_IsAvailable_Always_ReturnsTrue`, +`SherpaOnnxSpeechSynthesizerEngine_DisposeAsync_CalledTwice_DisposesBackendOnce`, +`SherpaOnnxSpeechSynthesizerEngine_DisposeAsync_WithActiveLeasedSession_DisposesSessionFirst` + +Verifies that `CreateSessionAsync` succeeds immediately when no lease is held; fails fast with +`SynthesisEngineBusyException` (never queueing or waiting) when a lease is already held by a +still-undisposed session; succeeds again once that prior session is disposed, proving the lease is +released exactly once from the session's own `DisposeAsync`; rejects a null device; always reports +itself available; disposes its owned backend exactly once across repeated `DisposeAsync` calls; +and, when disposed while a session is still actively leased, disposes that session first (which +also releases the lease) before disposing the backend. + +#### Engine: One-Shot Convenience Overloads + +**Tests**: `SherpaOnnxSpeechSynthesizerEngine_SpeakAsync_CalledTwice_CreatesAndDisposesASessionEachTime`, +`SherpaOnnxSpeechSynthesizerEngine_SynthesizeAsync_NoDeviceSupplied_ReturnsSegments` + +Verifies that the engine's one-shot `SpeakAsync` convenience overload creates and disposes a +fresh session per call (so repeated calls never conflict with the exclusivity lease), and that +`SynthesizeAsync` returns synthesized segments without requiring a playback device at all. + +#### Session: Lifecycle State Machine and StateChanged Event + +**Tests**: `SherpaOnnxSynthesisSession_StateChanged_OneSuccessfulOperation_RaisesExpectedTransitionsInOrder`, +`SherpaOnnxSynthesisSession_StateChanged_HandlerThrows_IsIsolatedAndDoesNotPropagate`, +`SherpaOnnxSynthesisSession_IsAvailable_BeforeAndAfterDispose_ReflectsLifecycle` + +Verifies that one successful `SpeakAsync`/`SynthesizeAsync` call raises the documented +`Created → Starting → Running → Stopping → Stopped` sequence via `StateChanged`, in order; that a +subscriber's handler exception is caught and routed to diagnostics rather than propagated or +destabilizing the session; and that `IsAvailable` correctly reflects the session's lifecycle +before and after disposal. + +#### Session: Overlap Rule - No Concurrent Operations + +**Tests**: `SherpaOnnxSynthesisSession_SpeakAsync_CalledWhileAlreadySpeaking_ThrowsInvalidOperationException`, +`SherpaOnnxSynthesisSession_SynthesizeAsync_CalledWhileAlreadySpeaking_ThrowsInvalidOperationException` + +Verifies that `SpeakAsync`/`SynthesizeAsync` never overlap on the same session instance: a second +call of either method made while one is already in flight throws `InvalidOperationException` +immediately rather than queueing or waiting, for every combination of the two methods. + +#### Session: Hot Reuse Across Repeated Calls + +**Test**: `SherpaOnnxSynthesisSession_SpeakAsync_CalledTwiceOnSameInstance_ReusesSameInstanceWithoutReconstruction` + +Verifies the key Engine/Session redesign goal: a session returns to `Stopped` after one operation +completes and can be reused for a second, sequential `SpeakAsync` call on the very same instance, +with no reconstruction - closing the class of bug where a host reconstructed a synthesizer per +utterance. + +#### Session: Faulted Is Terminal + +**Tests**: `SherpaOnnxSynthesisSession_SynthesizeAsync_BackendThrows_ReportsFaultAndTransitionsToFaulted`, +`SherpaOnnxSynthesisSession_SynthesizeAsync_AfterFault_ThrowsSynthesisSessionFaultedException` + +Verifies that a non-cancellation failure during an operation reports the fault through +diagnostics and transitions the session to the terminal `Faulted` state, and that every +subsequent operation on that same session throws `SynthesisSessionFaultedException` wrapping the +original fault rather than attempting to run again. + +#### Session: SynthesizeAsync Full-Fidelity Segment List + +**Test**: `SherpaOnnxSynthesisSession_SynthesizeAsync_ReturnsFullFidelitySegmentListIncludingSilence` + +Verifies that `SynthesizeAsync` returns the complete, ordered segment list the rendered +`SpeechPlan` produced, including a pure-silence `SynthesizedSpeech` segment for a rendered pause +tag - not only the segments containing synthesized speech - so a caller that saves or otherwise +processes the returned segments gets a faithful, lossless reconstruction of the plan. + +#### Session: Chunked Synthesis and Ordered Playback + +**Tests**: `SherpaOnnxSynthesisSession_SynthesizeAsync_PlainText_YieldsAudioSegment`, +`SherpaOnnxSynthesisSession_SpeakAsync_PlainText_StartsWritesAndStopsDevice`, +`SherpaOnnxSynthesisSession_SynthesizeAsync_LongMultiSentenceInput_ProducesOrderedSegmentsSequentially` + +Verifies that plain text yields a synthesized audio segment, that `SpeakAsync` starts the +playback device, writes segments in order, and stops the device, and that a long, multi-sentence +input still yields segments in the correct order with the backend never called concurrently. + +#### Session: Fault Containment + +**Tests**: `SherpaOnnxSynthesisSession_SpeakAsync_PlaybackDeviceWriteThrows_PropagatesAndStillStopsDevice`, +`SherpaOnnxSynthesisSession_SpeakAsync_PlaybackDeviceUnavailable_ThrowsRatherThanHanging`, +`SherpaOnnxSynthesisSession_SpeakAsync_PlaybackDeviceStartThrows_StillCallsStop` + +Verifies that a playback write failure still stops the device before propagating, that an +unavailable playback device fails promptly rather than hanging, and that a failure to start the +device still results in a stop attempt during teardown. + +#### Session: Genuine Playback Drain Before Stopping + +**Test**: `SherpaOnnxSynthesisSession_SpeakAsync_PlaybackDeviceReportsPendingSamples_WaitsForDrainBeforeStopping` + +Verifies that `SpeakAsync` genuinely waits for the playback device to report a drained queue +before stopping it, rather than stopping as soon as every segment has been written. Uses an +NSubstitute playback device whose `PendingSampleCount` getter signals a semaphore on every read, +so the test deterministically observes the wait has genuinely begun (rather than racing a sleep) +before asserting the awaited task has not completed and `Stop()` has not been called; only after +the test sets the reported pending count to zero does the task complete and the device get +stopped. + +#### Session: Cancellation and Disposal Lifecycle + +**Tests**: `SherpaOnnxSynthesisSession_StopAsync_WhileSpeaking_CancelsInFlightOperationOnlyAfterInFlightGenerateReturns`, +`SherpaOnnxSynthesisSession_StopAsync_NoOperationInFlight_IsNoOp`, +`SherpaOnnxSynthesisSession_DisposeAsync_CalledTwice_ReleasesLeaseOnce`, +`SherpaOnnxSynthesisSession_SynthesizeAsync_AfterDispose_ThrowsObjectDisposedException` + +Verifies that `StopAsync` cancels an in-flight operation deterministically, only after the +in-flight `DedicatedWorker`-routed `Generate` call has genuinely returned (never orphaning it); is +a safe no-op when no operation is in flight; releases the engine's exclusivity lease exactly once +across repeated `DisposeAsync` calls; and that operating on a disposed session throws +`ObjectDisposedException`. + +#### Session: Voice/Speaker Selection + +**Tests**: `SherpaOnnxSynthesisSession_SynthesizeAsync_NoParameterValues_ResolvesDefaultSpeakerIdFromModel`, +`SherpaOnnxSynthesisSession_SynthesizeAsync_ParameterValuesSupplied_ResolvesSpeakerIdFromBag`, +`SherpaOnnxSynthesisSession_SynthesizeAsync_ParameterValuesSuppliedAlongsideSpeedTag_BothMechanismsApplyIndependently` + +Verifies that a session with no supplied `parameterValues` resolves to the model's own default +speaker id, that a supplied `parameterValues` bag resolves to the correct speaker id via +`ISynthesisModel.ResolveSpeakerId`, and that a session-level selected voice and an independent, +per-segment `[fast]` Natural Language Audio Tag speed override both apply correctly in the same +call, proving the two mechanisms coexist without either regressing the other. + +#### DedicatedWorker: Cooperative Cancellation and Non-Cooperative Abandonment + +**Tests**: `DedicatedWorker_Run_CooperativeCancellation_CompletesPromptly`, +`DedicatedWorker_Run_NonCooperativeDelegate_AbandonsAfterTimeoutAndReportsDiagnostics`, +`DedicatedWorker_Run_UsesLongRunningTaskCreationOption` + +Verifies that a delegate honoring cancellation promptly lets `Run` complete promptly rather than +waiting out its full abandon timeout; that a delegate which never observes cancellation is +abandoned once the (injectable, test-shortened) abandon timeout elapses, with a `Warning` +diagnostic reported and the awaited call still returning control to its caller as cancelled; and +that the delegate always runs on a `TaskCreationOptions.LongRunning` task. #### Playback Format Conversion: Resampling, Anti-Aliasing, and Upmix @@ -375,16 +459,26 @@ fast path that skips the upmix step entirely. #### Unavailable Fallback: Honest Degradation -**Tests**: `UnavailableSpeechSynthesizer_IsAvailable_Read_ReturnsFalse`, -`UnavailableSpeechSynthesizer_SynthesizeStreamAsync_Always_ThrowsSpeechSynthesizerUnavailableException`, -`UnavailableSpeechSynthesizer_PlayStreamAsync_Always_ThrowsSpeechSynthesizerUnavailableException`, -`UnavailableSpeechSynthesizer_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException`, -`UnavailableSpeechSynthesizer_Stop_Always_ThrowsSpeechSynthesizerUnavailableException`, -`UnavailableSpeechSynthesizer_Dispose_CalledTwice_DoesNotThrow`, +**Tests**: `UnavailableSpeechSynthesizerEngine_IsAvailable_Read_ReturnsFalse`, +`UnavailableSpeechSynthesizerEngine_CreateSessionAsync_Always_ReturnsUnavailableSession`, +`UnavailableSpeechSynthesizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException`, +`UnavailableSpeechSynthesizerEngine_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException`, +`UnavailableSpeechSynthesizerEngine_SynthesizeAsync_Always_ThrowsSpeechSynthesizerUnavailableException`, +`UnavailableSpeechSynthesizerEngine_DisposeAsync_CalledTwice_DoesNotThrow`, +`UnavailableSynthesisSession_IsAvailable_Read_ReturnsFalse`, +`UnavailableSynthesisSession_State_Read_ReturnsCreated`, +`UnavailableSynthesisSession_StateChanged_SubscribeAndUnsubscribe_DoesNotThrow`, +`UnavailableSynthesisSession_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException`, +`UnavailableSynthesisSession_SpeakAsync_NullText_ThrowsArgumentNullException`, +`UnavailableSynthesisSession_SynthesizeAsync_Always_ThrowsSpeechSynthesizerUnavailableException`, +`UnavailableSynthesisSession_SynthesizeAsync_NullText_ThrowsArgumentNullException`, +`UnavailableSynthesisSession_StopAsync_NoSessionInFlight_IsNoOp`, +`UnavailableSynthesisSession_DisposeAsync_CalledTwice_DoesNotThrow`, `SpeechSynthesizerUnavailableException_Constructor_WithMessage_ExposesMessage`, `SpeechSynthesizerUnavailableException_Constructor_WithInnerException_ExposesBoth`, `SpeechSynthesizerUnavailableException_Constructor_Default_HasNonEmptyMessage` -Verifies that the shared fallback stays honest and safe to hold and dispose, that operational -misuse throws the documented exception, and that the exception conforms to the standard -three-constructor pattern. +Verifies that both the shared unavailable engine and the shared unavailable session stay honest +and safe to hold and dispose, that a device bound to an unavailable engine still succeeds (since +binding is an ordinary composition, not an error), that operational misuse throws the documented +exception, and that the exception conforms to the standard three-constructor pattern. diff --git a/docs/verification/speech/synthesis-subsystem/i-speech-synthesizer-engine.md b/docs/verification/speech/synthesis-subsystem/i-speech-synthesizer-engine.md new file mode 100644 index 0000000..8d85fff --- /dev/null +++ b/docs/verification/speech/synthesis-subsystem/i-speech-synthesizer-engine.md @@ -0,0 +1,30 @@ +### ISpeechSynthesizerEngine + +#### Verification Approach + +`ISpeechSynthesizerEngine` is verified indirectly through both shipped implementations: +`UnavailableSpeechSynthesizerEngine` proves the honest-unavailable path, and +`SherpaOnnxSpeechSynthesizerEngine` proves availability reporting, session exclusivity leasing, +and the one-shot `SpeakAsync`/`SynthesizeAsync` convenience overloads. The `SynthesizedSpeech` +value type is verified through the segments observed from `SynthesizeAsync`, since it carries no +behavior of its own. + +#### Test Environment + +- **Framework**: xUnit v3 running under the .NET SDK +- No additional setup beyond the standard test runner; no model, speakers, or native + speech-inference runtime is required. + +#### Acceptance Criteria + +The contract is considered verified when the unavailable implementation reports `false` +availability and throws on every operational member (while `CreateSessionAsync` itself still +succeeds, returning an unavailable session), and when the real implementation reports itself +available, enforces single-session exclusivity (`SynthesisEngineBusyException` on a concurrent +lease request), allows a new session once a prior one is disposed, and its one-shot convenience +overloads create and dispose a session per call. + +#### Test Scenarios + +See the SynthesisSubsystem-level scenarios "Engine: Session Exclusivity and Lease Lifecycle", +"Engine: One-Shot Convenience Overloads", and "Unavailable Fallback: Honest Degradation". diff --git a/docs/verification/speech/synthesis-subsystem/i-speech-synthesizer.md b/docs/verification/speech/synthesis-subsystem/i-speech-synthesizer.md deleted file mode 100644 index 441cd84..0000000 --- a/docs/verification/speech/synthesis-subsystem/i-speech-synthesizer.md +++ /dev/null @@ -1,30 +0,0 @@ -### ISpeechSynthesizer - -#### Verification Approach - -`ISpeechSynthesizer` is verified indirectly through both shipped implementations: -`UnavailableSpeechSynthesizer` proves the honest-unavailable path, and -`SherpaOnnxSpeechSynthesizer` proves availability reporting, chunked/pipelined synthesis and -playback, and cancellation. The `SynthesizedSpeech` value type is verified through the segments -observed from `SynthesizeStreamAsync`, since it carries no behavior of its own. - -#### Test Environment - -- **Framework**: xUnit v3 running under the .NET SDK -- No additional setup beyond the standard test runner; no model, speakers, or native - speech-inference runtime is required. - -#### Acceptance Criteria - -The contract is considered verified when the unavailable implementation reports `false` -availability and throws on every operational member, and when the real implementation reports -itself available, yields ordered synthesized segments with correct pre/post silence, plays them -in order, cancels an in-flight session deterministically on `Stop()`, and supports many -independent synthesize/play sessions on the same instance without needing to be reconstructed. - -#### Test Scenarios - -See the SynthesisSubsystem-level scenarios "Pipeline: Chunked Synthesis and Ordered Playback", -"Pipeline: Cancellation and Lifecycle", and "Unavailable Fallback: Honest Degradation". The -"construct once, reuse across many turns" contract is verified directly by -`PlayStreamAsync_CalledTwiceOnSameInstance_ReusesSameInstanceWithoutReconstruction`. diff --git a/docs/verification/speech/synthesis-subsystem/i-synthesis-session.md b/docs/verification/speech/synthesis-subsystem/i-synthesis-session.md new file mode 100644 index 0000000..d9273ad --- /dev/null +++ b/docs/verification/speech/synthesis-subsystem/i-synthesis-session.md @@ -0,0 +1,37 @@ +### ISynthesisSession + +#### Verification Approach + +`ISynthesisSession` is verified indirectly through both shipped implementations: +`UnavailableSynthesisSession` proves the honest-unavailable path, and `SherpaOnnxSynthesisSession` +proves the state machine, the overlap rule, hot reuse across calls, the terminal `Faulted` state, +`SynthesizeAsync`'s full-fidelity segment list, cancellation, and fault containment. +`SynthesisSessionState` and `SessionStateChangedEventArgs` are verified through the `StateChanged` +transition sequence observed during real operations, since neither carries independent behavior +of its own. + +#### Test Environment + +- **Framework**: xUnit v3 running under the .NET SDK +- No additional setup beyond the standard test runner; no model, speakers, or native + speech-inference runtime is required. + +#### Acceptance Criteria + +The contract is considered verified when the unavailable implementation reports `false` +availability, always reports `State` as `Created`, and throws on every operational member; and +when the real implementation reports itself available until disposed/faulted, never allows two +operations to overlap, reuses correctly across repeated calls, raises the documented state +sequence via `StateChanged`, transitions to and stays `Faulted` after a non-cancellation failure +(with every subsequent call throwing `SynthesisSessionFaultedException`), and +`SynthesizeAsync`/`SpeakAsync` behave correctly under cancellation, playback failure, and +playback drain. + +#### Test Scenarios + +See the SynthesisSubsystem-level scenarios "Session: Lifecycle State Machine and StateChanged +Event", "Session: Overlap Rule - No Concurrent Operations", "Session: Hot Reuse Across Repeated +Calls", "Session: Faulted Is Terminal", "Session: SynthesizeAsync Full-Fidelity Segment List", +"Session: Chunked Synthesis and Ordered Playback", "Session: Fault Containment", "Session: +Genuine Playback Drain Before Stopping", "Session: Cancellation and Disposal Lifecycle", "Session: +Voice/Speaker Selection", and "Unavailable Fallback: Honest Degradation". diff --git a/docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md b/docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md new file mode 100644 index 0000000..c52e26c --- /dev/null +++ b/docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer-engine.md @@ -0,0 +1,35 @@ +### SherpaOnnxSpeechSynthesizerEngine + +#### Verification Approach + +`SherpaOnnxSpeechSynthesizerEngine` is verified with a deterministic fake +`ISynthesisBackend`/`ISynthesisBackendFactory` pair standing in for the real sherpa-onnx backend, +so no downloaded model and no platform-specific native speech-inference binary is ever needed. Its +exclusivity lease is verified by driving two sequential `CreateSessionAsync` calls - the second +before the first session is disposed, then again after - and asserting +`SynthesisEngineBusyException` in the first case and success in the second. Disposal ordering +(active leased session disposed before the owned backend) is verified by asserting the fake +backend's dispose call is observed only after the session's own teardown has run. + +#### Test Environment + +- **Framework**: xUnit v3 running under the .NET SDK +- **Execution**: `dotnet test` invoked by `build.ps1` and the CI pipeline +- **Dependencies**: No external services, no downloaded model, no native speech-inference + runtime, and no physical audio hardware +- **Test doubles**: A fake `ISynthesisBackend`/`ISynthesisBackendFactory` pair and an NSubstitute + `IAudioPlaybackDevice` + +#### Acceptance Criteria + +The unit is considered verified when `CreateSessionAsync` succeeds immediately with no lease +held, fails fast with `SynthesisEngineBusyException` while a lease is held, succeeds again once +the leased session is disposed, rejects a null device, always reports itself available, disposes +its backend exactly once across repeated `DisposeAsync` calls, and disposes an active leased +session before disposing the backend. The one-shot `SpeakAsync`/`SynthesizeAsync` convenience +overloads are verified to create and dispose a fresh session per call. + +#### Test Scenarios + +See the SynthesisSubsystem-level scenarios "Engine: Session Exclusivity and Lease Lifecycle" and +"Engine: One-Shot Convenience Overloads". diff --git a/docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md b/docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md deleted file mode 100644 index 797667c..0000000 --- a/docs/verification/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.md +++ /dev/null @@ -1,105 +0,0 @@ -### SherpaOnnxSpeechSynthesizer - -#### Verification Approach - -The streaming synthesizer, the Layer 2 rendering strategy, the sentence chunker, the -synthesis-engine seam, and the playback-format converter are verified together, since the -synthesizer's whole purpose is to move text through rendering and chunking, into the engine, and -the resulting audio through resampling onto the playback device, while pipelining synthesis with -playback. - -The engine seam is replaced by a deterministic fake that records every `Generate` call and -returns a fixed, predictable amount of audio, so no downloaded model and no platform-specific -native speech-inference binary is ever needed. The playback device is an NSubstitute -`IAudioPlaybackDevice`, letting tests assert the exact ordered sequence of `Start`/`Write`/`Stop` -calls via `Received.InOrder`. Diagnostics are an NSubstitute sink, so fault containment is -verified by observing what was reported rather than by asserting an absence of exceptions alone. - -The genuine-drain-wait fix (`PlayStreamAsync` no longer treats "every segment enqueued" as -"finished playing") is verified the same deterministic way: the NSubstitute playback device's -`PendingSampleCount` getter is stubbed with a callback that signals a semaphore on every read, so -the test can `await` that semaphore to know the drain-wait loop has genuinely started polling - -no sleep, and no race between the test and the implementation - before asserting the awaited -`PlayStreamAsync` task is not yet complete and `Stop()` has not yet been called. - -Cancellation is verified deterministically, not by timing: a `BlockingSynthesisEngine` test -double uses `SemaphoreSlim`s to hold the producer mid-segment until the test explicitly releases -it, letting a test assert that neither `SpeakAsync` nor `SynthesizeStreamAsync`'s enumeration -completes while `Generate` is still in flight - proving the producer task is never orphaned - and -then, once released, that the pipeline unwound via cancellation rather than completing normally -and that disposing the synthesizer immediately afterward is safe. This is the regression coverage -for a fixed `AccessViolationException` crash: `SynthesizeStreamCore` previously could return -control to its caller (who could then dispose the owned engine) while the producer's native -`Generate` call was still genuinely running on a background thread; the assertions themselves -synchronize deterministically via semaphores and awaited tasks, with no polling-based sleeps or -waits anywhere in this coverage. Each test does carry a `[Fact(Timeout = ...)]` attribute, but only -as a safety-net deadlock guard that fails the test fast if the fix ever regressed, not as part of -the synchronization logic. - -A further test proves the fix for the crash cannot itself hang: it drives a fast, non-blocking -fake engine through more sentences than the producer's bounded look-ahead capacity, consumes only -the first synthesized segment, then abandons enumeration by disposing the enumerator directly - -exactly what the compiler's `await foreach` cleanup does when a consumer's loop body throws for an -unrelated reason, without the stream's own `cancellationToken` ever being cancelled - and asserts -that disposal still completes promptly rather than hanging on the producer's now-permanently-full -channel write. This is regression coverage for a hang that an earlier, narrower version of the fix -could otherwise have introduced. - -Speaker-id resolution is verified against a `FakeSynthesisModel` whose injectable -`resolveSpeakerId` delegate lets a test assert exactly which speaker id -`ISynthesisModel.ResolveSpeakerId` returns for a given `parameterValues` bag, and that the fake -engine's recorded `Generate` call received that same id - proving the resolved value genuinely -reaches the engine call rather than merely being computed and discarded. A further test supplies -both a `parameterValues` bag and a `[fast]` Natural Language Audio Tag on the same input, -asserting that the resolved speaker id and the tag-driven speed override both apply correctly and -independently in the same call, proving the two mechanisms coexist without either regressing the -other. - -`DefaultModelCapabilityProfile`, `SentenceChunker`, and `PlaybackAudioResampler` are each verified -separately as pure functions over plain inputs (a `StubSpeechModel` test double for the -capability-profile tests, plain strings for the chunker, and plain float arrays with a -tolerance-based comparer for the resampler), so their behavior can be asserted directly rather -than only observed indirectly through the pipeline. - -The real sherpa-onnx engine adapter itself is **not** covered by automated tests: exercising it -requires loading a real model through the native runtime, which is manual/local verification. The -seam is what keeps that uncovered surface as small as possible - it contains only the interop -calls, with all policy above it fully covered. - -#### Test Environment - -- **Framework**: xUnit v3 running under the .NET SDK -- **Execution**: `dotnet test` invoked by `build.ps1` and the CI pipeline -- **Dependencies**: No external services, no downloaded model, no native speech-inference - runtime, and no physical audio hardware -- **Test doubles**: Fake `ISynthesisEngine`/`ISynthesisEngineFactory` implementations (including - a semaphore-driven blocking variant for cancellation), NSubstitute playback devices (including - a semaphore-signaling `PendingSampleCount` stub for the genuine-drain-wait test) and - diagnostics sinks, and a `StubSpeechModel` test double for capability-profile tests - -#### Acceptance Criteria - -The units are considered verified when text is chunked, rendered, and synthesized into ordered -audio segments; a pause tag yields silence without an engine call; segments are played in order -with correct pre/post silence while a later chunk synthesizes during an earlier chunk's playback; -`PlayStreamAsync` genuinely waits for the playback device to report a drained queue before -stopping it, rather than stopping as soon as every segment has been enqueued; `Stop()` cancels an -in-flight session deterministically and is a safe no-op when idle; `SynthesizeStreamAsync` never -returns control to its caller while the producer's in-flight `Generate` call is still running, on -every exit path (normal completion, cancellation, or any other exception), so a caller can never -dispose the engine out from under a still-executing native call; engine faults and an unavailable -playback device fail the caller's task honestly rather than hanging; a playback write failure -still stops the device; a session's `parameterValues` bag resolves to the correct speaker -id via `ISynthesisModel.ResolveSpeakerId` once per segment, coexisting correctly with an -independent per-segment Natural Language Audio Tag speed override in the same call; and -`PlaybackAudioResampler` produces the documented output for -identity, up/down conversion, upmix, and boundary inputs while rejecting a non-positive channel -count. - -#### Test Scenarios - -See the SynthesisSubsystem-level scenarios "Layer 2 Rendering: Pauses, Native, Parameter-Mapped, -and Stripped Tags", "Chunking: Boundary Splitting and Edge Cases", "Pipeline: Chunked Synthesis -and Ordered Playback", "Pipeline: Fault Containment", "Pipeline: Genuine Playback Drain Before -Stopping", "Pipeline: Cancellation and Lifecycle", "Pipeline: Voice/Speaker Selection", and -"Playback Format Conversion: Resampling and Upmix". diff --git a/docs/verification/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md b/docs/verification/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md new file mode 100644 index 0000000..f7dc482 --- /dev/null +++ b/docs/verification/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.md @@ -0,0 +1,125 @@ +### SherpaOnnxSynthesisSession + +#### Verification Approach + +The real session, the Layer 2 rendering strategy, the sentence chunker, the synthesis-backend +seam, the dedicated-worker cancellation policy, and the playback-format converter are verified +together, since the session's whole purpose is to move text through rendering and chunking, into +the backend (via a dedicated worker), and the resulting audio through resampling onto the bound +playback device. + +The backend seam is replaced by a deterministic fake that records every `Generate` call and +returns a fixed, predictable amount of audio, so no downloaded model and no platform-specific +native speech-inference binary is ever needed. The playback device is an NSubstitute +`IAudioPlaybackDevice`, letting tests assert the exact ordered sequence of `Start`/`Write`/`Stop` +calls via `Received.InOrder`. Diagnostics are an NSubstitute sink, so fault containment is +verified by observing what was reported rather than by asserting an absence of exceptions alone. + +The session's `SynthesisSessionState` lifecycle is verified by subscribing to `StateChanged` and +asserting the exact ordered sequence of transitions raised for one successful operation +(`Created → Starting → Running → Stopping → Stopped`), and separately for a faulting operation +(ending in `Faulted`, reported through diagnostics). A handler that throws from `StateChanged` is +asserted to be isolated: the exception is caught and reported through diagnostics rather than +propagated or allowed to destabilize the session. The overlap rule is verified deterministically +with a `BlockingSynthesisEngine`-style test double that holds the first call's `Generate` open on +a `SemaphoreSlim` until the test releases it, letting a test assert that a second +`SpeakAsync`/`SynthesizeAsync` call made while the first is still in flight throws +`InvalidOperationException` immediately. Hot reuse across calls is verified by calling +`SpeakAsync` twice in sequence on the same session instance and asserting both calls succeed with +no reconstruction. The terminal `Faulted` state is verified by making the fake backend throw, +asserting the session transitions to `Faulted` and the fault is reported, and then asserting a +further call throws `SynthesisSessionFaultedException` wrapping that same fault. +`SynthesizeAsync`'s full-fidelity segment list is verified by rendering text containing a pause +tag and asserting the returned list includes a pure-silence segment for it, not only +speech-bearing segments. + +The genuine-drain-wait behavior (not treating "every segment written" as "finished playing") is +verified deterministically: the NSubstitute playback device's `PendingSampleCount` getter is +stubbed with a callback that signals a semaphore on every read, so the test can `await` that +semaphore to know the drain-wait loop has genuinely started polling - no sleep, and no race +between the test and the implementation - before asserting the awaited `SpeakAsync` task is not +yet complete and `Stop()` has not yet been called. + +Cancellation is verified deterministically, not by timing: a `BlockingSynthesisEngine` test +double uses `SemaphoreSlim`s to hold the backend call open mid-segment until the test explicitly +releases it, letting a test assert that `StopAsync` cancels the in-flight operation only after the +in-flight `Generate` call (routed through `DedicatedWorker`) has genuinely returned - proving the +native call is never orphaned - and then, once released, that the operation unwound via +cancellation rather than completing normally and that disposing the session immediately +afterward is safe. + +Speaker-id resolution is verified against a `FakeSynthesisModel` whose injectable +`resolveSpeakerId` delegate lets a test assert exactly which speaker id +`ISynthesisModel.ResolveSpeakerId` returns for a given `parameterValues` bag, and that the fake +backend's recorded `Generate` call received that same id - proving the resolved value genuinely +reaches the backend call rather than merely being computed and discarded. A further test supplies +both a `parameterValues` bag and a `[fast]` Natural Language Audio Tag on the same input, +asserting that the resolved speaker id and the tag-driven speed override both apply correctly and +independently in the same call, proving the two mechanisms coexist without either regressing the +other. + +`DedicatedWorker` is verified directly and in isolation, with an injectable abandon timeout so +the non-cooperative-abandonment test does not need to wait out the real two-second default: a +cooperative delegate (one that observes and honors its cancellation token promptly) lets `Run` +complete promptly on cancellation, while a non-cooperative delegate (one that ignores +cancellation) is abandoned once the shortened timeout elapses, reporting a `Warning` diagnostic +through an NSubstitute sink while still returning control to its caller as cancelled; a separate +test asserts the delegate always runs on a `TaskCreationOptions.LongRunning` task by inspecting +the running thread's characteristics from inside the delegate. + +`DefaultModelCapabilityProfile`, `SentenceChunker`, and `PlaybackAudioResampler` are each verified +separately as pure functions over plain inputs (a `StubSpeechModel` test double for the +capability-profile tests, plain strings for the chunker, and plain float arrays with a +tolerance-based comparer for the resampler), so their behavior can be asserted directly rather +than only observed indirectly through the session. + +The real sherpa-onnx engine adapter itself is **not** covered by automated tests: exercising it +requires loading a real model through the native runtime, which is manual/local verification. The +seam is what keeps that uncovered surface as small as possible - it contains only the interop +calls, with all policy above it fully covered. + +#### Test Environment + +- **Framework**: xUnit v3 running under the .NET SDK +- **Execution**: `dotnet test` invoked by `build.ps1` and the CI pipeline +- **Dependencies**: No external services, no downloaded model, no native speech-inference + runtime, and no physical audio hardware +- **Test doubles**: Fake `ISynthesisBackend`/`ISynthesisBackendFactory` implementations + (including a semaphore-driven blocking variant for cancellation/overlap tests), NSubstitute + playback devices (including a semaphore-signaling `PendingSampleCount` stub for the + genuine-drain-wait test) and diagnostics sinks, and a `StubSpeechModel`/`FakeSynthesisModel` + test double for capability-profile and speaker-id-resolution tests + +#### Acceptance Criteria + +The units are considered verified when text is chunked, rendered, and synthesized into ordered +audio segments; a pause tag yields silence without a backend call; `SpeakAsync` plays segments in +order with correct pre/post silence; `SpeakAsync` genuinely waits for the playback device to +report a drained queue before stopping it; `StopAsync` cancels an in-flight operation +deterministically, only after the in-flight `Generate` call has genuinely returned, and is a safe +no-op when idle; a second `SpeakAsync`/`SynthesizeAsync` call while one is already in flight +throws `InvalidOperationException`; a session is reusable across repeated calls without +reconstruction; the `StateChanged` event raises the documented transition sequence for both a +successful and a faulting operation, with a subscriber's handler exception isolated; a faulted +session throws `SynthesisSessionFaultedException` on every subsequent call; +`SynthesizeAsync` returns the full-fidelity segment list including silence; backend faults and an +unavailable playback device fail the caller's task honestly rather than hanging; a playback write +failure still stops the device; a session's `parameterValues` bag resolves to the correct speaker +id via `ISynthesisModel.ResolveSpeakerId` once per segment, coexisting correctly with an +independent per-segment Natural Language Audio Tag speed override in the same call; +`DedicatedWorker` completes promptly on cooperative cancellation and abandons a non-cooperative +delegate after its timeout while always using a long-running task; and `PlaybackAudioResampler` +produces the documented output for identity, up/down conversion, upmix, and boundary inputs while +rejecting a non-positive channel count. + +#### Test Scenarios + +See the SynthesisSubsystem-level scenarios "Layer 2 Rendering: Pauses, Native, Parameter-Mapped, +and Stripped Tags", "Chunking: Boundary Splitting and Edge Cases", "Session: Lifecycle State +Machine and StateChanged Event", "Session: Overlap Rule - No Concurrent Operations", "Session: +Hot Reuse Across Repeated Calls", "Session: Faulted Is Terminal", "Session: SynthesizeAsync +Full-Fidelity Segment List", "Session: Chunked Synthesis and Ordered Playback", "Session: Fault +Containment", "Session: Genuine Playback Drain Before Stopping", "Session: Cancellation and +Disposal Lifecycle", "Session: Voice/Speaker Selection", "DedicatedWorker: Cooperative +Cancellation and Non-Cooperative Abandonment", and "Playback Format Conversion: Resampling, +Anti-Aliasing, and Upmix". diff --git a/docs/verification/speech/synthesis-subsystem/speech-synthesizer-factory.md b/docs/verification/speech/synthesis-subsystem/speech-synthesizer-factory.md index 62a26ff..090bd3e 100644 --- a/docs/verification/speech/synthesis-subsystem/speech-synthesizer-factory.md +++ b/docs/verification/speech/synthesis-subsystem/speech-synthesizer-factory.md @@ -3,7 +3,7 @@ #### Verification Approach `SpeechSynthesizerFactory` is verified through unit tests that inject a fake -`ISynthesisEngineFactory`, so every composition branch can be exercised without a downloaded +`ISynthesisBackendFactory`, so every composition branch can be exercised without a downloaded model or a native speech-inference runtime. Model installation is simulated with a real scratch directory created and removed per test instance, so the "is it installed?" check is exercised against the file system rather than a mock. The store-based overload is verified against a real @@ -12,9 +12,9 @@ against the file system rather than a mock. The store-based overload is verified through the store rather than hard-coded, and the catalog-based overload is verified against a `SpeechModelCatalog` wrapping that same real store (via the internal test constructor with an empty known-model list), proving `catalog.Store` genuinely resolves through to the same store -rather than a second, disconnected one. Playback-device availability is supplied by an -NSubstitute `IAudioPlaybackDevice` and by the shared unavailable playback device. Diagnostics are -verified with an NSubstitute sink where the reported reason matters. +rather than a second, disconnected one. Diagnostics are verified with an NSubstitute sink where +the reported reason matters. Unlike the former synchronous factory, no playback device is +involved in composition at all - this type no longer takes one. #### Test Environment @@ -25,17 +25,18 @@ verified with an NSubstitute sink where the reported reason matters. #### Acceptance Criteria -Composition is considered verified when a real synthesizer is returned only for the fully -available case, every unavailable state returns the shared fallback without throwing, no engine -is loaded once an earlier check has failed, an engine load failure is caught and reported, null -arguments throw, an optional `parameterValues` bag supplied by the caller is forwarded unchanged -to the constructed synthesizer, an unrecognized `parameterValues` key composes successfully with -only an `Info` diagnostic reported, and an invalid value for a parameter the model does declare -throws `ArgumentException` synchronously from `Create()`. +Composition is considered verified when a real engine is returned only for the fully available +case, every unavailable state returns the shared fallback without throwing, no backend is loaded +once an earlier check has failed, a backend load failure is caught and reported, a null model +throws, a cancelled token throws `OperationCanceledException`, an optional `parameterValues` bag +supplied by the caller is forwarded unchanged to the constructed engine, an unrecognized +`parameterValues` key composes successfully with only an `Info` diagnostic reported, and an +invalid value for a parameter the model does declare throws `ArgumentException` synchronously +from `LoadAsync()`. #### Test Scenarios -See the SynthesisSubsystem-level scenarios "Composition: Real Synthesizer for an Installed Model -and Available Device", "Composition: Honest Fallback for Every Unavailable State", "Composition: -Null Arguments Are Programming Errors", "Composition: Voice Selection Value Bag Forwarding", and -"Composition: Parameter Value Validation". +See the SynthesisSubsystem-level scenarios "Composition: Real Engine for an Installed Model", +"Composition: Honest Fallback for Every Unavailable State", "Composition: Null Arguments and +Cancellation Are Programming Errors", and "Composition: Parameter Value Forwarding and +Validation". diff --git a/docs/verification/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.md b/docs/verification/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.md new file mode 100644 index 0000000..1843258 --- /dev/null +++ b/docs/verification/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.md @@ -0,0 +1,28 @@ +### UnavailableSpeechSynthesizerEngine + +#### Verification Approach + +`UnavailableSpeechSynthesizerEngine` is verified directly against its shared instance; it has no +dependencies to mock. `CreateSessionAsync` is verified to succeed, returning the shared +unavailable session, since binding a device to an already unavailable engine is an ordinary (if +useless) composition, not an error. `SpeechSynthesizerUnavailableException` is verified through +its three constructors. The safe-dispose test deliberately disposes the shared instance twice, +proving that a host wrapping its engine in a disposal scope cannot invalidate the fallback for +anyone else. + +#### Test Environment + +- **Framework**: xUnit v3 running under the .NET SDK +- No additional setup beyond the standard test runner. + +#### Acceptance Criteria + +The unit is considered verified when the shared instance reports `false` availability, always +succeeds at `CreateSessionAsync` (rejecting only a null device), throws +`SpeechSynthesizerUnavailableException` from `SpeakAsync`/`SynthesizeAsync`, treats repeated +disposal as a no-op, and the exception type exposes the message and inner exception supplied to +each of its constructors. + +#### Test Scenarios + +See the SynthesisSubsystem-level scenario "Unavailable Fallback: Honest Degradation". diff --git a/docs/verification/speech/synthesis-subsystem/unavailable-speech-synthesizer.md b/docs/verification/speech/synthesis-subsystem/unavailable-speech-synthesizer.md deleted file mode 100644 index 652835e..0000000 --- a/docs/verification/speech/synthesis-subsystem/unavailable-speech-synthesizer.md +++ /dev/null @@ -1,25 +0,0 @@ -### UnavailableSpeechSynthesizer - -#### Verification Approach - -`UnavailableSpeechSynthesizer` is verified directly against its shared instance; it has no -dependencies to mock. `SpeechSynthesizerUnavailableException` is verified through its three -constructors. The safe-dispose test deliberately disposes the shared instance twice, proving -that a host wrapping its synthesizer in a disposal scope cannot invalidate the fallback for -anyone else. - -#### Test Environment - -- **Framework**: xUnit v3 running under the .NET SDK -- No additional setup beyond the standard test runner. - -#### Acceptance Criteria - -The unit is considered verified when the shared instance reports `false` availability, throws -`SpeechSynthesizerUnavailableException` from every operational member, treats repeated disposal -as a no-op, and the exception type exposes the message and inner exception supplied to each of -its constructors. - -#### Test Scenarios - -See the SynthesisSubsystem-level scenario "Unavailable Fallback: Honest Degradation". diff --git a/docs/verification/speech/synthesis-subsystem/unavailable-synthesis-session.md b/docs/verification/speech/synthesis-subsystem/unavailable-synthesis-session.md new file mode 100644 index 0000000..bbf7651 --- /dev/null +++ b/docs/verification/speech/synthesis-subsystem/unavailable-synthesis-session.md @@ -0,0 +1,25 @@ +### UnavailableSynthesisSession + +#### Verification Approach + +`UnavailableSynthesisSession` is verified directly against its shared instance; it has no +dependencies to mock. `State` is verified to always report `Created`, since this session never +transitions. The safe-dispose test deliberately disposes the shared instance twice, proving that +a host wrapping its session in a disposal scope cannot invalidate the fallback for anyone else. + +#### Test Environment + +- **Framework**: xUnit v3 running under the .NET SDK +- No additional setup beyond the standard test runner. + +#### Acceptance Criteria + +The unit is considered verified when the shared instance reports `false` availability and +`SynthesisSessionState.Created`, throws `SpeechSynthesizerUnavailableException` from +`SpeakAsync`/`SynthesizeAsync` (and `ArgumentNullException` for a null `text`), treats `StopAsync` +as a safe no-op and repeated `DisposeAsync` calls as a no-op, and subscribing/unsubscribing +`StateChanged` never throws. + +#### Test Scenarios + +See the SynthesisSubsystem-level scenario "Unavailable Fallback: Honest Degradation". diff --git a/requirements.yaml b/requirements.yaml index 72020f6..94f77e0 100644 --- a/requirements.yaml +++ b/requirements.yaml @@ -58,16 +58,24 @@ includes: - docs/reqstream/speech/model-management-subsystem/sherpa-onnx-vits-libritts-en-synthesis-model.yaml - docs/reqstream/speech/model-management-subsystem/sherpa-onnx-kokoro-en-synthesis-model.yaml - docs/reqstream/speech/recognition-subsystem.yaml - - docs/reqstream/speech/recognition-subsystem/i-speech-recognizer.yaml + - docs/reqstream/speech/recognition-subsystem/i-speech-recognizer-engine.yaml + - docs/reqstream/speech/recognition-subsystem/i-recognition-session.yaml - docs/reqstream/speech/recognition-subsystem/speech-recognizer-factory.yaml - - docs/reqstream/speech/recognition-subsystem/sherpa-onnx-speech-recognizer.yaml - - docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer.yaml + - docs/reqstream/speech/recognition-subsystem/recognition-session-lease.yaml + - docs/reqstream/speech/recognition-subsystem/sherpa-onnx-recognition-session.yaml + - docs/reqstream/speech/recognition-subsystem/dedicated-worker.yaml + - docs/reqstream/speech/recognition-subsystem/unavailable-speech-recognizer-engine.yaml + - docs/reqstream/speech/recognition-subsystem/unavailable-recognition-session.yaml - docs/reqstream/speech/synthesis-subsystem.yaml - docs/reqstream/speech/synthesis-subsystem/audio-tag-parser.yaml - - docs/reqstream/speech/synthesis-subsystem/i-speech-synthesizer.yaml + - docs/reqstream/speech/synthesis-subsystem/i-speech-synthesizer-engine.yaml + - docs/reqstream/speech/synthesis-subsystem/i-synthesis-session.yaml - docs/reqstream/speech/synthesis-subsystem/speech-synthesizer-factory.yaml - - docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-speech-synthesizer.yaml - - docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer.yaml + - docs/reqstream/speech/synthesis-subsystem/synthesis-session-lease.yaml + - docs/reqstream/speech/synthesis-subsystem/sherpa-onnx-synthesis-session.yaml + - docs/reqstream/speech/synthesis-subsystem/dedicated-worker.yaml + - docs/reqstream/speech/synthesis-subsystem/unavailable-speech-synthesizer-engine.yaml + - docs/reqstream/speech/synthesis-subsystem/unavailable-synthesis-session.yaml - docs/reqstream/speech-demo.yaml - docs/reqstream/speech-demo/shell-subsystem.yaml - docs/reqstream/speech-demo/device-selection-subsystem.yaml diff --git a/src/DemaConsulting.Speech.Cli/Cli/ParameterBagParser.cs b/src/DemaConsulting.Speech.Cli/Cli/ParameterBagParser.cs index 5ff847c..67b71c3 100644 --- a/src/DemaConsulting.Speech.Cli/Cli/ParameterBagParser.cs +++ b/src/DemaConsulting.Speech.Cli/Cli/ParameterBagParser.cs @@ -27,11 +27,11 @@ namespace DemaConsulting.Speech.Cli.Cli; /// Parses and validates the speak/recognize/ask subcommands' repeatable /// --tts-param key=value/--stt-param key=value flags against a resolved model's declared /// set, into the untyped, boxed key-value bag a -/// SpeechSynthesizerFactory.Create or SpeechRecognizerFactory.Create call +/// SpeechSynthesizerFactory.LoadAsync or SpeechRecognizerFactory.LoadAsync call /// expects. /// /// -/// Deliberately stricter than SpeechSynthesizerFactory.Create/SpeechRecognizerFactory.Create's +/// Deliberately stricter than SpeechSynthesizerFactory.LoadAsync/SpeechRecognizerFactory.LoadAsync's /// own "unrecognized key silently ignored, Info-logged" library-level contract: an unrecognized /// --tts-param/--stt-param key is a CLI operator typo, which should fail loudly /// at the command line rather than silently mistune synthesis or recognition. This divergence is diff --git a/src/DemaConsulting.Speech.Cli/Commands/ConversationCommandSubsystem/AskCommand.cs b/src/DemaConsulting.Speech.Cli/Commands/ConversationCommandSubsystem/AskCommand.cs index 8e0e65a..8e6d3f9 100644 --- a/src/DemaConsulting.Speech.Cli/Commands/ConversationCommandSubsystem/AskCommand.cs +++ b/src/DemaConsulting.Speech.Cli/Commands/ConversationCommandSubsystem/AskCommand.cs @@ -70,35 +70,37 @@ namespace DemaConsulting.Speech.Cli.Commands.ConversationCommandSubsystem; /// does not carry the same tag-stripping concern speak's general-purpose text does. /// /// -/// Recognizer pre-warming. Phase 2's recognizer - including resolving its capture -/// device - is constructed concurrently with Phase 1's speak/playback wait, via a background -/// kicked off immediately after both phases' parameters are resolved, -/// rather than only once playback finishes: loading a recognition model into native memory -/// is the expensive step (see 's own remarks), so -/// overlapping that load with the time a person spends listening to the -/// prompt removes an otherwise-unavoidable turnaround gap between finishing speaking and -/// starting to listen. Only the model *load* is pre-warmed: -/// (which begins real microphone capture) is still called only once Phase 2 genuinely -/// begins, so pre-warming never captures audio while the prompt is still being spoken. A -/// pre-warm failure (an unknown/unavailable capture device, an invalid --stt-param -/// value, etc.) is not thrown from the background task itself; it surfaces when RunAsync -/// awaits the pre-warm task at the start of Phase 2, with the same exception type and message -/// a synchronous failure would have produced. +/// Recognizer pre-warming. Phase 2's recognizer engine and session - including +/// resolving its capture device - are constructed concurrently with Phase 1's +/// speak/playback wait, via a background kicked off immediately after both +/// phases' parameters are resolved, rather than only once playback finishes: loading a +/// recognition model into native memory is the expensive step (see +/// 's own remarks), so overlapping that load with the +/// time a person spends listening to the prompt removes an otherwise-unavoidable turnaround +/// gap between finishing speaking and starting to listen. Only the model *load* and session +/// creation are pre-warmed: (which begins real +/// microphone capture) is still called only once Phase 2 genuinely begins, so pre-warming +/// never captures audio while the prompt is still being spoken. A pre-warm failure (an +/// unknown/unavailable capture device, an invalid --stt-param value, etc.) is not +/// thrown from the background task itself; it surfaces when RunAsync awaits the +/// pre-warm task at the start of Phase 2, with the same exception type and message a +/// synchronous failure would have produced. /// /// /// Cancellation and disposal. A single handler, /// installed once for the whole call, cancels a shared -/// (aborting an in-flight SpeakAsync during Phase 1) and, once the recognizer has been -/// constructed, also calls and signals the same +/// (aborting an in-flight SpeakAsync during Phase 1) and, once the session has been +/// constructed, also calls (fire-and-forget, +/// since the handler itself is synchronous) and signals the same /// the mic-wait blocks on, so Ctrl+C cancels /// cleanly whether it lands during playback or during listening. When Phase 1 is canceled or -/// throws, the pre-warmed recognizer - which may already have been constructed by the +/// throws, the pre-warmed engine/session - which may already have been constructed by the /// concurrent background task, or may still be in flight - is never awaited synchronously by /// RunAsync: doing so would block Ctrl+C/fast-failure responsiveness on the /// expensive model-load step finishing. Instead a background continuation observes the -/// pre-warm task's eventual result (or exception) and disposes any recognizer it produces, -/// so RunAsync returns promptly while the recognizer is still guaranteed to be disposed -/// - just asynchronously - rather than leaked or left as an unobserved faulted +/// pre-warm task's eventual result (or exception) and disposes any engine/session it +/// produces, so RunAsync returns promptly while they are still guaranteed to be +/// disposed - just asynchronously - rather than leaked or left as an unobserved faulted /// . Because a final recognition result, a silence/start timeout, and /// Ctrl+C all unblock the same identically, /// Listen re-checks the shared (only ever canceled by @@ -110,10 +112,10 @@ namespace DemaConsulting.Speech.Cli.Commands.ConversationCommandSubsystem; /// writing/printing the recognized text, to close a narrow race where Ctrl+C lands /// after Listen has already unblocked with listenWasCanceled == false but before /// the result is written out; this final check is reported and handled identically to the -/// Phase 1/Phase 2 cancellation cases. The synthesizer, playback device, recognizer, -/// silence-timeout session, and output writer are each disposed exactly once via nested -/// finally blocks, mirroring 's and -/// 's own disposal ordering. +/// Phase 1/Phase 2 cancellation cases. The synthesis engine/session, playback device, +/// recognition engine/session, silence-timeout wrapper, and output writer are each disposed +/// exactly once via nested await using/finally blocks, mirroring +/// 's and 's own disposal ordering. /// /// internal static class AskCommand @@ -202,12 +204,12 @@ internal static void Run( using var cancellationSource = new CancellationTokenSource(); using var stopSignal = new ManualResetEventSlim(initialState: false); - // Tracks the live recognizer (once Phase 2 has constructed one) so a Ctrl+C landing - // during listening can stop it; guarded because the recognizer is created on this same + // Tracks the live session (once Phase 2 has constructed one) so a Ctrl+C landing + // during listening can stop it; guarded because the session is created on this same // thread but a Ctrl+C handler runs on a separate thread and could otherwise observe a // torn/stale reference. - ISpeechRecognizer? recognizer = null; - var recognizerGate = new object(); + IRecognitionSession? session = null; + var sessionGate = new object(); // Ask an in-flight speak/listen session to cancel cooperatively rather than letting the // runtime kill the process outright, so partially-played audio, an in-flight capture, and @@ -218,9 +220,25 @@ internal static void Run( e.Cancel = true; cancellationSource.Cancel(); - lock (recognizerGate) + // Fire-and-forget: this handler must stay synchronous. StopAsync is idempotent and + // safe to call concurrently per IRecognitionSession's own contract. The lock is only + // used to read the torn/stale-reference-safe snapshot; the async call itself is + // started outside the lock. + IRecognitionSession? currentSession; + lock (sessionGate) { - recognizer?.Stop(); + currentSession = session; + } + + if (currentSession is not null) + { + // Intentional fire-and-forget: this handler must stay synchronous (the + // ConsoleCancelEventHandler delegate signature is non-async), and StopAsync is + // idempotent and safe to call concurrently per IRecognitionSession's own + // contract, so neither the result nor any exception needs to be observed here. +#pragma warning disable S1854 + _ = currentSession.StopAsync(CancellationToken.None); +#pragma warning restore S1854 } stopSignal.Set(); @@ -235,11 +253,11 @@ internal static void Run( deviceSource, captureSource, stopSignal, - r => + s => { - lock (recognizerGate) + lock (sessionGate) { - recognizer = r; + session = s; } }, cancellationSource.Token).GetAwaiter().GetResult(); @@ -265,6 +283,15 @@ private readonly record struct AskInvocation( AskOptions Options, CancellationToken CancellationToken); + /// + /// Bundles a pre-warmed Phase 2 recognizer engine and the session created from it, so + /// / can + /// hand both back as one unit without an intermediate tuple at every call site. + /// + /// The loaded recognition engine. Owned by RunAsync once adopted. + /// The session created from , bound to the resolved capture device. + private readonly record struct PrewarmedRecognizer(ISpeechRecognizerEngine Engine, IRecognitionSession Session); + /// /// Runs the ask subcommand's full text-source resolution, model resolution, /// Phase 1 (speak) synthesis/playback, and Phase 2 (listen) capture/recognition logic. @@ -291,7 +318,7 @@ internal static async Task RunAsync( ICliPlaybackDeviceSource deviceSource, ICliCaptureDeviceSource captureSource, ManualResetEventSlim stopSignal, - Action onRecognizerCreated, + Action onRecognizerCreated, CancellationToken cancellationToken) { ArgumentNullException.ThrowIfNull(context); @@ -315,24 +342,25 @@ internal static async Task RunAsync( var invocation = new AskInvocation(context, catalog, options, cancellationToken); - // Kick off Phase 2's recognizer construction (including its capture-device resolution) - // on a background task now, so the expensive model-load step overlaps with Phase 1's - // speak/playback wait below instead of only starting once playback finishes - closing the - // turnaround-gap latency between finishing speaking and starting to listen. Real - // microphone capture (Start()) still only begins once Phase 2 genuinely starts, inside - // Listen. Only the "adopt the pre-warm result" exit path below (Phase 2 genuinely - // starting) awaits this task directly. The "Phase 1 canceled" and "Phase 1 (or anything - // between here and adopting the pre-warm result) threw" exit paths deliberately do NOT - // await this task - blocking RunAsync's return on the expensive model-load finishing - // would delay Ctrl+C/fast-failure responsiveness for no benefit - and instead hand it to - // DisposePrewarmedRecognizerAsync fire-and-forget: that method's own continuation still - // always observes this task's result/exception and disposes any recognizer it produces, - // just asynchronously in the background rather than before RunAsync returns. + // Kick off Phase 2's recognizer engine/session construction (including its + // capture-device resolution) on a background task now, so the expensive model-load step + // overlaps with Phase 1's speak/playback wait below instead of only starting once + // playback finishes - closing the turnaround-gap latency between finishing speaking and + // starting to listen. Real microphone capture (StartAsync) still only begins once Phase + // 2 genuinely starts, inside Listen. Only the "adopt the pre-warm result" exit path + // below (Phase 2 genuinely starting) awaits this task directly. The "Phase 1 canceled" + // and "Phase 1 (or anything between here and adopting the pre-warm result) threw" exit + // paths deliberately do NOT await this task - blocking RunAsync's return on the expensive + // model-load finishing would delay Ctrl+C/fast-failure responsiveness for no benefit - + // and instead hand it to DisposePrewarmedRecognizerAsync fire-and-forget: that method's + // own continuation still always observes this task's result/exception and disposes any + // engine/session it produces, just asynchronously in the background rather than before + // RunAsync returns. var prewarmTask = Task.Run( - () => PrewarmRecognizer(invocation, captureSource, sttDescriptor, sttParameterValues), + () => PrewarmRecognizerAsync(invocation, captureSource, sttDescriptor, sttParameterValues), CancellationToken.None); - ISpeechRecognizer recognizer; + PrewarmedRecognizer prewarmed; try { // Phase 1: speak the prompt through a real playback device and wait for it to finish. @@ -342,9 +370,9 @@ internal static async Task RunAsync( if (wasCanceled) { // Phase 1 was canceled, so Phase 2 never runs: the concurrently pre-warmed - // recognizer (whether already constructed, still in flight, or never going to - // succeed) must still eventually be disposed, but RunAsync must not block its own - // return on the (possibly still-loading) model finishing - that would delay + // engine/session (whether already constructed, still in flight, or never going + // to succeed) must still eventually be disposed, but RunAsync must not block its + // own return on the (possibly still-loading) model finishing - that would delay // Ctrl+C responsiveness for exactly the scenario where instant feedback matters // most. Hand it off fire-and-forget: DisposePrewarmedRecognizerAsync's own // continuation disposes it once construction eventually completes. @@ -353,38 +381,41 @@ internal static async Task RunAsync( } // Adopt the pre-warm task's result now that Phase 2 is genuinely starting: this - // re-throws (with its original type and message) any failure PrewarmRecognizer + // re-throws (with its original type and message) any failure PrewarmRecognizerAsync // raised - an unknown/unavailable capture device, an invalid --stt-param value, // etc. - at Phase 2's start, exactly as a synchronous call to the same logic would // have. Once assigned here, Listen's own finally block takes ownership of disposing - // this recognizer exactly once. - recognizer = await prewarmTask.ConfigureAwait(false); + // this session exactly once; the engine is disposed by RunAsync itself afterward. + prewarmed = await prewarmTask.ConfigureAwait(false); } // Intentionally broad: this is the command-level fail-fast boundary around Phase 1 and // recognizer pre-warm adoption, so any exception must preserve its original type/message - // while still handing off the concurrently created recognizer for asynchronous cleanup. + // while still handing off the concurrently created engine/session for asynchronous + // cleanup. catch (Exception) { // SpeakPromptAsync (or any resolution logic reachable above, e.g. an unknown - // --playback-device, or PrewarmRecognizer's own failure surfacing when its result is - // adopted) threw instead of returning a normal wasCanceled result: the pre-warmed - // recognizer - whether already constructed, still in flight, or never going to - // succeed - must still eventually be disposed and prewarmTask's result/exception must - // still eventually be observed, but RunAsync must not block this fast-failure exit on - // the (possibly still-loading) model finishing. Hand it off fire-and-forget, same as - // the Phase 1 cancellation path above; this is a no-op once it runs if prewarmTask - // never produces a recognizer. Rethrow immediately to preserve the original - // exception's type, message, and stack trace unchanged. + // --playback-device, or PrewarmRecognizerAsync's own failure surfacing when its + // result is adopted) threw instead of returning a normal wasCanceled result: the + // pre-warmed engine/session - whether already constructed, still in flight, or never + // going to succeed - must still eventually be disposed and prewarmTask's + // result/exception must still eventually be observed, but RunAsync must not block + // this fast-failure exit on the (possibly still-loading) model finishing. Hand it off + // fire-and-forget, same as the Phase 1 cancellation path above; this is a no-op once + // it runs if prewarmTask never produces an engine/session. Rethrow immediately to + // preserve the original exception's type, message, and stack trace unchanged. _ = DisposePrewarmedRecognizerAsync(prewarmTask); throw; } - // Phase 2: listen for the reply through the already-constructed recognizer. - var (recognizedText, listenWasCanceled) = Listen( + await using var engineLease = prewarmed.Engine; + + // Phase 2: listen for the reply through the already-constructed session. + var (recognizedText, listenWasCanceled) = await Listen( invocation, - recognizer, + prewarmed.Session, stopSignal, - onRecognizerCreated); + onRecognizerCreated).ConfigureAwait(false); if (listenWasCanceled) { @@ -474,11 +505,13 @@ private static async Task SpeakPromptAsync( } using var playbackDeviceLease = playbackDevice as IDisposable; - using var synthesizer = catalog.CreateSynthesizer(ttsDescriptor, playbackDevice, parameterValues); + await using var engine = await catalog.CreateSynthesizerEngineAsync(ttsDescriptor, parameterValues, cancellationToken) + .ConfigureAwait(false); + await using var session = await engine.CreateSessionAsync(playbackDevice, cancellationToken).ConfigureAwait(false); try { - await synthesizer.SpeakAsync(text, cancellationToken).ConfigureAwait(false); + await session.SpeakAsync(text, cancellationToken).ConfigureAwait(false); return false; } catch (OperationCanceledException) @@ -489,18 +522,18 @@ private static async Task SpeakPromptAsync( } /// - /// Constructs Phase 2's recognizer - resolving a real capture device from - /// (honoring --capture-device) first, since - /// requires an already-constructed + /// Constructs Phase 2's recognition engine and session - resolving a real capture device + /// from (honoring --capture-device) first, since + /// requires an already-constructed /// - without starting real microphone /// capture. Run on a background concurrently with Phase 1's /// speak/playback wait (see the class remarks' "Recognizer pre-warming" paragraph) so the /// expensive model-load step overlaps with the time a person spends listening to the /// prompt, rather than only starting once playback finishes. /// - /// The constructed recognizer, not yet started. + /// The constructed engine and a session bound to the resolved capture device, not yet started. /// Thrown when no real capture device is available. - private static ISpeechRecognizer PrewarmRecognizer( + private static async Task PrewarmRecognizerAsync( AskInvocation invocation, ICliCaptureDeviceSource captureSource, SpeechModelDescriptor sttDescriptor, @@ -508,6 +541,7 @@ private static ISpeechRecognizer PrewarmRecognizer( { var catalog = invocation.Catalog; var options = invocation.Options; + var cancellationToken = invocation.CancellationToken; var knownDevices = captureSource.CaptureProbe.Enumerate(); var selection = DevicesTestCommand.ResolveDeviceSelectionOrThrow(knownDevices, options.CaptureDeviceName, "capture"); @@ -519,12 +553,25 @@ private static ISpeechRecognizer PrewarmRecognizer( "No audio capture device is available on this machine; cannot run 'ask'."); } - return catalog.CreateRecognizer(sttDescriptor, captureDevice, parameterValues); + var engine = await catalog.CreateRecognizerEngineAsync(sttDescriptor, parameterValues, cancellationToken) + .ConfigureAwait(false); + try + { + var session = await engine.CreateSessionAsync(captureDevice, cancellationToken).ConfigureAwait(false); + return new PrewarmedRecognizer(engine, session); + } + catch (Exception) + { + // Session creation failed: the engine would otherwise be leaked since no + // PrewarmedRecognizer is ever returned to adopt it. + await engine.DisposeAsync().ConfigureAwait(false); + throw; + } } /// - /// Fire-and-forget cleanup for a concurrently pre-warmed recognizer, for use when Phase 1 - /// is canceled or throws and Phase 2 will never run: the recognizer + /// Fire-and-forget cleanup for a concurrently pre-warmed recognizer engine/session, for + /// use when Phase 1 is canceled or throws and Phase 2 will never run: the engine/session /// produces (if construction succeeds at all) would /// otherwise be leaked, and the task itself would otherwise go unobserved. /// @@ -535,29 +582,31 @@ private static ISpeechRecognizer PrewarmRecognizer( /// until that load finishes, even though the caller has already decided to exit. Instead, /// this method's own below attaches a continuation that runs /// whenever eventually completes - immediately, if it has - /// already finished by the time this method is called - disposing any recognizer it + /// already finished by the time this method is called - disposing any engine/session it /// produced and observing (via the below) any exception it /// raised, so is never left as an unobserved faulted - /// and a successfully constructed recognizer is never leaked - just + /// and a successfully constructed engine/session is never leaked - just /// disposed asynchronously in the background instead of before the caller returns. /// - private static async Task DisposePrewarmedRecognizerAsync(Task prewarmTask) + private static async Task DisposePrewarmedRecognizerAsync(Task prewarmTask) { try { - using var recognizer = await prewarmTask.ConfigureAwait(false); + var prewarmed = await prewarmTask.ConfigureAwait(false); + await prewarmed.Session.DisposeAsync().ConfigureAwait(false); + await prewarmed.Engine.DisposeAsync().ConfigureAwait(false); } // Intentionally broad: this asynchronous cleanup boundary exists only to observe the - // pre-warm task and dispose any successful recognizer result after the command has + // pre-warm task and dispose any successful engine/session result after the command has // already decided to exit, so a failed pre-warm is non-actionable noise here. catch (Exception) { - // No recognizer was produced, so there is nothing to dispose. + // No engine/session was produced, so there is nothing to dispose. } } /// - /// Runs Phase 2: listens through the already-constructed + /// Runs Phase 2: listens through the already-constructed /// until the first final recognition result arrives, a silence/start timeout fires, or /// Ctrl+C is pressed (signaled externally via ). /// @@ -568,28 +617,28 @@ private static async Task DisposePrewarmedRecognizerAsync(Task and the returned text /// must not be written to --output-text/stdout. /// - private static (string Text, bool WasCanceled) Listen( + private static async Task<(string Text, bool WasCanceled)> Listen( AskInvocation invocation, - ISpeechRecognizer recognizer, + IRecognitionSession session, ManualResetEventSlim stopSignal, - Action onRecognizerCreated) + Action onRecognizerCreated) { var context = invocation.Context; var options = invocation.Options; var cancellationToken = invocation.CancellationToken; - onRecognizerCreated(recognizer); - using var recognizerLease = recognizer; + onRecognizerCreated(session); + await using var sessionLease = session; try { if (stopSignal.IsSet) { // Ctrl+C already landed during Phase 1, during the concurrent recognizer // pre-warm, or in the narrow window between pre-warm finishing and Phase 2 - // starting: skip listening entirely rather than starting a recognizer session - // that would be stopped again immediately. Only report this as a cancellation - // when the shared token confirms Ctrl+C is the reason - the flag is otherwise - // never set before this point. + // starting: skip listening entirely rather than starting a session that would be + // stopped again immediately. Only report this as a cancellation when the shared + // token confirms Ctrl+C is the reason - the flag is otherwise never set before + // this point. if (cancellationToken.IsCancellationRequested) { context.WriteError(SpeechCanceledMessage); @@ -600,76 +649,65 @@ private static (string Text, bool WasCanceled) Listen( } var recognizedText = string.Empty; - EventHandler onResultReceived = (_, e) => - { - if (!e.Result.IsFinal) - { - return; - } - // "ask" models a single bounded question/answer turn: the first final result is - // always enough to end the turn, unlike "recognize --mic"'s open-ended listening. - // Do not call recognizer.Stop()/Dispose() here: ResultReceived is raised from the - // recognizer's own background decoding thread, and Stop() blocks its caller until - // that same thread finishes draining - calling it from this handler would wait on - // itself. Signal stopSignal instead and let the code below stop the recognizer - // from the calling thread once Wait() returns. - recognizedText = e.Result.Text; - stopSignal.Set(); - }; - - recognizer.ResultReceived += onResultReceived; + // A silence-timeout wrapper is always constructed - even when both flags are omitted - + // so a reply that never starts or never finishes cannot block "ask" forever with only + // Ctrl+C as an escape hatch. + var sessionStartTimeout = TimeSpan.FromSeconds(options.StartTimeoutSeconds ?? DefaultStartTimeoutSeconds); + var timeoutSession = new SilenceTimeoutRecognizerSession( + session, + TimeSpan.FromSeconds(options.SilenceTimeoutSeconds ?? DefaultSilenceTimeoutSeconds), + startTimeout: sessionStartTimeout); + + timeoutSession.TimedOut += (_, _) => stopSignal.Set(); + + // StartAsync blocks the calling thread until the session stops (it drives the real + // microphone capture loop), so it is started as a background Task here rather than + // awaited directly - mirroring RecognizeCommand's own mic-mode pump-before-start + // ordering - letting this method concurrently pump GetResultsAsync below. + var startTask = session.StartAsync(cancellationToken); try { - // A silence-timeout session is always constructed - even when both flags are - // omitted - so a reply that never starts or never finishes cannot block "ask" - // forever with only Ctrl+C as an escape hatch. - var sessionStartTimeout = TimeSpan.FromSeconds(options.StartTimeoutSeconds ?? DefaultStartTimeoutSeconds); - using var session = new SilenceTimeoutRecognizerSession( - recognizer, - TimeSpan.FromSeconds(options.SilenceTimeoutSeconds ?? DefaultSilenceTimeoutSeconds), - startTimeout: sessionStartTimeout); - - session.TimedOut += (_, _) => stopSignal.Set(); - - recognizer.Start(); - try - { - stopSignal.Wait(cancellationToken); - } - catch (OperationCanceledException) + await foreach (var result in timeoutSession.GetResultsAsync(cancellationToken).ConfigureAwait(false)) { - // Ctrl+C canceled the shared token while waiting; whether or not stopSignal - // itself has already been set by the same handler is irrelevant here - - // cancellationToken.IsCancellationRequested below is what distinguishes this - // from a legitimate empty result. Stop() is called unconditionally below - // regardless of which of the three ways this wait can end, so nothing - // further is needed here. + if (!result.Result.IsFinal) + { + continue; + } + + // "ask" models a single bounded question/answer turn: the first final result + // is always enough to end the turn, unlike "recognize --mic"'s open-ended + // listening. + recognizedText = result.Result.Text; + stopSignal.Set(); + break; } + } + catch (OperationCanceledException) + { + // Ctrl+C canceled the shared token while pumping results; cancellationToken's own + // IsCancellationRequested below is what distinguishes this from a legitimate + // empty result. + } - // Detach before stopping: Stop() drains and flushes any trailing audio, which can - // raise one more final result. If the handler were still attached, that result - // would pass the IsFinal check and overwrite recognizedText, breaking the - // first-final-wins rule above. The finally block's own detach is then a harmless - // no-op cleanup for the paths that throw before reaching here. - recognizer.ResultReceived -= onResultReceived; - - // Stop() is idempotent and always called here - whether the wait ended because - // of a final result, a silence/start timeout, or Ctrl+C - rather than from - // onResultReceived, because that handler runs on the recognizer's own background - // decoding thread and Stop() blocks its caller until that same thread finishes - // draining; calling it from the handler would wait on itself. Calling it here, - // from this thread, is always safe. - recognizer.Stop(); + // StopAsync is idempotent and always called here - whether the loop ended because of + // a final result, a silence/start timeout (which already called it internally inside + // timeoutSession), or Ctrl+C - so this is always safe, including the already-stopped + // case. + await session.StopAsync(CancellationToken.None).ConfigureAwait(false); + + try + { + await startTask.ConfigureAwait(false); } - finally + catch (OperationCanceledException) { - recognizer.ResultReceived -= onResultReceived; + // Ctrl+C canceled the shared token passed into StartAsync; already handled above. } - // stopSignal.Wait() above unblocks identically for a final result, a silence/start - // timeout, or Ctrl+C: only Ctrl+C ever cancels the shared token, so this is the sole - // reliable way to distinguish a genuine cancellation from a legitimate empty result. + // The loop above can end identically for a final result, a silence/start timeout, or + // Ctrl+C: only Ctrl+C ever cancels the shared token, so this is the sole reliable way + // to distinguish a genuine cancellation from a legitimate empty result. if (cancellationToken.IsCancellationRequested) { context.WriteError(SpeechCanceledMessage); diff --git a/src/DemaConsulting.Speech.Cli/Commands/ModelCommandsSubsystem/ICliModelCatalog.cs b/src/DemaConsulting.Speech.Cli/Commands/ModelCommandsSubsystem/ICliModelCatalog.cs index 7f6f3ad..ee3a84f 100644 --- a/src/DemaConsulting.Speech.Cli/Commands/ModelCommandsSubsystem/ICliModelCatalog.cs +++ b/src/DemaConsulting.Speech.Cli/Commands/ModelCommandsSubsystem/ICliModelCatalog.cs @@ -90,36 +90,40 @@ Task DownloadAsync( AudioFormat GetPreferredAudioFormat(SpeechModelDescriptor descriptor); /// - /// Constructs a real for a resolved synthesis-role model - /// descriptor, playing back through the supplied device. + /// Loads a real for a resolved synthesis-role + /// model descriptor, with no playback device bound yet. /// + /// + /// Loading is the expensive step; callers should load one engine per model/parameter + /// combination and reuse it to create sessions (via + /// ) rather than reloading per + /// turn - mirroring ISpeechSynthesizerEngine's own guidance. + /// /// /// The descriptor identifying the model to load. Must not be null and must describe a /// synthesis-role model. /// - /// The playback device to speak through. Must not be null. /// /// An optional session-level parameter value bag resolved from --param, or /// to use every model's own default parameter values. /// + /// A token to observe for cancellation of the load. /// - /// A real synthesizer when the model is installed, its role is synthesis, and the - /// playback device is available; otherwise an honestly unavailable synthesizer per - /// 's + /// A task that completes with a real engine when the model is installed and its role is + /// synthesis; otherwise an honestly unavailable engine per 's /// own contract. /// - /// - /// Thrown when or is . - /// + /// Thrown when is . /// /// Thrown when does not describe a synthesis-role model, or /// when contains an invalid value for a parameter the /// model declares. /// - ISpeechSynthesizer CreateSynthesizer( + /// Thrown when is cancelled before loading completes. + Task CreateSynthesizerEngineAsync( SpeechModelDescriptor descriptor, - IAudioPlaybackDevice playbackDevice, - IReadOnlyDictionary? parameterValues); + IReadOnlyDictionary? parameterValues, + CancellationToken cancellationToken = default); /// /// Resolves the mono audio format a recognition-role model descriptor's engine requires @@ -136,34 +140,38 @@ ISpeechSynthesizer CreateSynthesizer( AudioFormat GetAudioFormat(SpeechModelDescriptor descriptor); /// - /// Constructs a real for a resolved recognition-role model - /// descriptor, streaming audio from the supplied capture device. + /// Loads a real for a resolved recognition-role + /// model descriptor, with no capture device bound yet. /// + /// + /// Loading is the expensive step; callers should load one engine per model/parameter + /// combination and reuse it to create sessions (via + /// ) rather than reloading per + /// turn - mirroring ISpeechRecognizerEngine's own guidance. + /// /// /// The descriptor identifying the model to load. Must not be null and must describe a /// recognition-role model. /// - /// The capture device to stream audio from. Must not be null. /// /// An optional session-level parameter value bag resolved from --param, or /// to use every model's own default parameter values. /// + /// A token to observe for cancellation of the load. /// - /// A real recognizer when the model is installed, its role is recognition, and the - /// capture device is available; otherwise an honestly unavailable recognizer per - /// 's + /// A task that completes with a real engine when the model is installed and its role is + /// recognition; otherwise an honestly unavailable engine per 's /// own contract. /// - /// - /// Thrown when or is . - /// + /// Thrown when is . /// /// Thrown when does not describe a recognition-role model, or /// when contains an invalid value for a parameter the /// model declares. /// - ISpeechRecognizer CreateRecognizer( + /// Thrown when is cancelled before loading completes. + Task CreateRecognizerEngineAsync( SpeechModelDescriptor descriptor, - IAudioCaptureDevice captureDevice, - IReadOnlyDictionary? parameterValues); + IReadOnlyDictionary? parameterValues, + CancellationToken cancellationToken = default); } diff --git a/src/DemaConsulting.Speech.Cli/Commands/ModelCommandsSubsystem/SpeechModelCatalogAdapter.cs b/src/DemaConsulting.Speech.Cli/Commands/ModelCommandsSubsystem/SpeechModelCatalogAdapter.cs index 51dd148..7f5f814 100644 --- a/src/DemaConsulting.Speech.Cli/Commands/ModelCommandsSubsystem/SpeechModelCatalogAdapter.cs +++ b/src/DemaConsulting.Speech.Cli/Commands/ModelCommandsSubsystem/SpeechModelCatalogAdapter.cs @@ -79,16 +79,15 @@ public AudioFormat GetPreferredAudioFormat(SpeechModelDescriptor descriptor) } /// - public ISpeechSynthesizer CreateSynthesizer( + public Task CreateSynthesizerEngineAsync( SpeechModelDescriptor descriptor, - IAudioPlaybackDevice playbackDevice, - IReadOnlyDictionary? parameterValues) + IReadOnlyDictionary? parameterValues, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(descriptor); - ArgumentNullException.ThrowIfNull(playbackDevice); var synthesisModel = RequireSynthesisModel(descriptor); - return SpeechSynthesizerFactory.Create(synthesisModel, _catalog, playbackDevice, null, parameterValues); + return SpeechSynthesizerFactory.LoadAsync(synthesisModel, _catalog, null, parameterValues, cancellationToken); } /// @@ -123,16 +122,15 @@ public AudioFormat GetAudioFormat(SpeechModelDescriptor descriptor) } /// - public ISpeechRecognizer CreateRecognizer( + public Task CreateRecognizerEngineAsync( SpeechModelDescriptor descriptor, - IAudioCaptureDevice captureDevice, - IReadOnlyDictionary? parameterValues) + IReadOnlyDictionary? parameterValues, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(descriptor); - ArgumentNullException.ThrowIfNull(captureDevice); var recognitionModel = RequireRecognitionModel(descriptor); - return SpeechRecognizerFactory.Create(recognitionModel, _catalog, captureDevice, null, parameterValues); + return SpeechRecognizerFactory.LoadAsync(recognitionModel, _catalog, null, parameterValues, cancellationToken); } /// diff --git a/src/DemaConsulting.Speech.Cli/Commands/RecognitionCommandSubsystem/RecognizeCommand.cs b/src/DemaConsulting.Speech.Cli/Commands/RecognitionCommandSubsystem/RecognizeCommand.cs index 01b5d9b..dfb30a7 100644 --- a/src/DemaConsulting.Speech.Cli/Commands/RecognitionCommandSubsystem/RecognizeCommand.cs +++ b/src/DemaConsulting.Speech.Cli/Commands/RecognitionCommandSubsystem/RecognizeCommand.cs @@ -36,43 +36,46 @@ namespace DemaConsulting.Speech.Cli.Commands.RecognitionCommandSubsystem; /// /// /// File-input mode (--input) requires no explicit "wait until done" loop. -/// calls the supplied capture device's own -/// Start() synchronously (not on a background thread); a +/// calls the bound capture device's own +/// Start() synchronously on the awaiting thread (not on a background thread); a /// 's own Start() is itself fully -/// synchronous and blocking, delivering every frame before returning. This command subscribes -/// its own handler directly to the device instance's own EndOfFileReached event (a -/// member additional to , not the recognizer's internal -/// subscription) and that handler calls recognizer.Stop() - which runs reentrantly, -/// on the same thread, from inside 's own call to the -/// device's Start(), immediately after the last frame is delivered and immediately -/// before the device's own Start() returns. 's -/// drain is a genuine, synchronous block (confirmed by reading -/// SherpaOnnxSpeechRecognizer.StopCore/WaitForConsumer, which calls -/// consumerTask.GetAwaiter().GetResult()), so by the time -/// returns to this command, every result derived from -/// the whole file has already been raised and the recognizer has already fully stopped. No -/// settle-wait or fixed sleep is added anywhere in this design; none is needed. +/// synchronous and blocking, delivering every frame - and raising its own internal +/// EndOfFileReached event - before returning. By the time +/// await session.StartAsync(...) returns to this command in file mode, the whole file +/// has already been delivered into the recognition backend, so this command no longer needs +/// to subscribe to the device's own EndOfFileReached event at all (the previous, +/// pre-redesign version of this type did); it simply follows StartAsync with an +/// explicit await session.StopAsync(...) to drain any remaining buffered-but-not-yet- +/// decoded audio through the backend, exactly mirroring 's +/// own documented drain contract. /// /// -/// Mic-input mode (--mic) blocks the calling thread on a -/// set either by a Ctrl+C handler or by a -/// 's TimedOut event (the session -/// itself already called Stop() before raising that event). A session is always -/// constructed in mic mode, even when --silence-timeout/--start-timeout are -/// both omitted, so listening cannot block forever. The session enforces two distinct idle -/// windows: --start-timeout (defaulting to 8 seconds when omitted) governs the -/// grace period before any result has arrived, and --silence-timeout (defaulting to -/// 5 seconds when omitted) governs every re-arm from the first result onward. +/// Mic-input mode (--mic) relies on a +/// wrapper, which itself calls on an idle timeout - +/// which completes the session's result stream, which in turn ends the await foreach +/// pump below naturally - or on a Ctrl+C handler that stops the session directly. A +/// wrapper is always constructed in mic mode, even when --silence-timeout/ +/// --start-timeout are both omitted, so listening cannot block forever. It enforces two +/// distinct idle windows: --start-timeout (defaulting to 8 seconds when omitted) +/// governs the grace period before any result has arrived, and --silence-timeout +/// (defaulting to 5 seconds when omitted) governs every re-arm from the first result onward. /// /// -/// Disposal. Neither nor -/// nor the real PortAudio-backed -/// capture device implement (confirmed directly from all three -/// source files), so - unlike speak's playback-device side - this command never needs -/// a conditional capture-device disposal cast. Only the recognizer and, in mic mode, the -/// need disposal, both -/// handled in nested finally blocks so every exit path (EOF stop, silence-timeout stop, -/// Ctrl+C, or an error) disposes them exactly once. +/// Live interim printing. The result-pumping loop is started - as a genuinely async, +/// non-blocking - before is +/// awaited, not after. The session's own result buffer coalesces rapid interim/provisional +/// updates down to only the latest one whenever its single consumer is not actively reading; +/// starting the pump first ensures every interim update is still observed roughly as it is +/// produced, preserving the same "live hypothesis redraw" console UX the old synchronous +/// ResultReceived event gave for free. +/// +/// +/// Disposal. Both the recognizer engine and session are +/// and disposed via await using, which also guarantees their disposal happens even when +/// an exception propagates out of this method. and its +/// concrete implementations do not implement , so - unlike +/// speak's playback-device side - this command never needs a conditional +/// capture-device disposal cast. /// /// /// --interim/--final-only console UX. The default (neither flag) prints @@ -120,6 +123,25 @@ internal static class RecognizeCommand /// private const double DefaultSilenceTimeoutSeconds = 5.0; + /// + /// Resolves the idle-timeout window used to re-arm the mic-mode timer after every yielded + /// result: 's own --silence-timeout value, or + /// when omitted. + /// + internal static TimeSpan ResolveSilenceTimeout(RecognizeOptions options) => + TimeSpan.FromSeconds(options.SilenceTimeoutSeconds ?? DefaultSilenceTimeoutSeconds); + + /// + /// Resolves the idle-timeout window used only for the initial mic-mode timer arming, + /// before any result has arrived: 's own --start-timeout + /// value, or when omitted. This default is a + /// fixed value independent of 's own result - it never + /// borrows --silence-timeout's value, even when that flag is given and + /// --start-timeout is not. + /// + internal static TimeSpan ResolveStartTimeout(RecognizeOptions options) => + TimeSpan.FromSeconds(options.StartTimeoutSeconds ?? DefaultStartTimeoutSeconds); + /// /// Runs the recognize subcommand against a real, composed /// and . @@ -140,6 +162,15 @@ public static void Run(Context context) /// factory, for unit testing without a real model catalog, network access, or audio /// hardware. /// + /// + /// This overload stays synchronous at its boundary, mirroring SpeakCommand's own + /// established "sync entry wraps async inner via GetAwaiter().GetResult() once" + /// convention: it owns a and + /// subscription for the whole call, then blocks on + /// exactly once. CommandDispatch's own handler signature + /// (Action<Context>) is unchanged by this redesign, so this remains the one + /// place a blocking wait is unavoidable. + /// /// The invocation context. Must not be null. /// The catalog seam to resolve the model through. Must not be null. /// The audio device factory to resolve a real capture device through. Must not be null. @@ -159,6 +190,53 @@ internal static void Run(Context context, ICliModelCatalog catalog, AudioDeviceF ArgumentNullException.ThrowIfNull(catalog); ArgumentNullException.ThrowIfNull(factory); + using var cancellationSource = new CancellationTokenSource(); + ConsoleCancelEventHandler onCancelKeyPress = (_, e) => + { + e.Cancel = true; + cancellationSource.Cancel(); + }; + + Console.CancelKeyPress += onCancelKeyPress; + try + { + RunAsync(context, catalog, factory, cancellationSource.Token).GetAwaiter().GetResult(); + } + finally + { + Console.CancelKeyPress -= onCancelKeyPress; + } + } + + /// + /// Runs the recognize subcommand's full async implementation: resolves the model + /// and capture device, loads a recognition engine, binds a session to the device, and + /// pumps results until the session ends (file-drained, idle-timed-out, or cancelled). + /// + /// The invocation context. Must not be null. + /// The catalog seam to resolve the model through. Must not be null. + /// The audio device factory to resolve a real capture device through. Must not be null. + /// A token that, when cancelled, stops the session cooperatively. + /// + /// Thrown when , , or + /// is . + /// + /// + /// Thrown for any usage error: missing/unknown/wrong-role/not-downloaded model, + /// conflicting or missing input source, conflicting --interim/--final-only, + /// malformed/invalid --stt-param value, or unknown --capture-device. + /// + /// Thrown when no real capture device is available in mic mode. + internal static async Task RunAsync( + Context context, + ICliModelCatalog catalog, + AudioDeviceFactory factory, + CancellationToken cancellationToken) + { + ArgumentNullException.ThrowIfNull(context); + ArgumentNullException.ThrowIfNull(catalog); + ArgumentNullException.ThrowIfNull(factory); + var options = ParseArguments(context.CommandArgs); // Flag mutual-exclusion checks run before model resolution: a malformed flag shape is a @@ -171,70 +249,76 @@ internal static void Run(Context context, ICliModelCatalog catalog, AudioDeviceF var descriptor = ResolveModel(catalog, options.ModelId); var parameterValues = ParameterBagParser.Resolve(options.RawParameters, descriptor.Model.Parameters, SttParamFlag); - using var stopSignal = new ManualResetEventSlim(initialState: false); var captureDevice = ResolveCaptureDevice(factory, options); using var outputWriter = options.OutputPath is null ? null : new StreamWriter(options.OutputPath, append: false) { AutoFlush = true }; - using var recognizer = catalog.CreateRecognizer(descriptor, captureDevice, parameterValues); + await using var engine = await catalog + .CreateRecognizerEngineAsync(descriptor, parameterValues, cancellationToken) + .ConfigureAwait(false); + await using var session = await engine + .CreateSessionAsync(captureDevice, cancellationToken) + .ConfigureAwait(false); + var consoleLine = new ConsoleLineState(); - EventHandler onResultReceived = - (_, e) => HandleResult(outputWriter, options, e, consoleLine); - recognizer.ResultReceived += onResultReceived; - try + + // In mic mode, a silence-timeout wrapper is always constructed - even when both flags + // are omitted - so listening cannot block "recognize --mic" forever with only Ctrl+C as + // an escape hatch. File-input mode needs no wrapper: it terminates on its own once + // StartAsync (which blocks until the whole file is delivered) and the explicit drain + // StopAsync below both complete. + SilenceTimeoutRecognizerSession? silenceTimeout = null; + if (options.Mic) { - // In mic mode, a silence-timeout session is always constructed - even when both - // flags are omitted - so listening cannot block "recognize --mic" forever with only - // Ctrl+C as an escape hatch. File-input mode still needs no session: it terminates on - // its own via EndOfFileReached. - var sessionStartTimeout = TimeSpan.FromSeconds(options.StartTimeoutSeconds ?? DefaultStartTimeoutSeconds); - using var session = options.Mic - ? new SilenceTimeoutRecognizerSession( - recognizer, - TimeSpan.FromSeconds(options.SilenceTimeoutSeconds ?? DefaultSilenceTimeoutSeconds), - startTimeout: sessionStartTimeout) - : null; - - if (session is not null) - { - session.TimedOut += (_, _) => stopSignal.Set(); - } + silenceTimeout = new SilenceTimeoutRecognizerSession( + session, + ResolveSilenceTimeout(options), + startTimeout: ResolveStartTimeout(options)); + } - if (options.InputPath is not null && captureDevice is WavFileAudioCaptureDevice fileDevice) - { - fileDevice.EndOfFileReached += (_, _) => recognizer.Stop(); - } + // Ctrl+C stops the session cooperatively (fire-and-forget: this handler is synchronous + // and must not block). Stopping the session completes its result stream, which ends the + // pump below naturally; cancellationToken is also forwarded into GetResultsAsync as a + // belt-and-suspenders safety net for the documented edge case where StopAsync called + // while the session never started does not itself complete that stream. + ConsoleCancelEventHandler onCancelKeyPress = (_, _) => _ = session.StopAsync(CancellationToken.None); - ConsoleCancelEventHandler onCancelKeyPress = (_, e) => - { - e.Cancel = true; - recognizer.Stop(); - stopSignal.Set(); - }; + var results = silenceTimeout?.GetResultsAsync(cancellationToken) ?? session.GetResultsAsync(cancellationToken); - Console.CancelKeyPress += onCancelKeyPress; + // Started before StartAsync is awaited (see this type's remarks on live interim + // printing): this is a genuinely async, non-blocking call, not a background thread - it + // suspends immediately since no result exists yet, handing control straight back here. + var pumpTask = PumpResultsAsync(results, outputWriter, options, consoleLine, cancellationToken); + + Console.CancelKeyPress += onCancelKeyPress; + try + { try { - // File mode: Start() itself only returns once the whole file has been - // delivered and drained (see this type's remarks), so no wait is needed - // here. Mic mode: Start() returns quickly, so this thread blocks until - // Ctrl+C or a silence timeout signals stopSignal. - recognizer.Start(); - if (options.Mic) - { - stopSignal.Wait(); - } + await session.StartAsync(cancellationToken).ConfigureAwait(false); + } + catch (OperationCanceledException) + { + // Ctrl+C (or an already-cancelled token) landed before/while starting; ensure the + // session still reaches a terminal state so the pump below is guaranteed to end. + await session.StopAsync(CancellationToken.None).ConfigureAwait(false); } - finally + + if (!options.Mic) { - Console.CancelKeyPress -= onCancelKeyPress; + // File mode: StartAsync already blocked until the whole file was delivered (see + // this type's remarks); drain any buffered-but-not-yet-decoded audio now so every + // result is produced before the pump below settles. + await session.StopAsync(cancellationToken).ConfigureAwait(false); } + + await pumpTask.ConfigureAwait(false); } finally { - recognizer.ResultReceived -= onResultReceived; + Console.CancelKeyPress -= onCancelKeyPress; } // Settle any interim line still pending (e.g. a session that ends right after an @@ -247,6 +331,37 @@ internal static void Run(Context context, ICliModelCatalog catalog, AudioDeviceF : $"Recognition results written to '{options.OutputPath}'."); } + /// + /// Consumes via await foreach, calling + /// for each one, until the sequence ends or + /// is cancelled. + /// + /// + /// A cancellation that ends the enumeration is expected and graceful here - the same + /// Ctrl+C handler that cancels also calls + /// session.StopAsync() directly, so this is a belt-and-suspenders safety net, not + /// the primary termination path. + /// + private static async Task PumpResultsAsync( + IAsyncEnumerable results, + StreamWriter? outputWriter, + RecognizeOptions options, + ConsoleLineState consoleLine, + CancellationToken cancellationToken) + { + try + { + await foreach (var e in results.WithCancellation(cancellationToken).ConfigureAwait(false)) + { + HandleResult(outputWriter, options, e, consoleLine); + } + } + catch (OperationCanceledException) + { + // Expected, graceful termination triggered by Ctrl+C; nothing further to do. + } + } + /// /// Prints (and, for a final result, optionally writes to the output file) one recognition /// result, honoring --interim/--final-only. Any carriage-return-overwritten diff --git a/src/DemaConsulting.Speech.Cli/Commands/RecognitionCommandSubsystem/SilenceTimeoutRecognizerSession.cs b/src/DemaConsulting.Speech.Cli/Commands/RecognitionCommandSubsystem/SilenceTimeoutRecognizerSession.cs index 8855292..deb0702 100644 --- a/src/DemaConsulting.Speech.Cli/Commands/RecognitionCommandSubsystem/SilenceTimeoutRecognizerSession.cs +++ b/src/DemaConsulting.Speech.Cli/Commands/RecognitionCommandSubsystem/SilenceTimeoutRecognizerSession.cs @@ -18,261 +18,189 @@ // OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE // SOFTWARE. +using System.Runtime.CompilerServices; using DemaConsulting.Speech.RecognitionSubsystem; namespace DemaConsulting.Speech.Cli.Commands.RecognitionCommandSubsystem; /// -/// Observes an 's -/// event and calls , then raises , -/// when no result (partial or final) has arrived within a configured idle window - the -/// mic-mode implementation of recognize --mic --silence-timeout <seconds> (and its -/// companion --start-timeout <seconds> flag). +/// Wraps an 's +/// stream, calling , then raising +/// , when no result (partial or final) has arrived within a configured +/// idle window - the mic-mode implementation of recognize --mic --silence-timeout +/// <seconds> (and its companion --start-timeout <seconds> flag). /// /// /// -/// This is a pure event-driven observer composed alongside a recognizer, not a decorator -/// around its lifecycle API: it never intercepts or -/// calls made by its owner, and calls -/// itself only proactively, on timeout. +/// Redesign from the old event-observer shape. Before the Engine/Session redesign, +/// this type observed ISpeechRecognizer.ResultReceived - an event raised from the +/// recognizer's own background decode thread - and had to serialize that callback against a +/// concurrent call with a hand-rolled gate/wait-handle pair, +/// because a timer callback and Dispose could race on two different threads with no +/// other synchronization available. The new +/// contract is single-consumer and fully async: there is no longer a separate +/// background thread raising events into this type at arbitrary times, so there is nothing +/// left to race. This type is instead implemented as a stateless +/// decorator: is itself an async-iterator method whose entire +/// idle-timeout bookkeeping lives on its own call stack (a local await using enumerator +/// and a per-iteration ), so +/// it needs no locks, no wait handles, and - since it owns no state beyond one call's local +/// variables - no / of its own either; +/// an abandoned enumeration (for example a caller that stops awaiting MoveNextAsync +/// partway through) is cleaned up the same way any other async-iterator method's locals are, +/// through the compiler-generated await using around the inner enumerator. /// /// -/// The idle timer is implemented with the injectable -/// abstraction (available in the BCL since .NET 8, requiring no new package reference) rather -/// than a hard-coded /, so a unit test -/// can exercise the full idle/reset/timeout sequence deterministically with a fake -/// and no real wall-clock delay. +/// Each loop iteration races the next item +/// against a fresh idle +/// timer via . The idle timer always wins a race against +/// a result that has not yet arrived, since is +/// non-blocking and only ever observes tasks that are already pending; there is no window +/// where both could appear "ready" simultaneously in a way that silently drops a genuine +/// result - a result that arrives first simply completes the inner task first, exactly as a +/// single await foreach over the undecorated session would observe it. /// /// -/// This session enforces two distinct, sequential idle windows. Before any -/// event has arrived, the idle timer is armed -/// with the startTimeout value - a grace period for the user to begin speaking, which -/// is typically longer than the pause used to detect the end of an utterance. From the first -/// event onward (partial or final), the timer -/// is re-armed with idleTimeout (the --silence-timeout value) on every -/// subsequent event, which is the simplest correct semantics for "reset on every event". No -/// extra "first result seen" flag is needed for this phase transition: construction and -/// are already distinct call sites, so arming with -/// startTimeout once at construction and unconditionally re-arming with -/// idleTimeout on every call naturally implements the -/// two-phase contract. +/// This session enforces two distinct, sequential idle windows, identically to the prior +/// design: before any result has arrived, the idle timer is armed with the +/// startTimeout constructor parameter's value - a grace period for the user to begin +/// speaking, typically longer than the pause used to detect the end of an utterance. From +/// the first yielded result onward (partial or final), +/// the timer is re-armed with idleTimeout (the --silence-timeout value) for +/// every subsequent iteration. +/// +/// +/// On timeout, this type calls - which, per its +/// own contract, drains any already-captured-but-not-yet-decoded audio and completes the +/// session's result stream - then raises , then continues draining +/// itself so that any trailing result +/// produced by that drain is still yielded to this type's own caller before the enumeration +/// ends, exactly as it would have been without this decorator in place. /// /// -internal sealed class SilenceTimeoutRecognizerSession : IDisposable +internal sealed class SilenceTimeoutRecognizerSession { - /// The recognizer this session observes and, on timeout, stops. - private readonly ISpeechRecognizer _recognizer; + /// The session this type wraps and, on timeout, stops. + private readonly IRecognitionSession _session; - /// The idle window; re-armed on every event. + /// The idle window; re-armed after every yielded result. private readonly TimeSpan _idleTimeout; - /// The single-shot idle timer. - private readonly ITimer _timer; - - /// - /// Serializes , , and - /// against one another, so a timer/result callback racing a concurrent - /// call always either completes entirely before disposal starts or - /// observes already set and returns immediately - it never reads - /// a torn state or touches the timer after it has been disposed. - /// - private readonly object _gate = new(); + /// The idle window used only for the initial arming, before any result has arrived. + private readonly TimeSpan _startTimeout; - /// - /// Guards against a timer callback racing a concurrent call: once - /// set, the timer callback (which runs on a thread-pool thread independent of the - /// constructing/disposing thread) must not call or - /// raise once has already started and - /// observed the callback hasn't fired yet. Always read/written while holding - /// ; see for why is released - /// again before that call and raise actually happen. - /// - private bool _isDisposed; - - /// - /// Signaled whenever no invocation is currently between its - /// -protected disposed-check and the completion of its - /// call and raise. Starts - /// signaled (no callback in flight). waits on this - outside - /// - so it still happens-after any in-flight - /// call and raise, even though - /// no longer holds while making them (see - /// for why holding across those calls would - /// deadlock). Without this, a caller could observe return and then - /// tear down state a handler still in flight depends on (for - /// example disposing a synchronization primitive a handler closure calls back into, as - /// RecognizeCommand does with its shared stopSignal). - /// - /// - /// This assumes a subscriber never calls - /// synchronously from within its own handler - would deadlock - /// waiting on this event in that case, since only that same (blocked) thread's own - /// invocation could ever signal it. This session's sole caller, - /// RecognizeCommand, only ever sets a flag from its handler - /// and calls later, separately, after observing that flag on its own - /// thread, so this is not a real constraint in practice today. - /// - private readonly ManualResetEventSlim _idleCallbackDone = new(initialState: true); + /// The time source the idle timer is created from. + private readonly TimeProvider _timeProvider; /// - /// Initializes a new instance of the class, - /// arming its idle timer immediately. + /// Initializes a new instance of the class. /// - /// The recognizer to observe and, on timeout, stop. Must not be null. + /// The session to wrap and, on timeout, stop. Must not be null. /// - /// The idle window after which, with no - /// event, this session calls . Used to re-arm the - /// timer on every event, and as the - /// initial arming value when is . - /// Must be greater than . + /// The idle window after which, with no result yielded, this type calls + /// . Used to re-arm the timer after every + /// yielded result, and as the initial arming value when is + /// . Must be greater than . /// /// - /// The time source to create the idle timer from, or to use + /// The time source to race the idle timer against, or to use /// , mirroring AudioDeviceFactory's own "null seam /// parameter defaults to the real backend" convention. /// /// - /// The idle window used only for the initial timer arming, before any - /// event has arrived - a grace period for - /// the user to begin speaking, which is typically longer than the pause used to detect - /// the end of an utterance. Defaults to when + /// The idle window used only for the initial timer arming, before any result has arrived - + /// a grace period for the user to begin speaking, which is typically longer than the pause + /// used to detect the end of an utterance. Defaults to when /// . Callers such as RecognizeCommand/AskCommand /// instead resolve their own independent default before constructing this session, so /// this fallback is exercised only by callers that genuinely want one flag to imply the /// other. Must be greater than when supplied. /// - /// Thrown when is . + /// Thrown when is . /// /// Thrown when is not greater than , /// or when has a value that is not greater than /// . /// public SilenceTimeoutRecognizerSession( - ISpeechRecognizer recognizer, + IRecognitionSession session, TimeSpan idleTimeout, TimeProvider? timeProvider = null, TimeSpan? startTimeout = null) { - ArgumentNullException.ThrowIfNull(recognizer); + ArgumentNullException.ThrowIfNull(session); ArgumentOutOfRangeException.ThrowIfLessThanOrEqual(idleTimeout, TimeSpan.Zero); if (startTimeout.HasValue) { ArgumentOutOfRangeException.ThrowIfLessThanOrEqual(startTimeout.Value, TimeSpan.Zero); } - _recognizer = recognizer; + _session = session; _idleTimeout = idleTimeout; - - var provider = timeProvider ?? TimeProvider.System; - _timer = provider.CreateTimer(OnIdle, null, startTimeout ?? idleTimeout, Timeout.InfiniteTimeSpan); - - _recognizer.ResultReceived += OnResultReceived; + _startTimeout = startTimeout ?? idleTimeout; + _timeProvider = timeProvider ?? TimeProvider.System; } /// - /// Raised after this session has already called + /// Raised after this session has already called /// because no result arrived within the idle window. /// public event EventHandler? TimedOut; /// - /// Re-arms the idle timer on every result (partial or final), per this session's "reset - /// on every event" contract. + /// Streams every result from the wrapped session, calling + /// and raising once, the + /// first time the configured idle window elapses with no result yielded. /// - private void OnResultReceived(object? sender, SpeechRecognitionEvent e) + /// + /// A token that ends only this enumeration when cancelled - the wrapped session itself + /// keeps running, mirroring 's own + /// contract. + /// + /// An asynchronous sequence of recognition events. + public async IAsyncEnumerable GetResultsAsync( + [EnumeratorCancellation] CancellationToken cancellationToken = default) { - lock (_gate) - { - if (_isDisposed) - { - return; - } + var timeout = _startTimeout; + var timedOut = false; - // Holding _gate for the whole check-then-act guarantees Dispose cannot have disposed - // _timer between the _isDisposed check above and this call: Dispose only disposes - // _timer after it has both set _isDisposed and released _gate (see Dispose below), so - // observing _isDisposed == false here under the lock proves _timer is still live. - _timer.Change(_idleTimeout, Timeout.InfiniteTimeSpan); - } - } - - /// - /// Runs when the idle timer fires with no reset since it was last armed: stops the - /// recognizer, then raises . - /// - private void OnIdle(object? state) - { - lock (_gate) + await using var enumerator = _session.GetResultsAsync(cancellationToken).GetAsyncEnumerator(cancellationToken); + while (true) { - if (_isDisposed) + var moveNextTask = enumerator.MoveNextAsync().AsTask(); + var delayTask = Task.Delay(timeout, _timeProvider, cancellationToken); + var winner = await Task.WhenAny(moveNextTask, delayTask).ConfigureAwait(false); + + // A delayTask that "wins" because cancellationToken was cancelled (rather than + // because the idle window genuinely elapsed) is not a timeout: it is + // IsCompletedSuccessfully only when the idle window itself ran to completion, while a + // cancellation instead leaves it Canceled. Treating a cancellation-triggered delayTask + // as a timeout would wrongly call StopAsync here, racing the wrapped session's result + // stream completion against moveNextTask's own cancellation and sometimes swallowing + // the OperationCanceledException this enumeration's caller is entitled to observe. + if (winner == delayTask && delayTask.IsCompletedSuccessfully && !timedOut) { - return; + // The idle window elapsed with no result yielded since the previous iteration (or + // since this method started, for the first iteration). Stop the session - which + // completes its result stream - then raise TimedOut, then keep draining below so + // any trailing result the stop itself produced is still yielded to our own + // caller, exactly as an undecorated foreach over the session would observe it. + timedOut = true; + await _session.StopAsync(CancellationToken.None).ConfigureAwait(false); + TimedOut?.Invoke(this, EventArgs.Empty); } - // Reset while still holding _gate: this is the only place _idleCallbackDone is - // reset, and _isDisposed being false here (still checked under _gate) proves Dispose - // has not yet started waiting on it, so there is no race with the Wait() in Dispose. - _idleCallbackDone.Reset(); - } - - try - { - // Stop() and the TimedOut raise deliberately happen without holding _gate. Stop() - // drains already-captured audio and can block waiting for the recognizer's - // background decode thread to finish, and that thread may itself raise - // ResultReceived while draining - which needs _gate to reset the idle timer in - // OnResultReceived. Holding _gate across Stop() here would deadlock the two threads - // against each other (this thread blocked inside Stop() waiting for the decode - // thread, the decode thread blocked waiting to enter _gate). Releasing _gate first - // means a concurrent Dispose can now set _isDisposed and unsubscribe before Stop() - // returns; that is harmless because the recognizer is owned by this session's - // caller, not by the session itself, so calling Stop() (and raising TimedOut) after - // this session's own bookkeeping has been torn down is still safe. Dispose still - // waits for this call to finish (via _idleCallbackDone) before returning, so callers - // never observe Dispose complete while a TimedOut raise is still in flight. - _recognizer.Stop(); - TimedOut?.Invoke(this, EventArgs.Empty); - } - finally - { - _idleCallbackDone.Set(); - } - } - - /// - /// Disposes the idle timer and unsubscribes from . - /// Idempotent: a second call does nothing. - /// - public void Dispose() - { - lock (_gate) - { - if (_isDisposed) + // Propagate cancellation (either delayTask's own cancellation, or a fault surfaced + // through moveNextTask) the same way the standard compiler-generated "await foreach" + // would: by awaiting the task that is actually ready, letting its own exception (if + // any) propagate naturally. + if (!await moveNextTask.ConfigureAwait(false)) { - return; + yield break; } - _isDisposed = true; - _recognizer.ResultReceived -= OnResultReceived; + yield return enumerator.Current; + timeout = _idleTimeout; } - - // _timer.Dispose() runs outside _gate deliberately: some TimeProvider/ITimer - // implementations block a Dispose() call until any currently in-flight callback - // invocation has finished. Since OnIdle also acquires _gate, disposing the timer while - // still holding _gate here could deadlock (this thread blocked inside _timer.Dispose() - // waiting for OnIdle to return, while OnIdle is blocked waiting to acquire the very lock - // this thread holds). Releasing _gate first (having already set _isDisposed under it) - // still guarantees no new call to OnResultReceived/OnIdle acts on this session, since - // both re-check _isDisposed immediately after acquiring _gate. - // - // Wait for _idleCallbackDone before disposing the timer: an OnIdle invocation already - // past the _isDisposed check above (and therefore already committed to calling Stop() - // and raising TimedOut) is not tracked by _gate at all once it releases it, so without - // this wait, Dispose could return - and a caller could tear down state a still-in-flight - // TimedOut handler depends on - before that call actually happens (see _idleCallbackDone - // for the one assumption this relies on: no TimedOut subscriber calls Dispose() itself). - _idleCallbackDone.Wait(); - _timer.Dispose(); - _idleCallbackDone.Dispose(); } } diff --git a/src/DemaConsulting.Speech.Cli/Commands/SynthesisCommandSubsystem/SpeakCommand.cs b/src/DemaConsulting.Speech.Cli/Commands/SynthesisCommandSubsystem/SpeakCommand.cs index def9f45..44b1895 100644 --- a/src/DemaConsulting.Speech.Cli/Commands/SynthesisCommandSubsystem/SpeakCommand.cs +++ b/src/DemaConsulting.Speech.Cli/Commands/SynthesisCommandSubsystem/SpeakCommand.cs @@ -151,11 +151,16 @@ internal static async Task RunAsync( var playbackDevice = ResolvePlaybackDevice(catalog, deviceSource, descriptor, options); using var playbackDeviceLease = playbackDevice as IDisposable; - using var synthesizer = catalog.CreateSynthesizer(descriptor, playbackDevice, parameterValues); + await using var engine = await catalog + .CreateSynthesizerEngineAsync(descriptor, parameterValues, cancellationToken) + .ConfigureAwait(false); + await using var session = await engine + .CreateSessionAsync(playbackDevice, cancellationToken) + .ConfigureAwait(false); try { - await synthesizer.SpeakAsync(text, cancellationToken).ConfigureAwait(false); + await session.SpeakAsync(text, cancellationToken).ConfigureAwait(false); context.WriteLine(options.OutputPath is null ? "Speech playback finished." : $"Speech written to '{options.OutputPath}'."); diff --git a/src/DemaConsulting.Speech.Demo/App.axaml.cs b/src/DemaConsulting.Speech.Demo/App.axaml.cs index 1ddce97..55f5be1 100644 --- a/src/DemaConsulting.Speech.Demo/App.axaml.cs +++ b/src/DemaConsulting.Speech.Demo/App.axaml.cs @@ -31,6 +31,14 @@ public sealed class App : Application /// private SpeechModelCatalog? _catalog; + /// + /// Set once the deferred shutdown's async cleanup has been started, so the second, + /// genuine ShutdownRequested raised by this type's own call to + /// is allowed to proceed instead of + /// being deferred again. + /// + private bool _shuttingDown; + /// /// Loads the application's XAML-declared resources and styles. /// @@ -78,15 +86,67 @@ public override void OnFrameworkInitializationCompleted() desktop.MainWindow = new MainWindow { DataContext = viewModel }; // Release the catalog's download machinery and each panel ViewModel's own - // subscriptions/resources when the application exits. - desktop.ShutdownRequested += (_, _) => + // subscriptions/resources when the application exits. ShutdownRequested is a + // synchronous event with no async-aware overload, so this handler defers the + // shutdown Avalonia is about to perform (via ShutdownRequestedEventArgs.Cancel), + // synchronously, runs the async cleanup to genuine completion, and only then calls + // IClassicDesktopStyleApplicationLifetime.Shutdown() to let the real shutdown + // proceed - rather than an async void handler that would return control to Avalonia + // at its first incomplete await and let shutdown continue while cleanup (and any + // exception it throws) is still pending. + desktop.ShutdownRequested += (object? _, ShutdownRequestedEventArgs e) => { - viewModel.Synthesis.Dispose(); - viewModel.Recognition.Dispose(); - _catalog?.Dispose(); + if (_shuttingDown) + { + // This is the second, genuine ShutdownRequested raised by this handler's own + // call to Shutdown() below, once cleanup has already completed. + return; + } + + // Set synchronously, before the first await inside CompleteShutdownAsync runs + // (finding 29): a second ShutdownRequested racing in during that first await - + // for example, the user closing the window again while cleanup is still pending - + // must see this guard already set, rather than finding it still false and + // starting a second concurrent cleanup task against the same ViewModels/catalog. + _shuttingDown = true; + + e.Cancel = true; + CompleteShutdownAsync(desktop, viewModel).ConfigureAwait(false); }; } base.OnFrameworkInitializationCompleted(); } + + /// + /// Runs this application's asynchronous teardown to genuine completion, then lets the + /// deferred shutdown this type requested in + /// proceed. + /// + /// + /// is already set to by the caller + /// before this method starts (finding 29), so no field assignment is needed here; this + /// method's own only has to resume shutdown once cleanup has + /// settled, whether it succeeded or faulted. + /// + /// The desktop lifetime to resume shutdown on once cleanup has completed. + /// The main window's view model owning the panels to dispose. + private async Task CompleteShutdownAsync(IClassicDesktopStyleApplicationLifetime desktop, MainWindowViewModel viewModel) + { + try + { + await viewModel.Synthesis.DisposeAsync(); + await viewModel.Recognition.DisposeAsync(); + _catalog?.Dispose(); + } + catch + { + // Intentionally broad: a teardown fault must not prevent the application from + // actually exiting once shutdown has been requested. + } + finally + { + desktop.Shutdown(); + } + } } diff --git a/src/DemaConsulting.Speech.Demo/ModelSettingsSubsystem/ModelSettingsViewModel.cs b/src/DemaConsulting.Speech.Demo/ModelSettingsSubsystem/ModelSettingsViewModel.cs index e501836..0adf6ac 100644 --- a/src/DemaConsulting.Speech.Demo/ModelSettingsSubsystem/ModelSettingsViewModel.cs +++ b/src/DemaConsulting.Speech.Demo/ModelSettingsSubsystem/ModelSettingsViewModel.cs @@ -25,14 +25,15 @@ namespace DemaConsulting.Speech.Demo.ModelSettingsSubsystem; /// as the interface between a host's settings UI and per-model synthesis/recognition /// parameter handling. As of this pass, the synthesis panel's /// SynthesisPanelViewModel.PlayAsync passes this bag to -/// ISynthesizerSessionFactory.Create, which forwards it to the library's -/// SpeechSynthesizerFactory.Create and, from there, to a multi-speaker model's own +/// ISynthesizerSessionFactory.LoadAsync, which forwards it to the library's +/// SpeechSynthesizerFactory.LoadAsync and, from there, to a multi-speaker model's own /// ISynthesisModel.ResolveSpeakerId hook - so a selected value (for example /// 's voice choice) now reaches a real -/// synthesis call. The recognition panel's ISpeechRecognizer contract still accepts -/// no such bag on any per-call member, so for recognition this bag remains built and fully -/// exercised by unit tests only, with no real consumer yet - this is stated plainly here -/// rather than silently implied. +/// synthesis call. The recognition panel's IRecognizerSessionFactory.LoadAsync/ +/// ISpeechRecognizerEngine.CreateSessionAsync contract still accepts no such bag on +/// any per-call member, so for recognition this bag remains built and fully exercised by +/// unit tests only, with no real consumer yet - this is stated plainly here rather than +/// silently implied. /// /// /// Not thread-safe: expected to be used from the UI thread. @@ -124,7 +125,7 @@ public ModelSettingsViewModel(ISpeechModel? model = null) /// /// /// See the type-level remarks: as of this pass, the synthesis panel forwards this bag to - /// a real synthesis call (via ISynthesizerSessionFactory.Create); the recognition + /// a real synthesis call (via ISynthesizerSessionFactory.LoadAsync); the recognition /// panel still has no such consumer. /// public IReadOnlyDictionary BuildValueBag() => diff --git a/src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/IRecognizerSessionFactory.cs b/src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/IRecognizerSessionFactory.cs index 7b7dffb..c4dbf17 100644 --- a/src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/IRecognizerSessionFactory.cs +++ b/src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/IRecognizerSessionFactory.cs @@ -1,15 +1,14 @@ -using DemaConsulting.Speech.AudioSubsystem; using DemaConsulting.Speech.ModelManagementSubsystem; using DemaConsulting.Speech.RecognitionSubsystem; namespace DemaConsulting.Speech.Demo.RecognitionPanelSubsystem; /// -/// Demo-owned seam over the library's recognition-session composition surface. +/// Demo-owned seam over the library's recognition-engine composition surface. /// /// -/// The library exposes recognizer composition through the static -/// +/// The library exposes engine composition through the static +/// /// method, which requires an - an interface whose members are /// partly to the library, so only the library's own assemblies can /// implement it. This seam therefore accepts the common contract @@ -19,28 +18,27 @@ namespace DemaConsulting.Speech.Demo.RecognitionPanelSubsystem; /// substituted directly in a ViewModel unit test because it is static; the production /// implementation resolves the model's installed-files directory and delegates straight to /// that factory, while tests substitute a fake that returns a controlled -/// without a downloaded model, a native runtime, or a real -/// capture device. It adds no public API to DemaConsulting.Speech. +/// without a downloaded model, a native runtime, or a +/// real capture device. It adds no public API to DemaConsulting.Speech. +/// +/// Only the engine is composed here - a capture device is bound later, per run, via +/// directly against the returned +/// engine, so a host can load one engine per model and reuse it across many sessions instead +/// of reloading the model on every Start. +/// /// public interface IRecognizerSessionFactory { /// - /// Creates a speech recognizer for an installed model, streaming from the supplied - /// device. + /// Loads a speech recognizer engine for an installed model. /// /// The model to load. Must not be . - /// - /// The capture device to stream audio from. Must not be . - /// + /// A token to observe for cancellation of this call. /// - /// A real recognizer when the model is installed, implements the library's recognition - /// role, and the device is available; otherwise an honest IsAvailable == false - /// recognizer. + /// A task that completes with a real engine when the model is installed and implements + /// the library's recognition role; otherwise an honest IsAvailable == false engine. /// - /// - /// Thrown when or is - /// . - /// - /// Never throws; every honest unavailable state is reported through IsAvailable. - ISpeechRecognizer Create(ISpeechModel model, IAudioCaptureDevice captureDevice); + /// Thrown when is . + /// Never throws for an ordinary unavailable state; every such state is reported through IsAvailable. + Task LoadAsync(ISpeechModel model, CancellationToken cancellationToken = default); } diff --git a/src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/RecognitionPanelViewModel.cs b/src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/RecognitionPanelViewModel.cs index 077d437..5794029 100644 --- a/src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/RecognitionPanelViewModel.cs +++ b/src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/RecognitionPanelViewModel.cs @@ -20,7 +20,7 @@ namespace DemaConsulting.Speech.Demo.RecognitionPanelSubsystem; /// recognition model, starting and stopping a live transcription session, and rendering the /// progressive provisional ("partial") and final results the library raises while streaming /// - with honest reporting of every unavailable state (no installed recognition model, no -/// capture device, or an engine fault) rather than a silent failure or a crash. +/// capture device, or an engine/session fault) rather than a silent failure or a crash. /// /// Every library entry point this panel reaches is reached through an injected seam /// (, , @@ -29,34 +29,36 @@ namespace DemaConsulting.Speech.Demo.RecognitionPanelSubsystem; /// downloaded model, no native runtime, and no real microphone. /// /// -/// is documented to raise from the -/// recognizer's own background decoding thread, so this ViewModel captures the UI thread's -/// when a session starts and posts every event through -/// it before touching any observable property, mirroring the library's own -/// ConfigureAwait(true) marshaling used elsewhere in this demo. Otherwise not -/// thread-safe: expected to be used from the UI thread. +/// and the results from +/// are documented to raise/resume from the +/// session's own background thread, so captures the UI thread's +/// before the session starts and this ViewModel marshals +/// every event through it before touching any +/// observable property - the result pump below instead relies on each +/// await foreach continuation naturally resuming on that same captured context. +/// Otherwise not thread-safe: expected to be used from the UI thread. /// /// /// This ViewModel also subscribes to in its /// constructor so a newly downloaded recognition model installed from the Model Catalog panel /// is picked up automatically, without a manual Refresh click or app restart. The handler /// ignores events for any other and marshals onto the UI thread -/// (captured at construction) before calling , using the same -/// pattern as -/// handling above. unsubscribes this handler. +/// (captured at construction) before calling . +/// unsubscribes this handler. /// /// -/// Recognizer reuse. Per 's own "hot reuse" guidance, -/// this ViewModel constructs a recognizer at most once per selected model/capture-device pair -/// and calls / -/// repeatedly on that same instance across many Start/Stop clicks, instead of composing and -/// disposing a new one - and reloading its model - on every click. The cached recognizer is -/// invalidated (disposed and rebuilt on the next Start) only when the selected model changes, -/// the selected capture device changes, or the shared device-selection panel forces a device -/// refresh - each of which genuinely requires a different underlying recognizer/device pair. +/// Engine/session reuse. Per 's own reuse +/// guidance, this ViewModel loads an engine at most once per selected model and reuses it +/// across many Start/Stop cycles, instead of reloading its model on every click. Because an +/// is single-use (see its own remarks), a fresh session is +/// created from the cached engine for every Start and released before the next one is +/// created. The cached engine is invalidated (disposed and reloaded on the next Start) only +/// when the selected model changes; the cached session is invalidated (released, keeping the +/// engine) whenever the selected capture device changes or the shared device-selection panel +/// forces a device refresh - a session, not an engine, is bound to a capture device. /// /// -public sealed partial class RecognitionPanelViewModel : ObservableObject, IDisposable +public sealed partial class RecognitionPanelViewModel : ObservableObject, IAsyncDisposable { /// The message shown when no installed recognition model is available to choose. public const string NoModelsMessage = @@ -70,13 +72,16 @@ public sealed partial class RecognitionPanelViewModel : ObservableObject, IDispo public const string NoCaptureDeviceMessage = "No audio capture device is available. Choose one on the Audio Devices panel."; - /// The message shown when the composed recognizer honestly reports itself unavailable. + /// The message shown when the composed recognizer engine honestly reports itself unavailable. public const string RecognizerUnavailableMessage = "The selected model's speech recognizer is unavailable on this machine."; /// The message shown after listening is stopped by the user. public const string StoppedMessage = "Listening stopped."; + /// The message shown when the active session reports an unrecoverable fault. + public const string SessionFaultedMessage = "The recognition session reported an unrecoverable error."; + /// The seam supplying every installed recognition model. private readonly IModelCatalogService _catalogService; @@ -86,7 +91,7 @@ public sealed partial class RecognitionPanelViewModel : ObservableObject, IDispo /// The shared device-selection panel state supplying the chosen capture device. private readonly DeviceSelectionViewModel _deviceSelection; - /// The seam used to compose a recognizer for the selected model and device. + /// The seam used to load a recognizer engine for the selected model. private readonly IRecognizerSessionFactory _sessionFactory; /// @@ -97,15 +102,35 @@ public sealed partial class RecognitionPanelViewModel : ObservableObject, IDispo private readonly SynchronizationContext? _catalogEventUiContext = SynchronizationContext.Current; /// - /// The cached recognizer, bound to , reused across many - /// Start/Stop cycles; see the "Recognizer reuse" remarks above. - /// before the first Start, or after invalidation by a model/device change or a device - /// refresh. + /// The cached recognizer engine, reused across many Start/Stop cycles for as long as + /// matches ; see the + /// "Engine/session reuse" remarks above. before the first Start, or + /// after invalidation by a model change. + /// + private ISpeechRecognizerEngine? _engine; + + /// The model was loaded for, if any. + private ISpeechModel? _engineModel; + + /// + /// The active per-run session created from , if any. Single-use: + /// released via before the next Start can lease a new + /// one from the same engine. + /// + private IRecognitionSession? _session; + + /// + /// The background task pumping for + /// , stored so it can be awaited on Stop and canceled/awaited on + /// invalidation or disposal. /// - private ISpeechRecognizer? _recognizer; + private Task? _pumpTask; - /// The capture device was constructed with, if any. - private IAudioCaptureDevice? _captureDevice; + /// + /// Cancels only the result-pump enumeration for (never the session + /// itself) when that session is being released while still producing results. + /// + private CancellationTokenSource? _pumpCancellation; /// /// TEMPORARY diagnostic instrumentation for investigating a reported dropped-word bug @@ -121,7 +146,7 @@ public sealed partial class RecognitionPanelViewModel : ObservableObject, IDispo /// /// The pre-refresh hook registered with , retained so the - /// exact same delegate instance can be unregistered in . + /// exact same delegate instance can be unregistered in . /// private readonly Func _preRefreshHook; @@ -208,7 +233,7 @@ public sealed partial class RecognitionPanelViewModel : ObservableObject, IDispo /// The catalog seam to read installed models from. Must not be . /// The device seam used to create the capture device. Must not be . /// The shared device-selection panel state. Must not be . - /// The seam used to compose a recognizer. Must not be . + /// The seam used to load a recognizer engine. Must not be . /// Thrown when any parameter is . internal RecognitionPanelViewModel( IModelCatalogService catalogService, @@ -236,16 +261,22 @@ internal RecognitionPanelViewModel( } /// - /// Stops an actively listening session, if any, and invalidates the cached recognizer - /// when the user picks a different capture device - it was built against the previously - /// selected device and must not be reused against a new one. The capture picker is not - /// disabled while is - /// (unlike the model picker; see ), so a change can arrive - /// mid-session and must stop that session rather than leaving the cached recognizer - /// silently bound to the old device until some other trigger invalidates it. + /// Stops an actively listening session, if any, and invalidates the cached session (not + /// the cached engine) when the user picks a different capture device - the session was + /// bound to the previously selected device and must not be reused against a new one. The + /// capture picker is not disabled while is + /// (unlike the model picker; see + /// ), so a change can arrive mid-session and must stop that + /// session rather than leaving it silently bound to the old device. /// /// The raising device-selection panel. Unused. /// The event naming the property that changed. + /// + /// This handler's signature is fixed by + /// and cannot return a , so it fires the async work without awaiting it + /// - the same accepted fire-and-forget pattern used throughout this class for event + /// handlers that must perform async cleanup. + /// private void OnDeviceSelectionChanged(object? sender, PropertyChangedEventArgs e) { if (e.PropertyName != nameof(DeviceSelectionViewModel.CaptureSelection)) @@ -253,51 +284,162 @@ private void OnDeviceSelectionChanged(object? sender, PropertyChangedEventArgs e return; } + _ = HandleCaptureDeviceChangedAsync(); + } + + /// + /// Performs the async stop-then-invalidate-session work for . + /// + private async Task HandleCaptureDeviceChangedAsync() + { if (CanStop) { - Stop(); + await StopAsync().ConfigureAwait(true); } - InvalidateRecognizer(); + await InvalidateSessionAsync().ConfigureAwait(true); } /// - /// Stops an actively listening session, if any, and invalidates the cached recognizer - /// when the user picks a different recognition model - it was built against the - /// previously selected model and must not be reused against a new one. The model picker - /// is disabled in the view while listening (see ), but + /// Stops an actively listening session, if any, and invalidates the cached engine (and its + /// session) when the user picks a different recognition model - it was loaded for the + /// previously selected model and must not be reused against a new one. The model picker is + /// disabled in the view while listening (see ), but /// has a public setter and can still be set directly (for /// example, programmatically or via repopulating the catalog), so - /// this must not assume Stop has already run: disposing the recognizer without stopping - /// it first would leave stuck at - /// forever, since a later would see no cached recognizer and no-op. + /// this must not assume Stop has already run. /// /// The newly selected model. + /// + /// This source-generated partial method's signature is fixed to , so + /// it fires the async work without awaiting it - the same accepted fire-and-forget pattern + /// used throughout this class for handlers that must perform async cleanup. + /// partial void OnSelectedModelChanged(ISpeechModel? value) + { + _ = HandleSelectedModelChangedAsync(); + } + + /// + /// Performs the async stop-then-invalidate-engine work for . + /// + private async Task HandleSelectedModelChangedAsync() { if (CanStop) { - Stop(); + await StopAsync().ConfigureAwait(true); } - InvalidateRecognizer(); + await InvalidateEngineAsync().ConfigureAwait(true); } /// - /// Disposes and clears the cached recognizer and its bound capture device, if any, so the - /// next Start builds a fresh pair. Safe to call when nothing is cached. + /// Maps a to the corresponding + /// , giving every transition a single, predictable + /// source of truth instead of ad hoc assignment at each call site. /// - private void InvalidateRecognizer() + /// The session state to map. + /// The corresponding streaming state. + private static RecognitionStreamingState MapSessionState(RecognitionSessionState state) => state switch { - if (_recognizer is null) + RecognitionSessionState.Starting or RecognitionSessionState.Running or RecognitionSessionState.Stopping => + RecognitionStreamingState.Listening, + RecognitionSessionState.Faulted => RecognitionStreamingState.Error, + _ => RecognitionStreamingState.Idle, + }; + + /// + /// Marshals one event onto the UI thread and + /// applies . + /// + /// The raising session. Unused. + /// The event carrying the previous and current session state. + private void OnSessionStateChanged(object? sender, SessionStateChangedEventArgs e) + { + if (_uiContext is null) + { + ApplySessionStateChanged(e); + } + else + { + _uiContext.Post(state => ApplySessionStateChanged((SessionStateChangedEventArgs)state!), e); + } + } + + /// + /// Applies one session state transition to , additionally reporting + /// when the session has faulted. + /// + /// The event carrying the previous and current session state. + private void ApplySessionStateChanged(SessionStateChangedEventArgs e) + { + State = MapSessionState(e.Current); + + if (e.Current == RecognitionSessionState.Faulted) + { + StatusMessage = SessionFaultedMessage; + } + } + + /// + /// Releases and clears the cached session, if any, first canceling its result-pump + /// enumeration and awaiting that pump task, then disposing the session itself - which + /// releases the engine's exclusivity lease so a new session can be created. Safe to call + /// when nothing is cached. Does not touch the cached engine. + /// + private async Task InvalidateSessionAsync() + { + var session = _session; + if (session is null) { return; } - _recognizer.ResultReceived -= OnResultReceived; - _recognizer.Dispose(); - _recognizer = null; - _captureDevice = null; + session.StateChanged -= OnSessionStateChanged; + + if (_pumpCancellation is not null) + { + await _pumpCancellation.CancelAsync().ConfigureAwait(true); + } + + var pumpTask = _pumpTask; + if (pumpTask is not null) + { + try + { + await pumpTask.ConfigureAwait(true); + } + catch + { + // Already reported through the pump's own catch blocks; this await only drains it. + } + } + + await session.DisposeAsync().ConfigureAwait(true); + + _pumpCancellation?.Dispose(); + _pumpCancellation = null; + _pumpTask = null; + _session = null; + } + + /// + /// Releases and clears the cached session (see ), then + /// disposes and clears the cached engine, if any, so the next Start loads a fresh one. + /// Safe to call when nothing is cached. + /// + private async Task InvalidateEngineAsync() + { + await InvalidateSessionAsync().ConfigureAwait(true); + + if (_engine is null) + { + return; + } + + await _engine.DisposeAsync().ConfigureAwait(true); + _engine = null; + _engineModel = null; } /// @@ -329,11 +471,12 @@ public void Refresh() /// /// Starts a streaming transcription session for the currently selected model through the - /// currently selected capture device, building a new recognizer only if none is already - /// cached for this model/device pair (see the "Recognizer reuse" remarks above). + /// currently selected capture device, loading a new engine only if none is already cached + /// for this model (see the "Engine/session reuse" remarks above), and always creating a + /// fresh session since a session is single-use. /// [RelayCommand(CanExecute = nameof(CanStart))] - private void Start() + private async Task StartAsync() { var selectedModel = SelectedModel; if (selectedModel is null) @@ -343,28 +486,48 @@ private void Start() return; } - if (_recognizer is null) + // Check the cheap, synchronous precondition (a usable capture device) before paying for + // the comparatively expensive async engine load, so an honest "no device" outcome does + // not depend on what an engine-loading seam happens to return for this combination. + var captureDevice = _deviceService.CreateCaptureDevice(_deviceSelection.CaptureSelection); + if (!captureDevice.IsAvailable) { - var captureDevice = _deviceService.CreateCaptureDevice(_deviceSelection.CaptureSelection); - if (!captureDevice.IsAvailable) - { - StatusMessage = NoCaptureDeviceMessage; - State = RecognitionStreamingState.Error; - return; - } + StatusMessage = NoCaptureDeviceMessage; + State = RecognitionStreamingState.Error; + return; + } + + if (_engine is null || !ReferenceEquals(_engineModel, selectedModel)) + { + await InvalidateEngineAsync().ConfigureAwait(true); - var recognizer = _sessionFactory.Create(selectedModel, captureDevice); - if (!recognizer.IsAvailable) + var engine = await _sessionFactory.LoadAsync(selectedModel, CancellationToken.None).ConfigureAwait(true); + if (!engine.IsAvailable) { - recognizer.Dispose(); + await engine.DisposeAsync().ConfigureAwait(true); StatusMessage = RecognizerUnavailableMessage; State = RecognitionStreamingState.Error; return; } - recognizer.ResultReceived += OnResultReceived; - _recognizer = recognizer; - _captureDevice = captureDevice; + _engine = engine; + _engineModel = selectedModel; + } + + // A previous run's session is single-use and must be released before a new one can be + // leased from the engine (see IRecognitionSession's single-use/engine-exclusivity remarks). + await InvalidateSessionAsync().ConfigureAwait(true); + + IRecognitionSession session; + try + { + session = await _engine.CreateSessionAsync(captureDevice, CancellationToken.None).ConfigureAwait(true); + } + catch (RecognitionEngineBusyException ex) + { + StatusMessage = ex.Message; + State = RecognitionStreamingState.Error; + return; } Finals.Clear(); @@ -372,99 +535,123 @@ private void Start() StatusMessage = null; _uiContext = SynchronizationContext.Current; + session.StateChanged += OnSessionStateChanged; // TEMPORARY diagnostic instrumentation: an independent tap on the same capture device, // used only when DEMASPEECH_CAPTURE_DEBUG_DIR is set (see CaptureDebugRecorder), for // investigating a reported dropped-word bug after a mid-sentence pause. Subscribing here - // does not affect the recognizer above, which owns the device's Start/Stop lifecycle. - _captureDebugRecorder = CaptureDebugRecorder.TryStart(_captureDevice!); + // does not affect the session above, which owns the device's Start/Stop lifecycle. + _captureDebugRecorder = CaptureDebugRecorder.TryStart(captureDevice); try { - _recognizer.Start(); - State = RecognitionStreamingState.Listening; + await session.StartAsync(CancellationToken.None).ConfigureAwait(true); } catch (SpeechRecognizerUnavailableException ex) { _captureDebugRecorder?.Dispose(); _captureDebugRecorder = null; - // The cached recognizer itself reported a hard failure starting its device - treat it - // as unusable for the rest of its life rather than retrying the same broken instance; - // the next Start attempt builds a fresh recognizer/device pair. - InvalidateRecognizer(); + session.StateChanged -= OnSessionStateChanged; + await session.DisposeAsync().ConfigureAwait(true); + _uiContext = null; + StatusMessage = ex.Message; State = RecognitionStreamingState.Error; + return; } + + _pumpCancellation = new CancellationTokenSource(); + _session = session; + _pumpTask = PumpResultsAsync(session, _pumpCancellation.Token); } /// - /// Stops the in-flight streaming transcription session, if any, without discarding the - /// underlying recognizer - it remains cached, with its model still loaded, so the next - /// Start is cheap. A safe no-op when nothing is listening. + /// Stops the in-flight streaming transcription session, if any, awaiting its result pump + /// to fully drain before returning, without discarding the cached engine - it remains + /// cached, with its model still loaded, so the next Start is cheap. A safe no-op when + /// nothing is listening, and safe to call concurrently with itself. /// - [RelayCommand(CanExecute = nameof(CanStop))] - private void Stop() + [RelayCommand(CanExecute = nameof(CanStop), AllowConcurrentExecutions = true)] + private async Task StopAsync() { - if (_recognizer is null) + var session = _session; + if (session is null) { return; } - _recognizer.Stop(); - _uiContext = null; + await session.StopAsync(CancellationToken.None).ConfigureAwait(true); + + var pumpTask = _pumpTask; + if (pumpTask is not null) + { + try + { + await pumpTask.ConfigureAwait(true); + } + catch + { + // Already reported through the pump's own catch blocks; this await only drains it. + } + } // TEMPORARY diagnostic instrumentation: finalize the independent raw-capture recording // (if one was started) so its .wav header is patched and the file is closed _captureDebugRecorder?.Dispose(); _captureDebugRecorder = null; - State = RecognitionStreamingState.Idle; - StatusMessage = StoppedMessage; - } + _uiContext = null; - /// - /// Stops an actively listening session, if any, and invalidates the cached recognizer so - /// the shared device-selection panel can safely force the audio backend to re-scan its - /// device table: the recognizer's bound capture device would otherwise become stale the - /// instant the refresh completes, so it must not be reused once one is pending. - /// - /// - /// A synchronously completed task: is documented to be fully - /// synchronous down to the capture device's own closure (it blocks on draining the - /// recognizer before returning), so by the time this method returns the capture device is - /// already closed and no real awaiting ever occurs. The - /// shape exists only to satisfy the - /// - /// contract, which must support hooks that do need to await (see - /// 's equivalent hook). - /// - private Task StopBeforeDeviceRefreshAsync() - { - if (CanStop) + if (State != RecognitionStreamingState.Error) { - Stop(); + StatusMessage = StoppedMessage; } - - InvalidateRecognizer(); - - return Task.CompletedTask; } /// - /// Marshals one recognition result onto the UI thread and applies it to the transcript. + /// Stops an actively listening session, if any, and releases the cached session so the + /// shared device-selection panel can safely force the audio backend to re-scan its device + /// table: the session's bound capture device would otherwise become stale the instant the + /// refresh completes, so it must not be reused once one is pending. The cached engine is + /// left intact, since a device refresh does not affect the loaded model. /// - /// The raising recognizer. Unused. - /// The recognition event carrying the result to apply. - private void OnResultReceived(object? sender, SpeechRecognitionEvent e) + /// A task that completes once the in-flight session, if any, is fully released. + private Task StopBeforeDeviceRefreshAsync() => HandleCaptureDeviceChangedAsync(); + + /// + /// Pumps for one session, applying every + /// result to the transcript as it arrives. Runs as a background task stored on + /// ; relies on each await foreach continuation resuming on + /// the UI thread context captured by its caller () rather than + /// explicit marshaling. + /// + /// The session to pump results from. + /// + /// A token that ends only this enumeration (not the session) when the session is being + /// released while still producing results. + /// + private async Task PumpResultsAsync(IRecognitionSession session, CancellationToken cancellationToken) { - if (_uiContext is null) + try { - ApplyResult(e.Result); + await foreach (var recognitionEvent in session.GetResultsAsync(cancellationToken).ConfigureAwait(true)) + { + ApplyResult(recognitionEvent.Result); + } } - else + catch (OperationCanceledException) { - _uiContext.Post(state => ApplyResult((SpeechRecognitionResult)state!), e.Result); + // Enumeration was ended by invalidation (model/device change or disposal), not a + // session fault. + } + catch (RecognitionSessionFaultedException ex) + { + StatusMessage = ex.Message; + } + catch (SpeechRecognizerUnavailableException ex) + { + StatusMessage = ex.Message; } } @@ -532,19 +719,24 @@ private void OnModelInstalled(object? sender, ModelInstalledEventArgs e) } /// - /// Stops any in-flight session and releases the cached recognizer and capture device it - /// holds, and unsubscribes from and the + /// Stops any in-flight session and releases the cached session and engine it holds, and + /// unsubscribes from and the /// device-selection panel's change notifications. /// - public void Dispose() + public async ValueTask DisposeAsync() { _catalogService.ModelInstalled -= OnModelInstalled; _deviceSelection.PropertyChanged -= OnDeviceSelectionChanged; _deviceSelection.UnregisterPreRefreshHook(_preRefreshHook); + if (CanStop) + { + await StopAsync().ConfigureAwait(true); + } + _captureDebugRecorder?.Dispose(); _captureDebugRecorder = null; - InvalidateRecognizer(); + await InvalidateEngineAsync().ConfigureAwait(true); } } diff --git a/src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/RecognizerSessionFactory.cs b/src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/RecognizerSessionFactory.cs index cd992b6..7038201 100644 --- a/src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/RecognizerSessionFactory.cs +++ b/src/DemaConsulting.Speech.Demo/RecognitionPanelSubsystem/RecognizerSessionFactory.cs @@ -1,4 +1,3 @@ -using DemaConsulting.Speech.AudioSubsystem; using DemaConsulting.Speech.ModelManagementSubsystem; using DemaConsulting.Speech.RecognitionSubsystem; @@ -12,12 +11,12 @@ namespace DemaConsulting.Speech.Demo.RecognitionPanelSubsystem; /// This adapter resolves the model's installed-files directory from the shared /// , narrows to the /// the library's factory requires, and forwards to -/// , +/// , /// inheriting that factory's "nothing throws at composition" contract. A model that declares /// a role other than recognition (and therefore is not an ) is /// an honest unavailable outcome, exactly like a model that is not installed, rather than a /// defect: a host that lets a user choose an installed model with the wrong role must still -/// get a working, if unavailable, recognizer back. +/// get a working, if unavailable, engine back. /// public sealed class RecognizerSessionFactory : IRecognizerSessionFactory { @@ -37,16 +36,16 @@ public RecognizerSessionFactory(SpeechModelStore store) } /// - public ISpeechRecognizer Create(ISpeechModel model, IAudioCaptureDevice captureDevice) + public async Task LoadAsync(ISpeechModel model, CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(model); - ArgumentNullException.ThrowIfNull(captureDevice); if (model is not IRecognitionModel recognitionModel) { - return UnavailableSpeechRecognizer.Instance; + return UnavailableSpeechRecognizerEngine.Instance; } - return SpeechRecognizerFactory.Create(recognitionModel, _store, captureDevice); + return await SpeechRecognizerFactory.LoadAsync(recognitionModel, _store, cancellationToken: cancellationToken) + .ConfigureAwait(false); } } diff --git a/src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/ISynthesizerSessionFactory.cs b/src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/ISynthesizerSessionFactory.cs index 7959d21..17576f0 100644 --- a/src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/ISynthesizerSessionFactory.cs +++ b/src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/ISynthesizerSessionFactory.cs @@ -1,15 +1,14 @@ -using DemaConsulting.Speech.AudioSubsystem; using DemaConsulting.Speech.ModelManagementSubsystem; using DemaConsulting.Speech.SynthesisSubsystem; namespace DemaConsulting.Speech.Demo.SynthesisPanelSubsystem; /// -/// Demo-owned seam over the library's synthesis-session composition surface. +/// Demo-owned seam over the library's synthesis-engine composition surface. /// /// -/// The library exposes synthesizer composition through the static -/// +/// The library exposes engine composition through the static +/// /// method, which requires an - an interface whose members are /// partly to the library, so only the library's own assemblies can /// implement it. This seam therefore accepts the common contract @@ -20,37 +19,36 @@ namespace DemaConsulting.Speech.Demo.SynthesisPanelSubsystem; /// implementation supplies the shared itself and delegates /// straight to that factory, which resolves the model's installed-files directory /// internally, while tests substitute a fake that returns a controlled -/// without a downloaded model, a native runtime, or a real -/// playback device. It adds no public API to DemaConsulting.Speech. +/// without a downloaded model, a native runtime, or a +/// real playback device. It adds no public API to DemaConsulting.Speech. +/// +/// Only the engine is composed here - a playback device is bound later, per run, via +/// directly against the returned +/// engine, so a host can load one engine per model/parameter combination and reuse it across +/// many sessions instead of reloading the model on every Play. +/// /// public interface ISynthesizerSessionFactory { /// - /// Creates a speech synthesizer for an installed model, playing back through the supplied - /// device. + /// Loads a speech synthesizer engine for an installed model. /// /// The model to load. Must not be . - /// - /// The playback device to play synthesized audio through. Must not be . - /// /// /// An optional session-level parameter value bag (for example a selected voice, built by /// a host's settings UI from the model's declared ), - /// forwarded unchanged to the library's synthesizer composition, or - /// to use the model's own default voice/speaker. + /// forwarded unchanged to the library's engine composition, or to + /// use the model's own default voice/speaker. /// + /// A token to observe for cancellation of this call. /// - /// A real synthesizer when the model is installed, implements the library's synthesis - /// role, and the device is available; otherwise an honest IsAvailable == false - /// synthesizer. + /// A task that completes with a real engine when the model is installed and implements + /// the library's synthesis role; otherwise an honest IsAvailable == false engine. /// - /// - /// Thrown when or is - /// . - /// - /// Never throws; every honest unavailable state is reported through IsAvailable. - ISpeechSynthesizer Create( + /// Thrown when is . + /// Never throws for an ordinary unavailable state; every such state is reported through IsAvailable. + Task LoadAsync( ISpeechModel model, - IAudioPlaybackDevice playbackDevice, - IReadOnlyDictionary? parameterValues = null); + IReadOnlyDictionary? parameterValues, + CancellationToken cancellationToken = default); } diff --git a/src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/SynthesisPanelViewModel.cs b/src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/SynthesisPanelViewModel.cs index 50519d1..51c7975 100644 --- a/src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/SynthesisPanelViewModel.cs +++ b/src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/SynthesisPanelViewModel.cs @@ -38,10 +38,23 @@ namespace DemaConsulting.Speech.Demo.SynthesisPanelSubsystem; /// ignores events for any other and marshals onto the UI thread /// (captured at construction) before calling , mirroring /// 's identical pattern. -/// unsubscribes this handler. +/// unsubscribes this handler. +/// +/// +/// Engine/session reuse. Per 's and +/// 's own reuse guidance, this ViewModel loads an engine at +/// most once per selected model/parameter-value combination, and creates a session at most +/// once per engine/playback-device combination, reusing both across many Play calls instead +/// of reloading the model and recreating the session on every click (the bug this redesign +/// fixes). detects a reason to reload lazily, on its next call, +/// rather than proactively on a property change: the cached engine is reloaded only when the +/// selected model or ' built parameter values differ from the ones it +/// was last loaded with; the cached session is independently recreated whenever the engine +/// was just reloaded or the selected playback device differs from the one it was last bound +/// to. /// /// -public sealed partial class SynthesisPanelViewModel : ObservableObject, IDisposable +public sealed partial class SynthesisPanelViewModel : ObservableObject, IAsyncDisposable { /// The message shown when no installed synthesis model is available to choose. public const string NoModelsMessage = @@ -55,13 +68,16 @@ public sealed partial class SynthesisPanelViewModel : ObservableObject, IDisposa public const string NoPlaybackDeviceMessage = "No audio playback device is available. Choose one on the Audio Devices panel."; - /// The message shown when the composed synthesizer honestly reports itself unavailable. + /// The message shown when the composed synthesizer engine honestly reports itself unavailable. public const string SynthesizerUnavailableMessage = "The selected model's speech synthesizer is unavailable on this machine."; /// The message shown after playback is stopped by the user. public const string StoppedMessage = "Playback stopped."; + /// The message shown when the active session reports an unrecoverable fault. + public const string SessionFaultedMessage = "The synthesis session reported an unrecoverable error."; + /// Example Natural Language Audio Tags drawn from the library's closed, published vocabulary. public static IReadOnlyList ExampleTagHints { get; } = AudioTagCatalog.Tags.Select(descriptor => $"[{descriptor.Aliases[0]}]").ToArray(); @@ -75,7 +91,7 @@ public sealed partial class SynthesisPanelViewModel : ObservableObject, IDisposa /// The shared device-selection panel state supplying the chosen playback device. private readonly DeviceSelectionViewModel _deviceSelection; - /// The seam used to compose a synthesizer for the selected model and device. + /// The seam used to load a synthesizer engine for the selected model. private readonly ISynthesizerSessionFactory _sessionFactory; /// @@ -85,12 +101,39 @@ public sealed partial class SynthesisPanelViewModel : ObservableObject, IDisposa /// private readonly SynchronizationContext? _catalogEventUiContext = SynchronizationContext.Current; - /// The synthesizer currently in use by an in-flight Play, if any. - private ISpeechSynthesizer? _activeSynthesizer; + /// + /// The cached synthesizer engine, reused across many Play calls for as long as + /// and match the current + /// selection; see the "Engine/session reuse" remarks above. before + /// the first Play, or after invalidation by a model/parameter change. + /// + private ISpeechSynthesizerEngine? _engine; + + /// The model was loaded for, if any. + private ISpeechModel? _engineModel; + + /// The parameter value bag was loaded with, if any. + private IReadOnlyDictionary? _engineParameterValues; + + /// + /// The cached session created from , reused across many Play calls + /// for as long as it is bound to the currently selected playback device. + /// before the first Play, or after invalidation. + /// + private ISynthesisSession? _session; + + /// The playback device selection was created with, if any. + private AudioDeviceSelection? _sessionPlaybackSelection; + + /// + /// The UI thread context captured when was created, used to + /// marshal handling onto the UI thread. + /// + private SynchronizationContext? _uiContext; /// /// The pre-refresh hook registered with , retained so the - /// exact same delegate instance can be unregistered in . + /// exact same delegate instance can be unregistered in . /// private readonly Func _preRefreshHook; @@ -175,7 +218,7 @@ public sealed partial class SynthesisPanelViewModel : ObservableObject, IDisposa /// The catalog seam to read installed models from. Must not be . /// The device seam used to create the playback device. Must not be . /// The shared device-selection panel state. Must not be . - /// The seam used to compose a synthesizer. Must not be . + /// The seam used to load a synthesizer engine. Must not be . /// Thrown when any parameter is . internal SynthesisPanelViewModel( IModelCatalogService catalogService, @@ -232,11 +275,99 @@ public void Refresh() /// Applies a newly selected model's declared parameters to the embedded settings panel. /// /// The newly selected model. + /// + /// Does not proactively invalidate the cached engine/session: + /// detects the model change lazily on its next call (see the "Engine/session reuse" + /// remarks above) by comparing against . + /// partial void OnSelectedModelChanged(ISpeechModel? value) => Settings.Model = value; + /// + /// Maps a to the corresponding + /// , giving every transition a single, predictable + /// source of truth instead of ad hoc assignment at each call site. + /// + /// The session state to map. + /// The corresponding playback state. + private static SynthesisPlaybackState MapSessionState(SynthesisSessionState state) => state switch + { + SynthesisSessionState.Starting => SynthesisPlaybackState.Synthesizing, + SynthesisSessionState.Running or SynthesisSessionState.Stopping => SynthesisPlaybackState.Playing, + SynthesisSessionState.Faulted => SynthesisPlaybackState.Error, + _ => SynthesisPlaybackState.Idle, + }; + + /// + /// Marshals one event onto the UI thread and + /// applies . + /// + /// The raising session. Unused. + /// The event carrying the previous and current session state. + private void OnSessionStateChanged(object? sender, SessionStateChangedEventArgs e) + { + if (_uiContext is null) + { + ApplySessionStateChanged(e); + } + else + { + _uiContext.Post(state => ApplySessionStateChanged((SessionStateChangedEventArgs)state!), e); + } + } + + /// + /// Applies one session state transition to , additionally reporting + /// when the session has faulted. + /// + /// The event carrying the previous and current session state. + private void ApplySessionStateChanged(SessionStateChangedEventArgs e) + { + State = MapSessionState(e.Current); + + if (e.Current == SynthesisSessionState.Faulted) + { + StatusMessage = SessionFaultedMessage; + } + } + + /// + /// Compares two parameter value bags for equality by key and value, since + /// returns a freshly built dictionary + /// on every call rather than a stable cached instance. + /// + /// The first bag to compare, or . + /// The second bag to compare, or . + /// when both bags contain the same keys mapped to equal values. + private static bool ParameterValuesEqual( + IReadOnlyDictionary? first, + IReadOnlyDictionary? second) + { + if (ReferenceEquals(first, second)) + { + return true; + } + + if (first is null || second is null || first.Count != second.Count) + { + return false; + } + + foreach (var pair in first) + { + if (!second.TryGetValue(pair.Key, out var value) || !Equals(pair.Value, value)) + { + return false; + } + } + + return true; + } + /// /// Synthesizes and speaks through the currently selected playback - /// device, using the currently selected model. + /// device, using the currently selected model, reloading the cached engine and/or + /// recreating the cached session only when something they were built from has actually + /// changed (see the "Engine/session reuse" remarks above). /// /// /// The token the generated command supplies; canceling it (via ) @@ -254,34 +385,84 @@ private async Task PlayAsync(CancellationToken cancellationToken) return; } - var playbackDevice = _deviceService.CreatePlaybackDevice(_deviceSelection.PlaybackSelection); - if (!playbackDevice.IsAvailable) + var parameterValues = Settings.BuildValueBag(); + + // Check the cheap, synchronous precondition (a usable playback device) before paying for + // the comparatively expensive async engine load, so an honest "no device" outcome does + // not depend on what an engine-loading seam happens to return for this combination. + var desiredSelection = _deviceSelection.PlaybackSelection; + var precheckDevice = _deviceService.CreatePlaybackDevice(desiredSelection); + if (!precheckDevice.IsAvailable) { StatusMessage = NoPlaybackDeviceMessage; State = SynthesisPlaybackState.Error; return; } - using var synthesizer = _sessionFactory.Create(selectedModel, playbackDevice, Settings.BuildValueBag()); - if (!synthesizer.IsAvailable) + if (_engine is null || + !ReferenceEquals(_engineModel, selectedModel) || + !ParameterValuesEqual(_engineParameterValues, parameterValues)) { - StatusMessage = SynthesizerUnavailableMessage; - State = SynthesisPlaybackState.Error; - return; + await InvalidateEngineAsync().ConfigureAwait(true); + + var engine = await _sessionFactory.LoadAsync(selectedModel, parameterValues, cancellationToken).ConfigureAwait(true); + if (!engine.IsAvailable) + { + await engine.DisposeAsync().ConfigureAwait(true); + StatusMessage = SynthesizerUnavailableMessage; + State = SynthesisPlaybackState.Error; + return; + } + + _engine = engine; + _engineModel = selectedModel; + _engineParameterValues = parameterValues; } - _activeSynthesizer = synthesizer; + if (_session is null || !Equals(_sessionPlaybackSelection, desiredSelection)) + { + await ReleaseSessionAsync().ConfigureAwait(true); + + // Reuse the device already resolved by the precondition check above rather than + // resolving it a second time. + var playbackDevice = precheckDevice; + + ISynthesisSession session; + try + { + session = await _engine.CreateSessionAsync(playbackDevice, cancellationToken).ConfigureAwait(true); + } + catch (SynthesisEngineBusyException ex) + { + StatusMessage = ex.Message; + State = SynthesisPlaybackState.Error; + return; + } + + if (!session.IsAvailable) + { + await session.DisposeAsync().ConfigureAwait(true); + StatusMessage = SynthesizerUnavailableMessage; + State = SynthesisPlaybackState.Error; + return; + } + + _uiContext = SynchronizationContext.Current; + session.StateChanged += OnSessionStateChanged; + + _session = session; + _sessionPlaybackSelection = desiredSelection; + } + + StatusMessage = null; + try { - StatusMessage = null; - State = SynthesisPlaybackState.Synthesizing; - State = SynthesisPlaybackState.Playing; - await synthesizer.SpeakAsync(Text, cancellationToken).ConfigureAwait(true); - State = SynthesisPlaybackState.Idle; + await _session.SpeakAsync(Text, cancellationToken).ConfigureAwait(true); } catch (OperationCanceledException) { - State = SynthesisPlaybackState.Idle; + // The session's own StateChanged transitions already drove State back to Idle. StatusMessage = StoppedMessage; } catch (SpeechSynthesizerUnavailableException ex) @@ -289,12 +470,12 @@ private async Task PlayAsync(CancellationToken cancellationToken) State = SynthesisPlaybackState.Error; StatusMessage = ex.Message; } - finally + catch (SynthesisSessionFaultedException ex) { - // Disposal itself is left to the enclosing `using` (covering both this exit path and - // the early "unavailable" return above) so every exit path disposes exactly once - // through a single, unconditional mechanism instead of a duplicated manual call. - _activeSynthesizer = null; + // Faulted is terminal for a session (see ISynthesisSession's remarks): release it so + // the next Play creates a fresh one against the still-cached engine. + await ReleaseSessionAsync().ConfigureAwait(true); + StatusMessage = ex.Message; } } @@ -303,52 +484,93 @@ private async Task PlayAsync(CancellationToken cancellationToken) /// playing. /// [RelayCommand] - private void Stop() + private async Task StopAsync() { - _activeSynthesizer?.Stop(); + var session = _session; + if (session is not null) + { + await session.StopAsync().ConfigureAwait(true); + } + + // Defense-in-depth: also requests cancellation of PlayAsync's own cancellation token, in + // case the session itself cannot be stopped (for example an unavailable fallback). PlayCommand.Cancel(); } /// - /// Stops an in-flight Play, if any, and awaits its actual completion so the shared - /// device-selection panel can safely force the audio backend to re-scan its device table. + /// Stops an in-flight Play, if any, awaits its actual completion, and releases the cached + /// session (keeping the cached engine) so the shared device-selection panel can safely + /// force the audio backend to re-scan its device table - the session's bound playback + /// device would otherwise become stale the instant the refresh completes. /// /// /// A task that completes once the in-flight execution (if any) - /// has itself completed. + /// has itself completed and the cached session has been released. /// - /// - /// Unlike 's equivalent - /// hook, alone does not guarantee the playback device is actually - /// closed by the time it returns: only cancels the synthesizer's - /// internal token and requests cancellation of ; the device is - /// actually stopped later, one layer deeper inside the synthesizer's own streaming - /// playback implementation, as part of the already-in-flight task's own cleanup. So this - /// hook must also await to observe that - /// completion before returning, swallowing the expected - /// that 's cancellation causes. - /// private async Task StopBeforeDeviceRefreshAsync() { - if (!PlayCommand.IsRunning) + if (PlayCommand.IsRunning) + { + await StopAsync().ConfigureAwait(true); + + if (PlayCommand.ExecutionTask is { } executionTask) + { + try + { + await executionTask.ConfigureAwait(true); + } + catch (OperationCanceledException) + { + // Expected: StopAsync() cancels the in-flight PlayAsync, which surfaces as + // OperationCanceledException from its awaited ExecutionTask. + } + } + } + + await ReleaseSessionAsync().ConfigureAwait(true); + } + + /// + /// Releases and clears the cached session, if any, unsubscribing from + /// first so a subsequent disposal transition + /// cannot be mistaken for a fresh fault. Safe to call when nothing is cached. Does not + /// touch the cached engine. + /// + private async Task ReleaseSessionAsync() + { + var session = _session; + if (session is null) { return; } - Stop(); + session.StateChanged -= OnSessionStateChanged; + + await session.DisposeAsync().ConfigureAwait(true); + + _session = null; + _sessionPlaybackSelection = null; + _uiContext = null; + } + + /// + /// Releases the cached session (see ), then disposes and + /// clears the cached engine, if any, so the next Play loads a fresh one. Safe to call when + /// nothing is cached. + /// + private async Task InvalidateEngineAsync() + { + await ReleaseSessionAsync().ConfigureAwait(true); - if (PlayCommand.ExecutionTask is { } executionTask) + if (_engine is null) { - try - { - await executionTask.ConfigureAwait(true); - } - catch (OperationCanceledException) - { - // Expected: Stop() cancels the in-flight PlayAsync, which surfaces as - // OperationCanceledException from its awaited ExecutionTask. - } + return; } + + await _engine.DisposeAsync().ConfigureAwait(true); + _engine = null; + _engineModel = null; + _engineParameterValues = null; } /// @@ -376,16 +598,32 @@ private void OnModelInstalled(object? sender, ModelInstalledEventArgs e) } /// - /// Unsubscribes from and, defensively, - /// disposes an active synthesizer if one is still held (ordinarily released by - /// 's own finally block, but no longer guaranteed once this - /// ViewModel can be disposed independently of any in-flight Play). + /// Stops any in-flight Play, unsubscribes from + /// and the device-selection panel's pre-refresh hook, and releases the cached session and + /// engine, if any. /// - public void Dispose() + public async ValueTask DisposeAsync() { _catalogService.ModelInstalled -= OnModelInstalled; _deviceSelection.UnregisterPreRefreshHook(_preRefreshHook); - _activeSynthesizer?.Dispose(); - _activeSynthesizer = null; + + if (PlayCommand.IsRunning) + { + await StopAsync().ConfigureAwait(true); + + if (PlayCommand.ExecutionTask is { } executionTask) + { + try + { + await executionTask.ConfigureAwait(true); + } + catch (OperationCanceledException) + { + // Expected: StopAsync() cancels the in-flight PlayAsync. + } + } + } + + await InvalidateEngineAsync().ConfigureAwait(true); } } diff --git a/src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/SynthesizerSessionFactory.cs b/src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/SynthesizerSessionFactory.cs index b62996a..a0f8f55 100644 --- a/src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/SynthesizerSessionFactory.cs +++ b/src/DemaConsulting.Speech.Demo/SynthesisPanelSubsystem/SynthesizerSessionFactory.cs @@ -1,4 +1,3 @@ -using DemaConsulting.Speech.AudioSubsystem; using DemaConsulting.Speech.ModelManagementSubsystem; using DemaConsulting.Speech.SynthesisSubsystem; @@ -12,13 +11,13 @@ namespace DemaConsulting.Speech.Demo.SynthesisPanelSubsystem; /// This adapter narrows to the the /// library's factory requires and forwards it, together with the shared /// supplied at construction, to -/// , +/// , /// which resolves the model's installed-files directory itself, inheriting that factory's /// "nothing throws at composition" contract. A model that declares /// a role other than synthesis (and therefore is not an ) is an /// honest unavailable outcome, exactly like a model that is not installed, rather than a /// defect: a host that lets a user choose an installed model with the wrong role must still -/// get a working, if unavailable, synthesizer back. +/// get a working, if unavailable, engine back. /// public sealed class SynthesizerSessionFactory : ISynthesizerSessionFactory { @@ -38,19 +37,20 @@ public SynthesizerSessionFactory(SpeechModelStore store) } /// - public ISpeechSynthesizer Create( + public async Task LoadAsync( ISpeechModel model, - IAudioPlaybackDevice playbackDevice, - IReadOnlyDictionary? parameterValues = null) + IReadOnlyDictionary? parameterValues, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(model); - ArgumentNullException.ThrowIfNull(playbackDevice); if (model is not ISynthesisModel synthesisModel) { - return UnavailableSpeechSynthesizer.Instance; + return UnavailableSpeechSynthesizerEngine.Instance; } - return SpeechSynthesizerFactory.Create(synthesisModel, _store, playbackDevice, parameterValues: parameterValues); + return await SpeechSynthesizerFactory + .LoadAsync(synthesisModel, _store, parameterValues: parameterValues, cancellationToken: cancellationToken) + .ConfigureAwait(false); } } diff --git a/src/DemaConsulting.Speech/AudioSubsystem/WavFileAudioCaptureDevice.cs b/src/DemaConsulting.Speech/AudioSubsystem/WavFileAudioCaptureDevice.cs index 83d43fe..9ad8d5c 100644 --- a/src/DemaConsulting.Speech/AudioSubsystem/WavFileAudioCaptureDevice.cs +++ b/src/DemaConsulting.Speech/AudioSubsystem/WavFileAudioCaptureDevice.cs @@ -12,10 +12,12 @@ namespace DemaConsulting.Speech.AudioSubsystem; /// /// This is the correct layer for this capability (not a CLI-only helper) because /// is exactly the seam -/// SpeechRecognizerFactory.Create already requires, so any host application - not -/// only a future command-line tool - can drive speech recognition from a pre-recorded -/// file deterministically by supplying this class wherever a real capture device would -/// otherwise be used. +/// +/// already requires - the point at which a loaded recognizer engine is bound to a +/// capture device for a session's entire life - so any host application, including the +/// shipped recognize --input command-line tool, can drive speech recognition +/// from a pre-recorded file deterministically by supplying this class wherever a real +/// capture device would otherwise be used. /// /// /// Because the public contract has no "end of stream" diff --git a/src/DemaConsulting.Speech/AudioSubsystem/WavFileAudioPlaybackDevice.cs b/src/DemaConsulting.Speech/AudioSubsystem/WavFileAudioPlaybackDevice.cs index 711f696..f064aeb 100644 --- a/src/DemaConsulting.Speech/AudioSubsystem/WavFileAudioPlaybackDevice.cs +++ b/src/DemaConsulting.Speech/AudioSubsystem/WavFileAudioPlaybackDevice.cs @@ -9,10 +9,12 @@ namespace DemaConsulting.Speech.AudioSubsystem; /// /// This is the correct layer for this capability (not a CLI-only helper) because /// is exactly the seam -/// SpeechSynthesizerFactory.Create already requires, so any host application - -/// not only a future command-line tool - can capture synthesized speech to a file -/// deterministically (for example, for its own automated tests) by supplying this class -/// wherever a real playback device would otherwise be used. +/// 's +/// CreateSessionAsync/SpeakAsync +/// surface already requires, so any host application - including the shipped +/// speak --output-audio command-line tool - can capture synthesized speech to a +/// file deterministically (for example, for its own automated tests) by supplying this +/// class wherever a real playback device would otherwise be used. /// /// /// The RIFF/WAVE byte layout and clamp-before-scale rounding behavior deliberately match diff --git a/src/DemaConsulting.Speech/DemaConsulting.Speech.csproj b/src/DemaConsulting.Speech/DemaConsulting.Speech.csproj index 93afaaf..5f02195 100644 --- a/src/DemaConsulting.Speech/DemaConsulting.Speech.csproj +++ b/src/DemaConsulting.Speech/DemaConsulting.Speech.csproj @@ -17,10 +17,11 @@ DEMA Consulting Cross-platform .NET library providing local, offline speech-to-text and text-to-speech services for desktop applications. Start here: SpeechModelCatalog (enumerate - and download models), and AudioDeviceFactory, SpeechRecognizerFactory and - SpeechSynthesizerFactory (compose devices/recognizers/synthesizers). None of the latter - three throw for an ordinary machine state such as missing hardware or an uninstalled model; - check the returned instance's IsAvailable instead. See the README and user guide for a full + and download models), AudioDeviceFactory (create capture/playback devices), and + SpeechRecognizerFactory.LoadAsync/SpeechSynthesizerFactory.LoadAsync (async-load an engine, + then CreateSessionAsync a device-bound session). None of these throw for an ordinary + machine state such as missing hardware or an uninstalled model; check the returned + engine/session's IsAvailable instead. See the README and user guide for a full walkthrough. MIT https://github.com/demaconsulting/Speech @@ -149,14 +150,18 @@ assemblies). Links are absolute GitHub URLs because api.md is packed into the NuGet package, which has no access to repository-relative paths. --> DemaConsulting.Speech is a cross-platform .NET library providing local, offline - speech-to-text (STT) and text-to-speech (TTS) services for desktop applications. + speech-to-text (STT) and text-to-speech (TTS) services for desktop applications. The public + API is fully asynchronous. Start here: `SpeechModelCatalog` (enumerate/download STT/TTS models), `AudioDeviceFactory` - (create capture/playback audio devices), `SpeechRecognizerFactory` (compose a speech - recognizer from a model and capture device), and `SpeechSynthesizerFactory` (compose a - speech synthesizer from a model and playback device). None of the latter three throw for an - ordinary machine state such as missing hardware or an uninstalled model; check the returned - instance's `IsAvailable` instead. + (create capture/playback audio devices), and `SpeechRecognizerFactory.LoadAsync`/ + `SpeechSynthesizerFactory.LoadAsync` (async-load a speech engine from a model). Each loaded + engine's `CreateSessionAsync` then creates a cheap session bound to one concrete audio + device for its whole life; a recognition session streams results via + `IRecognitionSession.GetResultsAsync` (an `IAsyncEnumerable` of `SpeechRecognitionEvent`), + while a synthesis session exposes `SpeakAsync`/`SynthesizeAsync`. None of `LoadAsync`, + `CreateSessionAsync`, or the factories throw for an ordinary machine state such as missing + hardware or an uninstalled model; check the returned engine/session's `IsAvailable` instead. See the [README](https://github.com/demaconsulting/Speech/blob/main/README.md) and the [user guide](https://github.com/demaconsulting/Speech/blob/main/docs/user_guide/introduction.md) diff --git a/src/DemaConsulting.Speech/ModelManagementSubsystem/IRecognitionModel.cs b/src/DemaConsulting.Speech/ModelManagementSubsystem/IRecognitionModel.cs index ec025a1..63e1638 100644 --- a/src/DemaConsulting.Speech/ModelManagementSubsystem/IRecognitionModel.cs +++ b/src/DemaConsulting.Speech/ModelManagementSubsystem/IRecognitionModel.cs @@ -89,7 +89,7 @@ public interface IRecognitionModel : ISpeechModel /// absolute encoder/decoder/joiner/tokens paths the engine requires. /// /// - /// The untyped key-value bag supplied to SpeechRecognizerFactory.Create (for + /// The untyped key-value bag supplied to SpeechRecognizerFactory.LoadAsync (for /// example built from a host's settings UI via a declared /// entry), or when the caller supplied none. /// diff --git a/src/DemaConsulting.Speech/ModelManagementSubsystem/ISynthesisModel.cs b/src/DemaConsulting.Speech/ModelManagementSubsystem/ISynthesisModel.cs index 566c166..9b6860d 100644 --- a/src/DemaConsulting.Speech/ModelManagementSubsystem/ISynthesisModel.cs +++ b/src/DemaConsulting.Speech/ModelManagementSubsystem/ISynthesisModel.cs @@ -21,7 +21,7 @@ namespace DemaConsulting.Speech.ModelManagementSubsystem; /// Both members are deliberately rather than public, mirroring /// 's identical Phase 3 pattern: this library's "engine /// backend stays swappable at the public API surface" constraint is scoped to -/// ISpeechSynthesizer, while each per-model backing class is architecturally +/// ISpeechSynthesizerEngine, while each per-model backing class is architecturally /// responsible for "sherpa-onnx configuration for its own model architecture" - so returning /// a real here is consistent with the approved design. Keeping /// the members internal means this interface's public surface is unchanged, no @@ -43,7 +43,7 @@ public interface ISynthesisModel : ISpeechModel /// so the playback device attempts to open near the model's expected output format before /// synthesis starts, potentially reducing or eliminating later resampling work. The real /// source of truth remains the constructed engine's - /// , which is read only after + /// , which is read only after /// has been used to load the native engine. Callers must /// therefore still handle a mismatch by resampling playback audio after construction. /// @@ -100,11 +100,11 @@ public interface ISynthesisModel : ISpeechModel /// every existing model's previous hard-coded behavior. /// /// - /// The untyped key-value bag supplied to SpeechSynthesizerFactory.Create (for + /// The untyped key-value bag supplied to SpeechSynthesizerFactory.LoadAsync (for /// example built from a host's settings UI via a declared ), /// or when the caller supplied none. /// - /// The sherpa-onnx speaker id to pass to ISynthesisEngine.Generate. + /// The sherpa-onnx speaker id to pass to ISynthesisBackend.Generate. /// /// Added so a multi-speaker model (starting with /// ) can own its own string-to-int diff --git a/src/DemaConsulting.Speech/ModelManagementSubsystem/SherpaOnnxKokoroEnglishSynthesisModel.cs b/src/DemaConsulting.Speech/ModelManagementSubsystem/SherpaOnnxKokoroEnglishSynthesisModel.cs index 8bf58d9..8d05acf 100644 --- a/src/DemaConsulting.Speech/ModelManagementSubsystem/SherpaOnnxKokoroEnglishSynthesisModel.cs +++ b/src/DemaConsulting.Speech/ModelManagementSubsystem/SherpaOnnxKokoroEnglishSynthesisModel.cs @@ -268,7 +268,7 @@ OfflineTtsConfig ISynthesisModel.CreateEngineConfig(string installedModelDirecto /// to this archive's confirmed integer speaker id. /// /// - /// The untyped key-value bag supplied to SpeechSynthesizerFactory.Create, or + /// The untyped key-value bag supplied to SpeechSynthesizerFactory.LoadAsync, or /// . /// /// diff --git a/src/DemaConsulting.Speech/ModelManagementSubsystem/SherpaOnnxVitsLibriTtsEnglishSynthesisModel.cs b/src/DemaConsulting.Speech/ModelManagementSubsystem/SherpaOnnxVitsLibriTtsEnglishSynthesisModel.cs index 84d7a80..e4c1d3a 100644 --- a/src/DemaConsulting.Speech/ModelManagementSubsystem/SherpaOnnxVitsLibriTtsEnglishSynthesisModel.cs +++ b/src/DemaConsulting.Speech/ModelManagementSubsystem/SherpaOnnxVitsLibriTtsEnglishSynthesisModel.cs @@ -150,7 +150,7 @@ public SherpaOnnxVitsLibriTtsEnglishSynthesisModel() /// 's small, named voice set. The /// default speed/volume conventions /// already looks for are not declared here, since this class relies on the caller's own - /// speed/volume-scaling arguments to ISynthesisEngine.Generate rather than a + /// speed/volume-scaling arguments to ISynthesisBackend.Generate rather than a /// model-declared numeric parameter for those. /// public IReadOnlyList Parameters { get; } = @@ -241,7 +241,7 @@ OfflineTtsConfig ISynthesisModel.CreateEngineConfig(string installedModelDirecto /// to this model's plain numeric sherpa-onnx speaker id. /// /// - /// The untyped key-value bag supplied to SpeechSynthesizerFactory.Create (for + /// The untyped key-value bag supplied to SpeechSynthesizerFactory.LoadAsync (for /// example built from a host's settings UI via this model's declared /// ), or when the caller supplied /// none. diff --git a/src/DemaConsulting.Speech/ModelManagementSubsystem/SpeechModelCatalog.cs b/src/DemaConsulting.Speech/ModelManagementSubsystem/SpeechModelCatalog.cs index 02049c6..5480be7 100644 --- a/src/DemaConsulting.Speech/ModelManagementSubsystem/SpeechModelCatalog.cs +++ b/src/DemaConsulting.Speech/ModelManagementSubsystem/SpeechModelCatalog.cs @@ -148,7 +148,7 @@ internal SpeechModelCatalog( /// a second, potentially divergent . See /// and /// , whose catalog-based - /// Create overloads use this property internally. + /// LoadAsync overloads use this property internally. /// public SpeechModelStore Store => _store; diff --git a/src/DemaConsulting.Speech/ModelManagementSubsystem/SpeechModelParameterDiagnostics.cs b/src/DemaConsulting.Speech/ModelManagementSubsystem/SpeechModelParameterDiagnostics.cs index efc97f6..3936f68 100644 --- a/src/DemaConsulting.Speech/ModelManagementSubsystem/SpeechModelParameterDiagnostics.cs +++ b/src/DemaConsulting.Speech/ModelManagementSubsystem/SpeechModelParameterDiagnostics.cs @@ -13,9 +13,9 @@ namespace DemaConsulting.Speech.ModelManagementSubsystem; /// Per this library's "typed, self-describing descriptor set" grouping, this type is /// co-located with // /// /. It is called exactly once, -/// up front, from SpeechRecognizerFactory.Create and -/// SpeechSynthesizerFactory.Create, before either factory constructs a recognizer or -/// synthesizer - so an invalid recognized value is surfaced synchronously from Create +/// up front, from SpeechRecognizerFactory.LoadAsync and +/// SpeechSynthesizerFactory.LoadAsync, before either factory constructs a recognizer or +/// synthesizer - so an invalid recognized value is surfaced synchronously from LoadAsync /// rather than later, silently, from a per-call runtime hook. /// /// Deliberate split, and why: a supplied key with no matching declared parameter @@ -51,8 +51,8 @@ internal static class SpeechModelParameterDiagnostics /// /// The caller's untyped parameter value bag, or /empty to skip /// validation entirely (there is nothing to validate). Named to match the caller-facing - /// parameterValues parameter on every SpeechRecognizerFactory.Create/ - /// SpeechSynthesizerFactory.Create overload, so a thrown + /// parameterValues parameter on every SpeechRecognizerFactory.LoadAsync/ + /// SpeechSynthesizerFactory.LoadAsync overload, so a thrown /// 's names the /// argument the caller actually supplied. /// diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/DedicatedWorker.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/DedicatedWorker.cs new file mode 100644 index 0000000..6e7b85c --- /dev/null +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/DedicatedWorker.cs @@ -0,0 +1,175 @@ +using DemaConsulting.Speech.Diagnostics; + +namespace DemaConsulting.Speech.RecognitionSubsystem; + +/// +/// Internal utility that runs a delegate on a dedicated, non-pooled thread and applies a +/// cooperative-cancel-then-abandon policy when the caller requests cancellation. +/// +/// +/// An async signature does not make a blocking native call interruptible, so this utility +/// does not pretend otherwise: when 's cancellationToken is +/// cancelled, the delegate - which is expected to observe that same token cooperatively at +/// its own natural boundaries - is given to finish. If it does, +/// the returned task completes normally (or with whatever the delegate itself produced). If +/// it does not, the returned task completes as cancelled, the delegate's thread is detached +/// to finish in the background with its eventual result discarded, and the condition is +/// reported through the supplied sink at +/// . See Decision #4 in the engine/session async +/// redesign planning report for the full rationale, including why callers that abandon as +/// part of their own explicit stop/dispose treat that outcome as best-effort completion +/// rather than a fault. +/// +/// The delegate always runs via , which asks the +/// scheduler for a dedicated thread rather than a pooled one - appropriate here because the +/// delegate is expected to block for the life of a recognition session's pump loop, not +/// return quickly like ordinary thread-pool work. +/// +/// +internal sealed class DedicatedWorker +{ + /// The default abandon timeout used in production: 2 seconds. + internal static readonly TimeSpan DefaultAbandonTimeout = TimeSpan.FromSeconds(2); + + /// The diagnostics sink this worker reports abandonment to. + private readonly ISpeechDiagnostics _diagnostics; + + /// The diagnostics category used for every event this worker reports. + private readonly string _diagnosticsCategory; + + /// + /// Initializes a new instance of the class. + /// + /// + /// The duration to wait for the delegate to honor a cancellation request before + /// abandoning it, or to use . + /// Tests inject a near-zero value to keep abandonment tests fast. + /// + /// + /// The sink to report abandonment to, or to use + /// . + /// + /// The diagnostics category used for reported events. + internal DedicatedWorker( + TimeSpan? abandonTimeout = null, + ISpeechDiagnostics? diagnostics = null, + string diagnosticsCategory = "RecognitionSubsystem") + { + AbandonTimeout = abandonTimeout ?? DefaultAbandonTimeout; + _diagnostics = diagnostics ?? NullSpeechDiagnostics.Instance; + _diagnosticsCategory = diagnosticsCategory; + } + + /// + /// Gets the duration this worker waits for a cancelled delegate to finish before + /// abandoning it. + /// + internal TimeSpan AbandonTimeout { get; } + + /// + /// Runs to completion on a dedicated, long-running thread. + /// + /// + /// The delegate to run, given so it can observe + /// cancellation cooperatively at its own natural boundaries. Must not be null. + /// + /// + /// A token whose cancellation requests the delegate stop. The delegate is given + /// to honor the request before being abandoned. + /// + /// + /// A task that completes once the delegate finishes, or - if the delegate does not honor + /// a cancellation request within - completes as cancelled + /// while the delegate is abandoned to finish in the background. + /// + /// Thrown when is null. + internal Task RunAsync(Action action, CancellationToken cancellationToken = default) => + RunAsync(action, cancellationToken, out _); + + /// + /// Runs to completion on a dedicated, long-running thread, + /// additionally exposing the raw, non-abandon-aware completion of that thread. + /// + /// + /// The delegate to run, given so it can observe + /// cancellation cooperatively at its own natural boundaries. Must not be null. + /// + /// + /// A token whose cancellation requests the delegate stop. The delegate is given + /// to honor the request before being abandoned. + /// + /// + /// Set to the dedicated thread's own task, which completes only once + /// genuinely returns - even if it is abandoned. A caller that + /// must not touch a resource shares with another session (for + /// example, a "hot" backend reused across sessions) until that thread has truly exited - + /// regardless of whether the returned task completed early as abandoned - should await + /// this instead of (or as well as) the returned task. + /// + /// + /// A task that completes once the delegate finishes, or - if the delegate does not honor + /// a cancellation request within - completes as cancelled + /// while the delegate is abandoned to finish in the background. + /// + /// Thrown when is null. + internal Task RunAsync(Action action, CancellationToken cancellationToken, out Task completion) + { + ArgumentNullException.ThrowIfNull(action); + + var worker = Task.Factory.StartNew( + () => action(cancellationToken), + CancellationToken.None, + TaskCreationOptions.LongRunning, + TaskScheduler.Default); + + completion = worker; + return AwaitWithAbandonAsync(worker, cancellationToken); + } + + /// + /// Awaits , applying the cooperative-cancel-then-abandon policy + /// once is cancelled. + /// + private async Task AwaitWithAbandonAsync(Task worker, CancellationToken cancellationToken) + { + if (!cancellationToken.CanBeCanceled) + { + await worker.ConfigureAwait(false); + return; + } + + try + { + await worker.WaitAsync(cancellationToken).ConfigureAwait(false); + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) + { + // The caller asked the delegate to stop; give it AbandonTimeout to actually do so + // before giving up on waiting for it any further. + var abandonDelay = Task.Delay(AbandonTimeout, CancellationToken.None); + var finished = await Task.WhenAny(worker, abandonDelay).ConfigureAwait(false); + if (finished != worker) + { + _diagnostics.Report( + SpeechDiagnosticLevel.Warning, + _diagnosticsCategory, + "A dedicated worker did not honor a cooperative cancellation request within " + + $"the {AbandonTimeout.TotalSeconds:F1}s abandon timeout; it has been abandoned " + + "to finish in the background and its eventual result will be discarded."); + + // Observe and discard whatever the abandoned delegate eventually produces, so an + // unobserved-task-exception does not later surface unrelated to this call. + _ = worker.ContinueWith( + static completed => _ = completed.Exception, + CancellationToken.None, + TaskContinuationOptions.OnlyOnFaulted, + TaskScheduler.Default); + throw; + } + + // The delegate genuinely finished within the abandon bound; propagate whatever it + // produced (including rethrowing a fault) rather than the cancellation. + await worker.ConfigureAwait(false); + } + } +} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionEngine.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionBackend.cs similarity index 73% rename from src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionEngine.cs rename to src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionBackend.cs index adb0228..4db7a3b 100644 --- a/src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionEngine.cs +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionBackend.cs @@ -1,26 +1,27 @@ namespace DemaConsulting.Speech.RecognitionSubsystem; /// -/// Internal, mockable seam over one loaded streaming recognition engine: accepts mono audio +/// Internal, mockable seam over one loaded streaming recognition backend: accepts mono audio /// samples and reports the recognition results decoded from them. /// /// /// This seam exists for the same reason as IPortAudioApi (Phase 1b) and /// IModelDownloadClient (Phase 2a): it confines every sherpa-onnx interop call to a /// single implementation () so -/// 's threading, resampling, and event-emission logic -/// is fully unit-testable with a pure managed fake. No native sherpa-onnx runtime binary and -/// no downloaded model are ever required to run the recognition subsystem's tests. +/// 's threading, resampling, and event-emission +/// logic is fully unit-testable with a pure managed fake. No native sherpa-onnx runtime +/// binary and no downloaded model are ever required to run the recognition subsystem's +/// tests. /// -/// Implementations are not thread-safe: calls them -/// from exactly one background consumer thread at a time and never concurrently, which -/// matches the single-stream ownership model of the underlying engine. +/// Implementations are not thread-safe: calls them +/// from exactly one background pump thread at a time and never concurrently, which matches +/// the single-stream ownership model of the underlying engine. /// /// -internal interface IRecognitionEngine : IDisposable +internal interface IRecognitionBackend : IDisposable { /// - /// Feeds one block of mono audio into the engine's current utterance stream. + /// Feeds one block of mono audio into the backend's current utterance stream. /// /// /// Normalized single-channel samples in the range [-1.0, 1.0], already resampled to @@ -33,7 +34,7 @@ internal interface IRecognitionEngine : IDisposable void AcceptSamples(ReadOnlySpan monoSamples); /// - /// Decodes as much buffered audio as the engine currently can and reports the next + /// Decodes as much buffered audio as the backend currently can and reports the next /// recognition result, if any. /// /// @@ -42,7 +43,7 @@ internal interface IRecognitionEngine : IDisposable /// /// /// when a result is available; when the - /// engine has nothing new to report (no buffered audio, no text yet, or text unchanged + /// backend has nothing new to report (no buffered audio, no text yet, or text unchanged /// since the previous call). /// /// @@ -54,7 +55,7 @@ internal interface IRecognitionEngine : IDisposable bool TryDecode(out SpeechRecognitionResult? result); /// - /// Finalizes and decodes any buffered audio the engine has accepted but not yet decoded, + /// Finalizes and decodes any buffered audio the backend has accepted but not yet decoded, /// reporting one last result if that produced or completed any text. /// /// @@ -66,16 +67,16 @@ internal interface IRecognitionEngine : IDisposable /// when there was nothing buffered or finalizing it produced no text. /// /// - /// A streaming engine sometimes cannot decode the tail of an utterance without more audio - /// that a caller who has just stopped will never supply - for example a push-to-talk - /// release with no trailing silence. Call this once, at session end, before + /// A streaming backend sometimes cannot decode the tail of an utterance without more + /// audio that a caller who has just stopped will never supply - for example a + /// push-to-talk release with no trailing silence. Call this once, at session end, before /// discards the stream, so that trailing audio is finalized and /// delivered rather than silently discarded along with it. /// bool TryFlush(out SpeechRecognitionResult? result); /// - /// Discards any partially decoded utterance and returns the engine to its + /// Discards any partially decoded utterance and returns the backend to its /// start-of-utterance state. /// /// diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionEngineFactory.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionBackendFactory.cs similarity index 74% rename from src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionEngineFactory.cs rename to src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionBackendFactory.cs index a66c296..2c90a29 100644 --- a/src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionEngineFactory.cs +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionBackendFactory.cs @@ -3,20 +3,20 @@ namespace DemaConsulting.Speech.RecognitionSubsystem; /// -/// Internal, mockable factory seam that builds a loaded for +/// Internal, mockable factory seam that builds a loaded for /// one installed recognition model. /// /// -/// Separating engine construction from engine use lets -/// decide, without any sherpa-onnx knowledge of its own, whether a real engine could be -/// loaded, and lets tests inject a fake engine without a model directory, a native runtime, +/// Separating backend construction from backend use lets +/// decide, without any sherpa-onnx knowledge of its own, whether a real backend could be +/// loaded, and lets tests inject a fake backend without a model directory, a native runtime, /// or an inference session. It mirrors the IModelDownloadClient seam introduced in /// Phase 2a for exactly the same testability reason. /// -internal interface IRecognitionEngineFactory +internal interface IRecognitionBackendFactory { /// - /// Loads the recognition engine described by a model, from that model's installed files. + /// Loads the recognition backend described by a model, from that model's installed files. /// /// The installed recognition model to load. Must not be null. /// @@ -28,7 +28,7 @@ internal interface IRecognitionEngineFactory /// , /// or when the caller supplied none. /// - /// A loaded engine ready to accept samples. Never . + /// A loaded backend ready to accept samples. Never . /// Thrown when is null. /// /// Thrown when is null or empty. @@ -38,9 +38,10 @@ internal interface IRecognitionEngineFactory /// reasons - a missing native runtime binary, an unsupported RID, or corrupt model files. /// Implementations surface those failures as exceptions; /// converts them into the honest - /// fallback so composition still never throws. + /// fallback so composition still never + /// throws. /// - IRecognitionEngine Create( + IRecognitionBackend Create( IRecognitionModel model, string installedModelDirectory, IReadOnlyDictionary? parameterValues = null); diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionSession.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionSession.cs new file mode 100644 index 0000000..076fc4f --- /dev/null +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/IRecognitionSession.cs @@ -0,0 +1,121 @@ +namespace DemaConsulting.Speech.RecognitionSubsystem; + +/// +/// Mockable streaming speech-to-text session: cheap, bound to exactly one capture device +/// instance for its entire life, and single-use. +/// +/// +/// Obtained from rather than +/// constructed directly. Per this library's single-use session decision, a session that has +/// reached can never run again - +/// throws rather than +/// restarting; create a new session via +/// for another run. +/// +/// A session that cannot function honestly reports as +/// ; see for the canonical +/// fallback. Implementations own unmanaged inference resources reached through their owning +/// engine, so callers must dispose them; disposal is idempotent. Disposing a session that is +/// still running or stopping implicitly performs the same +/// -> +/// drain as an explicit +/// call before the engine's lease is released, so a caller never needs to call +/// before disposing. +/// +/// +public interface IRecognitionSession : IAsyncDisposable +{ + /// + /// Gets a value indicating whether this session is backed by a real, loaded recognition + /// engine and a usable capture device. When , + /// and throw + /// . + /// + bool IsAvailable { get; } + + /// Gets the session's current lifecycle state. + RecognitionSessionState State { get; } + + /// + /// Raised once for every transition this session + /// makes. + /// + /// + /// Raised from the session's own background pump thread and handlers are invoked + /// serially, never concurrently with each other - the same threading and fault-isolation + /// convention this library used for synchronous result delivery before the Engine/Session + /// split. + /// An exception thrown by a handler is caught and reported through the session's + /// diagnostics sink; it never propagates and never faults the session. + /// + event EventHandler? StateChanged; + + /// + /// Begins streaming audio from the bound capture device into the recognition backend, + /// after which results become available through . + /// + /// A token to observe for cancellation of this call. + /// A task that completes once the session is running. + /// + /// Thrown when is not - + /// most notably when the session has already reached + /// , since a session is single-use. + /// + /// + /// Thrown when is , or when the bound + /// capture device failed to start. + /// + Task StartAsync(CancellationToken cancellationToken = default); + + /// + /// Stops streaming audio and drains any already-captured audio through the backend, so + /// every result derived from audio accepted before this call is enumerable via + /// before the returned task completes. + /// + /// + /// A token to observe for cancellation of waiting for this call's own completion. Does + /// not abort the underlying drain/stop, which is shared with every other concurrent or + /// overlapping caller and keeps running to convergence (or abandonment) regardless of + /// whether this particular caller stopped waiting for it. + /// + /// A task that completes once the session has stopped. + /// + /// Idempotent and safe to call concurrently or while overlapping a prior call still in + /// flight: , + /// , and + /// all converge on + /// , and every caller's task completes once + /// that convergence happens. A fault while finalizing or resetting is reported through + /// diagnostics rather than thrown; this call still completes. + /// + /// Cancelling lets this call's own returned task + /// complete early with without waiting any + /// further, but it never aborts the shared teardown itself: one canceled caller must not + /// skip draining/stopping for every other caller (including ) + /// sharing the same in-flight operation. + /// + /// + Task StopAsync(CancellationToken cancellationToken = default); + + /// + /// Streams every provisional and final recognition result produced while the session + /// runs. + /// + /// + /// A token that ends only this enumeration when cancelled - the session itself keeps + /// running. + /// + /// An asynchronous sequence of recognition events. + /// + /// Thrown when a second concurrent call is made while one enumeration of this session's + /// results is already active; this contract is single-consumer. + /// + /// + /// Thrown when is . + /// + /// + /// Thrown from the enumerator when the session transitions to + /// while being enumerated. + /// + IAsyncEnumerable GetResultsAsync(CancellationToken cancellationToken = default); +} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/ISpeechRecognizer.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/ISpeechRecognizer.cs deleted file mode 100644 index 5a0a495..0000000 --- a/src/DemaConsulting.Speech/RecognitionSubsystem/ISpeechRecognizer.cs +++ /dev/null @@ -1,116 +0,0 @@ -namespace DemaConsulting.Speech.RecognitionSubsystem; - -/// -/// Mockable streaming speech-to-text contract: starts and stops consuming audio from a -/// capture device and raises progressive provisional and final recognition results while -/// running. -/// -/// -/// Per this library's "engine backend stays swappable at the public API surface" -/// decision, no member of this contract exposes a sherpa-onnx (or any other engine) type, so -/// a future non-sherpa-onnx backend can be added without a breaking change. Hosts obtain -/// implementations from rather than constructing them, -/// and tests can substitute this interface directly to exercise host transcript logic with no -/// model, no microphone, and no native runtime. -/// -/// A recognizer that cannot function honestly reports as -/// instead of throwing at composition time; see -/// for the canonical fallback. Implementations own -/// unmanaged inference resources, so callers must dispose them; disposal is idempotent and -/// implies . -/// -/// -/// Thread safety: , , and -/// may each be called concurrently, from any thread, -/// without external synchronization - implementations are responsible for serializing -/// their own internal state transitions so overlapping calls compose safely (each is -/// individually idempotent, as documented on that member). This contract does not -/// promise any particular outcome for the relative ordering of unrelated, racing -/// Start/Stop calls issued from different threads at the same time - only that each call -/// completes without corrupting the recognizer's internal state. -/// is always raised from the recognizer's own background decoding thread and handlers are -/// invoked serially, never concurrently with each other; see that event's remarks. -/// -/// -/// Reuse for low latency ("hot" recognition). Obtaining a recognizer via -/// is the expensive step - it loads the model into -/// native memory - while and are cheap and may be -/// called repeatedly on the same instance without reloading the model. For low-latency -/// repeated recognition (for example, many turns of a voice conversation), construct one -/// recognizer and reuse it across many / cycles rather -/// than disposing and recreating it per turn; only dispose and recreate to change model, -/// device, or parameters. See the user guide's "Hot TTS/STT: Reusing an Instance Across -/// Turns" section for a worked example. -/// -/// -public interface ISpeechRecognizer : IDisposable -{ - /// - /// Gets a value indicating whether this recognizer is backed by a real, loaded - /// recognition engine and a usable capture device. When , - /// and throw - /// rather than silently doing nothing, - /// because a caller that ignores this flag has made a programming error that should - /// surface immediately rather than silently recognize nothing. - /// - bool IsAvailable { get; } - - /// - /// Raised once for each provisional or final recognition result produced while the - /// recognizer is running. - /// - /// - /// Raised from the recognizer's own background decoding thread - never from the audio - /// capture callback thread - so a handler may do moderate work without risking audio - /// glitches. Handlers are still invoked serially, so a slow handler delays subsequent - /// results. An exception thrown by a handler is caught and reported through the - /// recognizer's diagnostics sink; it never propagates and never stops the recognizer. - /// Handlers must therefore not rely on exceptions escaping. - /// - /// Reentrantly calling , , or - /// from within a handler of this event is not supported - /// and can deadlock: and block their - /// caller until this same background thread finishes draining, so a handler that calls - /// back into one of them from that thread can end up waiting on itself. A host that needs - /// to stop or dispose the recognizer in response to a result must do so from another - /// thread rather than directly from this handler. - /// - /// - event EventHandler? ResultReceived; - - /// - /// Begins streaming audio from the configured capture device into the recognition engine, - /// after which is raised for each result until - /// is called. Calling this on an already-running recognizer is a safe - /// no-op. - /// - /// - /// Thrown when is , or when the - /// recognizer reported itself as available but its capture device failed to start. - /// - /// - /// Thrown when the recognizer has already been disposed. - /// - void Start(); - - /// - /// Stops streaming audio and drains any already-captured audio through the engine, so - /// every result derived from audio accepted before this call is raised before it returns. - /// No further events are raised until is - /// called again. Calling this on a recognizer that is not running is a safe no-op. - /// - /// - /// Guarantees zero carryover into the next session: even the tail of an utterance - /// released with no trailing silence - which a streaming engine cannot normally decode - /// without more audio that a caller who has just stopped will never supply - is finalized and - /// delivered as one last result here rather than left pending. This is best-effort: a - /// fault in the engine while finalizing or resetting is reported through diagnostics - /// rather than thrown, still completes, and a later - /// is still permitted, but the trailing audio and/or the engine's clean state can no - /// longer be guaranteed for that one call. - /// - /// - /// Thrown when is . - /// - void Stop(); -} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/ISpeechRecognizerEngine.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/ISpeechRecognizerEngine.cs new file mode 100644 index 0000000..3fd50ce --- /dev/null +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/ISpeechRecognizerEngine.cs @@ -0,0 +1,64 @@ +using DemaConsulting.Speech.AudioSubsystem; + +namespace DemaConsulting.Speech.RecognitionSubsystem; + +/// +/// Mockable contract for one loaded, native-backed recognition model, with no capture device +/// bound yet. +/// +/// +/// Obtained from . Loading the model into native memory +/// is the expensive step; an engine may be reused to create many sessions over its lifetime +/// (though only one at a time - see ), so a host should load +/// one engine per model/parameter combination and keep it for as long as it may recognize +/// speech, rather than reloading per turn. +/// +/// An engine that cannot function honestly reports as +/// instead of throwing at composition time; see +/// for the canonical fallback. +/// +/// +public interface ISpeechRecognizerEngine : IAsyncDisposable +{ + /// + /// Gets a value indicating whether this engine is backed by a real, loaded recognition + /// backend. When , never throws + /// for ordinary unavailability - it returns + /// instead. + /// + bool IsAvailable { get; } + + /// + /// Binds this engine to exactly one capture device for the entire life of the returned + /// session. + /// + /// + /// The capture device the returned session streams audio from for its entire life. Must + /// not be null. + /// + /// A token to observe for cancellation of this call. + /// + /// A new bound to , or + /// when is + /// . + /// + /// Thrown when is null. + /// + /// Thrown when a session created from this engine already holds this engine's + /// exclusivity lease - including while that prior session is still tearing down via its + /// own . This call never queues or waits for + /// that release; it fails fast instead. + /// + /// + /// Thrown when is cancelled before the call + /// completes. + /// + /// + /// Never faults for an ordinary unavailable-engine state; faults only for a null + /// argument, caller cancellation, or a concurrently leased session (see + /// ). + /// + Task CreateSessionAsync( + IAudioCaptureDevice device, + CancellationToken cancellationToken = default); +} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/NamespaceDoc.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/NamespaceDoc.cs index 3ead732..1c5eb9d 100644 --- a/src/DemaConsulting.Speech/RecognitionSubsystem/NamespaceDoc.cs +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/NamespaceDoc.cs @@ -1,16 +1,19 @@ namespace DemaConsulting.Speech.RecognitionSubsystem; /// -/// Streaming speech-to-text: the public recognizer contract and its composition root, and -/// the capture-to-engine pipeline that feeds a real speech-inference engine. +/// Streaming speech-to-text: the public Engine/Session contract and its composition root, and +/// the capture-to-backend pipeline that feeds a real speech-inference backend. /// /// -/// Contains and its result/event types, +/// Contains (the loaded, expensive, model-bound +/// composition unit) and (the cheap, single-use, +/// device-bound streaming unit) with their result/event/state types, /// - the composition root that returns either a -/// working recognizer or the honest fallback - the -/// capture-to-engine pipeline with its audio-format +/// working engine or the honest fallback (with +/// as the matching session-layer fallback) - the +/// capture-to-backend pipeline with its audio-format /// converter, and the internal mockable speech-inference seam backed by -/// . +/// . /// internal static class NamespaceDoc { diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionEngineBusyException.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionEngineBusyException.cs new file mode 100644 index 0000000..504ab9e --- /dev/null +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionEngineBusyException.cs @@ -0,0 +1,47 @@ +namespace DemaConsulting.Speech.RecognitionSubsystem; + +/// +/// Thrown when is called while a +/// previously created session's exclusivity lease is still held. +/// +/// +/// Per this library's engine-exclusivity decision, exactly one +/// may be leased on an at a time, from the moment it is +/// created until its own completes. A concurrent +/// call made while that lease is held +/// - including while the prior session is still tearing down - fails fast with this exception +/// rather than queuing or awaiting the prior session's release, so a caller's composition +/// latency never depends on an unrelated session's teardown. +/// +public sealed class RecognitionEngineBusyException : Exception +{ + /// + /// Initializes a new instance of the class + /// with a default message. + /// + public RecognitionEngineBusyException() + : base("The recognition engine already has a session leased.") + { + } + + /// + /// Initializes a new instance of the class + /// with a message describing the busy engine. + /// + /// A message describing why the engine is busy. + public RecognitionEngineBusyException(string message) + : base(message) + { + } + + /// + /// Initializes a new instance of the class + /// with a message and an inner exception describing the underlying cause. + /// + /// A message describing why the engine is busy. + /// The exception that is the cause of this exception. + public RecognitionEngineBusyException(string message, Exception innerException) + : base(message, innerException) + { + } +} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionResultBuffer.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionResultBuffer.cs new file mode 100644 index 0000000..e86e365 --- /dev/null +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionResultBuffer.cs @@ -0,0 +1,239 @@ +using System.Runtime.CompilerServices; +using DemaConsulting.Speech.Diagnostics; + +namespace DemaConsulting.Speech.RecognitionSubsystem; + +/// +/// Internal two-tier backpressure buffer feeding : +/// one overwritable "latest provisional" slot plus a byte-capped FIFO of final results. +/// +/// +/// Implements Decision #5 of the engine/session async redesign: a provisional result that +/// arrives before the consumer has read the previous one simply overwrites it (coalescing is +/// the documented common case and costs no allocation beyond the record itself), while final +/// results are queued and never silently coalesced - they are capped by total UTF-16 byte size +/// of their buffered (64 KiB) rather than by count, +/// and the oldest unread final is evicted only as a last-resort safety valve, reported through +/// at . A +/// provisional is also superseded (never delivered) once its own final is buffered, so a +/// consumer never sees a stale partial transcript trailing the final result for the same +/// utterance. Neither tier ever blocks the pump thread that calls : the +/// native decode/flush loop must never wait on a slow consumer. +/// +internal sealed class RecognitionResultBuffer +{ + /// The maximum total UTF-16 byte size of buffered final results' . + private const long MaxFinalResultBytes = 64 * 1024; + + /// Guards every field below. + private readonly object _lock = new(); + + /// The FIFO of buffered final results, capped by . + private readonly Queue _finals = new(); + + /// The sink to report a last-resort final-result eviction to. + private readonly ISpeechDiagnostics _diagnostics; + + /// The diagnostics category used for reported events. + private readonly string _diagnosticsCategory; + + /// The total UTF-16 byte size of every result currently queued in . + private long _finalBytes; + + /// The single overwritable "latest provisional" slot, or when empty. + private SpeechRecognitionEvent? _provisional; + + /// Whether no further results will ever be added. + private bool _completed; + + /// The cause of a session fault surfaced to an active enumerator, if any. + private Exception? _fault; + + /// Signaled whenever new data, completion, or a fault becomes available to a waiting consumer. + private TaskCompletionSource _signal = new(TaskCreationOptions.RunContinuationsAsynchronously); + + /// + /// Initializes a new instance of the class. + /// + /// The sink to report last-resort eviction to. Must not be null. + /// The diagnostics category used for reported events. + internal RecognitionResultBuffer(ISpeechDiagnostics diagnostics, string diagnosticsCategory) + { + ArgumentNullException.ThrowIfNull(diagnostics); + + _diagnostics = diagnostics; + _diagnosticsCategory = diagnosticsCategory; + } + + /// + /// Adds one recognition result to the buffer: a provisional result overwrites any + /// previously unread provisional result, while a final result is queued and also + /// supersedes (clears) any not-yet-consumed provisional result still buffered. + /// + /// The result to buffer. Must not be null. + internal void AddResult(SpeechRecognitionEvent result) + { + ArgumentNullException.ThrowIfNull(result); + + TaskCompletionSource toSignal; + lock (_lock) + { + if (_completed || _fault is not null) + { + return; + } + + if (result.Result.IsFinal) + { + // A final result supersedes any not-yet-consumed provisional: DequeueNext() + // always drains queued finals before the provisional slot, so an unconsumed + // provisional left in place here would otherwise be delivered after its own + // final - a stale, already-superseded partial transcript trailing the final + // result for the same utterance. Clearing it here means a consumer only ever + // sees the provisional if it reads before the final arrives, never after. + _provisional = null; + EnqueueFinal(result); + } + else + { + _provisional = result; + } + + toSignal = _signal; + _signal = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + } + + toSignal.TrySetResult(true); + } + + /// + /// Marks the buffer complete: every already-buffered result is still delivered, but no + /// further result will ever be added, and an active enumeration ends once it is drained. + /// + internal void Complete() + { + TaskCompletionSource toSignal; + lock (_lock) + { + _completed = true; + toSignal = _signal; + } + + toSignal.TrySetResult(true); + } + + /// + /// Marks the buffer faulted: an active or future enumeration throws + /// wrapping + /// once every already-buffered result has been delivered. + /// + /// The exception describing why the owning session faulted. Must not be null. + internal void Fault(Exception cause) + { + ArgumentNullException.ThrowIfNull(cause); + + TaskCompletionSource toSignal; + lock (_lock) + { + _fault ??= cause; + toSignal = _signal; + } + + toSignal.TrySetResult(true); + } + + /// + /// Streams every buffered result in order, waiting for new data as needed, until the + /// buffer is completed or faulted. + /// + /// A token that ends only this enumeration when cancelled. + internal async IAsyncEnumerable ReadAllAsync( + [EnumeratorCancellation] CancellationToken cancellationToken = default) + { + while (true) + { + SpeechRecognitionEvent? next; + bool completed; + Exception? fault; + Task waitTask; + lock (_lock) + { + next = DequeueNext(); + completed = _completed; + fault = _fault; + waitTask = _signal.Task; + } + + if (next is not null) + { + yield return next; + continue; + } + + if (fault is not null) + { + throw new RecognitionSessionFaultedException( + "The recognition session faulted while its results were being enumerated.", + fault); + } + + if (completed) + { + yield break; + } + + await waitTask.WaitAsync(cancellationToken).ConfigureAwait(false); + } + } + + /// + /// Removes and returns the next buffered result (a queued final takes priority over the + /// provisional slot), or when nothing is buffered. Must be called + /// under . + /// + private SpeechRecognitionEvent? DequeueNext() + { + if (_finals.Count > 0) + { + var next = _finals.Dequeue(); + _finalBytes -= ByteSize(next); + return next; + } + + if (_provisional is not null) + { + var next = _provisional; + _provisional = null; + return next; + } + + return null; + } + + /// + /// Queues a final result, evicting the oldest buffered final as a last resort if doing so + /// is the only way to respect . Must be called under + /// . + /// + private void EnqueueFinal(SpeechRecognitionEvent result) + { + _finals.Enqueue(result); + _finalBytes += ByteSize(result); + + while (_finalBytes > MaxFinalResultBytes && _finals.Count > 1) + { + var dropped = _finals.Dequeue(); + _finalBytes -= ByteSize(dropped); + + _diagnostics.Report( + SpeechDiagnosticLevel.Warning, + _diagnosticsCategory, + "A buffered final recognition result was evicted because the consumer of GetResultsAsync has " + + "not kept up and the final-result backlog exceeded its byte cap; this is a last-resort safety " + + "valve, not a routine occurrence."); + } + } + + /// Computes the UTF-16 byte size of a result's . + private static long ByteSize(SpeechRecognitionEvent result) => result.Result.Text.Length * 2L; +} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionSessionFaultedException.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionSessionFaultedException.cs new file mode 100644 index 0000000..bd141de --- /dev/null +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionSessionFaultedException.cs @@ -0,0 +1,47 @@ +namespace DemaConsulting.Speech.RecognitionSubsystem; + +/// +/// Thrown from 's enumerator when the +/// session transitions to while being +/// enumerated. +/// +/// +/// A session faults when an unrecoverable error occurs while +/// , , +/// or - for example the bound capture device +/// being lost mid-session. The fault is surfaced to a consumer currently enumerating +/// as this exception, with the underlying +/// cause available as . +/// +public sealed class RecognitionSessionFaultedException : Exception +{ + /// + /// Initializes a new instance of the + /// class with a default message. + /// + public RecognitionSessionFaultedException() + : base("The recognition session has faulted.") + { + } + + /// + /// Initializes a new instance of the + /// class with a message describing the fault. + /// + /// A message describing the fault. + public RecognitionSessionFaultedException(string message) + : base(message) + { + } + + /// + /// Initializes a new instance of the + /// class with a message and the inner exception that caused the session to fault. + /// + /// A message describing the fault. + /// The exception that caused the session to fault. + public RecognitionSessionFaultedException(string message, Exception innerException) + : base(message, innerException) + { + } +} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionSessionState.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionSessionState.cs new file mode 100644 index 0000000..f15a3bb --- /dev/null +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/RecognitionSessionState.cs @@ -0,0 +1,43 @@ +namespace DemaConsulting.Speech.RecognitionSubsystem; + +/// +/// The lifecycle states an moves through. +/// +/// +/// The forward-only progression is → → +/// → → → +/// → , with reachable as +/// a terminal state from , , or +/// . A session is single-use: once it reaches , +/// throws rather than permitting a back-edge to +/// again - see 's remarks. +/// +public enum RecognitionSessionState +{ + /// The session has been created but has not yet been called. + Created, + + /// The session is subscribing to and starting its bound capture device. + Starting, + + /// The session is streaming captured audio through the recognition backend. + Running, + + /// The session is draining buffered audio and tearing down its capture subscription. + Stopping, + + /// The session has stopped; it will never run again. + Stopped, + + /// The session is releasing its resources and the engine's exclusivity lease. + Disposing, + + /// The session has been fully disposed. + Disposed, + + /// + /// The session encountered an unrecoverable error while , + /// , or and can no longer be used. + /// + Faulted +} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/SessionStateChangedEventArgs.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/SessionStateChangedEventArgs.cs new file mode 100644 index 0000000..07a2080 --- /dev/null +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/SessionStateChangedEventArgs.cs @@ -0,0 +1,13 @@ +namespace DemaConsulting.Speech.RecognitionSubsystem; + +/// +/// Event data carrying an 's previous and current +/// around one state transition. +/// +/// The state the session transitioned from. +/// The state the session transitioned to. +/// +/// Raised by . See that event's remarks for the +/// threading and fault-isolation guarantees that apply to every raise of this event. +/// +public sealed record SessionStateChangedEventArgs(RecognitionSessionState Previous, RecognitionSessionState Current); diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionEngine.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionEngine.cs index b652db5..dba4c11 100644 --- a/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionEngine.cs +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionEngine.cs @@ -3,7 +3,7 @@ namespace DemaConsulting.Speech.RecognitionSubsystem; /// -/// Real implementation wrapping one sherpa-onnx +/// Real implementation wrapping one sherpa-onnx /// and the that carries the current /// utterance. /// @@ -17,7 +17,7 @@ namespace DemaConsulting.Speech.RecognitionSubsystem; /// /// Construction loads the model into native memory and therefore fails (throws) when the /// native runtime binary for the current RID is absent or the model files are unusable. -/// Callers convert that into the honest fallback; +/// Callers convert that into the honest fallback; /// see . /// /// @@ -33,7 +33,7 @@ namespace DemaConsulting.Speech.RecognitionSubsystem; /// after that flush cannot bleed into the next session's decoding. /// /// -internal sealed class SherpaOnnxRecognitionEngine : IRecognitionEngine +internal sealed class SherpaOnnxRecognitionEngine : IRecognitionBackend { /// /// Initializes a new instance of the class, diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionEngineFactory.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionEngineFactory.cs index a87c8c0..4ff7b9b 100644 --- a/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionEngineFactory.cs +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionEngineFactory.cs @@ -3,7 +3,7 @@ namespace DemaConsulting.Speech.RecognitionSubsystem; /// -/// Real implementation that builds a +/// Real implementation that builds a /// from a model's own declared engine /// configuration. /// @@ -17,14 +17,14 @@ namespace DemaConsulting.Speech.RecognitionSubsystem; /// Loading failures - a missing org.k2fsa.sherpa.onnx.runtime.{RID} native binary, an /// unsupported RID, or corrupt model files - propagate to the caller. /// catches them and returns -/// , so composition still never throws. +/// , so composition still never throws. /// /// The type is stateless and safe for concurrent use. /// -internal sealed class SherpaOnnxRecognitionEngineFactory : IRecognitionEngineFactory +internal sealed class SherpaOnnxRecognitionEngineFactory : IRecognitionBackendFactory { /// - public IRecognitionEngine Create( + public IRecognitionBackend Create( IRecognitionModel model, string installedModelDirectory, IReadOnlyDictionary? parameterValues = null) diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionSession.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionSession.cs new file mode 100644 index 0000000..252bfa6 --- /dev/null +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxRecognitionSession.cs @@ -0,0 +1,1124 @@ +using System.Runtime.CompilerServices; +using System.Threading.Channels; +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.Diagnostics; +using DemaConsulting.Speech.ModelManagementSubsystem; + +namespace DemaConsulting.Speech.RecognitionSubsystem; + +/// +/// Real implementation that streams one capture device's +/// audio through a resampler into a shared, "hot" recognition backend for the life of exactly +/// one run. +/// +/// +/// This type carries the per-session pump-loop logic that +/// SherpaOnnxSpeechRecognizer used to own directly before the Engine/Session split: the +/// capture device's FrameCaptured callback still does nothing but copy a block into a +/// bounded, drop-oldest , and a single dedicated background thread - +/// started via the subsystem's internal - still does all the +/// real work (downmix/resample, feed the backend, poll for results, buffer them for +/// ). The backend itself is owned and disposed by the owning +/// , not by this session, so that it stays +/// "hot" (loaded once, reused across sequential sessions) rather than being torn down at the +/// end of every run. +/// +/// completes the pending-frame channel and awaits the pump thread, so +/// every block accepted before the call has been decoded and every resulting event buffered +/// by the time it returns - including one last flushed result for a trailing utterance that +/// had not yet been decoded (see ). Both +/// and await that pump thread through the +/// cooperative-cancel-then-abandon policy documented on +/// (Decision #4): an abandoned stop/dispose still completes teardown as best-effort, while an +/// abandonment that happens for any other reason transitions this session to +/// . +/// +/// +internal sealed class SherpaOnnxRecognitionSession : IRecognitionSession +{ + /// + /// The maximum number of captured blocks held pending recognition before the oldest is + /// dropped. Sized for roughly a second of typical capture-callback blocks, mirroring the + /// bound SherpaOnnxSpeechRecognizer used before this split. + /// + private const int PendingFrameCapacity = 64; + + /// + /// The maximum number of results drained from the backend for a single captured block, so + /// that a faulty backend which always reports a result cannot livelock the pump thread. + /// + private const int MaxResultsPerFrame = 32; + + /// The diagnostics category used for every event this session reports. + private const string DiagnosticsCategory = "RecognitionSubsystem"; + + /// Guards every state transition and the fields touched by Start/Stop/Dispose. + private readonly object _syncRoot = new(); + + /// The recognition backend this session feeds; owned and disposed by the engine, not this session. + private readonly IRecognitionBackend _backend; + + /// The capture device supplying the audio to recognize, bound for this session's entire life. + private readonly IAudioCaptureDevice _device; + + /// The recognition model whose hook is applied to every result. + private readonly IRecognitionModel _model; + + /// The conversion from the device's capture format to the backend's required format. + private readonly AudioFrameResampler _resampler; + + /// The sink for structural lifecycle and fault events. + private readonly ISpeechDiagnostics _diagnostics; + + /// Runs the pump loop on a dedicated, non-pooled thread with a cooperative-cancel-then-abandon policy. + private readonly DedicatedWorker _worker; + + /// Invoked exactly once, at the very end of , to release the engine's exclusivity lease. + private readonly Action _releaseLease; + + /// Buffers provisional/final results for (Decision #5). + private readonly RecognitionResultBuffer _resultBuffer; + + /// This session's current lifecycle state. + private RecognitionSessionState _state = RecognitionSessionState.Created; + + /// The bounded hand-off between the audio callback thread and the pump thread. + private Channel? _pendingFrames; + + /// Signals the pump thread to stop cooperatively and bounds how long teardown waits for it. + private CancellationTokenSource? _pumpCts; + + /// The abandon-aware task returned by for the running pump loop. + private Task? _pumpTask; + + /// + /// The pump loop's own dedicated-thread completion, set alongside . + /// Unlike , this completes only once the pump thread has genuinely + /// exited - even if itself completed early as abandoned - so + /// teardown can safely wait for it before touching the shared backend again (Decision #4). + /// + private Task? _pumpRawCompletion; + + /// + /// The device-start worker's own raw completion, set alongside the abandon-aware task + /// itself awaits - exactly like + /// mirrors . Set under before + /// ever releases the lock, so can + /// await this - the device's own call genuinely + /// returning, even if 's own abandon-aware task completed early as + /// cancelled - before it ever stops the device: this is what still prevents a concurrent + /// / from converging this session to + /// and returning while the device is still + /// genuinely starting, now that no longer blocks the caller for + /// the life of that native call (see its own remarks). + /// + private Task? _deviceStartRawCompletion; + + /// + /// Set only when the pump worker was abandoned (finding 24): the continuation of + /// that safely resets the backend and stops the device + /// once that raw completion genuinely happens - deferred out of the teardown task itself + /// so / still complete promptly + /// (Decision #4) and this session's own state still converges immediately, while + /// still awaits this before releasing the engine's lease, + /// so the shared backend is never reset/touched while an abandoned pump thread may still + /// be inside it. + /// + private Task? _deferredBackendTeardown; + + /// + /// The single-flight teardown operation (drain the pump, reset the backend, stop the + /// device), lazily started by whichever of , , + /// or first needs it; every other caller awaits this same task + /// instead of repeating the (non-idempotent) teardown steps a second time. + /// + private Task? _teardownTask; + + /// + /// The single-flight disposal operation, shared by every concurrent + /// caller so a second call awaits the same real teardown rather than returning as soon as + /// the first call merely begins (Decision: see the review finding this closes). + /// + private Task? _disposeTask; + + /// 1 while a enumeration is active; enforces the single-consumer contract. + private int _consuming; + + /// + /// Initializes a new instance of the class, + /// bound to an already-loaded backend and an available capture device. + /// + /// The shared, "hot" recognition backend this session feeds. Must not be null. + /// The available capture device to stream audio from. Must not be null. + /// The rate, in Hz, the backend requires its input at. Must be greater than zero. + /// + /// The recognition model owning the backend, whose + /// hook is applied to every + /// result before it is buffered. Must not be null. + /// + /// + /// Invoked exactly once, at the very end of , to release the + /// owning engine's exclusivity lease. Must not be null. + /// + /// The sink for structural lifecycle and fault events. Must not be null. + /// The dedicated worker this session runs its pump loop through. Must not be null. + internal SherpaOnnxRecognitionSession( + IRecognitionBackend backend, + IAudioCaptureDevice device, + int targetSampleRate, + IRecognitionModel model, + Action releaseLease, + ISpeechDiagnostics diagnostics, + DedicatedWorker worker) + { + ArgumentNullException.ThrowIfNull(backend); + ArgumentNullException.ThrowIfNull(device); + ArgumentNullException.ThrowIfNull(model); + ArgumentNullException.ThrowIfNull(releaseLease); + ArgumentNullException.ThrowIfNull(diagnostics); + ArgumentNullException.ThrowIfNull(worker); + ArgumentOutOfRangeException.ThrowIfLessThanOrEqual(targetSampleRate, 0); + + _backend = backend; + _device = device; + _model = model; + _releaseLease = releaseLease; + _diagnostics = diagnostics; + _worker = worker; + _resultBuffer = new RecognitionResultBuffer(diagnostics, DiagnosticsCategory); + + var deviceSampleRate = device.SampleRate; + var deviceChannelCount = device.ChannelCount; + if (deviceSampleRate <= 0 || deviceChannelCount <= 0) + { + _diagnostics.Report( + SpeechDiagnosticLevel.Warning, + DiagnosticsCategory, + "Capture device reported an unusable audio format; assuming mono audio already at the model's rate."); + deviceSampleRate = targetSampleRate; + deviceChannelCount = 1; + } + + _resampler = new AudioFrameResampler(deviceSampleRate, deviceChannelCount, targetSampleRate); + } + + /// + /// + /// Always : this type is only ever created by + /// for a real, available device. + /// + public bool IsAvailable => true; + + /// + public RecognitionSessionState State + { + get + { + lock (_syncRoot) + { + return _state; + } + } + } + + /// + public event EventHandler? StateChanged; + + /// + /// + /// The state transition into - and starting + /// the pump thread and subscribing to - + /// still happens entirely under , but the actual, synchronous and + /// potentially slow call is run through + /// (the same dedicated-worker idiom and + /// already use) and awaited without holding + /// , so this method no longer blocks its caller for the life of that + /// native call (for example PortAudio's native stream open/start, or a file-backed device + /// synchronously replaying an entire file). The call is also given this method's own + /// , so the same cooperative-cancel-then-abandon policy + /// relies on applies here too: a caller that cancels while the device + /// is still starting is given before this call + /// gives up waiting and reports cancellation/failure, rather than ignoring the request + /// indefinitely. The worker's raw completion - which completes only once + /// genuinely returns, even if abandoned - is + /// published to before the lock is released, and + /// always awaits that before it ever stops the device: a + /// concurrent or call can therefore still + /// never converge this session to and return + /// while the device is still genuinely starting, even though the two calls no longer + /// literally serialize on one lock for the device call's entire duration. Only this call's + /// own continuation (once the abandon-aware device-start task completes) transitions this + /// session onward from to + /// /, + /// and only if a concurrent teardown has not already moved this session on first - so a + /// torn-down session is never forced back to . + /// + /// The pump is started before , not after: a + /// synchronous-replay device (for example a file-backed capture device) can emit an + /// entire file's worth of blocks before Start() returns, and the bounded, + /// drop-oldest channel below would silently discard its earliest blocks were nothing + /// already draining it. + /// + /// + public async Task StartAsync(CancellationToken cancellationToken = default) + { + cancellationToken.ThrowIfCancellationRequested(); + + SessionStateChangedEventArgs startingArgs; + Task deviceStartTask; + + lock (_syncRoot) + { + if (_state != RecognitionSessionState.Created) + { + throw new InvalidOperationException( + "Cannot start recognition: this session has already been started, stopped, or disposed. " + + "Sessions are single-use; create a new session via ISpeechRecognizerEngine.CreateSessionAsync to run again."); + } + + startingArgs = TransitionTo(RecognitionSessionState.Starting); + + var frames = Channel.CreateBounded( + new BoundedChannelOptions(PendingFrameCapacity) + { + FullMode = BoundedChannelFullMode.DropOldest, + SingleReader = true, + SingleWriter = false + }); + _pendingFrames = frames; + + // Start the pump thread first so the channel is already being drained the instant + // the device starts emitting blocks (see the remarks above). + _pumpCts = new CancellationTokenSource(); + _pumpTask = _worker.RunAsync(PumpLoop, _pumpCts.Token, out var pumpCompletion); + _pumpRawCompletion = pumpCompletion; + + _device.FrameCaptured += OnFrameCaptured; + + // Run the native, potentially slow device.Start() call through the same dedicated + // worker the pump loop uses, rather than inline on this caller's thread (see this + // method's remarks). Given this call's own cancellationToken so a cancelled start is + // genuinely abandoned (not silently ignored) rather than always running to completion + // regardless of the caller's request. The raw completion is published to + // _deviceStartRawCompletion before the lock is released so RunTeardownAsync can still + // await it before ever stopping the device. + deviceStartTask = _worker.RunAsync(_ => _device.Start(), cancellationToken, out var deviceStartRawCompletion); + _deviceStartRawCompletion = deviceStartRawCompletion; + } + + // Raised only after _syncRoot has been released (finding 21): invoking a host's + // StateChanged handler while still holding this session's state lock risks deadlock if + // that handler calls back into this session (for example StopAsync/DisposeAsync) and then + // synchronously blocks on the result, since any continuation that needs this same lock + // could never run while this thread holds it. + RaiseStateChanged(startingArgs); + + SessionStateChangedEventArgs? finalArgs; + SpeechRecognizerUnavailableException? startFailure = null; + try + { + await deviceStartTask.ConfigureAwait(false); + + lock (_syncRoot) + { + // Only advance from Starting: a concurrent StopAsync/DisposeAsync/FaultSession may + // already have moved this session on (to Stopping/Stopped, or a differently-caused + // Faulted) while the device was still starting, in which case that outcome stands. + finalArgs = _state == RecognitionSessionState.Starting + ? TransitionTo(RecognitionSessionState.Running) + : null; + } + } + catch (Exception ex) + { + lock (_syncRoot) + { + if (_state == RecognitionSessionState.Starting) + { + // Roll back the subscription and let the already-running pump drain out and + // exit on its own (it reacts only to the channel completing, never to + // _pumpCts - see PumpLoop's remarks) so this session leaks neither the + // subscription nor the dedicated pump thread even if nobody ever calls + // StopAsync/DisposeAsync on this now-Faulted session. Resetting the backend + // and stopping the device are still deferred to that eventual teardown call, + // exactly as for any other fault, since releasing the engine's lease still + // requires one. + _device.FrameCaptured -= OnFrameCaptured; + _pendingFrames.Writer.TryComplete(); + finalArgs = TransitionTo(RecognitionSessionState.Faulted); + } + else + { + // A concurrent teardown already moved this session on while the device was + // still starting; it already unsubscribed/completed the channel itself. + finalArgs = null; + } + } + + if (finalArgs is not null) + { + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"Failed to start recognition because the capture device could not start: {ex.Message}"); + _resultBuffer.Fault(ex); + startFailure = new SpeechRecognizerUnavailableException( + "Cannot start recognition: the capture device failed to start.", + ex); + } + } + + if (finalArgs is not null) + { + RaiseStateChanged(finalArgs); + } + + if (startFailure is not null) + { + throw startFailure; + } + + _diagnostics.Report(SpeechDiagnosticLevel.Info, DiagnosticsCategory, "Started streaming recognition."); + } + + /// + /// + /// Idempotent and safe to call concurrently, including while this session is + /// : every caller (and + /// and ) shares the single teardown operation started by + /// whichever of them gets there first, so the destructive steps - draining the pump, + /// resetting the shared backend, stopping the device - run exactly once no matter how many + /// callers are racing. A faulted session still runs this same teardown so the capture + /// device and shared backend are genuinely released, but its reported + /// stays rather than advancing to + /// , preserving the fault for + /// consumers. + /// + public Task StopAsync(CancellationToken cancellationToken = default) + { + Task teardownTask; + SessionStateChangedEventArgs? transition; + TeardownStart? start; + lock (_syncRoot) + { + if (_state is RecognitionSessionState.Disposing or RecognitionSessionState.Disposed) + { + teardownTask = Task.CompletedTask; + transition = null; + start = null; + } + else + { + (teardownTask, transition, start) = EnsureTeardownStartedLocked(preserveFault: _state == RecognitionSessionState.Faulted); + } + } + + // Raised only after _syncRoot has been released (finding 21/25): see StartAsync's remarks. + if (transition is not null) + { + RaiseStateChanged(transition); + } + + // Started only after the transition above has been raised, and only on this thread's own + // sequential program order (finding 32/33): RunTeardownAsync must never begin running - + // not even the portion of it that could complete synchronously given fast/trivial + // dependencies - until this caller is done raising its own transition, or a concurrent + // StateChanged subscriber could observe the eventual Stopped/Faulted-preserved transition + // before the Stopping transition that must logically precede it. + if (start is { } pending) + { + _ = RunTeardownAsync(pending); + } + + // cancellationToken only bounds this caller's own wait for teardown (finding 19/25): the + // shared teardown task above is never aborted by it, since it is shared with every other + // concurrent/overlapping StopAsync caller (and DisposeAsync), all of whom still need + // draining/resetting/stopping to genuinely happen regardless of whether this particular + // caller stopped waiting for it. This is the one, unconditional return statement in this + // method, so it is always reached regardless of which branch above produced teardownTask. + return cancellationToken.CanBeCanceled + ? teardownTask.WaitAsync(cancellationToken) + : teardownTask; + } + + /// + /// The captured inputs needs to actually run the teardown + /// sequence, plus the that backs the shared, + /// already-published - deliberately separated from starting + /// itself (findings 32/33): the caller that receives this + /// must invoke only after it has released + /// and raised its own state-transition event, never while still + /// holding the lock, so the eventual Stopped/Faulted-preserved transition that + /// raises can never be observed by a subscriber before the + /// Stopping transition that must logically precede it. + /// + private readonly record struct TeardownStart( + Task? PumpTask, + Task? PumpRawCompletion, + CancellationTokenSource? PumpCts, + Task? DeviceStartRawCompletion, + bool PreserveFault, + TaskCompletionSource CompletionSource); + + /// + /// Returns the single, lazily-created teardown task shared by , + /// , and , together with the + /// state-transition event (if any) that the caller must raise via + /// , and - only for whichever caller actually wins the + /// race to create it - the that caller alone must then pass to + /// once it has released and raised + /// its transition (findings 32/33). Must be called with held, so + /// creating it (including the synchronous unsubscribe/channel-completion below) is atomic + /// with every state check a concurrent caller might make - but, deliberately, this method + /// never itself invokes : doing so here, while + /// may be held reentrantly by an outer caller (, + /// , ), risks that method - an + /// ordinary async Task method - running synchronously to completion (because every + /// awaited dependency happens to already be complete) before this call returns, which would + /// let it raise the Stopped/Faulted-preserved transition before the Stopping transition + /// this method itself just produced has had a chance to be raised by the outer caller. + /// + /// + /// Whether the session was already when + /// teardown was first requested, so the final state transition below preserves it instead + /// of advancing to . + /// + private (Task Teardown, SessionStateChangedEventArgs? Transition, TeardownStart? Start) EnsureTeardownStartedLocked(bool preserveFault) + { + if (_teardownTask is not null) + { + return (_teardownTask, null, null); + } + + if (_state == RecognitionSessionState.Created) + { + // Never started: there is nothing to drain, reset, or stop, but single-use + // bookkeeping still requires converging on Stopped - and GetResultsAsync still must + // not hang forever waiting for a result buffer nobody ever completes (finding 23). + var createdArgs = TransitionTo(RecognitionSessionState.Stopped); + _resultBuffer.Complete(); + _teardownTask = Task.CompletedTask; + return (_teardownTask, createdArgs, null); + } + + SessionStateChangedEventArgs? stoppingArgs = null; + if (_state is RecognitionSessionState.Starting or RecognitionSessionState.Running or RecognitionSessionState.Stopping) + { + stoppingArgs = TransitionTo(RecognitionSessionState.Stopping); + } + + // Unsubscribing and completing the channel synchronously, while still holding the lock, + // closes the window in which a captured block could otherwise be queued after teardown + // has already decided to drain and stop. + _device.FrameCaptured -= OnFrameCaptured; + _pendingFrames?.Writer.TryComplete(); + + // Publish a not-yet-running placeholder task immediately so every concurrent caller + // still observes exactly one shared teardown operation; only the winning caller actually + // starts running it, and only after leaving this lock (see this method's remarks). + var completionSource = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + _teardownTask = completionSource.Task; + var start = new TeardownStart(_pumpTask, _pumpRawCompletion, _pumpCts, _deviceStartRawCompletion, preserveFault, completionSource); + return (_teardownTask, stoppingArgs, start); + } + + /// + /// Drains the pump loop, then finalizes this session's state - and, unless the pump worker + /// was abandoned, resets the shared recognition backend and stops the capture device + /// immediately - the one destructive teardown sequence shared by every caller through + /// . + /// + /// + /// The teardown inputs captured under by + /// , including the + /// backing the already-published, shared teardown task. + /// + /// + /// This session's own state still converges (to + /// or the preserved fault) and the result buffer still completes promptly here even when + /// the pump worker was abandoned, matching 's documented contract + /// that its returned task completing means this session itself has converged. Only + /// resetting the shared backend and stopping the capture device - which must never run + /// concurrently with the pump thread still being inside a blocking backend call (finding + /// 24) - is deferred to , a continuation of + /// 's pump raw completion that + /// awaits before releasing the engine's lease, which is what actually closes the reuse + /// race Decision #4 is about: a new session (or engine disposal) touching the shared + /// backend concurrently with an abandoned pump thread that has not yet genuinely exited. + /// + /// Must only ever be invoked after the caller that obtained from + /// has released and + /// raised its own Stopping/Faulted transition (findings 32/33): see + /// 's and 's remarks + /// for why. 's is always + /// completed, even if an earlier step reports a failure via diagnostics, since every step + /// below already converts its own failures to diagnostics rather than throwing. + /// + /// + private async Task RunTeardownAsync(TeardownStart start) + { + var (pumpTask, pumpRawCompletion, pumpCts, deviceStartRawCompletion, preserveFault, completionSource) = start; + try + { + // Every caller invokes this fire-and-forget (`_ = RunTeardownAsync(...)`), relying on + // control returning to it immediately so it never blocks on this method's own + // potentially-slow/native work (backend reset, device stop). Awaiting an + // already-completed task does not yield - it continues synchronously on the calling + // thread - so without this unconditional yield, a null/already-cancelled pumpCts or an + // already-completed pumpTask would let the whole method, including the blocking + // device.Stop() call below, run inline on whichever thread happened to call + // StopAsync/DisposeAsync/FaultSession, defeating their documented + // teardown-runs-in-the-background contract. + await Task.Yield(); + + // Awaited before anything below ever touches the device: StartAsync may still be + // mid-flight on its own native IAudioCaptureDevice.Start() call (run through _worker, + // never under _syncRoot - see its remarks), and device.Stop() below must never run + // concurrently with that still-in-progress device.Start() call. Awaiting the raw + // completion here (not StartAsync's own abandon-aware task) is deliberate: if a + // caller cancelled StartAsync while the device was still starting and the abandon + // timeout has already elapsed, StartAsync's own task completes early as cancelled/ + // faulted while the device-start worker thread may still genuinely be inside + // IAudioCaptureDevice.Start() - this raw completion only resolves once that thread has + // truly exited, exactly mirroring _pumpRawCompletion's role for the pump thread. This + // is also what keeps this session's documented invariant intact now that StartAsync no + // longer blocks its own caller for the device call's duration: this shared teardown + // task - and therefore any concurrent StopAsync/DisposeAsync awaiting it - cannot + // complete until the device has genuinely finished starting. Any exception from a + // failed/abandoned start is already reported and surfaced by StartAsync's own + // continuation; nothing further to do with it here. + if (deviceStartRawCompletion is not null) + { + try + { + await deviceStartRawCompletion.ConfigureAwait(false); + } + catch + { + // Already handled/reported by StartAsync's own continuation. + } + } + + // Cancelling here is purely the abandon-timeout deadline for DedicatedWorker (Decision + // #4): PumpLoop itself never observes this token while draining (see its own remarks), so + // every block already accepted is still decoded and buffered normally; this cancellation + // only matters if the pump thread is genuinely stuck inside a blocking backend call that + // never returns, in which case the worker is abandoned after its timeout rather than + // hanging this call forever. + if (pumpCts is not null) + { + await pumpCts.CancelAsync().ConfigureAwait(false); + } + + var abandoned = false; + if (pumpTask is not null) + { + try + { + await pumpTask.ConfigureAwait(false); + } + catch (OperationCanceledException) + { + // The pump thread may still be inside a blocking backend call + // (AcceptSamples/TryDecode/TryFlush) even though the abandon-aware wrapper above + // has given up waiting for it (finding 24): backend.Reset()/device.Stop() below + // must not run yet, or they would race that still-running call. + abandoned = true; + } + catch (Exception ex) + { + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"The recognition pump loop ended with a fault: {ex.Message}"); + } + } + + pumpCts?.Dispose(); + _resultBuffer.Complete(); + + if (abandoned) + { + // Defer only the backend reset/device stop until the pump thread's own raw + // completion genuinely happens (finding 24); DisposeCoreAsync awaits this before + // releasing the engine's lease. This call itself still completes promptly and + // best-effort, matching Decision #4's existing guarantee that an abandoned + // stop/dispose never hangs its caller. + lock (_syncRoot) + { + _deferredBackendTeardown = ResetBackendAndStopDeviceAsync(pumpRawCompletion); + } + } + else + { + // The pump thread already genuinely exited (it was not abandoned above, or there was + // no pump to begin with), so it is safe to reset the backend and stop the device now. + ResetBackendAndStopDeviceCore(); + } + + SessionStateChangedEventArgs? stoppedArgs = null; + lock (_syncRoot) + { + if (!preserveFault && _state == RecognitionSessionState.Stopping) + { + stoppedArgs = TransitionTo(RecognitionSessionState.Stopped); + } + } + + // Raised only after _syncRoot has been released (finding 21): see StartAsync's remarks. + if (stoppedArgs is not null) + { + RaiseStateChanged(stoppedArgs); + } + + _diagnostics.Report(SpeechDiagnosticLevel.Info, DiagnosticsCategory, "Stopped streaming recognition."); + } + finally + { + // Always completes the shared, previously-published teardown task (findings 32/33), + // regardless of which branch above ran or whether an unexpected exception escaped one + // of them, so no concurrent StopAsync/DisposeAsync/FaultSession caller can ever hang + // waiting on a teardown that silently stopped making progress. + completionSource.TrySetResult(); + } + } + + /// + /// Awaits (the pump thread's own raw completion), + /// then resets the shared recognition backend and stops the capture device - used only + /// when the pump worker was abandoned (finding 24), so the shared backend is never + /// touched while that thread may still be inside it. + /// + private async Task ResetBackendAndStopDeviceAsync(Task? pumpRawCompletion) + { + if (pumpRawCompletion is not null) + { + try + { + await pumpRawCompletion.ConfigureAwait(false); + } + catch + { + // Already reported by RunTeardownAsync if the pump task converged or faulted + // normally; if it was instead abandoned, this only proves the thread has + // genuinely exited - nothing further to report here beyond that guarantee. + } + } + + ResetBackendAndStopDeviceCore(); + } + + /// + /// Resets the shared recognition backend and stops the capture device, reporting (rather + /// than throwing) any failure from either step. + /// + private void ResetBackendAndStopDeviceCore() + { + try + { + _backend.Reset(); + } + catch (Exception ex) + { + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"Failed to reset the recognition backend after stopping: {ex.Message}"); + } + + try + { + _device.Stop(); + } + catch (Exception ex) + { + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"Failed to stop the capture device after recognition: {ex.Message}"); + } + } + + /// + public async IAsyncEnumerable GetResultsAsync( + [EnumeratorCancellation] CancellationToken cancellationToken = default) + { + if (Interlocked.CompareExchange(ref _consuming, 1, 0) != 0) + { + throw new InvalidOperationException( + "GetResultsAsync is single-consumer: a previous enumeration of this session's results is still active."); + } + + try + { + await foreach (var result in _resultBuffer.ReadAllAsync(cancellationToken).ConfigureAwait(false)) + { + yield return result; + } + } + finally + { + Volatile.Write(ref _consuming, 0); + } + } + + /// + /// + /// Every concurrent caller shares the same single-flight disposal task (Decision: see the + /// review finding this closes) rather than a second call returning the instant the first + /// merely begins: both wait for the exact same real teardown, state transitions, and + /// engine-lease release to complete. + /// + public ValueTask DisposeAsync() + { + lock (_syncRoot) + { + _disposeTask ??= DisposeCoreAsync(); + return new ValueTask(_disposeTask); + } + } + + /// + /// Runs the full Stopping -> Stopped (or Faulted-preserving) teardown, then additionally + /// awaits the deferred backend reset/device stop continuation set when the pump worker was + /// abandoned (finding 24) - which itself awaits the pump thread's own raw completion, not + /// merely the abandon-aware task the teardown above already waited for - before + /// transitioning through Disposing to Disposed and releasing the engine's exclusivity + /// lease. This closes Decision #4's race: even though deliberately + /// does not wait out a non-cooperative, abandoned native call, disposal must, because + /// releasing the lease is what permits a new session (or engine disposal) to touch or + /// dispose the shared backend - and that must never overlap an abandoned pump thread that + /// has not genuinely exited yet. Started at most once; see . + /// + private async Task DisposeCoreAsync() + { + Task teardown; + SessionStateChangedEventArgs? transition; + TeardownStart? start; + lock (_syncRoot) + { + (teardown, transition, start) = EnsureTeardownStartedLocked(preserveFault: _state == RecognitionSessionState.Faulted); + } + + // Raised only after _syncRoot has been released (finding 21): see StartAsync's remarks. + if (transition is not null) + { + RaiseStateChanged(transition); + } + + // Started only after the transition above has been raised (findings 32/33): see + // StopAsync's matching comment and TeardownStart's remarks for why. + if (start is { } pending) + { + _ = RunTeardownAsync(pending); + } + + await teardown.ConfigureAwait(false); + + Task? deferredBackendTeardown; + lock (_syncRoot) + { + deferredBackendTeardown = _deferredBackendTeardown; + } + + if (deferredBackendTeardown is not null) + { + try + { + await deferredBackendTeardown.ConfigureAwait(false); + } + catch + { + // FinishBackendTeardownAsync already reports every failure through diagnostics + // itself; this only confirms the abandoned pump thread - and the backend + // reset/device stop that had to wait for it (finding 24) - has genuinely finished + // before the lease below is released. + } + } + + SessionStateChangedEventArgs disposingArgs; + SessionStateChangedEventArgs disposedArgs; + lock (_syncRoot) + { + disposingArgs = TransitionTo(RecognitionSessionState.Disposing); + disposedArgs = TransitionTo(RecognitionSessionState.Disposed); + } + + RaiseStateChanged(disposingArgs); + RaiseStateChanged(disposedArgs); + + _releaseLease(); + } + + /// + /// Moves this session to and returns the event args the caller + /// must raise via once has been + /// released (finding 21). Must be called under , but deliberately + /// does not invoke itself: a host's handler calling back into + /// this session (for example /) and then + /// synchronously blocking on the result would otherwise risk deadlocking against this same + /// lock, since the resulting task's continuation could never run while this thread holds it. + /// + private SessionStateChangedEventArgs TransitionTo(RecognitionSessionState newState) + { + var previous = _state; + _state = newState; + return new SessionStateChangedEventArgs(previous, newState); + } + + /// + /// Invokes for . Must be called only + /// after releasing (finding 21); never called while holding it. + /// + private void RaiseStateChanged(SessionStateChangedEventArgs args) + { + try + { + StateChanged?.Invoke(this, args); + } + catch (Exception ex) + { + // Intentionally broad: a host's StateChanged handler must never be able to break this + // session's own state transition, mirroring ResultReceived's documented convention. + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"A StateChanged handler threw an exception: {ex.Message}"); + } + } + + /// + /// Enqueues one captured block for recognition. Runs on the capture device's audio + /// callback thread and therefore does no work beyond copying and queueing, and checks + /// whether the device has gone unavailable mid-session. + /// + private void OnFrameCaptured(object? sender, AudioCaptureFrameEventArgs e) + { + try + { + if (!_device.IsAvailable) + { + FaultSession(new SpeechRecognizerUnavailableException( + "The capture device became unavailable while a recognition session was running.")); + return; + } + + var frames = _pendingFrames; + if (frames is null || e.Samples.Count == 0) + { + return; + } + + frames.Writer.TryWrite([.. e.Samples]); + } + catch (Exception ex) + { + // Intentionally broad: this event runs on the native audio callback thread, and no + // exception may be allowed to escape back into that callback. + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"A captured audio block could not be queued for recognition: {ex.Message}"); + } + } + + /// + /// Transitions this session to , surfaces + /// through any active enumeration, + /// and starts the same teardown uses - preserving the fault - so + /// the capture device and shared backend are genuinely released even though this method + /// itself runs synchronously on the capture callback thread and cannot await the result. + /// + private void FaultSession(Exception cause) + { + SessionStateChangedEventArgs? faultedArgs = null; + SessionStateChangedEventArgs? teardownArgs = null; + TeardownStart? start = null; + lock (_syncRoot) + { + if (_state is RecognitionSessionState.Starting or RecognitionSessionState.Running or RecognitionSessionState.Stopping) + { + faultedArgs = TransitionTo(RecognitionSessionState.Faulted); + (_, teardownArgs, start) = EnsureTeardownStartedLocked(preserveFault: true); + } + } + + // Raised only after _syncRoot has been released (finding 21): see StartAsync's remarks. + if (faultedArgs is not null) + { + RaiseStateChanged(faultedArgs); + } + + if (teardownArgs is not null) + { + RaiseStateChanged(teardownArgs); + } + + // Started only after both transitions above have been raised (findings 32/33): see + // StopAsync's matching comment and TeardownStart's remarks for why. Fire-and-forget is + // safe here: RunTeardownAsync reports every failure through diagnostics itself rather + // than letting any step throw, so there is nothing this caller needs to observe beyond + // having started it. + if (start is { } pending) + { + _ = RunTeardownAsync(pending); + } + + _resultBuffer.Fault(cause); + } + + /// + /// Drains queued capture blocks through the recognition pipeline until the queue is + /// completed, then flushes any trailing audio. Runs synchronously on the dedicated pump + /// thread (see ). + /// + /// + /// Deliberately does not observe the cancellation token + /// passed to : that token is cancelled by + /// purely as the outer abandon-timeout deadline (Decision #4), and + /// reading it here as well would risk silently dropping buffered-but-undecoded frames in + /// a race with that cancellation, breaking the guarantee that every block accepted before + /// is still decoded. Completion of 's + /// writer is the only signal this loop reacts to. + /// + private void PumpLoop(CancellationToken _) + { + var reader = _pendingFrames!.Reader; + while (true) + { + float[] frame; + try + { + frame = reader.ReadAsync(CancellationToken.None).AsTask().GetAwaiter().GetResult(); + } + catch (ChannelClosedException) + { + break; + } + + if (!ProcessFrame(frame)) + { + // The backend itself failed (finding 22): continuing to pump further frames + // through a backend that just threw would only repeat the same failure for every + // subsequent block, while leaving this session Running and its result buffer open + // forever for a caller still awaiting GetResultsAsync. Stop draining; the session + // has already been faulted (result buffer included) from the pump thread itself. + break; + } + } + + FlushFinal(); + } + + /// + /// Finalizes and buffers any trailing audio the backend has accepted but not yet decoded, + /// as the very last action of the pump loop. + /// + private void FlushFinal() + { + try + { + if (!_backend.TryFlush(out var result) || result is null) + { + return; + } + + BufferResult(result); + } + catch (Exception ex) + { + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"Failed to flush the recognition backend's trailing audio during teardown: {ex.Message}"); + } + } + + /// + /// Converts one captured block to the backend's format, feeds it in, and buffers every + /// result it produced. + /// + /// + /// if the backend itself threw (finding 22), meaning this + /// session has already been faulted and the pump loop must not call this again; + /// otherwise . + /// + private bool ProcessFrame(float[] interleavedSamples) + { + try + { + var monoSamples = _resampler.Convert(interleavedSamples); + _backend.AcceptSamples(monoSamples); + + for (var i = 0; i < MaxResultsPerFrame; i++) + { + if (!_backend.TryDecode(out var result) || result is null) + { + return true; + } + + BufferResult(result); + } + + return true; + } + catch (Exception ex) + { + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"A captured audio block could not be recognized: {ex.Message}"); + + // Unlike the device-unavailable path (FaultSession), this runs ON the pump thread + // itself: FaultSession's teardown would await _pumpTask/_pumpRawCompletion, which + // represent this very thread, so starting teardown here would self-deadlock. Faulting + // the session's state and result buffer is safe to do directly; the pump thread is + // about to exit on its own (PumpLoop breaks right after this returns false), so the + // backend is not touched further, and whichever of StopAsync/DisposeAsync eventually + // runs still drives the real teardown (reset/stop/lease release) exactly as for any + // other fault. + FaultFromPumpThread(ex); + return false; + } + } + + /// + /// Faults this session's state and result buffer from the pump thread itself (finding 22), + /// without starting teardown: unlike , this is called while the + /// pump thread is still executing, so awaiting / + /// here (as starting teardown would) would await this very + /// thread. The pump thread is about to exit on its own right after this returns; the real + /// teardown (backend reset, device stop, lease release) still runs normally whenever a + /// caller eventually calls or . + /// + private void FaultFromPumpThread(Exception cause) + { + SessionStateChangedEventArgs? faultedArgs = null; + lock (_syncRoot) + { + if (_state is RecognitionSessionState.Starting or RecognitionSessionState.Running or RecognitionSessionState.Stopping) + { + faultedArgs = TransitionTo(RecognitionSessionState.Faulted); + } + } + + // Raised only after _syncRoot has been released (finding 21): see StartAsync's remarks. + if (faultedArgs is not null) + { + RaiseStateChanged(faultedArgs); + } + + _resultBuffer.Fault(cause); + } + + /// + /// Applies the owning model's text restoration and adds the result to + /// . + /// + private void BufferResult(SpeechRecognitionResult result) + { + var restoredText = _model.NormalizeText(result.Text, result.IsFinal); + var restoredResult = restoredText == result.Text ? result : result with { Text = restoredText }; + _resultBuffer.AddResult(new SpeechRecognitionEvent(restoredResult)); + } +} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxSpeechRecognizer.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxSpeechRecognizer.cs deleted file mode 100644 index 9a6f2fb..0000000 --- a/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxSpeechRecognizer.cs +++ /dev/null @@ -1,560 +0,0 @@ -using System.Threading.Channels; -using DemaConsulting.Speech.AudioSubsystem; -using DemaConsulting.Speech.Diagnostics; -using DemaConsulting.Speech.ModelManagementSubsystem; - -namespace DemaConsulting.Speech.RecognitionSubsystem; - -/// -/// Real implementation that streams a capture device's audio -/// through a resampler into a recognition engine and raises the resulting provisional and -/// final results. -/// -/// -/// The pipeline is deliberately split across two threads. The capture device raises -/// FrameCaptured from a high-priority audio callback thread where blocking work would -/// cause dropouts, so the frame handler does nothing but copy the block into a bounded -/// channel and return. A single background consumer task then does all the real work - -/// downmix/resample, feed the engine, poll for results, and raise -/// - so no recognition cost is ever paid on the audio thread and -/// handlers never run on it. -/// -/// The channel is bounded and drops the oldest queued block when full. Recognition that has -/// fallen behind live audio can never be caught up by queueing more of it, so bounding the -/// backlog keeps memory flat and latency honest instead of growing both without limit. -/// -/// -/// completes the channel and waits for the consumer to finish, so every -/// block accepted before the call has been decoded and every resulting event raised by the -/// time it returns - including one last flushed result for a trailing utterance that had not -/// yet been decoded, for example a push-to-talk release with no trailing silence (see -/// ). That makes the pipeline deterministic for -/// both hosts and tests, with no polling or timing assumptions anywhere. -/// then also resets the owned engine's decoder state, so a subsequent on -/// the same "hot" engine never inherits a partially decoded utterance or stale hypothesis -/// left over from before the stop. Flushing and resetting are each best-effort: they cross -/// the native decoder boundary, and a fault there is reported through -/// rather than thrown, so still -/// completes and a later is still permitted, but neither the trailing -/// flush nor the engine's clean state can be guaranteed in that one failure case, and in -/// rare cases restart may not fully recover the ability to decode (see -/// for when). This restart guarantee is specific to -/// : also performs the same flush and reset as part -/// of its teardown, and reports the same faults the same way, but is terminal regardless of -/// whether either succeeds - it disposes the engine and permanently blocks a later -/// , so there is nothing further to guarantee about restart in that case. -/// -/// -/// Exceptions raised anywhere in the pipeline - by a frame handler, by the engine, or by a -/// host's own handler - are caught and reported through the -/// diagnostics sink. They are never rethrown into the PortAudio callback, where an escaping -/// exception would tear down the audio stream, and never stop the recognizer. -/// -/// -/// Every decoded result is passed through the owning 's -/// hook before -/// is raised, so per-model text restoration (e.g. casing/ -/// contraction/punctuation restoration for a model whose raw output "yells") is applied -/// uniformly regardless of which model is in use - a model that already produces properly -/// cased/punctuated output relies on the hook's identity default and pays no cost beyond one -/// delegating call. -/// -/// -internal sealed class SherpaOnnxSpeechRecognizer : ISpeechRecognizer -{ - /// - /// The maximum number of captured blocks held pending recognition before the oldest is - /// dropped. Sized for roughly a second of typical capture-callback blocks: long enough to - /// ride out a decoding hiccup, short enough that a persistently overloaded machine sheds - /// audio instead of accumulating unbounded latency. - /// - private const int PendingFrameCapacity = 64; - - /// - /// The maximum number of results drained from the engine for a single captured block. - /// A correct engine reports at most a partial and a final per block; this bound exists so - /// that a faulty engine which always reports a result cannot livelock the consumer. - /// - private const int MaxResultsPerFrame = 32; - - /// The diagnostics category used for every event this recognizer reports. - private const string DiagnosticsCategory = "RecognitionSubsystem"; - - /// - /// Initializes a new instance of the class over - /// an already-loaded engine and an available capture device. - /// - /// - /// The loaded recognition engine this recognizer owns and disposes. Must not be null. - /// - /// - /// The available capture device to stream audio from. Must not be null and must report - /// IsAvailable as ; - /// guarantees both. - /// - /// - /// The rate, in Hz, the engine requires its input at. Must be greater than zero. - /// - /// - /// The recognition model owning this engine, whose - /// hook is applied to every - /// result before it is raised. Must not be null. Required (not optional), mirroring - /// SherpaOnnxSpeechSynthesizer's existing required ISynthesisModel - /// constructor parameter. - /// - /// - /// The sink for structural lifecycle and fault events, or to use - /// . - /// - /// - /// Thrown when , , or - /// is null. - /// - /// - /// Thrown when is less than or equal to zero. - /// - /// - /// Construction never starts capture and never touches hardware; it only records the - /// conversion the running pipeline will need. - /// - internal SherpaOnnxSpeechRecognizer( - IRecognitionEngine engine, - IAudioCaptureDevice captureDevice, - int targetSampleRate, - IRecognitionModel model, - ISpeechDiagnostics? diagnostics = null) - { - ArgumentNullException.ThrowIfNull(engine); - ArgumentNullException.ThrowIfNull(captureDevice); - ArgumentNullException.ThrowIfNull(model); - ArgumentOutOfRangeException.ThrowIfLessThanOrEqual(targetSampleRate, 0); - - _engine = engine; - _captureDevice = captureDevice; - _model = model; - _diagnostics = diagnostics ?? NullSpeechDiagnostics.Instance; - - // Resolve the device's actual capture format now. A device that reports a non-positive - // rate or channel count cannot be honored, so fall back to a pass-through conversion and - // say so, rather than throwing at composition time or silently corrupting the audio. - var deviceSampleRate = captureDevice.SampleRate; - var deviceChannelCount = captureDevice.ChannelCount; - if (deviceSampleRate <= 0 || deviceChannelCount <= 0) - { - _diagnostics.Report( - SpeechDiagnosticLevel.Warning, - DiagnosticsCategory, - "Capture device reported an unusable audio format; assuming mono audio already at the model's rate."); - deviceSampleRate = targetSampleRate; - deviceChannelCount = 1; - } - - _resampler = new AudioFrameResampler(deviceSampleRate, deviceChannelCount, targetSampleRate); - } - - /// Guards the running-state transitions performed by Start, Stop, and Dispose. - private readonly object _syncRoot = new(); - - /// The recognition engine this recognizer owns, uses, and disposes. - private readonly IRecognitionEngine _engine; - - /// The capture device supplying the audio to recognize. - private readonly IAudioCaptureDevice _captureDevice; - - /// The recognition model whose hook is applied to every result. - private readonly IRecognitionModel _model; - - /// The conversion from the device's capture format to the engine's required format. - private readonly AudioFrameResampler _resampler; - - /// The sink for structural lifecycle and fault events. - private readonly ISpeechDiagnostics _diagnostics; - - /// The bounded hand-off between the audio callback thread and the consumer task. - private Channel? _pendingFrames; - - /// The background task draining while running. - private Task? _consumerTask; - - /// Whether capture is currently subscribed and the consumer task is running. - private bool _isRunning; - - /// Whether has already run. - private bool _isDisposed; - - /// - /// - /// Always : this type is only ever created by - /// after the engine loaded successfully and the - /// capture device reported itself available, so its existence is itself the availability - /// guarantee. Every unavailable case is represented by - /// instead. - /// - public bool IsAvailable => true; - - /// - public event EventHandler? ResultReceived; - - /// - public void Start() - { - Channel frames; - lock (_syncRoot) - { - ObjectDisposedException.ThrowIf(_isDisposed, this); - - // Starting an already-running recognizer is a no-op rather than an error, matching - // PortAudioCaptureDevice.Start and letting a host call it defensively. - if (_isRunning) - { - return; - } - - // Drop the oldest pending block when the consumer falls behind: stale audio is worth - // less than bounded memory and bounded latency. - frames = Channel.CreateBounded( - new BoundedChannelOptions(PendingFrameCapacity) - { - FullMode = BoundedChannelFullMode.DropOldest, - SingleReader = true, - SingleWriter = false - }); - _pendingFrames = frames; - _consumerTask = Task.Run(() => ConsumeAsync(frames.Reader)); - _isRunning = true; - } - - // Subscribe before starting so no captured block can be raised before there is a handler - // to enqueue it. - _captureDevice.FrameCaptured += OnFrameCaptured; - try - { - _captureDevice.Start(); - } - catch (Exception ex) - { - // Intentionally broad: the first device start crosses the native audio boundary, and - // any failure there must be contained and surfaced as the recognizer's documented - // unavailable exception rather than escaping with partial pipeline state left behind. - // The device claimed to be available but failed on first use. Unwind everything this - // call set up so a later retry starts from a clean state, then surface the failure. - _captureDevice.FrameCaptured -= OnFrameCaptured; - Task? consumerToDrain; - lock (_syncRoot) - { - _isRunning = false; - _pendingFrames = null; - consumerToDrain = _consumerTask; - _consumerTask = null; - } - - frames.Writer.TryComplete(); - WaitForConsumer(consumerToDrain); - - try - { - // The consumer's last action before exiting was FlushFinal, which - on the real - // engine - marks the current stream finished even though capture never actually - // started and no genuine audio was ever accepted. Without this reset, a later - // retry of Start() on the same "hot" engine would feed that already-finished - // stream and never decode anything again. See StopCore's identical Reset() call - // for why this is best-effort and reported rather than thrown. - _engine.Reset(); - } - catch (Exception resetEx) - { - _diagnostics.Report( - SpeechDiagnosticLevel.Error, - DiagnosticsCategory, - $"Failed to reset the recognition engine after a failed start: {resetEx.Message}"); - } - - _diagnostics.Report( - SpeechDiagnosticLevel.Error, - DiagnosticsCategory, - $"Failed to start recognition because the capture device could not start: {ex.Message}"); - throw new SpeechRecognizerUnavailableException( - "Cannot start recognition: the capture device failed to start.", - ex); - } - - _diagnostics.Report(SpeechDiagnosticLevel.Info, DiagnosticsCategory, "Started streaming recognition."); - } - - /// - public void Stop() - { - StopCore(reportStopped: true); - } - - /// - /// - /// Stops the pipeline (if running) and disposes the owned engine. Idempotent: a second - /// call does nothing, so a host may safely dispose a recognizer it has already disposed. - /// - public void Dispose() - { - lock (_syncRoot) - { - if (_isDisposed) - { - return; - } - - _isDisposed = true; - } - - StopCore(reportStopped: false); - _engine.Dispose(); - } - - /// - /// Performs the shared stop sequence: unsubscribe, complete the queue, drain the - /// consumer, reset the engine, and stop the capture device. - /// - /// - /// Whether to report a structural "stopped" event. Suppressed during disposal, where the - /// stop is an implementation detail of tearing the object down rather than a lifecycle - /// transition a host asked for. - /// - /// - /// The running state is captured under the lock but the consumer is awaited outside it, - /// so a handler that calls back into the recognizer from the - /// consumer thread cannot deadlock against a concurrent stop. - /// - private void StopCore(bool reportStopped) - { - Channel? frames; - Task? consumerTask; - lock (_syncRoot) - { - if (!_isRunning) - { - return; - } - - _isRunning = false; - frames = _pendingFrames; - consumerTask = _consumerTask; - _pendingFrames = null; - _consumerTask = null; - } - - // Detach first so no further blocks are queued, then let the consumer drain what is - // already queued and exit. The consumer's last action before exiting is flushing and - // delivering any trailing audio it could not otherwise decode - see FlushFinal. - _captureDevice.FrameCaptured -= OnFrameCaptured; - frames?.Writer.TryComplete(); - WaitForConsumer(consumerTask); - - try - { - // Discard whatever the flush above could not recover, and any stale hypothesis left - // over from this session, so a subsequent Start() on the same "hot" engine always - // begins decoding from a clean start-of-utterance state rather than inheriting audio - // the caller has already abandoned. - _engine.Reset(); - } - catch (Exception ex) - { - // Intentionally broad: resetting crosses the native decoder boundary, and teardown - // must complete even if that boundary faults while resetting. - // Resetting the engine is best-effort during teardown: the recognizer is already - // detached, so an engine fault here must not prevent Stop/Dispose from completing. - _diagnostics.Report( - SpeechDiagnosticLevel.Error, - DiagnosticsCategory, - $"Failed to reset the recognition engine after stopping: {ex.Message}"); - } - - try - { - _captureDevice.Stop(); - } - catch (Exception ex) - { - // Intentionally broad: stop runs during teardown against the native audio backend, - // and teardown must complete even if that backend faults while stopping. - // Stopping the device is best-effort during teardown: the recognizer is already - // detached, so a device fault here must not prevent Stop/Dispose from completing. - _diagnostics.Report( - SpeechDiagnosticLevel.Error, - DiagnosticsCategory, - $"Failed to stop the capture device after recognition: {ex.Message}"); - } - - if (reportStopped) - { - _diagnostics.Report(SpeechDiagnosticLevel.Info, DiagnosticsCategory, "Stopped streaming recognition."); - } - } - - /// - /// Blocks until the background consumer task has finished draining, reporting rather than - /// propagating any failure it ended with. - /// - /// The consumer task to await, or when none was started. - private void WaitForConsumer(Task? consumerTask) - { - if (consumerTask is null) - { - return; - } - - try - { - consumerTask.GetAwaiter().GetResult(); - } - catch (Exception ex) - { - // Intentionally broad: any consumer fault, whether from engine inference or a host - // callback, must be reported here without crashing the caller performing stop/drain. - _diagnostics.Report( - SpeechDiagnosticLevel.Error, - DiagnosticsCategory, - $"The recognition consumer loop ended with a fault: {ex.Message}"); - } - } - - /// - /// Enqueues one captured block for recognition. Runs on the capture device's audio - /// callback thread and therefore does no work beyond copying and queueing. - /// - /// The capture device raising the event; unused. - /// The captured block of interleaved samples. - /// - /// Every failure is swallowed and reported: an exception escaping this handler would - /// propagate into the native PortAudio callback and tear down the audio stream. - /// - private void OnFrameCaptured(object? sender, AudioCaptureFrameEventArgs e) - { - try - { - // Re-read the field once: a concurrent Stop may have cleared it between the event - // being raised and this handler running. - var frames = _pendingFrames; - if (frames is null || e.Samples.Count == 0) - { - return; - } - - frames.Writer.TryWrite([.. e.Samples]); - } - catch (Exception ex) - { - // Intentionally broad: this event runs on the native audio callback thread, and no - // exception may be allowed to escape back into that callback. - _diagnostics.Report( - SpeechDiagnosticLevel.Error, - DiagnosticsCategory, - $"A captured audio block could not be queued for recognition: {ex.Message}"); - } - } - - /// - /// Drains queued capture blocks through the recognition pipeline until the queue is - /// completed and empty. - /// - /// The queue to drain. - /// A task that completes once the queue is completed and fully drained. - private async Task ConsumeAsync(ChannelReader reader) - { - await foreach (var frame in reader.ReadAllAsync().ConfigureAwait(false)) - { - ProcessFrame(frame); - } - - FlushFinal(); - } - - /// - /// Finalizes and delivers any trailing audio the engine has accepted but not yet decoded, - /// as the very last action of the consumer loop, before the queue is reported drained. - /// - /// - /// Runs on this same background decoding thread, preserving 's - /// documented "always raised from the recognizer's own background decoding thread" - /// guarantee. Without this, a streaming engine's inability to decode the tail of an - /// utterance without more audio it will now never receive - for example a push-to-talk - /// release with no trailing silence - would mean that tail is simply lost when - /// subsequently resets the engine for the next session. All - /// failures are contained here for the same reason as : this - /// crosses the native decoder boundary and may invoke a host's own handler, and a fault - /// from either must not prevent the consumer loop - and therefore or - /// - from completing. - /// - private void FlushFinal() - { - try - { - if (!_engine.TryFlush(out var result) || result is null) - { - return; - } - - // Apply the owning model's own text restoration, exactly as ProcessFrame does for - // every other decoded result, so this final flushed result is normalized the same way. - var restoredText = _model.NormalizeText(result.Text, result.IsFinal); - var restoredResult = restoredText == result.Text ? result : result with { Text = restoredText }; - - ResultReceived?.Invoke(this, new SpeechRecognitionEvent(restoredResult)); - } - catch (Exception ex) - { - // Intentionally broad: finalizing crosses the native decoder boundary, and a fault - // there must not prevent the consumer loop (and therefore Stop/Dispose) from - // completing. - _diagnostics.Report( - SpeechDiagnosticLevel.Error, - DiagnosticsCategory, - $"Failed to flush the recognition engine's trailing audio during teardown: {ex.Message}"); - } - } - - /// - /// Converts one captured block to the engine's format, feeds it in, and raises every - /// result it produced. - /// - /// The captured block, interleaved by channel. - /// - /// All failures are contained here so that one bad block, one engine fault, or one - /// throwing host handler cannot end the session. - /// - private void ProcessFrame(float[] interleavedSamples) - { - try - { - // Convert the device's native format into the single format the model accepts. - var monoSamples = _resampler.Convert(interleavedSamples); - _engine.AcceptSamples(monoSamples); - - // Drain every result this block produced, bounded so a misbehaving engine cannot - // livelock the consumer. - for (var i = 0; i < MaxResultsPerFrame; i++) - { - if (!_engine.TryDecode(out var result) || result is null) - { - return; - } - - // Apply the owning model's own text restoration (e.g. casing/contraction/ - // punctuation restoration for a model whose raw output "yells") before the - // result reaches any consumer, mirroring SherpaOnnxSpeechSynthesizer's own - // NormalizeText call on the synthesis side. - var restoredText = _model.NormalizeText(result.Text, result.IsFinal); - var restoredResult = restoredText == result.Text ? result : result with { Text = restoredText }; - - ResultReceived?.Invoke(this, new SpeechRecognitionEvent(restoredResult)); - } - } - catch (Exception ex) - { - // Intentionally broad: model inference and downstream result handlers both execute in - // this containment boundary, and a fault from either must not terminate recognition. - _diagnostics.Report( - SpeechDiagnosticLevel.Error, - DiagnosticsCategory, - $"A captured audio block could not be recognized: {ex.Message}"); - } - } -} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxSpeechRecognizerEngine.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxSpeechRecognizerEngine.cs new file mode 100644 index 0000000..fd8285f --- /dev/null +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/SherpaOnnxSpeechRecognizerEngine.cs @@ -0,0 +1,214 @@ +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.Diagnostics; +using DemaConsulting.Speech.ModelManagementSubsystem; + +namespace DemaConsulting.Speech.RecognitionSubsystem; + +/// +/// Real implementation holding one loaded, "hot" +/// sherpa-onnx recognition backend and enforcing single-session exclusivity over it. +/// +/// +/// Loading the backend into native memory is the expensive step +/// performs once; this engine then reuses that same backend across as many sequential +/// sessions as a host creates, which is why only one session may be active at a time +/// ( fails fast with +/// rather than queuing - Decision #2). The exclusivity lease spans the full life of a session, +/// through its own completing, so a caller that +/// wants low-latency reuse should keep this engine instance around for as long as it may +/// recognize speech and create a fresh session per run, rather than reloading the backend per +/// turn. +/// +internal sealed class SherpaOnnxSpeechRecognizerEngine : ISpeechRecognizerEngine +{ + /// The diagnostics category used for every event this engine reports. + private const string DiagnosticsCategory = "RecognitionSubsystem"; + + /// Guards access to and . + private readonly object _syncRoot = new(); + + /// The loaded, shared backend this engine owns and disposes. + private readonly IRecognitionBackend _backend; + + /// The recognition model owning . + private readonly IRecognitionModel _model; + + /// The sink for structural lifecycle and fault events. + private readonly ISpeechDiagnostics _diagnostics; + + /// The single-session exclusivity lease over (Decision #2). + private readonly SemaphoreSlim _lease = new(1, 1); + + /// The currently active session, if any, so can dispose it first. + private IRecognitionSession? _currentSession; + + /// Whether this engine has already been disposed. + private bool _isDisposed; + + /// + /// Initializes a new instance of the class + /// over an already-loaded backend. + /// + /// The loaded recognition backend this engine owns and disposes. Must not be null. + /// The recognition model owning . Must not be null. + /// The sink for structural lifecycle and fault events. Must not be null. + internal SherpaOnnxSpeechRecognizerEngine( + IRecognitionBackend backend, + IRecognitionModel model, + ISpeechDiagnostics diagnostics) + { + ArgumentNullException.ThrowIfNull(backend); + ArgumentNullException.ThrowIfNull(model); + ArgumentNullException.ThrowIfNull(diagnostics); + + _backend = backend; + _model = model; + _diagnostics = diagnostics; + } + + /// + /// + /// Always : this type is only ever created by + /// after the backend loaded successfully. + /// + public bool IsAvailable => true; + + /// + /// + /// Thrown when a previously created session still holds this engine's exclusivity lease, + /// including while that session is still tearing down via its own + /// . + /// + /// + /// The disposed check, lease acquisition, session construction, and registration as + /// all happen inside one critical + /// section - this method does no awaiting, so holding the lock for its entire body is + /// safe and closes the race where a concurrent could otherwise + /// observe "not yet disposed", dispose the shared backend and lease, and let this call + /// continue on to acquire a now-disposed lease or hand out a session over a disposed + /// backend. + /// + public Task CreateSessionAsync( + IAudioCaptureDevice device, + CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(device); + cancellationToken.ThrowIfCancellationRequested(); + + lock (_syncRoot) + { + ObjectDisposedException.ThrowIf(_isDisposed, this); + + // No microphone (or no working audio backend) is an ordinary machine state, exactly + // like a model not being installed - there is nothing to stream, so return the + // honest fallback rather than a session doomed to fail the instant it is started. + if (!device.IsAvailable) + { + _diagnostics.Report( + SpeechDiagnosticLevel.Warning, + DiagnosticsCategory, + "Cannot create a recognition session because the supplied capture device is unavailable."); + return Task.FromResult(UnavailableRecognitionSession.Instance); + } + + // Fail fast rather than queue: waiting here would make this call's latency depend on + // an unrelated session's teardown, with no way to bound that wait distinctly from + // the caller's own cancellationToken (Decision #2). + if (!_lease.Wait(0, CancellationToken.None)) + { + throw new RecognitionEngineBusyException( + "Cannot create a recognition session: this engine's backend is already leased by another " + + "active session. Dispose that session before creating a new one."); + } + + var released = 0; + void ReleaseLease() + { + // Idempotent: DisposeAsync's own best-effort teardown could in principle invoke + // this more than once on some error paths, and the lease must only ever be + // released once. + if (Interlocked.Exchange(ref released, 1) != 0) + { + return; + } + + lock (_syncRoot) + { + _currentSession = null; + + // Released inside the same lock DisposeAsync uses to snapshot _currentSession + // and decide whether to dispose _lease: releasing it only after leaving this + // lock would let a concurrent DisposeAsync observe _currentSession already + // cleared, dispose _lease, and then have this call's own Release() below throw + // ObjectDisposedException on the now-disposed semaphore. + _lease.Release(); + } + } + + SherpaOnnxRecognitionSession session; + try + { + session = new SherpaOnnxRecognitionSession( + _backend, + device, + _model.AudioFormat.SampleRate, + _model, + ReleaseLease, + _diagnostics, + new DedicatedWorker(diagnostics: _diagnostics, diagnosticsCategory: DiagnosticsCategory)); + } + catch + { + // Construction failed before the session could ever release its own lease + // itself; roll the acquired lease back so a construction failure cannot strand + // this engine permanently busy. + _lease.Release(); + throw; + } + + _currentSession = session; + return Task.FromResult(session); + } + } + + /// Releases resources held by this engine. + /// + /// Disposes the currently active session first (so its teardown completes and the lease + /// is released cleanly), then disposes the shared backend. Idempotent: a second call does + /// nothing. + /// + public async ValueTask DisposeAsync() + { + IRecognitionSession? session; + lock (_syncRoot) + { + if (_isDisposed) + { + return; + } + + _isDisposed = true; + session = _currentSession; + } + + if (session is not null) + { + try + { + await session.DisposeAsync().ConfigureAwait(false); + } + catch (Exception ex) + { + // Intentionally broad: the active session's own teardown must not prevent this + // engine's backend from being disposed. + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"Failed to dispose the active recognition session while disposing its engine: {ex.Message}"); + } + } + + _backend.Dispose(); + _lease.Dispose(); + } +} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognitionEvent.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognitionEvent.cs index d4df7ad..5e0fb77 100644 --- a/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognitionEvent.cs +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognitionEvent.cs @@ -1,8 +1,8 @@ namespace DemaConsulting.Speech.RecognitionSubsystem; /// -/// Event data carrying one raised by an -/// . +/// Event data carrying one produced by an +/// . /// /// /// The recognition result this event reports. Never . diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognitionResult.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognitionResult.cs index d9284af..7b2e897 100644 --- a/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognitionResult.cs +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognitionResult.cs @@ -2,7 +2,7 @@ namespace DemaConsulting.Speech.RecognitionSubsystem; /// /// One recognition result produced while streaming audio through an -/// . +/// . /// /// /// The recognized text for the current utterance. Never ; may be an diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognizerFactory.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognizerFactory.cs index b32725c..c74b281 100644 --- a/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognizerFactory.cs +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognizerFactory.cs @@ -1,46 +1,45 @@ -using DemaConsulting.Speech.AudioSubsystem; using DemaConsulting.Speech.Diagnostics; using DemaConsulting.Speech.ModelManagementSubsystem; namespace DemaConsulting.Speech.RecognitionSubsystem; /// -/// Composition entry point for obtaining an for one installed -/// recognition model and one capture device. +/// Composition entry point for obtaining an for one +/// installed recognition model. /// /// -/// Hosts call , -/// , or -/// -/// rather than constructing a recognizer directly, so all of the "can this machine actually +/// Hosts call one of this factory's three LoadAsync overloads - taking an +/// installedModelDirectory path directly +/// (), +/// a +/// (), +/// or a +/// () +/// - rather than constructing an engine directly, so all of the "can this machine actually /// recognize speech right now?" logic lives in one reviewable place. Per this library's -/// "nothing throws at composition" decision this method never throws for an ordinary machine -/// state - a model that is not installed, a model whose role is not recognition, a machine -/// with no capture device, and a machine missing the sherpa-onnx native runtime all return -/// and are reported through the -/// diagnostics sink. Only a null argument, which is a programming error rather than a machine -/// state, throws. +/// "nothing throws at composition" decision, these methods never fault their returned task +/// for an ordinary machine state - a model that is not installed, a model whose role is not +/// recognition, and a machine missing the sherpa-onnx native runtime all return +/// and are reported through the +/// diagnostics sink. Only a null argument, an invalid parameter value, and caller +/// cancellation - all programming errors or explicit caller requests rather than machine +/// states - fault the returned task. /// -/// Nothing in this type's public signature names a sherpa-onnx type, keeping this library's -/// "engine backend stays swappable at the public API surface" promise intact. +/// Loading a model is the expensive step (it touches the native runtime and can block), so it +/// runs on a dedicated worker thread and is awaited by the returned rather +/// than blocking the calling thread synchronously. Nothing in this type's public signature +/// names a sherpa-onnx type, keeping this library's "engine backend stays swappable at the +/// public API surface" promise intact. A capture device is bound later, per session, via +/// - not here - so one loaded engine +/// can be reused across many devices or many sequential sessions over its life. /// /// -/// See 's own remarks for guidance on reusing one recognizer -/// across many / -/// cycles for low-latency, repeated recognition, since a call to this factory is the -/// expensive step a host typically wants to make only once. -/// -/// -/// Concurrent pre-warming. Because this method's model-load work is the expensive -/// step above, a host may call it from a background task while other unrelated work -/// proceeds concurrently (for example, to overlap loading the next turn's recognizer with -/// the current turn's speech playback). This is safe with respect to this factory's own -/// state, since it holds none. It is only safe with respect to any caller-supplied -/// collaborator passed to Create - most notably diagnostics - when -/// that collaborator is itself safe for concurrent use from multiple threads; a -/// sink passed as diagnostics that is not -/// thread-safe must not be shared between a concurrent pre-warming call and any other -/// concurrent work that reports to the same sink. +/// Reuse for low latency ("hot" recognition). Calling this factory is the expensive +/// step; and the resulting session's +/// / +/// cycle are comparatively cheap. For low-latency repeated recognition (for example, many +/// turns of a voice conversation), load one engine and reuse it across many sessions rather +/// than reloading the model per turn; only load a new engine to change model or parameters. /// /// public static class SpeechRecognizerFactory @@ -49,8 +48,7 @@ public static class SpeechRecognizerFactory private const string DiagnosticsCategory = "RecognitionSubsystem"; /// - /// Creates a speech recognizer for an installed recognition model, streaming from the - /// supplied capture device. + /// Loads a speech recognizer engine for an installed recognition model. /// /// /// The recognition model to load. Must not be null and must already be installed. @@ -60,11 +58,6 @@ public static class SpeechRecognizerFactory /// SpeechModelStore.GetCurrentDirectory(model.Id). A path that is null, empty, or /// does not exist is treated as "model not installed", not as an error. /// - /// - /// The capture device to stream audio from, as returned by - /// AudioDeviceFactory.CreateCaptureDevice(...). Must not be null; a device - /// reporting IsAvailable == false is treated as "no microphone", not as an error. - /// /// /// The sink to report structural composition, lifecycle, and fault events to, or /// to use . @@ -74,53 +67,53 @@ public static class SpeechRecognizerFactory /// language or tuning value, built from the model's declared /// ), forwarded to /// - /// when the engine is constructed, or to use every model's own + /// when the backend is constructed, or to use every model's own /// default behavior. Neither shipped recognition model declares a parameter today, so this /// argument is a safe no-op for them. /// + /// A token to observe for cancellation of this call. /// - /// A real streaming recognizer when the model is installed, its role is recognition, the - /// capture device is available, and the engine loaded successfully; otherwise - /// . + /// A task that completes with a real engine when the model is installed, its role is + /// recognition, and the backend loaded successfully; otherwise + /// . The task itself is faulted + /// only for a null argument, an invalid parameter value, or cancellation. /// /// - /// Thrown when or is + /// Thrown (faulting the returned task) when is /// . /// /// - /// Thrown when contains a value for a parameter - /// declares that is invalid for it (wrong type, out of range, - /// non-integral for an integer-only parameter, or an unrecognized choice/boolean value). - /// An unrecognized parameter id is not an error - it is reported at - /// and silently ignored, preserving - /// this library's cross-model settings-dictionary-reuse contract. + /// Thrown (faulting the returned task) when contains a + /// value for a parameter declares that is invalid for it (wrong + /// type, out of range, non-integral for an integer-only parameter, or an unrecognized + /// choice/boolean value). An unrecognized parameter id is not an error - it is reported at + /// and silently ignored, preserving this + /// library's cross-model settings-dictionary-reuse contract. /// - /// - /// Loads the model into native memory when it succeeds, so the returned recognizer owns - /// unmanaged resources and must be disposed. Reports every fallback decision through the - /// diagnostics sink as a structural fact, never including recognized text. - /// - /// Recommended composition pattern: create the capture device first with - /// new AudioDeviceFactory().CreateCaptureDevice(selection, model.AudioFormat), then - /// pass that device here. When the backend honors the preferred format, the device opens - /// already matching the recognition model's mono input rate and - /// falls through to its existing no-op fast path. - /// - /// - public static ISpeechRecognizer Create( + /// + /// Thrown (faulting the returned task) when is + /// cancelled before loading completes. + /// + public static Task LoadAsync( IRecognitionModel model, string installedModelDirectory, - IAudioCaptureDevice captureDevice, ISpeechDiagnostics? diagnostics = null, - IReadOnlyDictionary? parameterValues = null) + IReadOnlyDictionary? parameterValues = null, + CancellationToken cancellationToken = default) { - return Create(model, installedModelDirectory, captureDevice, diagnostics, new SherpaOnnxRecognitionEngineFactory(), parameterValues); + return LoadAsync( + model, + installedModelDirectory, + diagnostics, + new SherpaOnnxRecognitionEngineFactory(), + parameterValues, + cancellationToken); } /// - /// Creates a speech recognizer for an installed recognition model, resolving the model's - /// installed-files directory from the supplied model store rather than requiring the caller - /// to know anything about the store's on-disk directory layout. + /// Loads a speech recognizer engine for an installed recognition model, resolving the + /// model's installed-files directory from the supplied model store rather than requiring + /// the caller to know anything about the store's on-disk directory layout. /// /// /// The recognition model to load. Must not be null and must already be installed. @@ -129,60 +122,52 @@ public static ISpeechRecognizer Create( /// The store to resolve 's installed-files directory from, via /// . Must not be null. /// - /// - /// The capture device to stream audio from, as returned by - /// AudioDeviceFactory.CreateCaptureDevice(...). Must not be null; a device - /// reporting IsAvailable == false is treated as "no microphone", not as an error. - /// /// /// The sink to report structural composition, lifecycle, and fault events to, or /// to use . /// + /// + /// An optional session-level parameter value bag, forwarded to + /// + /// when the backend is constructed, or to use every model's own + /// default behavior. + /// + /// A token to observe for cancellation of this call. /// - /// A real streaming recognizer when the model is installed, its role is recognition, the - /// capture device is available, and the engine loaded successfully; otherwise - /// . + /// A task that completes with a real engine, or + /// for any honest unavailable state. See the + /// + /// overload for the full behavior. /// /// - /// Thrown when , , or - /// is . + /// Thrown synchronously, before the returned task is created, when + /// or is . /// /// - /// Thrown when contains an invalid value for a - /// parameter declares. See the - /// - /// overload's matching remark for the full behavior. + /// Thrown (faulting the returned task) when contains an + /// invalid value for a parameter declares. /// - /// - /// Equivalent to calling - /// with - /// store.GetCurrentDirectory(model.Id) as the installed-model-directory argument, so - /// callers never need to know 's on-disk directory-naming - /// scheme just to compose a recognizer. - /// - /// - /// An optional session-level parameter value bag, forwarded to - /// - /// when the engine is constructed, or to use every model's own - /// default behavior. - /// - public static ISpeechRecognizer Create( + /// + /// Thrown (faulting the returned task) when is + /// cancelled before loading completes. + /// + public static Task LoadAsync( IRecognitionModel model, SpeechModelStore store, - IAudioCaptureDevice captureDevice, ISpeechDiagnostics? diagnostics = null, - IReadOnlyDictionary? parameterValues = null) + IReadOnlyDictionary? parameterValues = null, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(model); ArgumentNullException.ThrowIfNull(store); - return Create(model, store.GetCurrentDirectory(model.Id), captureDevice, diagnostics, parameterValues); + return LoadAsync(model, store.GetCurrentDirectory(model.Id), diagnostics, parameterValues, cancellationToken); } /// - /// Creates a speech recognizer for an installed recognition model, resolving the model's - /// installed-files directory from the supplied catalog's own store rather than requiring - /// the caller to construct a separate . + /// Loads a speech recognizer engine for an installed recognition model, resolving the + /// model's installed-files directory from the supplied catalog's own store rather than + /// requiring the caller to construct a separate . /// /// /// The recognition model to load. Must not be null and must already be installed. @@ -191,187 +176,191 @@ public static ISpeechRecognizer Create( /// The catalog whose resolves /// 's installed-files directory. Must not be null. /// - /// - /// The capture device to stream audio from, as returned by - /// AudioDeviceFactory.CreateCaptureDevice(...). Must not be null; a device - /// reporting IsAvailable == false is treated as "no microphone", not as an error. - /// /// /// The sink to report structural composition, lifecycle, and fault events to, or /// to use . /// + /// + /// An optional session-level parameter value bag, forwarded to + /// + /// when the backend is constructed, or to use every model's own + /// default behavior. + /// + /// A token to observe for cancellation of this call. /// - /// A real streaming recognizer when the model is installed, its role is recognition, the - /// capture device is available, and the engine loaded successfully; otherwise - /// . + /// A task that completes with a real engine, or + /// for any honest unavailable state. See the + /// + /// overload for the full behavior. /// /// - /// Thrown when , , or - /// is . + /// Thrown synchronously, before the returned task is created, when + /// or is . /// /// - /// Thrown when contains an invalid value for a - /// parameter declares. See the - /// - /// overload's matching remark for the full behavior. + /// Thrown (faulting the returned task) when contains an + /// invalid value for a parameter declares. /// - /// - /// Equivalent to calling - /// - /// with catalog.Store as the store argument, so a host that already owns a - /// for enumeration and download can compose a recognizer - /// through that same catalog instance, without constructing a second, potentially - /// divergent . - /// - /// - /// An optional session-level parameter value bag, forwarded to - /// - /// when the engine is constructed, or to use every model's own - /// default behavior. - /// - public static ISpeechRecognizer Create( + /// + /// Thrown (faulting the returned task) when is + /// cancelled before loading completes. + /// + public static Task LoadAsync( IRecognitionModel model, SpeechModelCatalog catalog, - IAudioCaptureDevice captureDevice, ISpeechDiagnostics? diagnostics = null, - IReadOnlyDictionary? parameterValues = null) + IReadOnlyDictionary? parameterValues = null, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(model); ArgumentNullException.ThrowIfNull(catalog); - return Create(model, catalog.Store, captureDevice, diagnostics, parameterValues); + return LoadAsync(model, catalog.Store, diagnostics, parameterValues, cancellationToken); } /// - /// Creates a speech recognizer using an injected engine factory and a model store, for tests - /// that need a deterministic engine while still exercising store-based directory resolution. + /// Loads a speech recognizer engine using an injected backend factory and a model store, + /// for tests that need a deterministic backend while still exercising store-based + /// directory resolution. /// /// The recognition model to load. Must not be null. /// The store to resolve the model's installed-files directory from. Must not be null. - /// The capture device to stream audio from. Must not be null. /// The diagnostics sink, or for the null sink. - /// The engine factory to load the model through. Must not be null. + /// The backend factory to load the model through. Must not be null. /// /// An optional session-level parameter value bag forwarded to - /// 's Create call, or to use + /// 's Create call, or to use /// every model's own default behavior. /// + /// A token to observe for cancellation of this call. /// - /// A real streaming recognizer, or for - /// any honest unavailable state. + /// A task that completes with a real engine, or + /// for any honest unavailable state. /// /// - /// Thrown when , , - /// , or is . + /// Thrown synchronously, before the returned task is created, when + /// , , or + /// is . /// /// - /// Thrown when contains an invalid value for a - /// parameter declares. + /// Thrown (faulting the returned task) when contains an + /// invalid value for a parameter declares. /// - internal static ISpeechRecognizer Create( + internal static Task LoadAsync( IRecognitionModel model, SpeechModelStore store, - IAudioCaptureDevice captureDevice, ISpeechDiagnostics? diagnostics, - IRecognitionEngineFactory engineFactory, - IReadOnlyDictionary? parameterValues = null) + IRecognitionBackendFactory backendFactory, + IReadOnlyDictionary? parameterValues = null, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(model); ArgumentNullException.ThrowIfNull(store); - return Create(model, store.GetCurrentDirectory(model.Id), captureDevice, diagnostics, engineFactory, parameterValues); + return LoadAsync( + model, + store.GetCurrentDirectory(model.Id), + diagnostics, + backendFactory, + parameterValues, + cancellationToken); } /// - /// Creates a speech recognizer using an injected engine factory and a model catalog, for - /// tests that need a deterministic engine while still exercising catalog-based directory - /// resolution. + /// Loads a speech recognizer engine using an injected backend factory and a model catalog, + /// for tests that need a deterministic backend while still exercising catalog-based + /// directory resolution. /// /// The recognition model to load. Must not be null. /// The catalog whose store resolves the model's installed-files directory. Must not be null. - /// The capture device to stream audio from. Must not be null. /// The diagnostics sink, or for the null sink. - /// The engine factory to load the model through. Must not be null. + /// The backend factory to load the model through. Must not be null. /// /// An optional session-level parameter value bag forwarded to - /// 's Create call, or to use + /// 's Create call, or to use /// every model's own default behavior. /// + /// A token to observe for cancellation of this call. /// - /// A real streaming recognizer, or for - /// any honest unavailable state. + /// A task that completes with a real engine, or + /// for any honest unavailable state. /// /// - /// Thrown when , , - /// , or is . + /// Thrown synchronously, before the returned task is created, when + /// , , or + /// is . /// /// - /// Thrown when contains an invalid value for a - /// parameter declares. + /// Thrown (faulting the returned task) when contains an + /// invalid value for a parameter declares. /// - internal static ISpeechRecognizer Create( + internal static Task LoadAsync( IRecognitionModel model, SpeechModelCatalog catalog, - IAudioCaptureDevice captureDevice, ISpeechDiagnostics? diagnostics, - IRecognitionEngineFactory engineFactory, - IReadOnlyDictionary? parameterValues = null) + IRecognitionBackendFactory backendFactory, + IReadOnlyDictionary? parameterValues = null, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(model); ArgumentNullException.ThrowIfNull(catalog); - return Create(model, catalog.Store, captureDevice, diagnostics, engineFactory, parameterValues); + return LoadAsync(model, catalog.Store, diagnostics, backendFactory, parameterValues, cancellationToken); } /// - /// Creates a speech recognizer using an injected engine factory, for tests that need a - /// deterministic engine with no model files and no native runtime. + /// Loads a speech recognizer engine using an injected backend factory, for tests that need + /// a deterministic backend with no model files and no native runtime. /// /// The recognition model to load. Must not be null. /// The model's installed-files directory. - /// The capture device to stream audio from. Must not be null. /// The diagnostics sink, or for the null sink. - /// The engine factory to load the model through. Must not be null. + /// The backend factory to load the model through. Must not be null. /// /// An optional session-level parameter value bag forwarded to - /// 's Create call, which in turn passes it to + /// 's Create call, which in turn passes it to /// , /// or to use every model's own default behavior. /// + /// A token to observe for cancellation of this call. /// - /// A real streaming recognizer, or for - /// any honest unavailable state. + /// A task that completes with a real engine, or + /// for any honest unavailable state. /// /// - /// Thrown when , , or - /// is . + /// Thrown (faulting the returned task) when or + /// is . /// /// - /// Thrown when contains an invalid value for a - /// parameter declares - wrong CLR type, out of range, a - /// non-integral value for an integer-only parameter, or an unrecognized choice/boolean - /// value. An unrecognized parameter id is reported at - /// and silently ignored instead. + /// Thrown (faulting the returned task) when contains an + /// invalid value for a parameter declares - wrong CLR type, out + /// of range, a non-integral value for an integer-only parameter, or an unrecognized + /// choice/boolean value. An unrecognized parameter id is reported at + /// and silently ignored instead. + /// + /// + /// Thrown (faulting the returned task) when is + /// cancelled before loading completes. /// - internal static ISpeechRecognizer Create( + internal static async Task LoadAsync( IRecognitionModel model, string installedModelDirectory, - IAudioCaptureDevice captureDevice, ISpeechDiagnostics? diagnostics, - IRecognitionEngineFactory engineFactory, - IReadOnlyDictionary? parameterValues = null) + IRecognitionBackendFactory backendFactory, + IReadOnlyDictionary? parameterValues = null, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(model); - ArgumentNullException.ThrowIfNull(captureDevice); - ArgumentNullException.ThrowIfNull(engineFactory); + ArgumentNullException.ThrowIfNull(backendFactory); + cancellationToken.ThrowIfCancellationRequested(); var sink = diagnostics ?? NullSpeechDiagnostics.Instance; // Validate the caller's parameter value bag against this model's own declared // parameters before anything else: an unrecognized id is reported and silently ignored // (preserving cross-model settings-dictionary reuse), while an invalid value for a - // parameter this model does declare throws synchronously from this call, rather than - // degrading silently deep inside the loaded engine. + // parameter this model does declare faults this call synchronously, rather than + // degrading silently deep inside the loaded backend. SpeechModelParameterDiagnostics.ValidateAndReport( model.Id, model.Parameters, parameterValues, sink, DiagnosticsCategory); @@ -383,60 +372,88 @@ internal static ISpeechRecognizer Create( SpeechDiagnosticLevel.Warning, DiagnosticsCategory, $"Speech recognition is unavailable because model '{model.Id}' is not installed."); - return UnavailableSpeechRecognizer.Instance; + return UnavailableSpeechRecognizerEngine.Instance; } // Guard against a model whose declared role contradicts the recognition interface it - // implements: loading it would build a recognizer that could never produce text. + // implements: loading it would build an engine that could never produce text. if (model.Role != SpeechModelRole.Recognition) { sink.Report( SpeechDiagnosticLevel.Warning, DiagnosticsCategory, $"Speech recognition is unavailable because model '{model.Id}' does not declare the recognition role."); - return UnavailableSpeechRecognizer.Instance; - } - - // No microphone (or no working audio backend) is an ordinary machine state too; there is - // nothing to stream, so report it honestly rather than loading a model that cannot be fed. - if (!captureDevice.IsAvailable) - { - sink.Report( - SpeechDiagnosticLevel.Warning, - DiagnosticsCategory, - "Speech recognition is unavailable because no audio capture device is available."); - return UnavailableSpeechRecognizer.Instance; + return UnavailableSpeechRecognizerEngine.Instance; } // Loading is the one step that touches the native runtime, so it is also the one step // that can fail for a missing org.k2fsa.sherpa.onnx.runtime.{RID} binary or unusable - // model files. Both degrade exactly like a missing model rather than crashing start-up. - IRecognitionEngine engine; + // model files. It is also the one step that genuinely blocks, so it runs on a dedicated + // worker thread and is awaited here rather than run synchronously on the caller's thread. + IRecognitionBackend? backend = null; + Exception? loadFailure = null; + var worker = new DedicatedWorker(diagnostics: sink, diagnosticsCategory: DiagnosticsCategory); + Task completion = Task.CompletedTask; try { - engine = engineFactory.Create(model, installedModelDirectory, parameterValues); + await worker.RunAsync( + _ => + { + try + { + backend = backendFactory.Create(model, installedModelDirectory, parameterValues); + } + catch (Exception ex) + { + // Intentionally broad: backend creation crosses the native runtime/model-file + // boundary, and every load failure must degrade to the documented unavailable + // engine rather than crash composition. + loadFailure = ex; + } + }, + cancellationToken, + out completion).ConfigureAwait(false); } - catch (Exception ex) + catch (OperationCanceledException) + { + // Abandoned: backendFactory.Create ignored cancellation past the worker's abandon + // timeout, so it may still create and assign a native backend after this call has + // already faulted with cancellation. Nothing else observes that eventual backend, so + // dispose it here once it genuinely arrives rather than leaking native model + // resources on a canceled load. + _ = completion.ContinueWith( + _ => backend?.Dispose(), + CancellationToken.None, + TaskContinuationOptions.ExecuteSynchronously, + TaskScheduler.Default); + throw; + } + + if (cancellationToken.IsCancellationRequested) + { + // Finding 30: the worker's delegate finished within the abandon grace period - so + // RunAsync above returned normally with no exception and backend creation may have + // genuinely succeeded - but cancellation was still requested before that happened. + // The caller must not receive a loaded engine for a call it asked to cancel; dispose + // whatever backend was created so this does not leak native model resources, then + // honor the cancellation rather than falling through to report success. + backend?.Dispose(); + cancellationToken.ThrowIfCancellationRequested(); + } + + if (loadFailure is not null) { - // Intentionally broad: engine creation crosses the native runtime/model-file - // boundary, and every load failure must degrade to the documented unavailable - // recognizer rather than crash composition. sink.Report( SpeechDiagnosticLevel.Error, DiagnosticsCategory, - $"Speech recognition is unavailable because the engine for model '{model.Id}' could not be loaded: {ex.Message}"); - return UnavailableSpeechRecognizer.Instance; + $"Speech recognition is unavailable because the engine for model '{model.Id}' could not be loaded: {loadFailure.Message}"); + return UnavailableSpeechRecognizerEngine.Instance; } sink.Report( SpeechDiagnosticLevel.Info, DiagnosticsCategory, - $"Composed a streaming speech recognizer for model '{model.Id}'."); - return new SherpaOnnxSpeechRecognizer( - engine, - captureDevice, - model.AudioFormat.SampleRate, - model, - sink); + $"Loaded a speech recognizer engine for model '{model.Id}'."); + return new SherpaOnnxSpeechRecognizerEngine(backend!, model, sink); } } diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognizerUnavailableException.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognizerUnavailableException.cs index 491caa9..93a63a4 100644 --- a/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognizerUnavailableException.cs +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/SpeechRecognizerUnavailableException.cs @@ -7,12 +7,15 @@ namespace DemaConsulting.Speech.RecognitionSubsystem; /// /// /// Per this library's "nothing throws at composition" decision, obtaining and holding an -/// never throws - -/// returns for every ordinary +/// or an never +/// throws - returns +/// , and its +/// returns +/// , for every ordinary /// "cannot recognize on this machine right now" state (model not installed, native runtime /// absent, no capture device). This exception is reserved for the two genuine error cases: /// a caller that ignored IsAvailable == false and invoked an operational member -/// anyway, and a recognizer whose underlying capture device failed when actually started. +/// anyway, and a session whose underlying capture device failed when actually started. /// It mirrors AudioDeviceUnavailableException so both subsystems signal misuse the /// same way. /// diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableRecognitionSession.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableRecognitionSession.cs new file mode 100644 index 0000000..46d50ea --- /dev/null +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableRecognitionSession.cs @@ -0,0 +1,79 @@ +namespace DemaConsulting.Speech.RecognitionSubsystem; + +/// +/// Honest fallback used when no real recognition session +/// can be composed, reporting as rather +/// than letting a caller operate a session that cannot function. +/// +/// +/// Per this library's "nothing throws at composition" decision, only the operational members +/// (, ) throw +/// , and only when actually invoked - +/// and are safe no-ops. The type is +/// stateless and holds no resources, so the shared is safe for +/// concurrent use by any number of callers. This mirrors +/// and the AudioSubsystem's UnavailableAudioCaptureDevice exactly. +/// +public sealed class UnavailableRecognitionSession : IRecognitionSession +{ + /// + /// Prevents external construction; callers use the shared instead + /// since the type carries no state and multiple instances would provide no value. + /// + private UnavailableRecognitionSession() + { + } + + /// Gets the single shared unavailable recognition session. + public static UnavailableRecognitionSession Instance { get; } = new(); + + /// + /// Backing field for . Never invoked, because this session can + /// never enter a running state; kept only so subscribe/unsubscribe are safe no-ops + /// rather than throwing. + /// + private EventHandler? _stateChanged; + + /// + public bool IsAvailable => false; + + /// + /// Always ; this session never runs. + public RecognitionSessionState State => RecognitionSessionState.Created; + + /// + /// Never raised, since this session can never enter a running state. + public event EventHandler? StateChanged + { + add => _stateChanged += value; + remove => _stateChanged -= value; + } + + /// + /// + /// Always thrown; this session has no real engine or capture device to start. + /// + public Task StartAsync(CancellationToken cancellationToken = default) => + throw new SpeechRecognizerUnavailableException( + "Cannot start recognition: no speech recognition session is available."); + + /// + /// A safe no-op: this session was never running. + public Task StopAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + /// + /// + /// Always thrown; this session has no real engine or capture device to stream results + /// from. + /// + public IAsyncEnumerable GetResultsAsync(CancellationToken cancellationToken = default) => + throw new SpeechRecognizerUnavailableException( + "Cannot stream recognition results: no speech recognition session is available."); + + /// Releases resources held by this session. + /// + /// A no-op: this session owns no engine, thread, or native resource. Disposal must not + /// throw or invalidate . + /// + public ValueTask DisposeAsync() => ValueTask.CompletedTask; +} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableSpeechRecognizer.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableSpeechRecognizer.cs deleted file mode 100644 index 4cd714e..0000000 --- a/src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableSpeechRecognizer.cs +++ /dev/null @@ -1,85 +0,0 @@ -namespace DemaConsulting.Speech.RecognitionSubsystem; - -/// -/// Honest fallback used when no real recognition engine can -/// be composed, reporting as rather than -/// letting a caller build against a recognizer that cannot function. -/// -/// -/// Per this library's "nothing throws at composition" decision, obtaining and holding this -/// instance never throws: a missing model, an absent native runtime, and a machine with no -/// microphone are ordinary machine states at application start-up, not programming errors. -/// Only the operational members (, ) throw -/// , and only when actually invoked - a -/// caller that checks first, as documented, never triggers them. -/// The type is stateless and holds no resources, so the shared is safe -/// for concurrent use by any number of callers and is a no-op that -/// never invalidates it. This mirrors UnavailableAudioCaptureDevice exactly. -/// -public sealed class UnavailableSpeechRecognizer : ISpeechRecognizer -{ - /// - /// Prevents external construction; callers use the shared instead - /// since the type carries no state and multiple instances would provide no value. - /// - private UnavailableSpeechRecognizer() - { - } - - /// - /// Gets the single shared unavailable speech recognizer. - /// - public static UnavailableSpeechRecognizer Instance { get; } = new(); - - /// - /// Backing field for . Never invoked, because this recognizer - /// can never enter a running state; kept only so subscribe/unsubscribe are safe no-ops - /// rather than throwing. - /// - private EventHandler? _resultReceived; - - /// - public bool IsAvailable => false; - - /// - /// - /// Never raised, since this recognizer can never enter a running state; subscribing and - /// unsubscribing are still safe no-ops. - /// - public event EventHandler? ResultReceived - { - add => _resultReceived += value; - remove => _resultReceived -= value; - } - - /// - /// - /// Always thrown; this recognizer has no real engine or capture device to start. - /// - public void Start() - { - throw new SpeechRecognizerUnavailableException( - "Cannot start recognition: no speech recognizer is available."); - } - - /// - /// - /// Always thrown; this recognizer has no real engine or capture device to stop. - /// - public void Stop() - { - throw new SpeechRecognizerUnavailableException( - "Cannot stop recognition: no speech recognizer is available."); - } - - /// Releases resources held by this recognizer. - /// - /// A no-op: this recognizer owns no engine, thread, or native resource. Disposal must not - /// throw or invalidate , because a host that wraps its recognizer in - /// a using block gets the shared instance here and may dispose it many times. - /// - public void Dispose() - { - // Intentionally empty - see the remarks above. - } -} diff --git a/src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableSpeechRecognizerEngine.cs b/src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableSpeechRecognizerEngine.cs new file mode 100644 index 0000000..1e72fdb --- /dev/null +++ b/src/DemaConsulting.Speech/RecognitionSubsystem/UnavailableSpeechRecognizerEngine.cs @@ -0,0 +1,64 @@ +using DemaConsulting.Speech.AudioSubsystem; + +namespace DemaConsulting.Speech.RecognitionSubsystem; + +/// +/// Honest fallback used when no real recognition +/// backend can be composed, reporting as +/// rather than letting a caller build against an engine that cannot function. +/// +/// +/// Per this library's "nothing throws at composition" decision, obtaining and holding this +/// instance never throws, and always succeeds against an +/// ordinary machine state - a missing model, an absent native runtime, and a machine with no +/// microphone are not programming errors - returning +/// rather than throwing for any of them. +/// A null device argument or an already-cancelled cancellation token is a caller/programming +/// error rather than a machine state, and still throws +/// synchronously for either, exactly like the real engine. The type is stateless and holds +/// no resources, so the shared is safe for concurrent use by any +/// number of callers and is a no-op that never invalidates it. +/// This mirrors exactly. +/// +public sealed class UnavailableSpeechRecognizerEngine : ISpeechRecognizerEngine +{ + /// + /// Prevents external construction; callers use the shared instead + /// since the type carries no state and multiple instances would provide no value. + /// + private UnavailableSpeechRecognizerEngine() + { + } + + /// Gets the single shared unavailable speech recognizer engine. + public static UnavailableSpeechRecognizerEngine Instance { get; } = new(); + + /// + public bool IsAvailable => false; + + /// + /// + /// Succeeds, returning , for any + /// non-null device with a non-cancelled token - this engine has nothing to bind a device + /// to, but that is an ordinary unavailable state, not an error. A null + /// or an already-cancelled + /// is still a caller error and throws synchronously, same as the real engine. + /// + public Task CreateSessionAsync( + IAudioCaptureDevice device, + CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(device); + cancellationToken.ThrowIfCancellationRequested(); + + return Task.FromResult(UnavailableRecognitionSession.Instance); + } + + /// Releases resources held by this engine. + /// + /// A no-op: this engine owns no backend, thread, or native resource. Disposal must not + /// throw or invalidate , because a host that wraps its engine in an + /// await using block gets the shared instance here and may dispose it many times. + /// + public ValueTask DisposeAsync() => ValueTask.CompletedTask; +} diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/DedicatedWorker.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/DedicatedWorker.cs new file mode 100644 index 0000000..ac96d9f --- /dev/null +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/DedicatedWorker.cs @@ -0,0 +1,178 @@ +using DemaConsulting.Speech.Diagnostics; + +namespace DemaConsulting.Speech.SynthesisSubsystem; + +/// +/// Internal utility that runs a delegate on a dedicated, long-running background thread, +/// applying a cooperative-cancel-then-abandon policy for native calls that do not honor +/// cancellation promptly. +/// +/// +/// Dedicated per subsystem (this is the synthesis subsystem's own copy; the recognition +/// subsystem carries an independently unit-tested, identically-shaped copy) rather than +/// factored into a new shared internal subsystem, since introducing a third subsystem (with +/// its own full companion-artifact set) for one ~100-line utility would be disproportionate. +/// +/// Every call runs on its own task (a dedicated +/// OS thread, not a pooled thread-pool thread) so a blocking native call never starves the +/// thread pool. On cancellation, the worker is given a +/// bounded grace period (the abandon timeout, injectable for testability, +/// defaulting to ) to stop cooperatively; if it has not +/// finished by then, the returned task completes as cancelled, a +/// is reported, and the worker thread is detached +/// (left to finish in the background; its eventual result or exception is discarded). An +/// async signature alone cannot make a blocking P/Invoke call interruptible, so this is a +/// bounded, honest compromise rather than a true cancellation guarantee. +/// +/// +internal static class DedicatedWorker +{ + /// + /// The default grace period given to a worker to stop cooperatively after cancellation is + /// requested, before it is abandoned. See the type-level remarks for the full policy. + /// + public static readonly TimeSpan DefaultAbandonTimeout = TimeSpan.FromSeconds(2); + + /// + /// Runs on a dedicated long-running thread, applying the + /// cooperative-cancel-then-abandon policy. + /// + /// The type of result produces. + /// + /// The delegate to run, receiving a token it should check cooperatively to stop early. + /// Must not be . + /// + /// A token requesting cooperative cancellation of . + /// The sink to report an abandonment warning to, or for the null sink. + /// The diagnostics category to report an abandonment warning under. + /// + /// The grace period to wait for cooperative completion after cancellation before + /// abandoning the worker, or to use . + /// Exposed for deterministic testing. + /// + /// The result of . + /// Thrown when is . + /// + /// Thrown when is cancelled and + /// either observes it and stops, or does not stop within + /// and is abandoned. + /// + public static async Task Run( + Func work, + CancellationToken cancellationToken, + ISpeechDiagnostics? diagnostics, + string diagnosticsCategory, + TimeSpan? abandonTimeout = null) => + await Start(work, cancellationToken, diagnostics, diagnosticsCategory, abandonTimeout).Task.ConfigureAwait(false); + + /// + /// Runs on a dedicated long-running thread, applying the + /// cooperative-cancel-then-abandon policy, and additionally exposes the dedicated + /// thread's own raw completion alongside the abandon-aware task also + /// returns. + /// + /// The type of result produces. + /// + /// The delegate to run, receiving a token it should check cooperatively to stop early. + /// Must not be . + /// + /// A token requesting cooperative cancellation of . + /// The sink to report an abandonment warning to, or for the null sink. + /// The diagnostics category to report an abandonment warning under. + /// + /// The grace period to wait for cooperative completion after cancellation before + /// abandoning the worker, or to use . + /// Exposed for deterministic testing. + /// + /// + /// The abandon-aware task (see ) together with the + /// dedicated thread's own completion (see ), + /// which a caller that must not reuse or dispose a resource + /// shares with another session - even an abandoned one - should await instead of (or as + /// well as) the abandon-aware task. + /// + /// Thrown when is . + internal static DedicatedWorkerRun Start( + Func work, + CancellationToken cancellationToken, + ISpeechDiagnostics? diagnostics, + string diagnosticsCategory, + TimeSpan? abandonTimeout = null) + { + ArgumentNullException.ThrowIfNull(work); + + var workerTask = Task.Factory.StartNew( + () => work(cancellationToken), + CancellationToken.None, + TaskCreationOptions.LongRunning, + TaskScheduler.Default); + + return new DedicatedWorkerRun(workerTask, AwaitWithAbandonAsync(workerTask, cancellationToken, diagnostics, diagnosticsCategory, abandonTimeout)); + } + + /// + /// Applies the cooperative-cancel-then-abandon policy to , + /// the shared implementation behind both and . + /// + private static async Task AwaitWithAbandonAsync( + Task workerTask, + CancellationToken cancellationToken, + ISpeechDiagnostics? diagnostics, + string diagnosticsCategory, + TimeSpan? abandonTimeout) + { + var sink = diagnostics ?? NullSpeechDiagnostics.Instance; + var timeout = abandonTimeout ?? DefaultAbandonTimeout; + + if (!cancellationToken.CanBeCanceled) + { + return await workerTask.ConfigureAwait(false); + } + + var cancelSignal = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + using var registration = cancellationToken.Register(static state => ((TaskCompletionSource)state!).TrySetResult(true), cancelSignal); + + var firstCompleted = await Task.WhenAny(workerTask, cancelSignal.Task).ConfigureAwait(false); + if (firstCompleted == workerTask) + { + return await workerTask.ConfigureAwait(false); + } + + // Cancellation was requested: give the worker a bounded grace period to stop cooperatively. + var abandonDelay = Task.Delay(timeout, CancellationToken.None); + var raceResult = await Task.WhenAny(workerTask, abandonDelay).ConfigureAwait(false); + if (raceResult == workerTask) + { + return await workerTask.ConfigureAwait(false); + } + + // Abandoned: the worker did not finish cooperatively within the grace period. + sink.Report( + SpeechDiagnosticLevel.Warning, + diagnosticsCategory, + $"A dedicated worker did not stop cooperatively within {timeout.TotalSeconds:F1}s of cancellation and was abandoned; its eventual result will be discarded."); + + // Observe the worker's eventual completion (success or fault) so it never becomes an + // unobserved task exception, without ever awaiting it here. + _ = workerTask.ContinueWith(static t => _ = t.Exception, CancellationToken.None, TaskContinuationOptions.ExecuteSynchronously, TaskScheduler.Default); + + throw new OperationCanceledException(cancellationToken); + } +} + +/// +/// The pair of tasks returned by : the abandon-aware +/// task ordinary callers await, and the dedicated thread's own raw completion that a caller +/// protecting a shared resource from an abandoned worker must await before touching that +/// resource again. +/// +/// The type of result the dedicated work produces. +/// +/// The dedicated thread's own task, which completes only once the work genuinely returns - +/// even if itself already completed early as abandoned. +/// +/// +/// The abandon-aware task with the same semantics as 's +/// return value. +/// +internal readonly record struct DedicatedWorkerRun(Task Completion, Task Task); diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/EngineAudio.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/EngineAudio.cs index 24d836f..362695f 100644 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/EngineAudio.cs +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/EngineAudio.cs @@ -1,7 +1,7 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// -/// The plain-data audio result of one call. +/// The plain-data audio result of one call. /// /// /// The synthesized mono samples, normalized to [-1.0, 1.0]. Never null; may be empty. diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/ISpeechSynthesizer.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/ISpeechSynthesizer.cs deleted file mode 100644 index 4daa069..0000000 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/ISpeechSynthesizer.cs +++ /dev/null @@ -1,124 +0,0 @@ -namespace DemaConsulting.Speech.SynthesisSubsystem; - -/// -/// Mockable chunked/streaming text-to-speech contract: synthesizes text (including inline -/// Natural Language Audio Tags) into ordered audio segments and plays them back, with -/// playback of an earlier segment beginning while later segments are still being -/// synthesized. -/// -/// -/// Per this library's "engine backend stays swappable at the public API surface" -/// decision, no member of this contract exposes a sherpa-onnx (or any other engine) type, so -/// a future non-sherpa-onnx backend can be added without a breaking change. Hosts obtain -/// implementations from rather than constructing them, -/// and tests can substitute this interface directly to exercise host playback logic with no -/// model, no speakers, and no native runtime. -/// -/// A synthesizer that cannot function honestly reports as -/// instead of throwing at composition time; see -/// for the canonical fallback. Implementations own -/// unmanaged inference resources, so callers must dispose them; disposal is idempotent and -/// implies . -/// -/// -/// Thread safety: and may be -/// called concurrently, from any thread, without external synchronization - this is the -/// intended way to cancel an in-flight or -/// session from another thread (for example, a UI thread -/// reacting to a "stop" control while synthesis runs in the background). Only one -/// synthesis/playback session (, , -/// or ) is supported in flight at a time per instance; -/// starting a second session on the same instance while one is already running is not -/// supported and produces undefined interleaving of playback - a caller needing to speak -/// concurrently must use separate synthesizer instances. -/// -/// -/// Reuse for low latency ("hot" synthesis). Obtaining a synthesizer via -/// is the expensive step - it loads the model into -/// native memory - while a // -/// session is cheap and may be run repeatedly on the same -/// instance without reloading the model. For low-latency repeated synthesis (for example, -/// many turns of a voice conversation), construct one synthesizer and reuse it across many -/// sessions rather than disposing and recreating it per turn; only dispose and recreate to -/// change model, device, or parameters. See the user guide's "Hot TTS/STT: Reusing an -/// Instance Across Turns" section for a worked example. -/// -/// -public interface ISpeechSynthesizer : IDisposable -{ - /// - /// Gets a value indicating whether this synthesizer is backed by a real, loaded synthesis - /// engine and a usable playback device. When , every operational - /// member throws rather than silently - /// doing nothing, because a caller that ignores this flag has made a programming error - /// that should surface immediately rather than silently speak nothing. - /// - bool IsAvailable { get; } - - /// - /// Synthesizes text (which may contain inline Natural Language Audio Tags) into an - /// ordered, asynchronously produced stream of audio segments. - /// - /// The text to synthesize. Must not be . - /// A token to cancel the synthesis session. - /// - /// An ordered asynchronous sequence of segments. Segments - /// later in the sequence may still be being synthesized while earlier ones are already - /// available, per this library's chunked, low-latency streaming design. - /// - /// - /// Thrown when is . - /// - /// Thrown when is . - /// - /// Recognized tag syntax and the full closed vocabulary are described on - /// and ; a bracketed - /// span that does not match a recognized tag is passed through as literal text, never - /// thrown as an error. - /// - IAsyncEnumerable SynthesizeStreamAsync(string text, CancellationToken cancellationToken = default); - - /// - /// Plays an ordered stream of audio segments through the configured playback device, - /// writing each segment's silence and samples in order as they become available, and - /// only returns once the playback device has genuinely finished rendering every sample - /// written - not merely once every segment has been handed off to it. - /// - /// The ordered segment stream to play. Must not be . - /// A token to cancel playback. - /// - /// A task that completes once every segment in has been played - /// and the playback device has drained everything written to it (or cancellation ends the - /// wait early). - /// - /// - /// Thrown when is . - /// - /// Thrown when is . - Task PlayStreamAsync(IAsyncEnumerable stream, CancellationToken cancellationToken = default); - - /// - /// Convenience method composing and - /// : synthesizes and speaks , with - /// playback of earlier segments beginning while later segments are still being - /// synthesized. - /// - /// The text to speak. Must not be . - /// A token to cancel the session. - /// A task that completes once the entire text has been played. - /// - /// Thrown when is . - /// - /// Thrown when is . - Task SpeakAsync(string text, CancellationToken cancellationToken = default); - - /// - /// Cancels an in-flight or session - /// deterministically, stopping playback and ending the session's task. Calling this when - /// no session is in flight is a safe no-op. - /// - /// - /// Thrown when is . - /// - void Stop(); -} diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/ISpeechSynthesizerEngine.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/ISpeechSynthesizerEngine.cs new file mode 100644 index 0000000..b2f6e63 --- /dev/null +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/ISpeechSynthesizerEngine.cs @@ -0,0 +1,107 @@ +using DemaConsulting.Speech.AudioSubsystem; + +namespace DemaConsulting.Speech.SynthesisSubsystem; + +/// +/// Layer 3: the loaded, expensive, native-backed text-to-speech model. No playback device is +/// bound yet - obtain an via +/// to bind one and perform repeated, low-latency synthesis against it. +/// +/// +/// Hosts obtain implementations from rather than +/// constructing them directly. Per this library's "nothing throws at composition" decision, +/// +/// never faults for an ordinary machine state (model not installed, role mismatch, native +/// runtime absent) - it returns +/// instead, which honestly reports as . +/// +/// Obtaining this engine is the expensive step: it loads the model into native memory. A +/// session obtained from it is cheap and may be created and run repeatedly without reloading +/// the model - construct one engine per model/parameter combination and reuse it across many +/// sessions rather than disposing and recreating it per turn. +/// +/// +/// Exclusivity. At most one may be leased from this +/// engine at a time, with the lease held for the session's entire life through +/// completion. A concurrent +/// call while a lease is held fails fast with +/// . +/// +/// +public interface ISpeechSynthesizerEngine : IAsyncDisposable +{ + /// + /// Gets a value indicating whether this engine is backed by a real, loaded synthesis + /// model. When , every operational member throws + /// rather than silently doing nothing. + /// + bool IsAvailable { get; } + + /// + /// Binds this engine to exactly one playback device for the entire life of the returned + /// session. + /// + /// + /// The playback device to bind the session to. Must not be . A + /// device reporting IsAvailable == false is an ordinary machine state, not an + /// error; the returned session degrades honestly when an operation that needs playback is + /// actually attempted. + /// + /// A token to cancel this call. + /// A session bound to . + /// Thrown when is . + /// + /// Thrown when this engine's exclusivity lease is already held by another session. + /// + /// Thrown when is cancelled. + Task CreateSessionAsync( + IAudioPlaybackDevice device, + CancellationToken cancellationToken = default); + + /// + /// One-shot convenience: creates a session bound to , speaks + /// through it, and disposes the session, for callers with no need + /// to reuse a session across multiple calls. + /// + /// The playback device to speak through. Must not be . + /// The text to speak. Must not be . + /// A token to cancel the operation. + /// A task that completes once the text has been fully played. + /// + /// Thrown when or is . + /// + /// + /// Thrown when is . + /// + /// + /// Thrown when this engine's exclusivity lease is already held by another session. + /// + Task SpeakAsync( + IAudioPlaybackDevice device, + string text, + CancellationToken cancellationToken = default); + + /// + /// One-shot convenience: creates a session, synthesizes into its + /// full-fidelity ordered segments without playing it, and disposes the session. + /// + /// The text to synthesize. Must not be . + /// A token to cancel the operation. + /// + /// The ordered segments produced for . + /// + /// Thrown when is . + /// + /// Thrown when is . + /// + /// + /// Thrown when this engine's exclusivity lease is already held by another session. + /// + /// + /// This overload never plays audio, so it never binds a real playback device internally; + /// no speakers are required to call it. + /// + Task> SynthesizeAsync( + string text, + CancellationToken cancellationToken = default); +} diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisEngine.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisBackend.cs similarity index 66% rename from src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisEngine.cs rename to src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisBackend.cs index 96c2bdc..c392b24 100644 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisEngine.cs +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisBackend.cs @@ -5,20 +5,25 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// segment of text at a time into raw audio. /// /// -/// This seam exists for the same reason as -/// (Phase 3): it confines every sherpa-onnx TTS interop call to a single implementation -/// () so 's +/// This seam exists for the same reason as RecognitionSubsystem.IRecognitionBackend: it +/// confines every sherpa-onnx TTS interop call to a single implementation +/// () so 's /// chunking, pipelining, and playback logic is fully unit-testable with a pure managed fake. /// No native sherpa-onnx runtime binary and no downloaded model are ever required to run the /// synthesis subsystem's tests. /// +/// Renamed from ISynthesisEngine so the "engine" vocabulary is reserved for the public +/// Layer 3 contract, removing the naming collision +/// between the public, loaded-model-level type and this internal, per-segment native seam. +/// +/// /// Implementations are not thread-safe with respect to concurrent calls, but -/// may be called from a background producer thread while a caller +/// may be called from a background worker thread while a caller /// concurrently disposes the engine from another thread to stop an in-flight session; see -/// for how that race is contained. +/// for how that race is contained. /// /// -internal interface ISynthesisEngine : IDisposable +internal interface ISynthesisBackend : IDisposable { /// /// Gets the sample rate, in Hz, of the audio produces. @@ -35,7 +40,7 @@ internal interface ISynthesisEngine : IDisposable /// /// /// The text to synthesize. May be empty; an empty segment is never generated by - /// , which handles pure-silence segments without + /// , which handles pure-silence segments without /// calling this method at all, but implementations must still behave predictably if given /// one. /// @@ -48,8 +53,10 @@ internal interface ISynthesisEngine : IDisposable /// Thrown when the engine has been disposed. /// /// This is a blocking, synchronous, whole-segment call: it returns only once the segment - /// has been fully synthesized. Callers wishing to overlap synthesis with playback must run - /// it on a background thread, which does. + /// has been fully synthesized. Callers wishing to overlap synthesis with playback, or to + /// bound how long they wait for this call, must run it on a background thread with its own + /// cooperative-cancel-then-abandon policy, which + /// does via . /// EngineAudio Generate(string text, float speed, int speakerId); } diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisEngineFactory.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisBackendFactory.cs similarity index 74% rename from src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisEngineFactory.cs rename to src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisBackendFactory.cs index 5e3f3e6..79bb16c 100644 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisEngineFactory.cs +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisBackendFactory.cs @@ -3,17 +3,18 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// -/// Internal, mockable factory seam that builds a loaded for one -/// installed synthesis model. +/// Internal, mockable factory seam that builds a loaded for +/// one installed synthesis model. /// /// /// Separating engine construction from engine use lets /// decide, without any sherpa-onnx knowledge of its own, whether a real engine could be /// loaded, and lets tests inject a fake engine without a model directory, a native runtime, or -/// an inference session. It mirrors -/// for exactly the same testability reason. +/// an inference session. It mirrors RecognitionSubsystem.IRecognitionBackendFactory for +/// exactly the same testability reason. Renamed from ISynthesisEngineFactory alongside +/// ; members unchanged. /// -internal interface ISynthesisEngineFactory +internal interface ISynthesisBackendFactory { /// /// Loads the synthesis engine described by a model, from that model's installed files. @@ -33,7 +34,8 @@ internal interface ISynthesisEngineFactory /// reasons - a missing native runtime binary, an unsupported RID, or corrupt model files. /// Implementations surface those failures as exceptions; /// converts them into the honest - /// fallback so composition still never throws. + /// fallback so composition still never + /// throws. /// - ISynthesisEngine Create(ISynthesisModel model, string installedModelDirectory); + ISynthesisBackend Create(ISynthesisModel model, string installedModelDirectory); } diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisSession.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisSession.cs new file mode 100644 index 0000000..13298b9 --- /dev/null +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/ISynthesisSession.cs @@ -0,0 +1,91 @@ +namespace DemaConsulting.Speech.SynthesisSubsystem; + +/// +/// Layer 5: a cheap text-to-speech session, bound to exactly one playback device instance for +/// its entire life, obtained from . +/// +/// +/// Binding a session to one concrete device instance for its whole life neutralizes the +/// device-remap hazard of AudioSubsystem.AudioDeviceFactory.RefreshDevices(): a session +/// never silently starts talking to a different physical device than the one it was created +/// with. +/// +/// Reuse for low latency. A session is cheap to create once per model/device +/// combination and cheap to run repeatedly: call or +/// as many times as needed rather than disposing and recreating +/// the session per call; only dispose and recreate to change model, device, or parameters. +/// +/// +/// Concurrency. and do not +/// overlap on the same session: a second call while one is already in flight throws +/// rather than producing undefined interleaving. +/// may be called at any time, from any thread, to request cancellation +/// of whichever operation is currently in flight. +/// +/// +public interface ISynthesisSession : IAsyncDisposable +{ + /// + /// Gets a value indicating whether this session is backed by a real, loaded synthesis + /// engine and has not faulted or been disposed. When , + /// and throw + /// (or, once faulted, + /// ) rather than silently doing nothing. + /// + bool IsAvailable { get; } + + /// Gets this session's current lifecycle state. + SynthesisSessionState State { get; } + + /// + /// Raised every time changes. Handler exceptions are caught and + /// reported through diagnostics rather than propagated, mirroring this library's other + /// event-exception-isolation conventions. + /// + event EventHandler? StateChanged; + + /// + /// Synthesizes and plays through this session's bound playback + /// device. + /// + /// The text to speak. Must not be . + /// A token to cancel this operation. + /// A task that completes once the text has been fully played. + /// Thrown when is . + /// Thrown when is . + /// Thrown when this session has faulted. + /// + /// Thrown when another or call is + /// already in flight on this session. + /// + Task SpeakAsync(string text, CancellationToken cancellationToken = default); + + /// + /// Synthesizes into its full-fidelity ordered segments without + /// playing it. + /// + /// The text to synthesize. Must not be . + /// A token to cancel this operation. + /// The ordered segments produced for . + /// Thrown when is . + /// Thrown when is . + /// Thrown when this session has faulted. + /// + /// Thrown when another or call is + /// already in flight on this session. + /// + Task> SynthesizeAsync(string text, CancellationToken cancellationToken = default); + + /// + /// Requests cancellation of any in-flight or + /// operation. Idempotent and safe to call with no operation + /// in flight, and safe to call concurrently, from any thread, without external + /// synchronization. + /// + /// A token to cancel waiting for the in-flight operation to stop. + /// + /// A task that completes once the in-flight operation (if any) has stopped, or immediately + /// if none was in flight. + /// + Task StopAsync(CancellationToken cancellationToken = default); +} diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/NamespaceDoc.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/NamespaceDoc.cs index f4f2b06..6fea849 100644 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/NamespaceDoc.cs +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/NamespaceDoc.cs @@ -1,17 +1,18 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// -/// Text-to-speech: the closed Natural Language Audio Tag vocabulary, the streaming -/// synthesis/playback pipeline, and the real sherpa-onnx synthesis engine. +/// Text-to-speech: the closed Natural Language Audio Tag vocabulary, the async +/// Engine/Session synthesis API, and the real sherpa-onnx synthesis backend. /// /// /// Contains the model-independent Layer 1 audio-tag parser (, /// ) alongside per-model Layer 2 rendering and sentence -/// chunking, the public streaming/playback pipeline with its -/// composition root and honest -/// fallback, and the internal mockable synthesis -/// seam backed by and its playback-format -/// converter. +/// chunking, the public / +/// Engine/Session API with its composition root and +/// honest / +/// fallbacks, and the internal mockable native seam (, +/// ) backed by , +/// , and their playback-format converter. /// internal static class NamespaceDoc { diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SessionStateChangedEventArgs.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SessionStateChangedEventArgs.cs new file mode 100644 index 0000000..7aaeab0 --- /dev/null +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/SessionStateChangedEventArgs.cs @@ -0,0 +1,16 @@ +namespace DemaConsulting.Speech.SynthesisSubsystem; + +/// +/// Event arguments describing an state transition. +/// +/// The state the session transitioned from. +/// The state the session transitioned to. +/// +/// Raised by . This record is deliberately +/// declared separately from RecognitionSubsystem.SessionStateChangedEventArgs (which +/// carries RecognitionSessionState instead) rather than shared, matching this +/// library's existing "no cross-subsystem public type" boundary (for example +/// SpeechRecognizerUnavailableException/ +/// are similarly duplicated per subsystem). +/// +public sealed record SessionStateChangedEventArgs(SynthesisSessionState Previous, SynthesisSessionState Current); diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSpeechSynthesizer.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSpeechSynthesizer.cs deleted file mode 100644 index 8cfc8cb..0000000 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSpeechSynthesizer.cs +++ /dev/null @@ -1,601 +0,0 @@ -using System.Numerics.Tensors; -using System.Runtime.CompilerServices; -using System.Threading.Channels; -using DemaConsulting.Speech.AudioSubsystem; -using DemaConsulting.Speech.Diagnostics; -using DemaConsulting.Speech.ModelManagementSubsystem; - -namespace DemaConsulting.Speech.SynthesisSubsystem; - -/// -/// Real implementation that chunks text into -/// s, synthesizes each one on a background task while an earlier -/// one plays, and plays the resulting audio through a playback device. -/// -/// -/// The pipeline mirrors but runs -/// in the opposite direction: rather than a native audio callback thread producing captured -/// blocks for a background consumer to recognize, a background producer task synthesizes -/// ordered segments into a bounded channel while the awaiting -/// caller (or ) plays them in order, so synthesis of a later -/// segment overlaps with playback of an earlier one per this library's chunked, -/// low-latency streaming design. -/// -/// Unlike the recognizer's capture queue, the channel here never drops a segment: synthesized -/// audio that has already cost real inference time must never be silently discarded, so the -/// channel is bounded only to cap look-ahead memory and blocks the producer (rather than -/// dropping) when the consumer falls behind. -/// -/// -/// cancels the in-flight session deterministically; a playback device -/// failure or engine failure fails the session's task honestly rather than hanging or -/// crashing, and always stops the playback device in a -/// finally block so a fault never leaves it running. -/// -/// -internal sealed class SherpaOnnxSpeechSynthesizer : ISpeechSynthesizer -{ - /// - /// The maximum number of synthesized segments held pending playback before the producer - /// blocks. Raised from 2 to 5 to smooth pacing over long multi-sentence - /// text: with only 2, playback could catch up to and stall on a still-synthesizing - /// segment whenever one chunk took noticeably longer to synthesize than its predecessor - /// took to play, even though the pipeline as a whole was keeping up on average. A capacity - /// of 5 gives the strictly sequential background producer more chunks of slack to - /// stay ahead of playback, at the cost of a few more segments' worth of look-ahead memory - /// - still small enough to bound both memory and latency to a handful of segments. This is - /// purely a buffer-size tuning change: it does not reduce the latency before the very - /// first word is spoken, which remains bounded by however long the first chunk alone - /// takes to synthesize. - /// - /// Raised again from 5 to 8 once started - /// splitting clause punctuation (,/;/:) unconditionally rather than - /// only past its length budget: a typical multi-clause sentence now yields roughly 1.5-2x - /// as many, smaller chunks as before, so the same segment count now covers - /// noticeably less audio duration than it used to, eroding the anti-starvation margin the - /// 2-to-5 increase above was intended to provide. 8 restores an - /// equivalent real-time look-ahead margin at the new, smaller average chunk size, while - /// remaining a small, bounded constant fully within this field's "cap look-ahead memory" - /// design intent. - /// - /// - /// Bounded look-ahead concurrency (i.e. calling - /// for more than one segment at a time) was investigated and deliberately rejected, not - /// implemented: 's own remarks state that implementations - /// are not thread-safe with respect to concurrent calls, and - /// SherpaOnnxSpeechSynthesizerTests.SynthesizeStreamAsync_LongMultiSentenceInput_ProducesOrderedSegmentsSequentially - /// already asserts engine.MaxConcurrentGenerateCalls == 1 as a regression guard - /// that real parallel calls would immediately - /// break. already starts the next segment's - /// call immediately once the previous segment's - /// channel write completes (which itself only blocks once this capacity is exhausted), so - /// the "synthesize the next chunk while an earlier one plays" pipelining this capacity - /// exists for is already achieved without any concurrent engine calls; only the buffer - /// size above was tuned to compensate for smaller chunks, not the concurrency model. - /// - /// - private const int PendingSegmentCapacity = 8; - - /// The diagnostics category used for every event this synthesizer reports. - private const string DiagnosticsCategory = "SynthesisSubsystem"; - - /// - /// How often polls - /// while waiting for the playback - /// device to finish rendering every queued sample. Short enough that playback never lags - /// noticeably behind the hardware actually finishing, long enough to avoid busy-spinning - /// the thread pool on a value that only a real-time audio callback can change. - /// - private static readonly TimeSpan DrainPollInterval = TimeSpan.FromMilliseconds(15); - - /// - /// An additional wait applied once - /// first reports 0, before stops the device. Real - /// playback backends (notably PortAudio) may still hold a small amount of audio in an - /// internal host buffer that has already left the managed sample queue but has not yet - /// actually reached the speakers; this margin - roughly two callback buffers' worth at a - /// conservative estimate - covers that residual latency so the tail of the last segment - /// is not audibly clipped. - /// - private static readonly TimeSpan DrainTailMargin = TimeSpan.FromMilliseconds(40); - - /// The synthesis engine this synthesizer owns, uses, and disposes. - private readonly ISynthesisEngine _engine; - - /// The available playback device to play synthesized audio through. - private readonly IAudioPlaybackDevice _playbackDevice; - - /// The model driving text normalization, tag rendering, and parameter conventions. - private readonly ISynthesisModel _model; - - /// - /// The session-level parameter value bag (for example a selected voice), or - /// when the caller supplied none. Independent of, and never - /// conflated with, a segment's own per-tag - /// - see 's remarks. - /// - private readonly IReadOnlyDictionary? _parameterValues; - - /// The sink for structural lifecycle and fault events. - private readonly ISpeechDiagnostics _diagnostics; - - /// Guards across concurrent Stop/SpeakAsync calls. - private readonly object _syncRoot = new(); - - /// The cancellation source for the currently in-flight session, if any. - private CancellationTokenSource? _sessionCancellation; - - /// Whether has already run. - private bool _isDisposed; - - /// - /// Initializes a new instance of the class over - /// an already-loaded engine and an available playback device. - /// - /// The loaded synthesis engine this synthesizer owns and disposes. Must not be null. - /// - /// The available playback device to play synthesized audio through. Must not be null and - /// must report IsAvailable as ; - /// guarantees this. - /// - /// The model to normalize text and render tags with. Must not be null. - /// - /// The session-level parameter value bag (for example a selected voice) to resolve a - /// speaker id from once per synthesized segment, or when the - /// caller supplied none. - /// - /// - /// The sink for structural lifecycle and fault events, or to use - /// . - /// - /// - /// Thrown when , , or - /// is null. - /// - internal SherpaOnnxSpeechSynthesizer( - ISynthesisEngine engine, - IAudioPlaybackDevice playbackDevice, - ISynthesisModel model, - IReadOnlyDictionary? parameterValues = null, - ISpeechDiagnostics? diagnostics = null) - { - ArgumentNullException.ThrowIfNull(engine); - ArgumentNullException.ThrowIfNull(playbackDevice); - ArgumentNullException.ThrowIfNull(model); - - _engine = engine; - _playbackDevice = playbackDevice; - _model = model; - _parameterValues = parameterValues; - _diagnostics = diagnostics ?? NullSpeechDiagnostics.Instance; - } - - /// - /// - /// Always : this type is only ever created by - /// after the engine loaded successfully and the - /// playback device reported itself available, so its existence is itself the - /// availability guarantee. Every unavailable case is represented by - /// instead. - /// - public bool IsAvailable => true; - - /// - public IAsyncEnumerable SynthesizeStreamAsync( - string text, - CancellationToken cancellationToken = default) - { - ArgumentNullException.ThrowIfNull(text); - ObjectDisposedException.ThrowIf(_isDisposed, this); - - return SynthesizeStreamCore(text, cancellationToken); - } - - /// - /// The iterator body of , split out so parameter - /// validation happens eagerly rather than only on the first enumeration. - /// - /// The already-validated text to synthesize. - /// A token to cancel the session. - private async IAsyncEnumerable SynthesizeStreamCore( - string text, - [EnumeratorCancellation] CancellationToken cancellationToken) - { - var normalized = _model.NormalizeText(text); - var spans = AudioTagParser.Parse(normalized); - var plan = _model.CapabilityProfile.Render(spans, _model); - - var channel = Channel.CreateBounded( - new BoundedChannelOptions(PendingSegmentCapacity) - { - FullMode = BoundedChannelFullMode.Wait, - SingleReader = true, - SingleWriter = true - }); - - // The producer observes its own linked token, distinct from the caller's - // cancellationToken, so that abandoning enumeration for any reason (not just external - // cancellation) can always unblock it - see the finally block below. - using var producerCancellation = CancellationTokenSource.CreateLinkedTokenSource(cancellationToken); - var producerTask = Task.Run(() => ProduceAsync(plan, channel.Writer, producerCancellation.Token), CancellationToken.None); - - try - { - await foreach (var segment in channel.Reader.ReadAllAsync(cancellationToken).ConfigureAwait(false)) - { - yield return segment; - } - } - finally - { - // The loop above can exit early via an exception before ever reaching a - // normal-completion await - most commonly the reader observing cancellationToken - // cancellation and throwing OperationCanceledException, but also (via the compiler's - // await-foreach desugaring calling DisposeAsync on this iterator) whenever a - // *consumer* of this stream - e.g. PlayStreamAsync's own await foreach - throws for - // an unrelated reason, such as a playback device fault, without cancellationToken - // itself ever being cancelled. producerTask must still be awaited on every exit path - // - normal completion, cancellation, or any other exception - because ProduceAsync - // may be mid-way through a native engine call (GenerateSegment) when this iterator - // is abandoned: it only checks its token between segments, so the native call can - // keep running, untracked, after this method has otherwise returned control to its - // caller. If that caller then disposes the engine (as SpeakAsync's caller commonly - // does once cancellation propagates), the still-running native call touches freed - // native memory. Awaiting here unconditionally guarantees the producer has genuinely - // finished before this iterator ever yields control past this point. - // - // However, an unconditional await alone is not safe: ProduceAsync writes to a - // *bounded* channel via writer.WriteAsync, which only unblocks when either the - // reader keeps draining or its token is cancelled. If enumeration is abandoned for a - // reason other than cancellationToken being cancelled (the unrelated-exception case - // above), and the channel happens to be full at that moment, the reader will never - // drain again and cancellationToken was never cancelled, so WriteAsync - and this - // await - would hang forever. Cancelling the producer's own, always-owned - // producerCancellation here guarantees WriteAsync always has a way to unblock on - // every abandonment path, regardless of why the caller's cancellationToken was or - // was not cancelled. This preserves the previous behavior of surfacing any genuine - // (non-cancellation) producer fault to the caller on the normal-completion path, - // since that path reaches this finally block without needing to cancel anything - - // ProduceAsync has already completed the channel with that fault and this await then - // rethrows it before producerCancellation.Cancel() below can matter. - await producerCancellation.CancelAsync().ConfigureAwait(false); - - // ProduceAsync swallows OperationCanceledException internally (see its own catch - // block) and completes the channel normally instead of faulting on that path, so - // this await will not normally throw OperationCanceledException; the catch below - // only guards the unlikely case of a residual OperationCanceledException still - // escaping, which is expected and benign here and must not mask whatever exception - // (if any) is already propagating out of the try block above. - try - { - await producerTask.ConfigureAwait(false); - } - catch (OperationCanceledException) - { - // Benign on the cancellation path; do not let it mask another exception already - // propagating from the try block above. - } - } - } - - /// - public async Task PlayStreamAsync(IAsyncEnumerable stream, CancellationToken cancellationToken = default) - { - ArgumentNullException.ThrowIfNull(stream); - ObjectDisposedException.ThrowIf(_isDisposed, this); - - try - { - // Started inside the try block (rather than before it) so that a failure from - // Start() itself still runs the finally block below - otherwise a device that faults - // on Start() would never reach Stop(), leaking whatever partial resource it acquired. - _playbackDevice.Start(); - - var resampler = new PlaybackAudioResampler( - _engine.SampleRate, - _playbackDevice.SampleRate > 0 ? _playbackDevice.SampleRate : _engine.SampleRate, - _playbackDevice.ChannelCount > 0 ? _playbackDevice.ChannelCount : 1); - - await foreach (var segment in stream.WithCancellation(cancellationToken).ConfigureAwait(false)) - { - PlaySegment(segment, resampler); - } - - // Every segment has been enqueued, but Write() is fire-and-forget: the playback - // hardware may not have actually rendered any of it yet. Wait for genuine drain - // before falling into the finally block's Stop(), which would otherwise discard - // whatever is still queued and cut the audio off almost as soon as it started. - await WaitForPlaybackDrainAsync(cancellationToken).ConfigureAwait(false); - } - finally - { - try - { - _playbackDevice.Stop(); - } - catch (Exception ex) - { - // Intentionally broad: stop runs during teardown against the native playback - // backend, and a stop fault must not mask the earlier playback outcome. - // Stopping the device is best-effort during teardown: a fault here must not mask - // an earlier, more meaningful exception from playback itself. - _diagnostics.Report( - SpeechDiagnosticLevel.Error, - DiagnosticsCategory, - $"Failed to stop the playback device after synthesis: {ex.Message}"); - } - } - } - - /// - public async Task SpeakAsync(string text, CancellationToken cancellationToken = default) - { - ArgumentNullException.ThrowIfNull(text); - ObjectDisposedException.ThrowIf(_isDisposed, this); - - using var session = CancellationTokenSource.CreateLinkedTokenSource(cancellationToken); - lock (_syncRoot) - { - _sessionCancellation = session; - } - - try - { - var stream = SynthesizeStreamAsync(text, session.Token); - await PlayStreamAsync(stream, session.Token).ConfigureAwait(false); - } - finally - { - lock (_syncRoot) - { - if (ReferenceEquals(_sessionCancellation, session)) - { - _sessionCancellation = null; - } - } - } - } - - /// - public void Stop() - { - CancellationTokenSource? session; - lock (_syncRoot) - { - session = _sessionCancellation; - } - - session?.Cancel(); - } - - /// - /// - /// Stops any in-flight session and disposes the owned engine. Idempotent: a second call - /// does nothing, so a host may safely dispose a synthesizer it has already disposed. - /// - public void Dispose() - { - lock (_syncRoot) - { - if (_isDisposed) - { - return; - } - - _isDisposed = true; - } - - Stop(); - _engine.Dispose(); - } - - /// - /// Synthesizes every segment of a plan in order, writing each result to the channel, and - /// completes the channel normally on cancellation or with the fault on any other failure. - /// - /// The rendered plan to synthesize. - /// The channel to write completed segments to. - /// A token to cancel the session. - private async Task ProduceAsync(SpeechPlan plan, ChannelWriter writer, CancellationToken cancellationToken) - { - try - { - foreach (var segment in plan.Segments) - { - cancellationToken.ThrowIfCancellationRequested(); - var synthesized = GenerateSegment(segment); - await writer.WriteAsync(synthesized, cancellationToken).ConfigureAwait(false); - } - - writer.TryComplete(); - } - catch (OperationCanceledException) - { - writer.TryComplete(); - } - catch (Exception ex) - { - // Intentionally broad: segment synthesis crosses the native inference boundary, and - // any non-cancellation failure must fault the stream predictably instead of escaping - // the background producer unobserved. - _diagnostics.Report( - SpeechDiagnosticLevel.Error, - DiagnosticsCategory, - $"Speech synthesis stopped because a segment could not be synthesized: {ex.Message}"); - writer.TryComplete(ex); - } - } - - /// - /// Synthesizes one segment, applying any speed/volume parameter overrides it carries via - /// , resolving the session-level speaker id via - /// , or produces pure silence for an - /// empty-text pause segment without calling the engine at all. - /// - /// The segment to synthesize. - /// The resulting . - /// - /// Voice selection (, resolved via - /// ) and this segment's own per-tag - /// (resolved via - /// ) are two independent mechanisms with different - /// lifetimes - a session-level voice choice made once per synthesizer versus a - /// transient, per-segment Natural Language Audio Tag override - and this method combines - /// them without either one influencing the other: the resolved speaker id is - /// re-evaluated (cheaply; the hook is pure) for every segment rather than cached once for - /// the whole session, so it always reflects exactly, while - /// the speed/volume ratios continue to come solely from this segment's own overrides. - /// - private SynthesizedSpeech GenerateSegment(SpeechSegment segment) - { - if (segment.Text.Length == 0) - { - return new SynthesizedSpeech( - [], - _engine.SampleRate, - TimeSpan.FromMilliseconds(segment.PreSilenceMs), - TimeSpan.FromMilliseconds(segment.PostSilenceMs)); - } - - var (speedRatio, volumeRatio) = ResolveOverrideRatios(segment.ParameterOverrides); - var speakerId = _model.ResolveSpeakerId(_parameterValues); - - var generated = _engine.Generate(segment.Text, speedRatio, speakerId); - var samples = generated.Samples; - if (Math.Abs(volumeRatio - 1.0) > double.Epsilon) - { - samples = ApplyVolume(samples, volumeRatio); - } - - return new SynthesizedSpeech( - samples, - generated.SampleRate, - TimeSpan.FromMilliseconds(segment.PreSilenceMs), - TimeSpan.FromMilliseconds(segment.PostSilenceMs)); - } - - /// - /// Resolves a segment's parameter overrides into an engine speed ratio and a post-hoc - /// volume (amplitude) ratio, by matching each overridden parameter id against - /// and this model's own declared parameters. - /// - /// The segment's parameter overrides, or . - /// The speed ratio (default 1.0f) and volume ratio (default 1.0) to apply. - private (float SpeedRatio, double VolumeRatio) ResolveOverrideRatios(IReadOnlyDictionary? overrides) - { - var speedRatio = 1.0f; - var volumeRatio = 1.0; - if (overrides is null || overrides.Count == 0) - { - return (speedRatio, volumeRatio); - } - - foreach (var (parameterId, overriddenValue) in overrides) - { - var parameter = _model.Parameters - .OfType() - .FirstOrDefault(candidate => candidate.Id == parameterId); - if (parameter is null || Math.Abs(parameter.Default) <= double.Epsilon || overriddenValue is not double doubleValue) - { - continue; - } - - var ratio = doubleValue / parameter.Default; - if (SpeechParameterConventions.IsSpeedParameter(parameterId)) - { - speedRatio = (float)ratio; - } - else if (SpeechParameterConventions.IsVolumeParameter(parameterId)) - { - volumeRatio = ratio; - } - } - - return (speedRatio, volumeRatio); - } - - /// - /// Scales sample amplitude by a ratio, clamping to [-1.0, 1.0] so a boosted segment - /// never clips into an invalid sample value. - /// - /// The samples to scale. - /// The amplitude ratio to apply. - /// A newly allocated, scaled sample array. - private static float[] ApplyVolume(float[] samples, double ratio) - { - var scaled = new float[samples.Length]; - var ratioF = (float)ratio; - TensorPrimitives.Multiply(samples, ratioF, scaled); - TensorPrimitives.Clamp(scaled, -1.0f, 1.0f, scaled); - return scaled; - } - - /// - /// Polls until every sample written - /// during this session has genuinely been rendered by the playback hardware (not merely - /// enqueued), then applies a small additional tail wait for residual host buffering. - /// - /// - /// A token that, when cancelled, ends the wait promptly rather than waiting for the full - /// drain, consistent with cancellation elsewhere in this class. - /// - /// - /// Polls rather than busy-spins because - /// only changes when PortAudio's own real-time callback thread dequeues samples - there is - /// nothing productive this thread can do except wait for that to happen. - /// - private async Task WaitForPlaybackDrainAsync(CancellationToken cancellationToken) - { - while (_playbackDevice.PendingSampleCount > 0) - { - await Task.Delay(DrainPollInterval, cancellationToken).ConfigureAwait(false); - } - - await Task.Delay(DrainTailMargin, cancellationToken).ConfigureAwait(false); - } - - /// - /// Writes one segment's pre-silence, resampled audio, and post-silence to the playback - /// device in order. - /// - /// The segment to play. - /// The conversion from the engine's rate to the device's resolved format. - private void PlaySegment(SynthesizedSpeech segment, PlaybackAudioResampler resampler) - { - WriteSilence(segment.PreSilence); - - if (segment.Samples.Count > 0) - { - var interleaved = resampler.Convert(segment.Samples is float[] array ? array : [.. segment.Samples]); - if (interleaved.Length > 0) - { - _playbackDevice.Write(interleaved); - } - } - - WriteSilence(segment.PostSilence); - } - - /// - /// Writes a block of zero-valued samples to the playback device representing a duration - /// of real silence, sized for the device's resolved sample rate and channel count. - /// - /// The duration of silence to write. - private void WriteSilence(TimeSpan duration) - { - if (duration <= TimeSpan.Zero) - { - return; - } - - var sampleRate = _playbackDevice.SampleRate > 0 ? _playbackDevice.SampleRate : _engine.SampleRate; - var channelCount = _playbackDevice.ChannelCount > 0 ? _playbackDevice.ChannelCount : 1; - var frameCount = (int)(duration.TotalSeconds * sampleRate); - if (frameCount <= 0) - { - return; - } - - _playbackDevice.Write(new float[frameCount * channelCount]); - } -} diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerEngine.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerEngine.cs new file mode 100644 index 0000000..bff4410 --- /dev/null +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerEngine.cs @@ -0,0 +1,203 @@ +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.Diagnostics; +using DemaConsulting.Speech.ModelManagementSubsystem; + +namespace DemaConsulting.Speech.SynthesisSubsystem; + +/// +/// Real implementation holding one loaded +/// , enforcing single-session exclusivity, and constructing +/// instances bound to a caller-supplied playback +/// device. +/// +/// +/// Exclusivity is enforced with a binary acquired (fail-fast, no +/// waiting) in and released only once the resulting session's +/// has fully completed - not merely when an +/// operation stops - so the lease genuinely spans the session's entire life. +/// +internal sealed class SherpaOnnxSpeechSynthesizerEngine : ISpeechSynthesizerEngine +{ + /// The diagnostics category used for every event this engine reports. + private const string DiagnosticsCategory = "SynthesisSubsystem"; + + /// The loaded synthesis backend this engine owns, lends to sessions, and disposes. + private readonly ISynthesisBackend _backend; + + /// The model driving text normalization, tag rendering, and parameter conventions. + private readonly ISynthesisModel _model; + + /// The session-level parameter value bag, or when none was supplied. + private readonly IReadOnlyDictionary? _parameterValues; + + /// The sink for structural lifecycle and fault events. + private readonly ISpeechDiagnostics _diagnostics; + + /// The exclusivity lease: at most one session may hold it at a time. + private readonly SemaphoreSlim _lease = new(1, 1); + + /// Guards the create-versus-dispose transition and access to . + private readonly object _syncRoot = new(); + + /// Whether has already run. + private bool _isDisposed; + + /// The currently leased session, if any, so can dispose it first. + private ISynthesisSession? _activeSession; + + /// + /// Initializes a new instance of the class + /// over an already-loaded backend. + /// + /// The loaded synthesis backend this engine owns and disposes. Must not be null. + /// The model to normalize text and render tags with. Must not be null. + /// The session-level parameter value bag, or . + /// + /// The sink for structural lifecycle and fault events, or to use + /// . + /// + /// Thrown when or is null. + internal SherpaOnnxSpeechSynthesizerEngine( + ISynthesisBackend backend, + ISynthesisModel model, + IReadOnlyDictionary? parameterValues = null, + ISpeechDiagnostics? diagnostics = null) + { + ArgumentNullException.ThrowIfNull(backend); + ArgumentNullException.ThrowIfNull(model); + + _backend = backend; + _model = model; + _parameterValues = parameterValues; + _diagnostics = diagnostics ?? NullSpeechDiagnostics.Instance; + } + + /// + /// + /// Always : this type is only ever created by + /// after the backend loaded successfully, so its + /// existence is itself the availability guarantee. + /// + public bool IsAvailable => true; + + /// + /// + /// The disposed check, lease acquisition, session construction, and registration as + /// all happen inside one critical + /// section - this method does no awaiting, so holding the lock for its entire body is + /// safe and closes the race where a concurrent could otherwise + /// pass the disposed check, dispose the lease/backend, and let this call continue on to + /// acquire a now-disposed lease or hand out a session backed by already-disposed native + /// state. + /// + public Task CreateSessionAsync(IAudioPlaybackDevice device, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(device); + cancellationToken.ThrowIfCancellationRequested(); + + lock (_syncRoot) + { + ObjectDisposedException.ThrowIf(_isDisposed, this); + + if (!_lease.Wait(0, cancellationToken)) + { + throw new SynthesisEngineBusyException( + "Cannot create a synthesis session: this engine's exclusivity lease is already held by another session."); + } + + SherpaOnnxSynthesisSession session; + try + { + session = new SherpaOnnxSynthesisSession(_backend, device, _model, _parameterValues, _diagnostics, ReleaseLease); + } + catch + { + // Construction failed before the session could ever release its own lease + // itself; roll the acquired lease back so a construction failure cannot strand + // this engine permanently busy. + _lease.Release(); + throw; + } + + _activeSession = session; + return Task.FromResult(session); + } + } + + /// + public async Task SpeakAsync(IAudioPlaybackDevice device, string text, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(device); + ArgumentNullException.ThrowIfNull(text); + + await using var session = await CreateSessionAsync(device, cancellationToken).ConfigureAwait(false); + await session.SpeakAsync(text, cancellationToken).ConfigureAwait(false); + } + + /// + public async Task> SynthesizeAsync(string text, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(text); + + await using var session = await CreateSessionAsync(UnavailableAudioPlaybackDevice.Instance, cancellationToken) + .ConfigureAwait(false); + return await session.SynthesizeAsync(text, cancellationToken).ConfigureAwait(false); + } + + /// Releases the exclusivity lease. Invoked once by a leased session's . + private void ReleaseLease() + { + lock (_syncRoot) + { + _activeSession = null; + + // Released inside the same lock DisposeAsync uses to snapshot _activeSession and + // decide whether to dispose _lease: releasing it only after leaving this lock would + // let a concurrent DisposeAsync observe _activeSession already cleared, dispose + // _lease, and then have this call's own Release() below throw ObjectDisposedException + // on the now-disposed semaphore. + _lease.Release(); + } + } + + /// + /// + /// Disposes any still-active leased session first (best-effort), then disposes the owned + /// backend. Idempotent: a second call does nothing. + /// + public async ValueTask DisposeAsync() + { + ISynthesisSession? activeSession; + lock (_syncRoot) + { + if (_isDisposed) + { + return; + } + + _isDisposed = true; + activeSession = _activeSession; + } + + if (activeSession is not null) + { + try + { + await activeSession.DisposeAsync().ConfigureAwait(false); + } + catch (Exception ex) + { + // Intentionally broad: disposing an already-leased session during engine teardown + // is best-effort, and a fault here must not prevent the backend itself from being + // released. + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"Failed to dispose the active synthesis session while disposing the engine: {ex.Message}"); + } + } + + _lease.Dispose(); + _backend.Dispose(); + } +} diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisEngine.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisEngine.cs index 2f30fe1..792e1a9 100644 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisEngine.cs +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisEngine.cs @@ -3,7 +3,7 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// -/// Real implementation wrapping one sherpa-onnx +/// Real implementation wrapping one sherpa-onnx /// instance. /// /// @@ -13,8 +13,8 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// /// Construction loads the model into native memory and therefore fails (throws) when the /// native runtime binary for the current RID is absent or the model files are unusable. -/// Callers convert that into the honest fallback; -/// see . +/// Callers convert that into the honest +/// fallback; see . /// /// /// Instances own unmanaged resources and must be disposed. may safely @@ -24,7 +24,7 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// provides no cancellation primitive for a call already in progress. /// /// -internal sealed class SherpaOnnxSynthesisEngine : ISynthesisEngine +internal sealed class SherpaOnnxSynthesisEngine : ISynthesisBackend { /// The loaded native offline text-to-speech engine. private readonly OfflineTts _tts; diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisEngineFactory.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisEngineFactory.cs index 2617bee..6e0908f 100644 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisEngineFactory.cs +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisEngineFactory.cs @@ -3,7 +3,7 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// -/// Real implementation that builds a +/// Real implementation that builds a /// from a model's own declared engine configuration. /// /// @@ -15,14 +15,15 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// Loading failures - a missing org.k2fsa.sherpa.onnx.runtime.{RID} native binary, an /// unsupported RID, or corrupt model files - propagate to the caller. /// catches them and returns -/// , so composition still never throws. +/// , so composition still never +/// throws. /// /// The type is stateless and safe for concurrent use. /// -internal sealed class SherpaOnnxSynthesisEngineFactory : ISynthesisEngineFactory +internal sealed class SherpaOnnxSynthesisEngineFactory : ISynthesisBackendFactory { /// - public ISynthesisEngine Create(ISynthesisModel model, string installedModelDirectory) + public ISynthesisBackend Create(ISynthesisModel model, string installedModelDirectory) { ArgumentNullException.ThrowIfNull(model); ArgumentException.ThrowIfNullOrEmpty(installedModelDirectory); diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisSession.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisSession.cs new file mode 100644 index 0000000..eff767e --- /dev/null +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/SherpaOnnxSynthesisSession.cs @@ -0,0 +1,899 @@ +using System.Numerics.Tensors; +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.Diagnostics; +using DemaConsulting.Speech.ModelManagementSubsystem; + +namespace DemaConsulting.Speech.SynthesisSubsystem; + +/// +/// Real implementation that chunks text into +/// s, synthesizes each one on a dedicated worker thread (bounding +/// how long a non-cooperative native call is waited on), and - for - +/// plays the resulting audio through this session's bound playback device. +/// +/// +/// Moved here from the former SherpaOnnxSpeechSynthesizer: the chunking/pipelined- +/// generation logic and playback-device ownership are unchanged in substance, now wrapped in +/// an explicit state machine (), an overlap guard (at most +/// one / call in flight at a time), and +/// per-segment native calls routed through for the +/// cooperative-cancel-then-abandon policy. +/// +/// Unlike the former streaming pipeline, this session does not pipeline synthesis ahead of +/// playback across an unbounded channel: each segment is synthesized, then (for +/// ) written to the playback device, before the next segment's +/// synthesis begins. This keeps the per-call state machine simple (one worker call in flight +/// at a time) while still overlapping this call's own synthesis-then-playback work normally. +/// +/// +internal sealed class SherpaOnnxSynthesisSession : ISynthesisSession +{ + /// The diagnostics category used for every event this session reports. + private const string DiagnosticsCategory = "SynthesisSubsystem"; + + /// + /// How often polls + /// while waiting for the playback + /// device to finish rendering every queued sample. + /// + private static readonly TimeSpan DrainPollInterval = TimeSpan.FromMilliseconds(15); + + /// + /// An additional wait applied once + /// first reports 0, before playback is considered drained, covering a real + /// playback backend's residual host-buffer latency. + /// + private static readonly TimeSpan DrainTailMargin = TimeSpan.FromMilliseconds(40); + + /// The synthesis backend this session uses for every native call. + private readonly ISynthesisBackend _backend; + + /// The playback device this session is bound to for its entire life. + private readonly IAudioPlaybackDevice _device; + + /// The model driving text normalization, tag rendering, and parameter conventions. + private readonly ISynthesisModel _model; + + /// The session-level parameter value bag, or when none was supplied. + private readonly IReadOnlyDictionary? _parameterValues; + + /// The sink for structural lifecycle and fault events. + private readonly ISpeechDiagnostics _diagnostics; + + /// Invoked exactly once, from , to release the engine's exclusivity lease. + private readonly Action _releaseLease; + + /// Guards every field below against concurrent /// calls. + private readonly object _syncRoot = new(); + + /// This session's current lifecycle state. + private SynthesisSessionState _state = SynthesisSessionState.Created; + + /// The cancellation source for the currently in-flight operation, if any. + private CancellationTokenSource? _operationCancellation; + + /// + /// The backing the currently in-flight / + /// call, if any. Tracked so and + /// can await genuine completion of the operation - including + /// the underlying call it may still be waiting on - rather + /// than merely requesting cancellation and returning while native work is still running. + /// + private Task? _operationTask; + + /// + /// The dedicated worker thread's own raw completion for the most recently started native + /// Generate call, set alongside (but independently of) . + /// Unlike , this completes only once the native call has + /// genuinely returned - even if the operation's abandon-aware task already completed early + /// as abandoned - so can await it before releasing the engine's + /// exclusivity lease, closing the race where an abandoned native call could still be + /// running against a backend a new session (or engine disposal) is now free to touch. + /// + private Task? _pendingNativeCompletion; + + /// The exception that faulted this session, if is . + private Exception? _fault; + + /// + /// The single-flight disposal operation, shared by every concurrent + /// caller so a second call awaits the same real teardown rather than returning as soon as + /// the first call merely begins. Also doubles as the disposed flag: non-null means + /// disposal has started. + /// + private Task? _disposeTask; + + /// Whether the exclusivity lease has already been released. + private bool _leaseReleased; + + /// + /// Initializes a new instance of the class, bound + /// to one playback device for its entire life. + /// + /// The loaded synthesis backend to synthesize with. Must not be null. + /// The playback device this session is bound to. Must not be null. + /// The model to normalize text and render tags with. Must not be null. + /// The session-level parameter value bag, or . + /// The sink for structural lifecycle and fault events. Must not be null. + /// Invoked exactly once, from , to release the engine's exclusivity lease. + internal SherpaOnnxSynthesisSession( + ISynthesisBackend backend, + IAudioPlaybackDevice device, + ISynthesisModel model, + IReadOnlyDictionary? parameterValues, + ISpeechDiagnostics diagnostics, + Action releaseLease) + { + ArgumentNullException.ThrowIfNull(backend); + ArgumentNullException.ThrowIfNull(device); + ArgumentNullException.ThrowIfNull(model); + ArgumentNullException.ThrowIfNull(diagnostics); + ArgumentNullException.ThrowIfNull(releaseLease); + + _backend = backend; + _device = device; + _model = model; + _parameterValues = parameterValues; + _diagnostics = diagnostics; + _releaseLease = releaseLease; + } + + /// + /// + /// once this session has been disposed or has faulted; otherwise + /// always - this type is only ever constructed with a real, loaded + /// backend by . + /// + public bool IsAvailable + { + get + { + lock (_syncRoot) + { + return _disposeTask is null && _state != SynthesisSessionState.Faulted; + } + } + } + + /// + public SynthesisSessionState State + { + get + { + lock (_syncRoot) + { + return _state; + } + } + } + + /// + public event EventHandler? StateChanged; + + /// + public Task SpeakAsync(string text, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(text); + return StartOperation(text, playAfterSynthesis: true, cancellationToken); + } + + /// + public Task> SynthesizeAsync(string text, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(text); + return StartOperation(text, playAfterSynthesis: false, cancellationToken); + } + + /// + /// Validates the overlap rule and current state, then starts the operation and records it + /// as this session's in-flight operation (see ) atomically + /// with that validation. + /// + /// + /// Validation and registration both run inside one critical + /// section, including the call into itself: lock is + /// reentrant on the same thread, and calling an method runs it + /// synchronously up to its first genuine before control returns + /// here with a handle - so 's own state + /// transitions to have already happened by the + /// time is assigned below. This closes the window where a + /// concurrent could otherwise observe + /// as while the operation is already inside + /// /native Generate. + /// + private Task> StartOperation( + string text, + bool playAfterSynthesis, + CancellationToken callerToken) + { + var operationCancellation = CancellationTokenSource.CreateLinkedTokenSource(callerToken); + lock (_syncRoot) + { + if (_disposeTask is not null) + { + operationCancellation.Dispose(); + throw new ObjectDisposedException(nameof(SherpaOnnxSynthesisSession)); + } + + if (_state == SynthesisSessionState.Faulted) + { + operationCancellation.Dispose(); + throw new SynthesisSessionFaultedException( + "Cannot perform a synthesis operation: this session has faulted.", _fault!); + } + + if (_state is SynthesisSessionState.Starting or SynthesisSessionState.Running or SynthesisSessionState.Stopping) + { + operationCancellation.Dispose(); + throw new InvalidOperationException( + "session already has an operation in progress; await completion before calling again"); + } + + _operationCancellation = operationCancellation; + + var task = RunOperationAsync(text, playAfterSynthesis, operationCancellation); + _operationTask = task; + return task; + } + } + + /// + /// + /// Requests cancellation of any in-flight operation and awaits the tracked operation's own + /// completion, including its raw native-call completion (see + /// ) so an abandoned + /// native call cannot still be running once this returns (finding 27), before returning, + /// matching the documented contract that the + /// returned task completes once the in-flight operation has genuinely stopped. + /// bounds only this caller's own wait for that + /// teardown (finding 28), mirroring the recognition session's StopAsync: it never + /// aborts the underlying cancel-and-await work itself, which every other concurrent caller + /// (and ) still needs to complete regardless of whether this + /// particular caller stopped waiting for it. + /// + public Task StopAsync(CancellationToken cancellationToken = default) + { + CancellationTokenSource? operationCancellation; + Task? operationTask; + lock (_syncRoot) + { + operationCancellation = _operationCancellation; + operationTask = _operationTask; + } + + var stopTask = CancelAndAwaitOperationAsync(operationCancellation, operationTask); + + return cancellationToken.CanBeCanceled + ? stopTask.WaitAsync(cancellationToken) + : stopTask; + } + + /// + /// Requests cancellation of (if any), tolerating + /// the case where the in-flight operation has already completed and disposed it + /// concurrently. + /// + private static async Task CancelOperationAsync(CancellationTokenSource? operationCancellation) + { + try + { + if (operationCancellation is not null) + { + await operationCancellation.CancelAsync().ConfigureAwait(false); + } + } + catch (ObjectDisposedException) + { + // The in-flight operation completed and disposed its own cancellation source + // concurrently with this call; there is nothing left to cancel. + } + } + + /// + /// Requests cancellation of (if any), awaits + /// (if any), and then - regardless of whether + /// was present - re-reads and awaits the operation's raw + /// native-call completion (see ) to genuine + /// completion. + /// + /// The in-flight operation's cancellation source, if any. + /// The in-flight operation's abandon-aware task, if any. + /// + /// is deliberately re-read here - under + /// - only after has settled (when + /// present), rather than accepted as a snapshot taken by the caller before this method + /// began awaiting: always publishes that field on the + /// same execution path strictly before it awaits the native call itself, so it is + /// guaranteed to already reflect this operation's (possibly still-running, if abandoned) + /// native call by the time completes. A snapshot taken + /// any earlier - for example at the very start of /, + /// before has necessarily reached that publish - can + /// still be even though the operation is genuinely in flight, + /// letting / report the operation as + /// stopped while an abandoned native call is still demonstrably running. + /// + /// This re-read, and the subsequent await, must happen even when + /// is itself (finding 30): + /// 's abandon-fault handler clears _operationTask + /// (and _operationCancellation) as part of transitioning to + /// , while deliberately leaving + /// set to the still-running native call. A caller + /// that reads _operationTask as after that point (for + /// example / invoked after the fault has + /// already been observed) must not treat that as proof there is nothing left to await: + /// skipping the check in that case would let + /// release the engine's exclusivity lease (and let the backend + /// be reused) while the abandoned native call is still demonstrably executing. + /// + /// + private async Task CancelAndAwaitOperationAsync( + CancellationTokenSource? operationCancellation, + Task? operationTask) + { + await CancelOperationAsync(operationCancellation).ConfigureAwait(false); + + if (operationTask is not null) + { + try + { + await operationTask.ConfigureAwait(false); + } + catch + { + // RunOperationAsync already transitions state and reports diagnostics for its own + // failure/cancellation; the caller here only needs to know the operation has + // actually finished. + } + } + + // operationTask can already be null here even though a native call is still genuinely + // running: RunOperationAsync's abandon-fault handler clears _operationTask (and + // _operationCancellation) eagerly, alongside the Faulted transition, so a later + // Start can be distinguished from "this operation is still in flight" - but it + // deliberately leaves _pendingNativeCompletion set. Re-reading and awaiting that field + // unconditionally (not only when operationTask was non-null) is what actually proves the + // abandoned native call has genuinely returned before this method lets the caller + // (StopAsync/DisposeCoreAsync) release the engine lease or consider the backend reusable + // (finding 30). + Task? pendingNativeCompletion; + lock (_syncRoot) + { + pendingNativeCompletion = _pendingNativeCompletion; + } + + if (pendingNativeCompletion is null) + { + return; + } + + try + { + await pendingNativeCompletion.ConfigureAwait(false); + } + catch + { + // Already reported (if it genuinely faulted) by GenerateSegmentAsync's own + // caller; this await exists purely to prove the native call has genuinely + // returned, not to re-surface its outcome. + } + } + + /// + /// Runs one / operation end-to-end: + /// transitions through /, + /// synthesizes (and optionally plays) every segment, then transitions through + /// back to + /// on success or genuine, promptly-honored cancellation, or to + /// on any other failure - including a + /// cancellation request the native Generate call did not honor within + /// 's abandon timeout (finding 26), since that leaves the + /// shared backend not safely reusable until the abandoned call genuinely returns. + /// + /// The text to synthesize. + /// Whether to play each segment through the bound device as it is produced. + /// + /// This operation's cancellation source, already recorded as + /// by before this method runs. + /// + /// The ordered synthesized segments. + private async Task> RunOperationAsync( + string text, + bool playAfterSynthesis, + CancellationTokenSource operationCancellation) + { + TransitionTo(SynthesisSessionState.Starting); + TransitionTo(SynthesisSessionState.Running); + + try + { + var results = await GenerateAndOptionallyPlayAsync(text, playAfterSynthesis, operationCancellation.Token) + .ConfigureAwait(false); + + ClearOperation(); + + TransitionTo(SynthesisSessionState.Stopping); + TransitionTo(SynthesisSessionState.Stopped); + return results; + } + catch (OperationCanceledException) when (operationCancellation.IsCancellationRequested) + { + // Only an exception that corresponds to this operation's own cancellation request + // (from a caller's token, StopAsync, or DisposeAsync - all of which cancel this same + // linked source) is even a candidate for normal cancellation. An + // OperationCanceledException the backend throws on its own initiative, with no + // cancellation actually requested, falls through to the general fault handler below + // instead (finding 9), since silently treating it as a clean stop would let a + // genuinely broken backend be reused. + Task? pendingNativeCompletion; + lock (_syncRoot) + { + pendingNativeCompletion = _pendingNativeCompletion; + } + + if (pendingNativeCompletion is not null && !pendingNativeCompletion.IsCompleted) + { + // The native Generate call did not honor this request within DedicatedWorker's + // abandon timeout and is still running in the background (finding 26): the shared + // backend is not safely reusable until that raw completion genuinely finishes, so + // - unlike a native call that stopped promptly - this is not a clean cancellation + // this session can return to Stopped from. Fault instead, which both reports the + // condition and (via StartOperation's Faulted check) keeps this session from + // starting a second native call concurrently with the still-running abandoned + // one; _pendingNativeCompletion itself is left set so StopAsync/DisposeAsync still + // await its genuine completion. + var abandonFault = new TimeoutException( + "A native Generate call did not honor a cancellation request within the " + + "dedicated worker's abandon timeout and was abandoned while still running."); + + lock (_syncRoot) + { + _operationCancellation = null; + _operationTask = null; + _fault = abandonFault; + } + + TransitionTo(SynthesisSessionState.Faulted); + + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"Synthesis session faulted: {abandonFault.Message}"); + throw; + } + + ClearOperation(); + + TransitionTo(SynthesisSessionState.Stopping); + TransitionTo(SynthesisSessionState.Stopped); + throw; + } + catch (Exception ex) + { + lock (_syncRoot) + { + _operationCancellation = null; + _operationTask = null; + _fault = ex; + } + + TransitionTo(SynthesisSessionState.Faulted); + + // Intentionally broad: a synthesis or playback fault must transition the session to + // the terminal Faulted state and be reported as a structural fact rather than escape + // unobserved. + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"Synthesis session faulted: {ex.Message}"); + throw; + } + finally + { + operationCancellation.Dispose(); + } + } + + /// Clears and once an operation has settled. + private void ClearOperation() + { + lock (_syncRoot) + { + _operationCancellation = null; + _operationTask = null; + } + } + + /// + /// Synthesizes every segment of the rendered plan in order, writing each to the playback + /// device immediately when is , + /// and waiting for genuine playback drain once every segment has been written. + /// + private async Task> GenerateAndOptionallyPlayAsync( + string text, + bool playAfterSynthesis, + CancellationToken cancellationToken) + { + var normalized = _model.NormalizeText(text); + var spans = AudioTagParser.Parse(normalized); + var plan = _model.CapabilityProfile.Render(spans, _model); + + List results = []; + + if (!playAfterSynthesis) + { + foreach (var segment in plan.Segments) + { + cancellationToken.ThrowIfCancellationRequested(); + results.Add(await GenerateSegmentAsync(segment, cancellationToken).ConfigureAwait(false)); + } + + return results; + } + + DedicatedWorkerRun? deviceStartRun = null; + try + { + // Run the native, potentially slow IAudioPlaybackDevice.Start() through the same + // DedicatedWorker cooperative-cancel-then-abandon policy the per-segment native + // Generate call below uses, rather than a plain Task.Run with CancellationToken.None: + // a stuck/blocking device-open call must still be bounded by this operation's own + // cancellation, or a caller cancelling via StopAsync/DisposeAsync would await this + // operation's task forever despite having requested cancellation (the device-start + // call never observes a request it was never given). The raw completion is published + // to _pendingNativeCompletion - the same field GenerateSegmentAsync publishes its own + // native call's completion to - so an abandoned device-start is detected by + // RunOperationAsync's existing fault-vs-stopped check exactly like an abandoned + // Generate call, and so the finally block below can tell whether it is safe to call + // _device.Stop() now or must defer it until the device-start call genuinely finishes. + deviceStartRun = DedicatedWorker.Start( + _ => + { + _device.Start(); + return true; + }, + cancellationToken, + _diagnostics, + DiagnosticsCategory); + + lock (_syncRoot) + { + _pendingNativeCompletion = deviceStartRun.Value.Completion; + } + + await deviceStartRun.Value.Task.ConfigureAwait(false); + + var resampler = new PlaybackAudioResampler( + _backend.SampleRate, + _device.SampleRate > 0 ? _device.SampleRate : _backend.SampleRate, + _device.ChannelCount > 0 ? _device.ChannelCount : 1); + + foreach (var segment in plan.Segments) + { + cancellationToken.ThrowIfCancellationRequested(); + var synthesized = await GenerateSegmentAsync(segment, cancellationToken).ConfigureAwait(false); + results.Add(synthesized); + PlaySegment(synthesized, resampler); + } + + await WaitForPlaybackDrainAsync(cancellationToken).ConfigureAwait(false); + } + finally + { + if (deviceStartRun is not null && !deviceStartRun.Value.Completion.IsCompleted) + { + // The device-start call was abandoned (its own abandon-aware task above already + // gave up waiting for it) and may still genuinely be running: calling + // _device.Stop() now would race a still-in-progress _device.Start(). Defer the + // stop to a background continuation that waits for the device-start call's + // genuine completion before ever touching the device, and publish that + // continuation as this operation's pending native completion so + // CancelAndAwaitOperationAsync (shared by StopAsync/DisposeAsync) still waits for + // the device to genuinely stop before returning, exactly as it already does for an + // abandoned Generate call. + var deviceStartCompletion = deviceStartRun.Value.Completion; + lock (_syncRoot) + { + _pendingNativeCompletion = StopDeviceAfterGenuineStartCompletionAsync(deviceStartCompletion); + } + } + else + { + try + { + _device.Stop(); + } + catch (Exception ex) + { + // Intentionally broad: stopping the device during teardown is best-effort + // against the native playback backend, and a stop fault must not mask the + // earlier outcome. + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"Failed to stop the playback device after synthesis: {ex.Message}"); + } + } + } + + return results; + } + + /// + /// Awaits an abandoned playback device-start call's genuine raw completion, then stops the + /// device - never running the two concurrently. + /// + /// + /// The device-start call's raw completion (see ), + /// already known to be incomplete (the call was abandoned) when this was scheduled. + /// + private async Task StopDeviceAfterGenuineStartCompletionAsync(Task deviceStartCompletion) + { + try + { + await deviceStartCompletion.ConfigureAwait(false); + } + catch + { + // Already observed/reported via GenerateAndOptionallyPlayAsync's own await of the + // abandon-aware task; this await exists purely to prove the device-start call has + // genuinely returned before Stop() is ever called on it. + } + + try + { + _device.Stop(); + } + catch (Exception ex) + { + // Intentionally broad: stopping the device during teardown is best-effort against the + // native playback backend, and a stop fault must not mask the earlier outcome. + _diagnostics.Report( + SpeechDiagnosticLevel.Error, + DiagnosticsCategory, + $"Failed to stop the playback device after synthesis: {ex.Message}"); + } + } + + /// + /// Synthesizes one segment, applying any speed/volume parameter overrides it carries, + /// resolving the session-level speaker id, or producing pure silence for an empty-text + /// pause segment without calling the backend at all. The backend call itself runs on a + /// so a non-cooperative native call is bounded by the + /// cooperative-cancel-then-abandon policy rather than awaited indefinitely. + /// + private async Task GenerateSegmentAsync(SpeechSegment segment, CancellationToken cancellationToken) + { + if (segment.Text.Length == 0) + { + return new SynthesizedSpeech( + [], + _backend.SampleRate, + TimeSpan.FromMilliseconds(segment.PreSilenceMs), + TimeSpan.FromMilliseconds(segment.PostSilenceMs)); + } + + var (speedRatio, volumeRatio) = ResolveOverrideRatios(segment.ParameterOverrides); + var speakerId = _model.ResolveSpeakerId(_parameterValues); + + var run = DedicatedWorker.Start( + _ => _backend.Generate(segment.Text, speedRatio, speakerId), + cancellationToken, + _diagnostics, + DiagnosticsCategory); + + lock (_syncRoot) + { + _pendingNativeCompletion = run.Completion; + } + + var generated = await run.Task.ConfigureAwait(false); + + var samples = generated.Samples; + if (Math.Abs(volumeRatio - 1.0) > double.Epsilon) + { + samples = ApplyVolume(samples, volumeRatio); + } + + return new SynthesizedSpeech( + samples, + generated.SampleRate, + TimeSpan.FromMilliseconds(segment.PreSilenceMs), + TimeSpan.FromMilliseconds(segment.PostSilenceMs)); + } + + /// + /// Resolves a segment's parameter overrides into a backend speed ratio and a post-hoc + /// volume (amplitude) ratio, by matching each overridden parameter id against + /// . + /// + private (float SpeedRatio, double VolumeRatio) ResolveOverrideRatios(IReadOnlyDictionary? overrides) + { + var speedRatio = 1.0f; + var volumeRatio = 1.0; + if (overrides is null || overrides.Count == 0) + { + return (speedRatio, volumeRatio); + } + + foreach (var (parameterId, overriddenValue) in overrides) + { + var parameter = _model.Parameters + .OfType() + .FirstOrDefault(candidate => candidate.Id == parameterId); + if (parameter is null || Math.Abs(parameter.Default) <= double.Epsilon || overriddenValue is not double doubleValue) + { + continue; + } + + var ratio = doubleValue / parameter.Default; + if (SpeechParameterConventions.IsSpeedParameter(parameterId)) + { + speedRatio = (float)ratio; + } + else if (SpeechParameterConventions.IsVolumeParameter(parameterId)) + { + volumeRatio = ratio; + } + } + + return (speedRatio, volumeRatio); + } + + /// + /// Scales sample amplitude by a ratio, clamping to [-1.0, 1.0] so a boosted segment + /// never clips into an invalid sample value. + /// + private static float[] ApplyVolume(float[] samples, double ratio) + { + var scaled = new float[samples.Length]; + var ratioF = (float)ratio; + TensorPrimitives.Multiply(samples, ratioF, scaled); + TensorPrimitives.Clamp(scaled, -1.0f, 1.0f, scaled); + return scaled; + } + + /// + /// Polls until every sample written + /// during this operation has genuinely been rendered by the playback hardware, then + /// applies a small additional tail wait for residual host buffering. + /// + private async Task WaitForPlaybackDrainAsync(CancellationToken cancellationToken) + { + while (_device.PendingSampleCount > 0) + { + await Task.Delay(DrainPollInterval, cancellationToken).ConfigureAwait(false); + } + + await Task.Delay(DrainTailMargin, cancellationToken).ConfigureAwait(false); + } + + /// + /// Writes one segment's pre-silence, resampled audio, and post-silence to the playback + /// device in order. + /// + private void PlaySegment(SynthesizedSpeech segment, PlaybackAudioResampler resampler) + { + WriteSilence(segment.PreSilence); + + if (segment.Samples.Count > 0) + { + var interleaved = resampler.Convert(segment.Samples is float[] array ? array : [.. segment.Samples]); + if (interleaved.Length > 0) + { + _device.Write(interleaved); + } + } + + WriteSilence(segment.PostSilence); + } + + /// + /// Writes a block of zero-valued samples to the playback device representing a duration + /// of real silence, sized for the device's resolved sample rate and channel count. + /// + private void WriteSilence(TimeSpan duration) + { + if (duration <= TimeSpan.Zero) + { + return; + } + + var sampleRate = _device.SampleRate > 0 ? _device.SampleRate : _backend.SampleRate; + var channelCount = _device.ChannelCount > 0 ? _device.ChannelCount : 1; + var frameCount = (int)(duration.TotalSeconds * sampleRate); + if (frameCount <= 0) + { + return; + } + + _device.Write(new float[frameCount * channelCount]); + } + + /// + /// Updates and raises , with handler + /// exceptions caught and routed to diagnostics rather than propagated, mirroring this + /// library's other event-exception-isolation conventions. + /// + private void TransitionTo(SynthesisSessionState newState) + { + SynthesisSessionState previous; + lock (_syncRoot) + { + previous = _state; + _state = newState; + } + + if (previous == newState) + { + return; + } + + try + { + StateChanged?.Invoke(this, new SessionStateChangedEventArgs(previous, newState)); + } + catch (Exception ex) + { + // Intentionally broad: a host's StateChanged handler must never be able to destabilize + // this session's own lifecycle. + _diagnostics.Report( + SpeechDiagnosticLevel.Warning, + DiagnosticsCategory, + $"A StateChanged handler threw: {ex.Message}"); + } + } + + /// + /// + /// Every concurrent caller shares the same single-flight disposal task rather than a + /// second call returning the instant the first merely begins: both wait for the exact + /// same cancellation, operation completion (including the raw native-call completion - see + /// - so an abandoned + /// thread cannot still be touching the shared backend once this returns), state + /// transitions, and engine-lease release to complete. + /// + public ValueTask DisposeAsync() + { + lock (_syncRoot) + { + _disposeTask ??= DisposeCoreAsync(); + return new ValueTask(_disposeTask); + } + } + + /// + /// Cancels and awaits any in-flight operation to genuine completion (including the raw + /// native-call completion), then transitions through + /// to and releases the engine's exclusivity + /// lease exactly once - only after the operation has settled, so the lease is never + /// released while this session's own call into the backend is still demonstrably in + /// flight, whether or not it was abandoned. Started at most once; see . + /// + private async Task DisposeCoreAsync() + { + CancellationTokenSource? operationCancellation; + Task? operationTask; + lock (_syncRoot) + { + operationCancellation = _operationCancellation; + operationTask = _operationTask; + } + + await CancelAndAwaitOperationAsync(operationCancellation, operationTask).ConfigureAwait(false); + + TransitionTo(SynthesisSessionState.Disposing); + TransitionTo(SynthesisSessionState.Disposed); + + lock (_syncRoot) + { + if (_leaseReleased) + { + return; + } + + _leaseReleased = true; + } + + _releaseLease(); + } +} diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SpeechParameterConventions.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SpeechParameterConventions.cs index 75974b5..d0f1682 100644 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/SpeechParameterConventions.cs +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/SpeechParameterConventions.cs @@ -2,7 +2,7 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// /// The conservative, built-in naming conventions -/// and both use to recognize which of a model's own +/// and both use to recognize which of a model's own /// declared numeric parameters (if any) conventionally controls speaking rate or output /// volume. /// diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SpeechSynthesizerFactory.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SpeechSynthesizerFactory.cs index fef34bd..bdad840 100644 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/SpeechSynthesizerFactory.cs +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/SpeechSynthesizerFactory.cs @@ -5,30 +5,34 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// -/// Composition entry point for obtaining an for one -/// installed synthesis model and one playback device. +/// Composition entry point for obtaining an for one +/// installed synthesis model. /// /// -/// Hosts call , -/// , -/// or -/// rather than constructing a synthesizer directly, so all of the "can this machine actually -/// speak right now?" logic lives in one reviewable place. Per this library's "nothing -/// throws at composition" decision this method never throws for an ordinary machine state - a -/// model that is not installed, a model whose role is not synthesis, a machine with no -/// playback device, and a machine missing the sherpa-onnx native runtime all return -/// and are reported through the -/// diagnostics sink. Only a null argument, which is a programming error rather than a machine -/// state, throws. +/// Hosts call one of this factory's three LoadAsync overloads - taking an +/// installedModelDirectory path directly +/// (), +/// a +/// (), +/// or a +/// () +/// - rather than constructing an engine directly, so all of the "can this machine actually speak +/// right now?" logic lives in one reviewable place. Per this library's "nothing throws at +/// composition" decision this method never throws for an ordinary machine state - a model +/// that is not installed, a model whose role is not synthesis, and a machine missing the +/// sherpa-onnx native runtime all return +/// and are reported through the diagnostics sink. A null argument or an invalid enum value is a +/// programming error rather than a machine state, and still throws; so does cancelling the +/// supplied cancellation token via . /// -/// Nothing in this type's public signature names a sherpa-onnx type, keeping this library's -/// "engine backend stays swappable at the public API surface" promise intact. +/// This factory takes no playback device: an engine loaded here can create many sessions over +/// its life, each bound to its own device, via +/// . /// /// -/// See 's own remarks for guidance on reusing one synthesizer -/// across many sessions for low-latency, repeated -/// synthesis, since a call to this factory is the expensive step a host typically wants to -/// make only once. +/// The blocking native model load itself runs on a rather than +/// the calling thread, so awaiting any LoadAsync overload never blocks a caller's +/// synchronization context. /// /// public static class SpeechSynthesizerFactory @@ -37,8 +41,7 @@ public static class SpeechSynthesizerFactory private const string DiagnosticsCategory = "SynthesisSubsystem"; /// - /// Creates a speech synthesizer for an installed synthesis model, playing back through the - /// supplied playback device. + /// Loads a speech synthesizer engine for an installed synthesis model. /// /// /// The synthesis model to load. Must not be null and must already be installed. @@ -48,11 +51,6 @@ public static class SpeechSynthesizerFactory /// SpeechModelStore.GetCurrentDirectory(model.Id). A path that is null, empty, or /// does not exist is treated as "model not installed", not as an error. /// - /// - /// The playback device to play synthesized audio through, as returned by - /// AudioDeviceFactory.CreatePlaybackDevice(...). Must not be null; a device - /// reporting IsAvailable == false is treated as "no speakers", not as an error. - /// /// /// The sink to report structural composition, lifecycle, and fault events to, or /// to use . @@ -60,54 +58,41 @@ public static class SpeechSynthesizerFactory /// /// An optional session-level parameter value bag (for example a selected voice, built /// from the model's declared ), forwarded to the - /// returned synthesizer and re-resolved via - /// once per synthesized segment, or to use every model's own - /// default voice/speaker. + /// returned engine and re-resolved via once + /// per synthesized segment, or to use every model's own default + /// voice/speaker. /// + /// A token to cancel the load. Must not already be cancelled. /// - /// A real chunked/streaming synthesizer when the model is installed, its role is - /// synthesis, the playback device is available, and the engine loaded successfully; - /// otherwise . + /// A real engine when the model is installed, its role is synthesis, and the backend loaded + /// successfully; otherwise . /// /// - /// Thrown when or is - /// . + /// Thrown when is . /// /// /// Thrown when contains a value for a parameter /// declares that is invalid for it (wrong type, out of range, /// non-integral for an integer-only parameter, or an unrecognized choice/boolean value). /// An unrecognized parameter id is not an error - it is reported at - /// and silently ignored, preserving - /// this library's cross-model settings-dictionary-reuse contract. + /// and silently ignored, preserving this + /// library's cross-model settings-dictionary-reuse contract. /// - /// - /// Loads the model into native memory when it succeeds, so the returned synthesizer owns - /// unmanaged resources and must be disposed. Reports every fallback decision through the - /// diagnostics sink as a structural fact, never including synthesized text. - /// - /// Recommended composition pattern: create the playback device first with - /// new AudioDeviceFactory().CreatePlaybackDevice(selection, model.PreferredAudioFormat), - /// then pass that device here. This hint is best-effort only: the authoritative playback - /// rate remains the constructed engine's , so - /// remains the guaranteed fallback whenever the - /// loaded engine's actual output rate differs from the preferred hint. - /// - /// - public static ISpeechSynthesizer Create( + /// Thrown when is cancelled. + public static Task LoadAsync( ISynthesisModel model, string installedModelDirectory, - IAudioPlaybackDevice playbackDevice, ISpeechDiagnostics? diagnostics = null, - IReadOnlyDictionary? parameterValues = null) + IReadOnlyDictionary? parameterValues = null, + CancellationToken cancellationToken = default) { - return Create(model, installedModelDirectory, playbackDevice, diagnostics, new SherpaOnnxSynthesisEngineFactory(), parameterValues); + return LoadAsync(model, installedModelDirectory, diagnostics, new SherpaOnnxSynthesisEngineFactory(), parameterValues, cancellationToken); } /// - /// Creates a speech synthesizer for an installed synthesis model, resolving the model's - /// installed-files directory from the supplied model store rather than requiring the caller - /// to know anything about the store's on-disk directory layout. + /// Loads a speech synthesizer engine for an installed synthesis model, resolving the + /// model's installed-files directory from the supplied model store rather than requiring + /// the caller to know anything about the store's on-disk directory layout. /// /// /// The synthesis model to load. Must not be null and must already be installed. @@ -116,58 +101,51 @@ public static ISpeechSynthesizer Create( /// The store to resolve 's installed-files directory from, via /// . Must not be null. /// - /// - /// The playback device to play synthesized audio through, as returned by - /// AudioDeviceFactory.CreatePlaybackDevice(...). Must not be null; a device - /// reporting IsAvailable == false is treated as "no speakers", not as an error. - /// /// /// The sink to report structural composition, lifecycle, and fault events to, or /// to use . /// /// - /// An optional session-level parameter value bag, forwarded to the returned synthesizer, or + /// An optional session-level parameter value bag, forwarded to the returned engine, or /// to use every model's own default voice/speaker. /// + /// A token to cancel the load. Must not already be cancelled. /// - /// A real chunked/streaming synthesizer when the model is installed, its role is synthesis, - /// the playback device is available, and the engine loaded successfully; otherwise - /// . + /// A real engine when the model is installed, its role is synthesis, and the backend loaded + /// successfully; otherwise . /// /// - /// Thrown when , , or - /// is . + /// Thrown when or is . /// /// /// Thrown when contains an invalid value for a /// parameter declares. See the - /// + /// /// overload's matching remark for the full behavior. /// + /// Thrown when is cancelled. /// /// Equivalent to calling - /// - /// with store.GetCurrentDirectory(model.Id) as the installed-model-directory argument, - /// so callers never need to know 's on-disk directory-naming - /// scheme just to compose a synthesizer. + /// + /// with store.GetCurrentDirectory(model.Id) as the installed-model-directory argument. /// - public static ISpeechSynthesizer Create( + public static Task LoadAsync( ISynthesisModel model, SpeechModelStore store, - IAudioPlaybackDevice playbackDevice, ISpeechDiagnostics? diagnostics = null, - IReadOnlyDictionary? parameterValues = null) + IReadOnlyDictionary? parameterValues = null, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(model); ArgumentNullException.ThrowIfNull(store); - return Create(model, store.GetCurrentDirectory(model.Id), playbackDevice, diagnostics, parameterValues); + return LoadAsync(model, store.GetCurrentDirectory(model.Id), diagnostics, parameterValues, cancellationToken); } /// - /// Creates a speech synthesizer for an installed synthesis model, resolving the model's - /// installed-files directory from the supplied catalog's own store rather than requiring - /// the caller to construct a separate . + /// Loads a speech synthesizer engine for an installed synthesis model, resolving the + /// model's installed-files directory from the supplied catalog's own store rather than + /// requiring the caller to construct a separate . /// /// /// The synthesis model to load. Must not be null and must already be installed. @@ -176,174 +154,168 @@ public static ISpeechSynthesizer Create( /// The catalog whose resolves /// 's installed-files directory. Must not be null. /// - /// - /// The playback device to play synthesized audio through, as returned by - /// AudioDeviceFactory.CreatePlaybackDevice(...). Must not be null; a device - /// reporting IsAvailable == false is treated as "no speakers", not as an error. - /// /// /// The sink to report structural composition, lifecycle, and fault events to, or /// to use . /// /// - /// An optional session-level parameter value bag, forwarded to the returned synthesizer, or + /// An optional session-level parameter value bag, forwarded to the returned engine, or /// to use every model's own default voice/speaker. /// + /// A token to cancel the load. Must not already be cancelled. /// - /// A real chunked/streaming synthesizer when the model is installed, its role is synthesis, - /// the playback device is available, and the engine loaded successfully; otherwise - /// . + /// A real engine when the model is installed, its role is synthesis, and the backend loaded + /// successfully; otherwise . /// /// - /// Thrown when , , or - /// is . + /// Thrown when or is . /// /// /// Thrown when contains an invalid value for a /// parameter declares. See the - /// + /// /// overload's matching remark for the full behavior. /// + /// Thrown when is cancelled. /// /// Equivalent to calling - /// - /// with catalog.Store as the store argument, so a host that already owns a - /// for enumeration and download can compose a synthesizer - /// through that same catalog instance, without constructing a second, potentially - /// divergent . + /// + /// with catalog.Store as the store argument. /// - public static ISpeechSynthesizer Create( + public static Task LoadAsync( ISynthesisModel model, SpeechModelCatalog catalog, - IAudioPlaybackDevice playbackDevice, ISpeechDiagnostics? diagnostics = null, - IReadOnlyDictionary? parameterValues = null) + IReadOnlyDictionary? parameterValues = null, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(model); ArgumentNullException.ThrowIfNull(catalog); - return Create(model, catalog.Store, playbackDevice, diagnostics, parameterValues); + return LoadAsync(model, catalog.Store, diagnostics, parameterValues, cancellationToken); } /// - /// Creates a speech synthesizer using an injected engine factory and a model store, for - /// tests that need a deterministic engine while still exercising store-based directory + /// Loads a speech synthesizer engine using an injected engine factory and a model store, + /// for tests that need a deterministic engine while still exercising store-based directory /// resolution. /// /// The synthesis model to load. Must not be null. /// The store to resolve the model's installed-files directory from. Must not be null. - /// The playback device to play through. Must not be null. /// The diagnostics sink, or for the null sink. - /// The engine factory to load the model through. Must not be null. + /// The backend factory to load the model through. Must not be null. /// - /// An optional session-level parameter value bag forwarded to the returned synthesizer, or + /// An optional session-level parameter value bag forwarded to the returned engine, or /// to use every model's own default voice/speaker. /// + /// A token to cancel the load. Must not already be cancelled. /// - /// A real chunked/streaming synthesizer, or - /// for any honest unavailable state. + /// A real engine, or for any + /// honest unavailable state. /// /// - /// Thrown when , , - /// , or is . + /// Thrown when , , or + /// is . /// /// /// Thrown when contains an invalid value for a /// parameter declares. /// - internal static ISpeechSynthesizer Create( + /// Thrown when is cancelled. + internal static Task LoadAsync( ISynthesisModel model, SpeechModelStore store, - IAudioPlaybackDevice playbackDevice, ISpeechDiagnostics? diagnostics, - ISynthesisEngineFactory engineFactory, - IReadOnlyDictionary? parameterValues = null) + ISynthesisBackendFactory backendFactory, + IReadOnlyDictionary? parameterValues = null, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(model); ArgumentNullException.ThrowIfNull(store); - return Create(model, store.GetCurrentDirectory(model.Id), playbackDevice, diagnostics, engineFactory, parameterValues); + return LoadAsync(model, store.GetCurrentDirectory(model.Id), diagnostics, backendFactory, parameterValues, cancellationToken); } /// - /// Creates a speech synthesizer using an injected engine factory and a model catalog, for - /// tests that need a deterministic engine while still exercising catalog-based directory - /// resolution. + /// Loads a speech synthesizer engine using an injected engine factory and a model catalog, + /// for tests that need a deterministic engine while still exercising catalog-based + /// directory resolution. /// /// The synthesis model to load. Must not be null. /// The catalog whose store resolves the model's installed-files directory. Must not be null. - /// The playback device to play through. Must not be null. /// The diagnostics sink, or for the null sink. - /// The engine factory to load the model through. Must not be null. + /// The backend factory to load the model through. Must not be null. /// - /// An optional session-level parameter value bag forwarded to the returned synthesizer, or + /// An optional session-level parameter value bag forwarded to the returned engine, or /// to use every model's own default voice/speaker. /// + /// A token to cancel the load. Must not already be cancelled. /// - /// A real chunked/streaming synthesizer, or - /// for any honest unavailable state. + /// A real engine, or for any + /// honest unavailable state. /// /// - /// Thrown when , , - /// , or is . + /// Thrown when , , or + /// is . /// /// /// Thrown when contains an invalid value for a /// parameter declares. /// - internal static ISpeechSynthesizer Create( + /// Thrown when is cancelled. + internal static Task LoadAsync( ISynthesisModel model, SpeechModelCatalog catalog, - IAudioPlaybackDevice playbackDevice, ISpeechDiagnostics? diagnostics, - ISynthesisEngineFactory engineFactory, - IReadOnlyDictionary? parameterValues = null) + ISynthesisBackendFactory backendFactory, + IReadOnlyDictionary? parameterValues = null, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(model); ArgumentNullException.ThrowIfNull(catalog); - return Create(model, catalog.Store, playbackDevice, diagnostics, engineFactory, parameterValues); + return LoadAsync(model, catalog.Store, diagnostics, backendFactory, parameterValues, cancellationToken); } /// - /// Creates a speech synthesizer using an injected engine factory, for tests that need a - /// deterministic engine with no model files and no native runtime. + /// Loads a speech synthesizer engine using an injected engine factory, for tests that need + /// a deterministic engine with no model files and no native runtime. /// /// The synthesis model to load. Must not be null. /// The model's installed-files directory. - /// The playback device to play through. Must not be null. /// The diagnostics sink, or for the null sink. - /// The engine factory to load the model through. Must not be null. + /// The backend factory to load the model through. Must not be null. /// - /// An optional session-level parameter value bag forwarded to the returned synthesizer, - /// or to use every model's own default voice/speaker. + /// An optional session-level parameter value bag forwarded to the returned engine, or + /// to use every model's own default voice/speaker. /// + /// A token to cancel the load. Must not already be cancelled. /// - /// A real chunked/streaming synthesizer, or - /// for any honest unavailable state. + /// A real engine, or for any + /// honest unavailable state. /// /// - /// Thrown when , , or - /// is . + /// Thrown when or is . /// /// /// Thrown when contains an invalid value for a /// parameter declares - wrong CLR type, out of range, a /// non-integral value for an integer-only parameter, or an unrecognized choice/boolean - /// value. An unrecognized parameter id is reported at - /// and silently ignored instead. + /// value. An unrecognized parameter id is reported at + /// and silently ignored instead. /// - internal static ISpeechSynthesizer Create( + /// Thrown when is cancelled. + internal static async Task LoadAsync( ISynthesisModel model, string installedModelDirectory, - IAudioPlaybackDevice playbackDevice, ISpeechDiagnostics? diagnostics, - ISynthesisEngineFactory engineFactory, - IReadOnlyDictionary? parameterValues = null) + ISynthesisBackendFactory backendFactory, + IReadOnlyDictionary? parameterValues = null, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(model); - ArgumentNullException.ThrowIfNull(playbackDevice); - ArgumentNullException.ThrowIfNull(engineFactory); + ArgumentNullException.ThrowIfNull(backendFactory); + cancellationToken.ThrowIfCancellationRequested(); var sink = diagnostics ?? NullSpeechDiagnostics.Instance; @@ -363,56 +335,83 @@ internal static ISpeechSynthesizer Create( SpeechDiagnosticLevel.Warning, DiagnosticsCategory, $"Speech synthesis is unavailable because model '{model.Id}' is not installed."); - return UnavailableSpeechSynthesizer.Instance; + return UnavailableSpeechSynthesizerEngine.Instance; } // Guard against a model whose declared role contradicts the synthesis interface it - // implements: loading it would build a synthesizer that could never produce audio. + // implements: loading it would build an engine that could never produce audio. if (model.Role != SpeechModelRole.Synthesis) { sink.Report( SpeechDiagnosticLevel.Warning, DiagnosticsCategory, $"Speech synthesis is unavailable because model '{model.Id}' does not declare the synthesis role."); - return UnavailableSpeechSynthesizer.Instance; - } - - // No speakers (or no working audio backend) is an ordinary machine state too; there is - // nowhere to play audio, so report it honestly rather than loading a model that could - // never be heard. - if (!playbackDevice.IsAvailable) - { - sink.Report( - SpeechDiagnosticLevel.Warning, - DiagnosticsCategory, - "Speech synthesis is unavailable because no audio playback device is available."); - return UnavailableSpeechSynthesizer.Instance; + return UnavailableSpeechSynthesizerEngine.Instance; } // Loading is the one step that touches the native runtime, so it is also the one step // that can fail for a missing org.k2fsa.sherpa.onnx.runtime.{RID} binary or unusable // model files. Both degrade exactly like a missing model rather than crashing start-up. - ISynthesisEngine engine; + // The native load is a blocking call, so it runs on a dedicated worker rather than + // blocking whichever thread is awaiting this method. + ISynthesisBackend backend; + var run = DedicatedWorker.Start( + _ => backendFactory.Create(model, installedModelDirectory), + cancellationToken, + sink, + DiagnosticsCategory); try { - engine = engineFactory.Create(model, installedModelDirectory); + backend = await run.Task.ConfigureAwait(false); + } + catch (OperationCanceledException) + { + // Abandoned: backendFactory.Create ignored cancellation past the worker's abandon + // timeout, so it may still return a successfully created native backend after this + // call has already faulted with cancellation. Nothing else observes that eventual + // backend, so dispose it here once it genuinely arrives rather than leaking native + // model resources on a canceled load. + _ = run.Completion.ContinueWith( + static t => + { + if (t.Status == TaskStatus.RanToCompletion) + { + t.Result.Dispose(); + } + }, + CancellationToken.None, + TaskContinuationOptions.ExecuteSynchronously, + TaskScheduler.Default); + throw; } catch (Exception ex) { // Intentionally broad: engine creation crosses the native runtime/model-file // boundary, and every load failure must degrade to the documented unavailable - // synthesizer rather than crash composition. + // engine rather than crash composition. sink.Report( SpeechDiagnosticLevel.Error, DiagnosticsCategory, $"Speech synthesis is unavailable because the engine for model '{model.Id}' could not be loaded: {ex.Message}"); - return UnavailableSpeechSynthesizer.Instance; + return UnavailableSpeechSynthesizerEngine.Instance; + } + + if (cancellationToken.IsCancellationRequested) + { + // Finding 31: the worker's delegate finished within the abandon grace period - so + // run.Task above returned normally with a genuinely created backend - but + // cancellation was still requested before that happened. The caller must not + // receive a loaded engine for a call it asked to cancel; dispose the backend that + // was created so this does not leak native model resources, then honor the + // cancellation rather than falling through to report success. + backend.Dispose(); + cancellationToken.ThrowIfCancellationRequested(); } sink.Report( SpeechDiagnosticLevel.Info, DiagnosticsCategory, - $"Composed a streaming speech synthesizer for model '{model.Id}'."); - return new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, model, parameterValues, sink); + $"Loaded a speech synthesizer engine for model '{model.Id}'."); + return new SherpaOnnxSpeechSynthesizerEngine(backend, model, parameterValues, sink); } } diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SpeechSynthesizerUnavailableException.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SpeechSynthesizerUnavailableException.cs index 123e018..a7fbbae 100644 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/SpeechSynthesizerUnavailableException.cs +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/SpeechSynthesizerUnavailableException.cs @@ -6,15 +6,28 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// use. /// /// -/// Per this library's "nothing throws at composition" decision, obtaining and holding an -/// never throws - -/// returns for every ordinary -/// "cannot synthesize on this machine right now" state (model not installed, native runtime -/// absent, no playback device). This exception is reserved for the two genuine error cases: -/// a caller that ignored IsAvailable == false and invoked an operational member -/// anyway, and a synthesizer whose underlying engine or playback device failed when actually -/// used. It mirrors -/// so both subsystems signal misuse the same way. +/// Per this library's "nothing throws at composition" decision, loading an +/// never throws for an ordinary machine state - +/// returns +/// for every ordinary "cannot +/// synthesize on this machine right now" state (model not installed, native runtime absent). +/// and +/// report themselves honestly through IsAvailable == false rather than throwing this +/// exception on an operational call. +/// +/// This type is reserved for the same first-use-after-"unavailable" misuse case described +/// above - this exception is not what a real, available +/// throws for its own runtime failures: a playback +/// device failure (for example the speakers disconnecting mid-call) propagates as +/// , not this type, and a call +/// made on a session that has already reached its terminal +/// state throws +/// , not this type. Consumers that want to +/// catch a real session's runtime failures should catch those two types instead of this one; +/// it mirrors 's +/// equivalent "unavailable" boundary so both subsystems signal that specific misuse the same +/// way. +/// /// public sealed class SpeechSynthesizerUnavailableException : Exception { diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisEngineBusyException.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisEngineBusyException.cs new file mode 100644 index 0000000..546f99e --- /dev/null +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisEngineBusyException.cs @@ -0,0 +1,52 @@ +namespace DemaConsulting.Speech.SynthesisSubsystem; + +/// +/// Thrown when is called while this +/// engine's exclusivity lease is already held by another session. +/// +/// +/// Per the engine-exclusivity design decision, at most one may +/// be leased from a given at a time, with the lease held +/// for the session's entire life through +/// completion. A concurrent call +/// while that lease is held fails fast with this exception rather than queueing or awaiting +/// release, since waiting would make this call's latency depend on an unrelated session's +/// teardown with no caller-visible way to bound that wait. A caller that needs to wait should +/// implement its own retry/backoff. +/// +public sealed class SynthesisEngineBusyException : Exception +{ + /// + /// Initializes a new instance of the class with + /// a default message. + /// + /// + /// Provided for standard .NET exception-type conformance; callers should prefer + /// to describe which engine was busy. + /// + public SynthesisEngineBusyException() + : base("The synthesis engine's exclusivity lease is already held by another session.") + { + } + + /// + /// Initializes a new instance of the class with + /// a message describing which engine was busy. + /// + /// A message describing the busy engine. + public SynthesisEngineBusyException(string message) + : base(message) + { + } + + /// + /// Initializes a new instance of the class with + /// a message and an inner exception describing the underlying cause. + /// + /// A message describing the busy engine. + /// The exception that is the cause of this exception. + public SynthesisEngineBusyException(string message, Exception innerException) + : base(message, innerException) + { + } +} diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisSessionFaultedException.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisSessionFaultedException.cs new file mode 100644 index 0000000..c7f3933 --- /dev/null +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisSessionFaultedException.cs @@ -0,0 +1,57 @@ +namespace DemaConsulting.Speech.SynthesisSubsystem; + +/// +/// Thrown when an operation is attempted on, or surfaced from, an +/// that has transitioned to . +/// +/// +/// A session faults when an in-flight or +/// operation fails for a reason other than its +/// own promptly-honored cancellation - including a native call that did not stop cooperatively +/// within the dedicated worker's abandon timeout and was detached, whether or not it was ever +/// requested to stop via or +/// : either way, the native call may still be +/// running against the shared backend, so the session cannot safely be treated as a clean, +/// reusable stop. +/// Once faulted, the session is terminal: every subsequent +/// / +/// call throws this exception (carrying the original fault as its inner exception) rather than +/// attempting to run; callers must dispose the faulted session and create a new one. +/// +public sealed class SynthesisSessionFaultedException : Exception +{ + /// + /// Initializes a new instance of the class + /// with a default message. + /// + /// + /// Provided for standard .NET exception-type conformance; callers should prefer + /// to carry the original + /// fault cause. + /// + public SynthesisSessionFaultedException() + : base("The synthesis session has faulted and can no longer perform operations.") + { + } + + /// + /// Initializes a new instance of the class + /// with a message describing the faulted session. + /// + /// A message describing the faulted session. + public SynthesisSessionFaultedException(string message) + : base(message) + { + } + + /// + /// Initializes a new instance of the class + /// with a message and the original exception that caused the session to fault. + /// + /// A message describing the faulted session. + /// The exception that caused the session to fault. + public SynthesisSessionFaultedException(string message, Exception innerException) + : base(message, innerException) + { + } +} diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisSessionState.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisSessionState.cs new file mode 100644 index 0000000..7518046 --- /dev/null +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesisSessionState.cs @@ -0,0 +1,52 @@ +namespace DemaConsulting.Speech.SynthesisSubsystem; + +/// +/// The lifecycle states an passes through. +/// +/// +/// Unlike RecognitionSubsystem.RecognitionSessionState, which describes a single +/// continuous capture window, , , and +/// here denote one discrete, in-flight +/// or +/// operation rather than a continuous stream: a session has no analogue of recognition's +/// continuous start/stop capture window, so it returns to after each +/// operation completes and is ready to accept another +/// or call - this repeatability is what lets a +/// host construct one session per model/device combination and reuse it across many calls for +/// low-latency, repeated synthesis, rather than recreating it per call. +/// +/// is terminal: once reached (an unrequested abandonment of a native +/// call, or any other non-cancellation failure), the session can no longer perform operations +/// and must be disposed and replaced. +/// +/// +public enum SynthesisSessionState +{ + /// The session has been created but has not yet performed any operation. + Created, + + /// A SpeakAsync/SynthesizeAsync call has begun but audio has not yet started generating. + Starting, + + /// A SpeakAsync/SynthesizeAsync call is actively generating (and, for SpeakAsync, playing) audio. + Running, + + /// The in-flight operation is winding down (draining playback or completing cancellation). + Stopping, + + /// The most recent operation has completed; the session is ready to accept another call. + Stopped, + + /// has begun but has not yet completed. + Disposing, + + /// The session has been fully disposed and can no longer be used. + Disposed, + + /// + /// An in-flight operation failed for a reason other than its own requested cancellation + /// (including an unrequested native-call abandonment). Terminal: the session must be + /// disposed and replaced. + /// + Faulted +} diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesizedSpeech.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesizedSpeech.cs index ae7fe1a..a405b74 100644 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesizedSpeech.cs +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/SynthesizedSpeech.cs @@ -1,12 +1,12 @@ namespace DemaConsulting.Speech.SynthesisSubsystem; /// -/// One segment of synthesized audio produced by , -/// ready to be played back in order by . +/// One segment of synthesized audio produced during a +/// or call, ready to be played back in order. /// /// /// This is the one Layer-2/engine output type named directly in -/// 's signature. It carries no sherpa-onnx +/// 's signature. It carries no sherpa-onnx /// type, per this library's "engine backend stays swappable at the public API surface" /// decision. The record is immutable and safe to share across threads. Validation happens /// eagerly in the constructor so an invalid instance is rejected the moment it is created. diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSpeechSynthesizer.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSpeechSynthesizer.cs deleted file mode 100644 index 1a0d2ef..0000000 --- a/src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSpeechSynthesizer.cs +++ /dev/null @@ -1,90 +0,0 @@ -namespace DemaConsulting.Speech.SynthesisSubsystem; - -/// -/// Honest fallback used when no real synthesis engine can be -/// composed, reporting as rather than -/// letting a caller build against a synthesizer that cannot function. -/// -/// -/// Per this library's "nothing throws at composition" decision, obtaining and holding this -/// instance never throws: a missing model, an absent native runtime, and a machine with no -/// speakers are ordinary machine states at application start-up, not programming errors. -/// Only the operational members (, -/// , , ) throw -/// , and only when actually invoked - a -/// caller that checks first, as documented, never triggers them. -/// The type is stateless and holds no resources, so the shared is safe -/// for concurrent use by any number of callers and is a no-op that -/// never invalidates it. This mirrors -/// exactly. -/// -public sealed class UnavailableSpeechSynthesizer : ISpeechSynthesizer -{ - /// - /// Prevents external construction; callers use the shared instead - /// since the type carries no state and multiple instances would provide no value. - /// - private UnavailableSpeechSynthesizer() - { - } - - /// - /// Gets the single shared unavailable speech synthesizer. - /// - public static UnavailableSpeechSynthesizer Instance { get; } = new(); - - /// - public bool IsAvailable => false; - - /// - /// - /// Always thrown; this synthesizer has no real engine to synthesize with. - /// - public IAsyncEnumerable SynthesizeStreamAsync(string text, CancellationToken cancellationToken = default) - { - throw new SpeechSynthesizerUnavailableException( - "Cannot synthesize speech: no speech synthesizer is available."); - } - - /// - /// - /// Always thrown; this synthesizer has no real playback device to play through. - /// - public Task PlayStreamAsync(IAsyncEnumerable stream, CancellationToken cancellationToken = default) - { - throw new SpeechSynthesizerUnavailableException( - "Cannot play synthesized speech: no speech synthesizer is available."); - } - - /// - /// - /// Always thrown; this synthesizer has no real engine or playback device to speak with. - /// - public Task SpeakAsync(string text, CancellationToken cancellationToken = default) - { - throw new SpeechSynthesizerUnavailableException( - "Cannot speak: no speech synthesizer is available."); - } - - /// - /// - /// Always thrown; this synthesizer has no in-flight session to stop. - /// - public void Stop() - { - throw new SpeechSynthesizerUnavailableException( - "Cannot stop: no speech synthesizer is available."); - } - - /// Releases resources held by this synthesizer. - /// - /// A no-op: this synthesizer owns no engine, thread, or native resource. Disposal must - /// not throw or invalidate , because a host that wraps its - /// synthesizer in a using block gets the shared instance here and may dispose it - /// many times. - /// - public void Dispose() - { - // Intentionally empty - see the remarks above. - } -} diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSpeechSynthesizerEngine.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSpeechSynthesizerEngine.cs new file mode 100644 index 0000000..eb2750f --- /dev/null +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSpeechSynthesizerEngine.cs @@ -0,0 +1,82 @@ +using DemaConsulting.Speech.AudioSubsystem; + +namespace DemaConsulting.Speech.SynthesisSubsystem; + +/// +/// Honest fallback used when no real synthesis engine +/// can be composed, reporting as rather than +/// letting a caller build against an engine that cannot function. +/// +/// +/// Per this library's "nothing throws at composition" decision, obtaining and holding this +/// instance never throws: a missing model, an absent native runtime, and a model whose role +/// does not declare synthesis are ordinary machine states at application start-up, not +/// programming errors. always succeeds, returning +/// , since binding a device to an already +/// unavailable engine is itself an ordinary (if useless) composition, not an error; only the +/// session's own operational members throw. The type is stateless and holds no resources, so +/// the shared is safe for concurrent use by any number of callers and +/// is a no-op that never invalidates it. +/// +public sealed class UnavailableSpeechSynthesizerEngine : ISpeechSynthesizerEngine +{ + /// + /// Prevents external construction; callers use the shared instead + /// since the type carries no state and multiple instances would provide no value. + /// + private UnavailableSpeechSynthesizerEngine() + { + } + + /// Gets the single shared unavailable speech synthesizer engine. + public static UnavailableSpeechSynthesizerEngine Instance { get; } = new(); + + /// + public bool IsAvailable => false; + + /// + /// + /// Always succeeds, returning ; only the + /// returned session's operational members throw. + /// + public Task CreateSessionAsync(IAudioPlaybackDevice device, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(device); + cancellationToken.ThrowIfCancellationRequested(); + + return Task.FromResult(UnavailableSynthesisSession.Instance); + } + + /// + /// + /// Always thrown; this engine has no real model to speak with. + /// + public async Task SpeakAsync(IAudioPlaybackDevice device, string text, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(device); + ArgumentNullException.ThrowIfNull(text); + + await using var session = await CreateSessionAsync(device, cancellationToken).ConfigureAwait(false); + await session.SpeakAsync(text, cancellationToken).ConfigureAwait(false); + } + + /// + /// + /// Always thrown; this engine has no real model to synthesize with. + /// + public async Task> SynthesizeAsync(string text, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(text); + + await using var session = await CreateSessionAsync(UnavailableAudioPlaybackDevice.Instance, cancellationToken) + .ConfigureAwait(false); + return await session.SynthesizeAsync(text, cancellationToken).ConfigureAwait(false); + } + + /// + /// + /// A no-op: this engine owns no model, thread, or native resource. Disposal must not throw + /// or invalidate . + /// + public ValueTask DisposeAsync() => ValueTask.CompletedTask; +} diff --git a/src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSynthesisSession.cs b/src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSynthesisSession.cs new file mode 100644 index 0000000..a539975 --- /dev/null +++ b/src/DemaConsulting.Speech/SynthesisSubsystem/UnavailableSynthesisSession.cs @@ -0,0 +1,80 @@ +namespace DemaConsulting.Speech.SynthesisSubsystem; + +/// +/// Honest fallback used when no real synthesis engine could be +/// loaded, reporting as rather than letting +/// a caller build against a session that cannot function. +/// +/// +/// The type is stateless and holds no resources, so the shared is safe +/// for concurrent use by any number of callers, and is a no-op that +/// never invalidates it. This mirrors and +/// RecognitionSubsystem.UnavailableRecognitionSession. +/// +public sealed class UnavailableSynthesisSession : ISynthesisSession +{ + /// + /// Prevents external construction; callers use the shared instead + /// since the type carries no state and multiple instances would provide no value. + /// + private UnavailableSynthesisSession() + { + } + + /// Gets the single shared unavailable synthesis session. + public static UnavailableSynthesisSession Instance { get; } = new(); + + /// + public bool IsAvailable => false; + + /// + /// + /// Always : this session never performs an + /// operation and so never transitions, and is never itself disposed away (the shared + /// remains reusable across many callers). + /// + public SynthesisSessionState State => SynthesisSessionState.Created; + + /// + /// Never raised: this session never transitions state. +#pragma warning disable S108 // Intentionally empty: this session never transitions state, so no handler is ever invoked and none needs to be retained. + public event EventHandler? StateChanged + { + add { } + remove { } + } +#pragma warning restore S108 + + /// + /// + /// Always thrown; this session has no real engine or playback device to speak with. + /// + public Task SpeakAsync(string text, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(text); + throw new SpeechSynthesizerUnavailableException( + "Cannot speak: no synthesis session is available."); + } + + /// + /// + /// Always thrown; this session has no real engine to synthesize with. + /// + public Task> SynthesizeAsync(string text, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(text); + throw new SpeechSynthesizerUnavailableException( + "Cannot synthesize speech: no synthesis session is available."); + } + + /// + /// A no-op: this session has no in-flight operation to stop. + public Task StopAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + /// + /// + /// A no-op: this session owns no engine, thread, or native resource. Disposal must not + /// throw or invalidate . + /// + public ValueTask DisposeAsync() => ValueTask.CompletedTask; +} diff --git a/test/DemaConsulting.Speech.Cli.Tests/Commands/ConversationCommandSubsystem/AskCommandTests.cs b/test/DemaConsulting.Speech.Cli.Tests/Commands/ConversationCommandSubsystem/AskCommandTests.cs index 3fcab72..eedc32a 100644 --- a/test/DemaConsulting.Speech.Cli.Tests/Commands/ConversationCommandSubsystem/AskCommandTests.cs +++ b/test/DemaConsulting.Speech.Cli.Tests/Commands/ConversationCommandSubsystem/AskCommandTests.cs @@ -28,13 +28,16 @@ using DemaConsulting.Speech.Cli.Tests.Commands.RecognitionCommandSubsystem; using DemaConsulting.Speech.Cli.Tests.Commands.SynthesisCommandSubsystem; using DemaConsulting.Speech.ModelManagementSubsystem; +using DemaConsulting.Speech.RecognitionSubsystem; +using DemaConsulting.Speech.SynthesisSubsystem; namespace DemaConsulting.Speech.Cli.Tests.Commands.ConversationCommandSubsystem; /// /// Unit tests for , using , -/// , , and fake audio -/// device probes/sources so every scenario runs deterministically with no real catalog, +/// /, +/// /, and fake +/// audio device probes/sources so every scenario runs deterministically with no real catalog, /// network access, native engine, or PortAudio hardware. Both the playback-side and /// capture-side device resolution reuse dedicated CLI-owned seams - /// (mirroring @@ -75,6 +78,34 @@ private static FakePlaybackDeviceSource CreatePlaybackSource() => private static FakeCaptureDeviceSource CreateCaptureSource() => new(new FakeAudioCaptureDeviceProbe([CaptureDevice])); + /// + /// Creates a wrapped in a , + /// and wires the catalog's + /// to return the engine. + /// + private static (FakeSynthesisSession Session, FakeSpeechSynthesizerEngine Engine) WireSynthesizer( + FakeCliModelCatalog catalog) + { + var session = new FakeSynthesisSession(); + var engine = new FakeSpeechSynthesizerEngine(session); + catalog.CreateSynthesizerEngineOverride = (_, _, _) => Task.FromResult(engine); + return (session, engine); + } + + /// + /// Creates a wrapped in a , + /// and wires the catalog's + /// to return the engine. + /// + private static (FakeRecognitionSession Session, FakeSpeechRecognizerEngine Engine) WireRecognizer( + FakeCliModelCatalog catalog) + { + var session = new FakeRecognitionSession(); + var engine = new FakeSpeechRecognizerEngine(session); + catalog.CreateRecognizerEngineOverride = (_, _, _) => Task.FromResult(engine); + return (session, engine); + } + /// /// Polls until it returns or /// elapses, for use asserting on state that @@ -347,21 +378,20 @@ public void AskCommand_Run_UnknownPlaybackDevice_ThrowsArgumentException() /// /// Test that no available playback device throws InvalidOperationException from Phase 1 /// (before Phase 2's SpeakPromptAsync call ever awaits anything), and that the - /// concurrently pre-warmed recognizer - constructed on a background task that may still be - /// racing with (or may have already completed ahead of) that synchronous throw - is - /// nonetheless eventually disposed rather than leaked or left as an unobserved faulted - /// task: this proves RunAsync observes and cleans up the pre-warm task on every - /// exit path, not only the wasCanceled path. Disposal is now driven by - /// DisposePrewarmedRecognizerAsync's fire-and-forget background continuation rather - /// than completing synchronously before Run returns, so this polls for it instead - /// of asserting immediately. + /// concurrently pre-warmed recognizer engine/session - constructed on a background task + /// that may still be racing with (or may have already completed ahead of) that + /// synchronous throw - is nonetheless eventually disposed rather than leaked or left as + /// an unobserved faulted task: this proves RunAsync observes and cleans up the + /// pre-warm task on every exit path, not only the wasCanceled path. Disposal is + /// now driven by DisposePrewarmedRecognizerAsync's fire-and-forget background + /// continuation rather than completing synchronously before Run returns, so this + /// polls for it instead of asserting immediately. /// [Fact] public async Task AskCommand_Run_NoPlaybackDeviceAvailable_ThrowsInvalidOperationException() { var catalog = CreateCatalogWithModels(); - var recognizer = new FakeSpeechRecognizer(); - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; + var (recognizerSession, recognizerEngine) = WireRecognizer(catalog); var deviceSource = new FakePlaybackDeviceSource(new FakeAudioPlaybackDeviceProbe()); using var context = Context.Create( ["ask", "--tts-model", "tts-model-1", "--stt-model", "stt-model-1", "--text", "hi"]); @@ -369,12 +399,14 @@ public async Task AskCommand_Run_NoPlaybackDeviceAvailable_ThrowsInvalidOperatio Assert.Throws( () => AskCommand.Run(context, catalog, deviceSource, CreateCaptureSource())); - // The concurrently pre-warmed recognizer must eventually be disposed by the background - // continuation, even though the exception that ended the call was thrown from Phase 1's - // SpeakPromptAsync rather than from the wasCanceled path. + // The concurrently pre-warmed engine/session must eventually be disposed by the + // background continuation, even though the exception that ended the call was thrown from + // Phase 1's SpeakPromptAsync rather than from the wasCanceled path. Assert.True( - await WaitForConditionAsync(() => recognizer.DisposeCallCount == 1, TimeSpan.FromSeconds(5)), - "The pre-warmed recognizer was never disposed by the background continuation."); + await WaitForConditionAsync( + () => recognizerSession.DisposeCallCount == 1 && recognizerEngine.DisposeCallCount == 1, + TimeSpan.FromSeconds(5)), + "The pre-warmed recognizer engine/session was never disposed by the background continuation."); } /// Test that an unknown --capture-device throws before any recognizer is created. @@ -382,8 +414,7 @@ await WaitForConditionAsync(() => recognizer.DisposeCallCount == 1, TimeSpan.Fro public void AskCommand_Run_UnknownCaptureDevice_ThrowsArgumentException() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + WireSynthesizer(catalog); var captureSource = new FakeCaptureDeviceSource(new FakeAudioCaptureDeviceProbe()); using var context = Context.Create( [ @@ -405,8 +436,7 @@ public void AskCommand_Run_UnknownCaptureDevice_ThrowsArgumentException() public void AskCommand_Run_NoCaptureDeviceAvailable_ThrowsInvalidOperationException() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + var (synthSession, _) = WireSynthesizer(catalog); var captureSource = new FakeCaptureDeviceSource(new FakeAudioCaptureDeviceProbe()); using var context = Context.Create( ["ask", "--tts-model", "tts-model-1", "--stt-model", "stt-model-1", "--text", "hi"]); @@ -414,7 +444,7 @@ public void AskCommand_Run_NoCaptureDeviceAvailable_ThrowsInvalidOperationExcept Assert.Throws( () => AskCommand.Run(context, catalog, CreatePlaybackSource(), captureSource)); - Assert.Equal(["hi"], synthesizer.SpeakAsyncCalls); + Assert.Equal(["hi"], synthSession.SpeakAsyncCalls); } // --- Success path --- @@ -422,21 +452,18 @@ public void AskCommand_Run_NoCaptureDeviceAvailable_ThrowsInvalidOperationExcept /// /// Test that a successful run speaks the prompt, then listens and stops on the first final /// result, printing the recognized text and running Start/Stop/Dispose exactly once for - /// both the synthesizer and the recognizer, and disposing the resolved playback device - /// exactly once too - proving SpeakPromptAsync's playback-device disposal, not only - /// the synthesizer's, runs on the ordinary success path. + /// both the synthesizer session and the recognizer session (and their engines), and + /// disposing the resolved playback device exactly once too - proving + /// SpeakPromptAsync's playback-device disposal, not only the synthesizer's, runs on + /// the ordinary success path. /// [Fact] public void AskCommand_Run_Success_SpeaksThenListensAndPrintsFinalResult() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; - var recognizer = new FakeSpeechRecognizer - { - OnStart = self => self.RaiseResult("hello there", isFinal: true) - }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; + var (synthSession, synthEngine) = WireSynthesizer(catalog); + var (recognitionSession, recognitionEngine) = WireRecognizer(catalog); + recognitionSession.OnStart = self => self.RaiseResult("hello there", isFinal: true); var playbackDevice = new FakeAudioPlaybackDevice(); var playbackSource = new FakePlaybackDeviceSource(new FakeAudioPlaybackDeviceProbe([OutputDevice]), playbackDevice); @@ -456,12 +483,14 @@ public void AskCommand_Run_Success_SpeaksThenListensAndPrintsFinalResult() Console.SetOut(originalOut); } - Assert.Equal(["How are you?"], synthesizer.SpeakAsyncCalls); - Assert.Equal(1, synthesizer.DisposeCallCount); + Assert.Equal(["How are you?"], synthSession.SpeakAsyncCalls); + Assert.Equal(1, synthSession.DisposeCallCount); + Assert.Equal(1, synthEngine.DisposeCallCount); Assert.Equal(1, playbackDevice.DisposeCallCount); - Assert.Equal(1, recognizer.StartCallCount); - Assert.Equal(1, recognizer.StopCallCount); - Assert.Equal(1, recognizer.DisposeCallCount); + Assert.Equal(1, recognitionSession.StartCallCount); + Assert.Equal(1, recognitionSession.StopCallCount); + Assert.Equal(1, recognitionSession.DisposeCallCount); + Assert.Equal(1, recognitionEngine.DisposeCallCount); var printed = writer.ToString(); Assert.Contains("hello there", printed); @@ -470,10 +499,10 @@ public void AskCommand_Run_Success_SpeaksThenListensAndPrintsFinalResult() // --- Recognizer pre-warming --- /// - /// Test that the STT recognizer is constructed concurrently with - not only after - - /// Phase 1's speak/playback wait: - /// holds Phase 1 "in flight" until a signal set by CreateRecognizer fires, with a - /// bounded wait. If a regression moves recognizer construction back to only after + /// Test that the STT recognizer engine is constructed concurrently with - not only after - + /// Phase 1's speak/playback wait: + /// holds Phase 1 "in flight" until a signal set by the recognizer-engine override fires, + /// with a bounded wait. If a regression moves recognizer construction back to only after /// playback finishes, the wait below times out and fails explicitly instead of this test /// silently passing (or the process deadlocking indefinitely). /// @@ -483,26 +512,28 @@ public void AskCommand_Run_PrewarmsRecognizerConcurrentlyWithPlayback_CreatesRec using var recognizerCreatedSignal = new ManualResetEventSlim(initialState: false); var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer + var synthSession = new FakeSynthesisSession { SpeakAsyncAwaiter = () => { Assert.True( recognizerCreatedSignal.Wait(TimeSpan.FromSeconds(5)), - "The recognizer was not created concurrently with Phase 1's playback wait."); + "The recognizer engine was not created concurrently with Phase 1's playback wait."); return Task.CompletedTask; } }; - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + var synthEngine = new FakeSpeechSynthesizerEngine(synthSession); + catalog.CreateSynthesizerEngineOverride = (_, _, _) => Task.FromResult(synthEngine); - var recognizer = new FakeSpeechRecognizer + var recognitionSession = new FakeRecognitionSession { OnStart = self => self.RaiseResult("hello", isFinal: true) }; - catalog.CreateRecognizerOverride = (_, _, _) => + var recognitionEngine = new FakeSpeechRecognizerEngine(recognitionSession); + catalog.CreateRecognizerEngineOverride = (_, _, _) => { recognizerCreatedSignal.Set(); - return recognizer; + return Task.FromResult(recognitionEngine); }; var originalOut = Console.Out; @@ -522,9 +553,9 @@ public void AskCommand_Run_PrewarmsRecognizerConcurrentlyWithPlayback_CreatesRec Console.SetOut(originalOut); } - Assert.Equal(["hi"], synthesizer.SpeakAsyncCalls); - Assert.Equal(1, recognizer.StartCallCount); - Assert.Equal(1, recognizer.DisposeCallCount); + Assert.Equal(["hi"], synthSession.SpeakAsyncCalls); + Assert.Equal(1, recognitionSession.StartCallCount); + Assert.Equal(1, recognitionSession.DisposeCallCount); Assert.Contains("hello", writer.ToString()); } @@ -533,13 +564,9 @@ public void AskCommand_Run_PrewarmsRecognizerConcurrentlyWithPlayback_CreatesRec public void AskCommand_Run_OutputText_WritesRecognizedTextToFile() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; - var recognizer = new FakeSpeechRecognizer - { - OnStart = self => self.RaiseResult("the reply", isFinal: true) - }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; + WireSynthesizer(catalog); + var (recognitionSession, _) = WireRecognizer(catalog); + recognitionSession.OnStart = self => self.RaiseResult("the reply", isFinal: true); var outputPath = Path.Join(Path.GetTempPath(), $"ask-test-{Guid.NewGuid():N}.txt"); try @@ -564,7 +591,7 @@ public void AskCommand_Run_OutputText_WritesRecognizedTextToFile() } } - /// Test that --tts-param and --stt-param each forward to their own model's CreateSynthesizer/CreateRecognizer call. + /// Test that --tts-param and --stt-param each forward to their own model's CreateSynthesizerEngine/CreateRecognizerEngine call. [Fact] public void AskCommand_Run_ValidParams_ForwardToRespectiveCreateCalls() { @@ -573,13 +600,9 @@ public void AskCommand_Run_ValidParams_ForwardToRespectiveCreateCalls() var beamParameter = new NumericParameter( "beam", "Beam", "Beam width", new NumericParameterBounds(1.0, 10.0, 1.0, 4.0)); var catalog = CreateCatalogWithModels(ttsParameters: [rateParameter], sttParameters: [beamParameter]); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; - var recognizer = new FakeSpeechRecognizer - { - OnStart = self => self.RaiseResult("ok", isFinal: true) - }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; + WireSynthesizer(catalog); + var (recognitionSession, _) = WireRecognizer(catalog); + recognitionSession.OnStart = self => self.RaiseResult("ok", isFinal: true); using var context = Context.Create( [ @@ -619,24 +642,20 @@ public void AskCommand_Run_InvalidTtsParam_ThrowsArgumentException() /// Test that --silence-timeout ends the listen phase with an empty recognized text when /// no final result ever arrives, using a real (small) wall-clock idle window so the /// production - /// genuinely fires and calls Stop(), exactly as it would for a real reply that + /// genuinely fires and calls StopAsync, exactly as it would for a real reply that /// never finishes. /// [Fact] public void AskCommand_Run_SilenceTimeoutWithNoFinalResult_EndsTurnWithEmptyText() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + WireSynthesizer(catalog); // OnStart synchronously raises only an interim (non-final) result, then never a final // one: the idle timer re-arms once on that event and then fires for real (a small but - // real wall-clock delay), calling Stop() and ending the turn with no recognized text. - var recognizer = new FakeSpeechRecognizer - { - OnStart = self => self.RaiseResult("still talking", isFinal: false) - }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; + // real wall-clock delay), calling StopAsync() and ending the turn with no recognized text. + var (recognitionSession, _) = WireRecognizer(catalog); + recognitionSession.OnStart = self => self.RaiseResult("still talking", isFinal: false); var outputPath = Path.Join(Path.GetTempPath(), $"ask-test-{Guid.NewGuid():N}.txt"); try @@ -649,10 +668,10 @@ public void AskCommand_Run_SilenceTimeoutWithNoFinalResult_EndsTurnWithEmptyText AskCommand.Run(context, catalog, CreatePlaybackSource(), CreateCaptureSource()); - // Two idempotent Stop() calls: one from the session's own timeout handler, one from - // Listen's unified post-wait call (which always runs from the calling thread, not - // from onResultReceived, to avoid a reentrant deadlock). - Assert.Equal(2, recognizer.StopCallCount); + // Two idempotent StopAsync() calls: one from the session's own timeout handler, one + // from Listen's unified post-wait call (which always runs from the calling thread, + // not from the result callback, to avoid a reentrant deadlock). + Assert.Equal(2, recognitionSession.StopCallCount); Assert.True(File.Exists(outputPath)); Assert.Equal(string.Empty, File.ReadAllText(outputPath)); } @@ -676,37 +695,33 @@ public void AskCommand_Run_SilenceTimeoutWithNoFinalResult_EndsTurnWithEmptyText public void AskCommand_Run_NoTimeoutFlagsGiven_StillEndsTurnViaDefaultSilenceTimeout() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + WireSynthesizer(catalog); // OnStart raises only an interim (non-final) result and never a final one; with no // --silence-timeout given, the command must still have armed a session using its // built-in default idle window rather than blocking stopSignal.Wait() forever. - var recognizer = new FakeSpeechRecognizer - { - OnStart = self => self.RaiseResult("still talking", isFinal: false) - }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; + var (recognitionSession, _) = WireRecognizer(catalog); + recognitionSession.OnStart = self => self.RaiseResult("still talking", isFinal: false); using var context = Context.Create( ["ask", "--tts-model", "tts-model-1", "--stt-model", "stt-model-1", "--text", "hi"]); AskCommand.Run(context, catalog, CreatePlaybackSource(), CreateCaptureSource()); - // Two idempotent Stop() calls: one from the session's own default-timeout handler, one - // from Listen's unified post-wait call (which always runs from the calling thread, not - // from onResultReceived, to avoid a reentrant deadlock). - Assert.Equal(2, recognizer.StopCallCount); + // Two idempotent StopAsync() calls: one from the session's own default-timeout handler, + // one from Listen's unified post-wait call (which always runs from the calling thread, + // not from the result callback, to avoid a reentrant deadlock). + Assert.Equal(2, recognitionSession.StopCallCount); } // --- Cancellation --- /// /// Test that a canceled Phase-1 speak session is reported cleanly, Phase 2 (listen) never - /// runs, the concurrently pre-warmed recognizer is eventually disposed rather than leaked, - /// and the resolved playback device is disposed exactly once too - proving - /// SpeakPromptAsync's playback-device disposal runs even when playback itself is - /// canceled, not only on the success path. Recognizer disposal is now driven by + /// runs, the concurrently pre-warmed recognizer engine/session is eventually disposed + /// rather than leaked, and the resolved playback device is disposed exactly once too - + /// proving SpeakPromptAsync's playback-device disposal runs even when playback + /// itself is canceled, not only on the success path. Recognizer disposal is now driven by /// DisposePrewarmedRecognizerAsync's fire-and-forget background continuation rather /// than completing synchronously before Run returns, so this polls for it instead /// of asserting immediately. @@ -715,10 +730,9 @@ public void AskCommand_Run_NoTimeoutFlagsGiven_StillEndsTurnViaDefaultSilenceTim public async Task AskCommand_Run_CanceledDuringSpeak_SkipsListenPhase() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer { SpeakAsyncException = new OperationCanceledException() }; - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; - var recognizer = new FakeSpeechRecognizer(); - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; + var (synthSession, synthEngine) = WireSynthesizer(catalog); + synthSession.SpeakAsyncException = new OperationCanceledException(); + var (recognizerSession, recognizerEngine) = WireRecognizer(catalog); var playbackDevice = new FakeAudioPlaybackDevice(); var playbackSource = new FakePlaybackDeviceSource(new FakeAudioPlaybackDeviceProbe([OutputDevice]), playbackDevice); @@ -729,12 +743,15 @@ public async Task AskCommand_Run_CanceledDuringSpeak_SkipsListenPhase() AskCommand.Run(context, catalog, playbackSource, CreateCaptureSource()); Assert.Equal(1, context.ExitCode); - Assert.Equal(1, synthesizer.DisposeCallCount); + Assert.Equal(1, synthSession.DisposeCallCount); + Assert.Equal(1, synthEngine.DisposeCallCount); Assert.Equal(1, playbackDevice.DisposeCallCount); - Assert.Equal(0, recognizer.StartCallCount); + Assert.Equal(0, recognizerSession.StartCallCount); Assert.True( - await WaitForConditionAsync(() => recognizer.DisposeCallCount == 1, TimeSpan.FromSeconds(5)), - "The pre-warmed recognizer was never disposed by the background continuation."); + await WaitForConditionAsync( + () => recognizerSession.DisposeCallCount == 1 && recognizerEngine.DisposeCallCount == 1, + TimeSpan.FromSeconds(5)), + "The pre-warmed recognizer engine/session was never disposed by the background continuation."); } /// @@ -751,10 +768,9 @@ await WaitForConditionAsync(() => recognizer.DisposeCallCount == 1, TimeSpan.Fro public async Task AskCommand_Run_ExceptionDuringSpeak_DisposesPlaybackDeviceAndRethrows() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer { SpeakAsyncException = new InvalidOperationException("synthesis engine failure") }; - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; - var recognizer = new FakeSpeechRecognizer(); - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; + var (synthSession, synthEngine) = WireSynthesizer(catalog); + synthSession.SpeakAsyncException = new InvalidOperationException("synthesis engine failure"); + var (recognizerSession, recognizerEngine) = WireRecognizer(catalog); var playbackDevice = new FakeAudioPlaybackDevice(); var playbackSource = new FakePlaybackDeviceSource(new FakeAudioPlaybackDeviceProbe([OutputDevice]), playbackDevice); @@ -765,11 +781,14 @@ public async Task AskCommand_Run_ExceptionDuringSpeak_DisposesPlaybackDeviceAndR Assert.Throws( () => AskCommand.Run(context, catalog, playbackSource, CreateCaptureSource())); - Assert.Equal(1, synthesizer.DisposeCallCount); + Assert.Equal(1, synthSession.DisposeCallCount); + Assert.Equal(1, synthEngine.DisposeCallCount); Assert.Equal(1, playbackDevice.DisposeCallCount); Assert.True( - await WaitForConditionAsync(() => recognizer.DisposeCallCount == 1, TimeSpan.FromSeconds(5)), - "The pre-warmed recognizer was never disposed by the background continuation."); + await WaitForConditionAsync( + () => recognizerSession.DisposeCallCount == 1 && recognizerEngine.DisposeCallCount == 1, + TimeSpan.FromSeconds(5)), + "The pre-warmed recognizer engine/session was never disposed by the background continuation."); } /// @@ -778,12 +797,13 @@ await WaitForConditionAsync(() => recognizer.DisposeCallCount == 1, TimeSpan.Fro /// flight - proving the fix for the reviewer-flagged responsiveness regression where /// DisposePrewarmedRecognizerAsync used to be awaited synchronously before /// RunAsync returned, delaying Ctrl+C/fast-failure responsiveness until the - /// expensive recognizer model-load finished. + /// expensive recognizer model-load finished. The /// blocks on a the test controls, simulating the /// model-load step still being in flight; if a regression reintroduces a synchronous /// await, this test times out waiting for RunAsync to return instead of passing - /// instantly. Once the hold is released, the recognizer is still proven to be eventually - /// disposed by the background continuation - never leaked - just asynchronously. + /// instantly. Once the hold is released, the recognizer engine/session is still proven to + /// be eventually disposed by the background continuation - never leaked - just + /// asynchronously. /// [Fact] public async Task AskCommand_RunAsync_CanceledDuringSpeak_ReturnsPromptlyWithoutAwaitingInFlightPrewarm() @@ -791,18 +811,19 @@ public async Task AskCommand_RunAsync_CanceledDuringSpeak_ReturnsPromptlyWithout using var holdPrewarm = new ManualResetEventSlim(initialState: false); var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer { SpeakAsyncException = new OperationCanceledException() }; - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + var (synthSession, _) = WireSynthesizer(catalog); + synthSession.SpeakAsyncException = new OperationCanceledException(); - var recognizer = new FakeSpeechRecognizer(); - catalog.CreateRecognizerOverride = (_, _, _) => + var recognitionSession = new FakeRecognitionSession(); + var recognitionEngine = new FakeSpeechRecognizerEngine(recognitionSession); + catalog.CreateRecognizerEngineOverride = (_, _, _) => { // Simulates the expensive recognizer model-load step still being in flight when // Phase 1 cancels. Assert.True( holdPrewarm.Wait(TimeSpan.FromSeconds(10)), "Test setup failure: the hold signal was never released."); - return recognizer; + return Task.FromResult(recognitionEngine); }; using var context = Context.Create( @@ -832,7 +853,7 @@ public async Task AskCommand_RunAsync_CanceledDuringSpeak_ReturnsPromptlyWithout await runTask; Assert.Equal(1, context.ExitCode); - Assert.Equal(0, recognizer.DisposeCallCount); + Assert.Equal(0, recognitionSession.DisposeCallCount); } finally { @@ -842,15 +863,17 @@ public async Task AskCommand_RunAsync_CanceledDuringSpeak_ReturnsPromptlyWithout } Assert.True( - await WaitForConditionAsync(() => recognizer.DisposeCallCount == 1, TimeSpan.FromSeconds(5)), - "The pre-warmed recognizer was never disposed by the background continuation once construction completed."); + await WaitForConditionAsync( + () => recognitionSession.DisposeCallCount == 1 && recognitionEngine.DisposeCallCount == 1, + TimeSpan.FromSeconds(5)), + "The pre-warmed recognizer engine/session was never disposed by the background continuation once construction completed."); } /// /// Test that a genuine Ctrl+C landing mid-listen (simulated by canceling the same /// and setting the same /// the real handler uses, from within the fake - /// recognizer's Start() callback) is reported as a cancellation - via + /// recognition session's StartAsync callback) is reported as a cancellation - via /// and a non-zero exit code - rather than silently /// printed as an empty, successful result. Drives /// directly (rather than the public Run entry point) because there is no @@ -860,25 +883,21 @@ await WaitForConditionAsync(() => recognizer.DisposeCallCount == 1, TimeSpan.Fro public async Task AskCommand_RunAsync_CtrlCDuringListen_ReportsCanceledAndDoesNotPrintText() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + WireSynthesizer(catalog); using var cancellationSource = new CancellationTokenSource(); using var stopSignal = new ManualResetEventSlim(initialState: false); - var recognizer = new FakeSpeechRecognizer + var (recognitionSession, _) = WireRecognizer(catalog); + // Simulates the exact interleaving AskCommand's own onCancelKeyPress handler produces + // when Ctrl+C lands during Phase 2: cancel the shared token, stop the session, then + // signal stopSignal - all without ever raising a final result. + recognitionSession.OnStart = self => { - // Simulates the exact interleaving AskCommand's own onCancelKeyPress handler - // produces when Ctrl+C lands during Phase 2: cancel the shared token, stop the - // recognizer, then signal stopSignal - all without ever raising a final result. - OnStart = self => - { - cancellationSource.Cancel(); - self.Stop(); - stopSignal.Set(); - } + cancellationSource.Cancel(); + _ = self.StopAsync(); + stopSignal.Set(); }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; var originalOut = Console.Out; using var writer = new StringWriter { NewLine = "\n" }; @@ -917,19 +936,12 @@ await AskCommand.RunAsync( public async Task AskCommand_RunAsync_CtrlCBeforeListenStarts_ReportsCanceledAndDoesNotPrintText() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; - var recognizer = new FakeSpeechRecognizer(); - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; + WireSynthesizer(catalog); + var (recognitionSession, _) = WireRecognizer(catalog); using var cancellationSource = new CancellationTokenSource(); using var stopSignal = new ManualResetEventSlim(initialState: false); - // Ctrl+C already landed (token canceled, stopSignal set) in the narrow window between - // Phase 1 finishing successfully and Phase 2 starting, before RunAsync is even invoked. - await cancellationSource.CancelAsync(); - stopSignal.Set(); - var originalOut = Console.Out; using var writer = new StringWriter { NewLine = "\n" }; Console.SetOut(writer); @@ -944,11 +956,25 @@ await AskCommand.RunAsync( CreatePlaybackSource(), CreateCaptureSource(), stopSignal, - _ => { }, + createdSession => + { + // Listen() calls onRecognizerCreated(session) as its very first step, before + // checking stopSignal.IsSet: canceling here (only once, on the non-null call) + // simulates Ctrl+C landing in the narrow window between the concurrently + // pre-warmed recognizer finishing and Phase 2 genuinely starting, so Listen + // observes stopSignal already set without ever starting the session - rather + // than canceling before RunAsync is even invoked, which would instead hit + // Phase 1's own (narrower) cancellation handling. + if (createdSession is not null) + { + cancellationSource.Cancel(); + stopSignal.Set(); + } + }, cancellationSource.Token); Assert.Equal(1, context.ExitCode); - Assert.Equal(0, recognizer.StartCallCount); + Assert.Equal(0, recognitionSession.StartCallCount); Assert.Equal(string.Empty, writer.ToString()); } finally @@ -966,19 +992,15 @@ await AskCommand.RunAsync( public async Task AskCommand_RunAsync_SilenceTimeoutWithNoCtrlC_ReportsSuccessNotCanceled() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + WireSynthesizer(catalog); using var cancellationSource = new CancellationTokenSource(); using var stopSignal = new ManualResetEventSlim(initialState: false); - var recognizer = new FakeSpeechRecognizer - { - // No Ctrl+C involved: the timeout session (armed by --silence-timeout) is what sets - // stopSignal here, exactly as the production TimedOut handler does. - OnStart = _ => stopSignal.Set() - }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; + var (recognitionSession, _) = WireRecognizer(catalog); + // No Ctrl+C involved: the timeout session (armed by --silence-timeout) is what sets + // stopSignal here, exactly as the production TimedOut handler does. + recognitionSession.OnStart = _ => stopSignal.Set(); var originalOut = Console.Out; using var writer = new StringWriter { NewLine = "\n" }; @@ -1022,23 +1044,19 @@ await AskCommand.RunAsync( public async Task AskCommand_RunAsync_CtrlCImmediatelyAfterListenReturns_ReportsCanceledAndDoesNotPrintText() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + WireSynthesizer(catalog); using var cancellationSource = new CancellationTokenSource(); using var stopSignal = new ManualResetEventSlim(initialState: false); - var recognizer = new FakeSpeechRecognizer + var (recognitionSession, _) = WireRecognizer(catalog); + // A legitimate final result, with no Ctrl+C involved yet: Listen() will observe + // cancellationToken.IsCancellationRequested == false and return (text, false). + recognitionSession.OnStart = self => { - // A legitimate final result, with no Ctrl+C involved yet: Listen() will observe - // cancellationToken.IsCancellationRequested == false and return (text, false). - OnStart = self => - { - self.RaiseResult("hello", isFinal: true); - stopSignal.Set(); - } + self.RaiseResult("hello", isFinal: true); + stopSignal.Set(); }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; var originalOut = Console.Out; using var writer = new StringWriter { NewLine = "\n" }; @@ -1096,23 +1114,19 @@ await AskCommand.RunAsync( public async Task AskCommand_RunAsync_CtrlCImmediatelyBeforeFileWrite_ReportsCanceledAndDoesNotWriteFile() { var catalog = CreateCatalogWithModels(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + WireSynthesizer(catalog); using var cancellationSource = new CancellationTokenSource(); using var stopSignal = new ManualResetEventSlim(initialState: false); - var recognizer = new FakeSpeechRecognizer + var (recognitionSession, _) = WireRecognizer(catalog); + // A legitimate final result, with no Ctrl+C involved yet: Listen() will observe + // cancellationToken.IsCancellationRequested == false and return (text, false). + recognitionSession.OnStart = self => { - // A legitimate final result, with no Ctrl+C involved yet: Listen() will observe - // cancellationToken.IsCancellationRequested == false and return (text, false). - OnStart = self => - { - self.RaiseResult("hello", isFinal: true); - stopSignal.Set(); - } + self.RaiseResult("hello", isFinal: true); + stopSignal.Set(); }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; var outputPath = Path.Join(Path.GetTempPath(), $"ask-test-{Guid.NewGuid():N}.txt"); var originalOut = Console.Out; diff --git a/test/DemaConsulting.Speech.Cli.Tests/Commands/ModelCommandsSubsystem/FakeCliModelCatalog.cs b/test/DemaConsulting.Speech.Cli.Tests/Commands/ModelCommandsSubsystem/FakeCliModelCatalog.cs index d0282e6..57cda5e 100644 --- a/test/DemaConsulting.Speech.Cli.Tests/Commands/ModelCommandsSubsystem/FakeCliModelCatalog.cs +++ b/test/DemaConsulting.Speech.Cli.Tests/Commands/ModelCommandsSubsystem/FakeCliModelCatalog.cs @@ -140,15 +140,15 @@ public void Uninstall(string modelId) public Func? GetPreferredAudioFormatOverride { get; set; } /// - /// Gets or sets the delegate forwards to, or + /// Gets or sets the delegate forwards to, or /// to throw a clear "not configured" exception if a test forgets /// to set it and the code path is reached. /// - public Func?, ISpeechSynthesizer>? - CreateSynthesizerOverride + public Func?, CancellationToken, Task>? + CreateSynthesizerEngineOverride { get; set; } - /// Gets the ordered list of parameter value bags passed to . + /// Gets the ordered list of parameter value bags passed to . public List?> CreateSynthesizerParameterValueCalls { get; } = []; /// @@ -162,19 +162,18 @@ public AudioFormat GetPreferredAudioFormat(SpeechModelDescriptor descriptor) } /// - public ISpeechSynthesizer CreateSynthesizer( + public Task CreateSynthesizerEngineAsync( SpeechModelDescriptor descriptor, - IAudioPlaybackDevice playbackDevice, - IReadOnlyDictionary? parameterValues) + IReadOnlyDictionary? parameterValues, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(descriptor); - ArgumentNullException.ThrowIfNull(playbackDevice); CreateSynthesizerParameterValueCalls.Add(parameterValues); - return CreateSynthesizerOverride?.Invoke(descriptor, playbackDevice, parameterValues) + return CreateSynthesizerEngineOverride?.Invoke(descriptor, parameterValues, cancellationToken) ?? throw new InvalidOperationException( - $"{nameof(FakeCliModelCatalog)}.{nameof(CreateSynthesizerOverride)} was not configured for this test."); + $"{nameof(FakeCliModelCatalog)}.{nameof(CreateSynthesizerEngineOverride)} was not configured for this test."); } /// @@ -185,15 +184,15 @@ public ISpeechSynthesizer CreateSynthesizer( public Func? GetAudioFormatOverride { get; set; } /// - /// Gets or sets the delegate forwards to, or + /// Gets or sets the delegate forwards to, or /// to throw a clear "not configured" exception if a test forgets /// to set it and the code path is reached. /// - public Func?, ISpeechRecognizer>? - CreateRecognizerOverride + public Func?, CancellationToken, Task>? + CreateRecognizerEngineOverride { get; set; } - /// Gets the ordered list of parameter value bags passed to . + /// Gets the ordered list of parameter value bags passed to . public List?> CreateRecognizerParameterValueCalls { get; } = []; /// @@ -207,18 +206,17 @@ public AudioFormat GetAudioFormat(SpeechModelDescriptor descriptor) } /// - public ISpeechRecognizer CreateRecognizer( + public Task CreateRecognizerEngineAsync( SpeechModelDescriptor descriptor, - IAudioCaptureDevice captureDevice, - IReadOnlyDictionary? parameterValues) + IReadOnlyDictionary? parameterValues, + CancellationToken cancellationToken = default) { ArgumentNullException.ThrowIfNull(descriptor); - ArgumentNullException.ThrowIfNull(captureDevice); CreateRecognizerParameterValueCalls.Add(parameterValues); - return CreateRecognizerOverride?.Invoke(descriptor, captureDevice, parameterValues) + return CreateRecognizerEngineOverride?.Invoke(descriptor, parameterValues, cancellationToken) ?? throw new InvalidOperationException( - $"{nameof(FakeCliModelCatalog)}.{nameof(CreateRecognizerOverride)} was not configured for this test."); + $"{nameof(FakeCliModelCatalog)}.{nameof(CreateRecognizerEngineOverride)} was not configured for this test."); } } diff --git a/test/DemaConsulting.Speech.Cli.Tests/Commands/ModelCommandsSubsystem/SpeechModelCatalogAdapterTests.cs b/test/DemaConsulting.Speech.Cli.Tests/Commands/ModelCommandsSubsystem/SpeechModelCatalogAdapterTests.cs index 7d81fd4..f12efa1 100644 --- a/test/DemaConsulting.Speech.Cli.Tests/Commands/ModelCommandsSubsystem/SpeechModelCatalogAdapterTests.cs +++ b/test/DemaConsulting.Speech.Cli.Tests/Commands/ModelCommandsSubsystem/SpeechModelCatalogAdapterTests.cs @@ -171,54 +171,36 @@ public void SpeechModelCatalogAdapter_GetPreferredAudioFormat_NullDescriptor_Thr } /// - /// Test that throws a clean - /// for a real, compiled-in recognition-role model (not a - /// synthesis model). + /// Test that throws a + /// clean for a real, compiled-in recognition-role model + /// (not a synthesis model). /// [Fact] - public void SpeechModelCatalogAdapter_CreateSynthesizer_RecognitionRoleModel_ThrowsArgumentException() + public async Task SpeechModelCatalogAdapter_CreateSynthesizerEngineAsync_RecognitionRoleModel_ThrowsArgumentException() { // Arrange using var adapter = new SpeechModelCatalogAdapter(new SpeechModelStoreOptions { RootPathOverride = _testRoot }); var descriptor = adapter.Enumerate().First(d => d.Role == SpeechModelRole.Recognition); - var wavPath = Path.Join(_testRoot, "output.wav"); - using var playbackDevice = new WavFileAudioPlaybackDevice(wavPath, 22050, 1); // Act & Assert - var exception = Assert.Throws( - () => adapter.CreateSynthesizer(descriptor, playbackDevice, null)); + var exception = await Assert.ThrowsAsync( + () => adapter.CreateSynthesizerEngineAsync(descriptor, null, TestContext.Current.CancellationToken)); Assert.Equal("descriptor", exception.ParamName); } /// - /// Test that rejects a null - /// descriptor. + /// Test that rejects a + /// null descriptor. /// [Fact] - public void SpeechModelCatalogAdapter_CreateSynthesizer_NullDescriptor_ThrowsArgumentNullException() + public async Task SpeechModelCatalogAdapter_CreateSynthesizerEngineAsync_NullDescriptor_ThrowsArgumentNullException() { // Arrange using var adapter = new SpeechModelCatalogAdapter(new SpeechModelStoreOptions { RootPathOverride = _testRoot }); - var wavPath = Path.Join(_testRoot, "output.wav"); - using var playbackDevice = new WavFileAudioPlaybackDevice(wavPath, 22050, 1); // Act & Assert - Assert.Throws(() => adapter.CreateSynthesizer(null!, playbackDevice, null)); - } - - /// - /// Test that rejects a null - /// playback device. - /// - [Fact] - public void SpeechModelCatalogAdapter_CreateSynthesizer_NullPlaybackDevice_ThrowsArgumentNullException() - { - // Arrange - using var adapter = new SpeechModelCatalogAdapter(new SpeechModelStoreOptions { RootPathOverride = _testRoot }); - var descriptor = adapter.Enumerate().First(d => d.Role == SpeechModelRole.Recognition); - - // Act & Assert - Assert.Throws(() => adapter.CreateSynthesizer(descriptor, null!, null)); + await Assert.ThrowsAsync( + () => adapter.CreateSynthesizerEngineAsync(null!, null, TestContext.Current.CancellationToken)); } /// @@ -253,81 +235,36 @@ public void SpeechModelCatalogAdapter_GetAudioFormat_NullDescriptor_ThrowsArgume } /// - /// Test that throws a clean - /// for a real, compiled-in synthesis-role model (not a - /// recognition model). + /// Test that throws a + /// clean for a real, compiled-in synthesis-role model (not + /// a recognition model). /// [Fact] - public void SpeechModelCatalogAdapter_CreateRecognizer_SynthesisRoleModel_ThrowsArgumentException() + public async Task SpeechModelCatalogAdapter_CreateRecognizerEngineAsync_SynthesisRoleModel_ThrowsArgumentException() { // Arrange using var adapter = new SpeechModelCatalogAdapter(new SpeechModelStoreOptions { RootPathOverride = _testRoot }); var descriptor = adapter.Enumerate().First(d => d.Role == SpeechModelRole.Synthesis); - var wavPath = Path.Join(_testRoot, "input.wav"); - WriteMinimalWavFile(wavPath); - var captureDevice = new WavFileAudioCaptureDevice(wavPath); // Act & Assert - var exception = Assert.Throws( - () => adapter.CreateRecognizer(descriptor, captureDevice, null)); + var exception = await Assert.ThrowsAsync( + () => adapter.CreateRecognizerEngineAsync(descriptor, null, TestContext.Current.CancellationToken)); Assert.Equal("descriptor", exception.ParamName); } /// - /// Test that rejects a null - /// descriptor. + /// Test that rejects a + /// null descriptor. /// [Fact] - public void SpeechModelCatalogAdapter_CreateRecognizer_NullDescriptor_ThrowsArgumentNullException() + public async Task SpeechModelCatalogAdapter_CreateRecognizerEngineAsync_NullDescriptor_ThrowsArgumentNullException() { // Arrange using var adapter = new SpeechModelCatalogAdapter(new SpeechModelStoreOptions { RootPathOverride = _testRoot }); - var wavPath = Path.Join(_testRoot, "input.wav"); - WriteMinimalWavFile(wavPath); - var captureDevice = new WavFileAudioCaptureDevice(wavPath); // Act & Assert - Assert.Throws(() => adapter.CreateRecognizer(null!, captureDevice, null)); - } - - /// - /// Test that rejects a null - /// capture device. - /// - [Fact] - public void SpeechModelCatalogAdapter_CreateRecognizer_NullCaptureDevice_ThrowsArgumentNullException() - { - // Arrange - using var adapter = new SpeechModelCatalogAdapter(new SpeechModelStoreOptions { RootPathOverride = _testRoot }); - var descriptor = adapter.Enumerate().First(d => d.Role == SpeechModelRole.Recognition); - - // Act & Assert - Assert.Throws(() => adapter.CreateRecognizer(descriptor, null!, null)); - } - - /// - /// Writes a minimal valid mono, 16-bit PCM RIFF/WAVE file (no sample data required) so a - /// test can construct a real without needing a - /// shared binary test fixture. - /// - private static void WriteMinimalWavFile(string path) - { - using var stream = new FileStream(path, FileMode.Create, FileAccess.Write); - using var writer = new BinaryWriter(stream); - - writer.Write("RIFF"u8); - writer.Write(36); // RIFF chunk size (no data) - writer.Write("WAVE"u8); - writer.Write("fmt "u8); - writer.Write(16); // fmt chunk size - writer.Write((short)1); // PCM - writer.Write((short)1); // mono - writer.Write(16000); // sample rate - writer.Write(32000); // byte rate - writer.Write((short)2); // block align - writer.Write((short)16); // bits per sample - writer.Write("data"u8); - writer.Write(0); // data chunk size (no samples) + await Assert.ThrowsAsync( + () => adapter.CreateRecognizerEngineAsync(null!, null, TestContext.Current.CancellationToken)); } // NOTE: No test exercises a real, downloaded synthesis-role model here. CI has no cached TTS diff --git a/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/FakeRecognitionSession.cs b/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/FakeRecognitionSession.cs new file mode 100644 index 0000000..fbbe5e4 --- /dev/null +++ b/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/FakeRecognitionSession.cs @@ -0,0 +1,133 @@ +// Copyright (c) DEMA Consulting +// +// Permission is hereby granted, free of charge, to any person obtaining a copy +// of this software and associated documentation files (the "Software"), to deal +// in the Software without restriction, including without limitation the rights +// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +// copies of the Software, and to permit persons to whom the Software is +// furnished to do so, subject to the following conditions: +// +// The above copyright notice and this permission notice shall be included in all +// copies or substantial portions of the Software. +// +// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +// SOFTWARE. + +using System.Threading.Channels; +using DemaConsulting.Speech.RecognitionSubsystem; + +namespace DemaConsulting.Speech.Cli.Tests.Commands.RecognitionCommandSubsystem; + +/// +/// Deterministic, in-memory fake used by recognize +/// and ask command unit tests, so no test depends on a real, native sherpa-onnx engine. +/// +/// +/// Results are buffered on an unbounded rather than raised through a +/// background-thread event, mirroring the real +/// single-consumer, fully async contract: a test calls to +/// enqueue a result, which a concurrently running await foreach over +/// then observes, with no real background thread involved. +/// +internal sealed class FakeRecognitionSession : IRecognitionSession +{ + /// The channel backing ; results are written via . + private readonly Channel _channel = Channel.CreateUnbounded(); + + /// Gets the number of times was called. + public int StartCallCount { get; private set; } + + /// Gets the number of times was called. + public int StopCallCount { get; private set; } + + /// Gets the number of times was called. + public int DisposeCallCount { get; private set; } + + /// Gets or sets an exception to throw from , or for none. + public Exception? StartException { get; set; } + + /// + /// Gets or sets an action invoked synchronously from , after + /// incrementing and transitioning to + /// . + /// + /// + /// Lets a test simulate a file-mode session by calling and/or + /// reentrantly from within this callback. + /// + public Action? OnStart { get; set; } + + /// + /// Gets or sets an action invoked synchronously from , after + /// incrementing but before the results channel is completed. + /// + public Action? OnStop { get; set; } + + /// + public bool IsAvailable { get; set; } = true; + + /// + public RecognitionSessionState State { get; private set; } = RecognitionSessionState.Created; + + /// + public event EventHandler? StateChanged; + + /// + /// Synthetically enqueues a result onto the results channel, letting a test simulate a + /// recognized result without a real engine. Safe to call from any thread, including + /// reentrantly from /. + /// + public void RaiseResult(string text, bool isFinal) => + _channel.Writer.TryWrite(new SpeechRecognitionEvent(new SpeechRecognitionResult(text, isFinal))); + + /// + public Task StartAsync(CancellationToken cancellationToken = default) + { + StartCallCount++; + SetState(RecognitionSessionState.Running); + + if (StartException is not null) + { + SetState(RecognitionSessionState.Faulted); + throw StartException; + } + + OnStart?.Invoke(this); + return Task.CompletedTask; + } + + /// + public Task StopAsync(CancellationToken cancellationToken = default) + { + StopCallCount++; + OnStop?.Invoke(this); + SetState(RecognitionSessionState.Stopped); + _channel.Writer.TryComplete(); + return Task.CompletedTask; + } + + /// + public IAsyncEnumerable GetResultsAsync(CancellationToken cancellationToken = default) => + _channel.Reader.ReadAllAsync(cancellationToken); + + /// + public ValueTask DisposeAsync() + { + DisposeCallCount++; + _channel.Writer.TryComplete(); + return ValueTask.CompletedTask; + } + + /// Updates and raises . + private void SetState(RecognitionSessionState newState) + { + var previous = State; + State = newState; + StateChanged?.Invoke(this, new SessionStateChangedEventArgs(previous, newState)); + } +} diff --git a/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/FakeSpeechRecognizer.cs b/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/FakeSpeechRecognizer.cs deleted file mode 100644 index 470fca4..0000000 --- a/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/FakeSpeechRecognizer.cs +++ /dev/null @@ -1,96 +0,0 @@ -// Copyright (c) DEMA Consulting -// -// Permission is hereby granted, free of charge, to any person obtaining a copy -// of this software and associated documentation files (the "Software"), to deal -// in the Software without restriction, including without limitation the rights -// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -// copies of the Software, and to permit persons to whom the Software is -// furnished to do so, subject to the following conditions: -// -// The above copyright notice and this permission notice shall be included in all -// copies or substantial portions of the Software. -// -// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -// SOFTWARE. - -using DemaConsulting.Speech.RecognitionSubsystem; - -namespace DemaConsulting.Speech.Cli.Tests.Commands.RecognitionCommandSubsystem; - -/// -/// Deterministic, in-memory fake used by recognize -/// command unit tests, so no test depends on a real, native sherpa-onnx engine. -/// -internal sealed class FakeSpeechRecognizer : ISpeechRecognizer -{ - /// Gets the number of times was called. - public int StartCallCount { get; private set; } - - /// Gets the number of times was called. - public int StopCallCount { get; private set; } - - /// Gets the number of times was called. - public int DisposeCallCount { get; private set; } - - /// Gets or sets an exception to throw from , or for none. - public Exception? StartException { get; set; } - - /// Gets or sets an action invoked synchronously from , after incrementing . - /// - /// Lets a test simulate a file-mode WavFileAudioCaptureDevice-driven session by - /// raising events and/or calling - /// reentrantly from within this callback, mirroring the real recognizer's own - /// synchronous, blocking Start() contract for file input. - /// - public Action? OnStart { get; set; } - - /// Gets or sets an action invoked synchronously from , after incrementing . - /// - /// Lets a test simulate the real recognizer's Stop() blocking while its own - /// background decode thread drains already-captured audio and raises - /// - the exact interleaving - /// SilenceTimeoutRecognizerSession.OnIdle must not deadlock against. - /// - public Action? OnStop { get; set; } - - /// - public bool IsAvailable { get; set; } = true; - - /// - public event EventHandler? ResultReceived; - - /// - /// Synthetically raises with the given text and finality, - /// letting a test simulate a recognized result without a real engine. - /// - public void RaiseResult(string text, bool isFinal) => - ResultReceived?.Invoke(this, new SpeechRecognitionEvent(new SpeechRecognitionResult(text, isFinal))); - - /// - public void Start() - { - StartCallCount++; - - if (StartException is not null) - { - throw StartException; - } - - OnStart?.Invoke(this); - } - - /// - public void Stop() - { - StopCallCount++; - OnStop?.Invoke(this); - } - - /// - public void Dispose() => DisposeCallCount++; -} diff --git a/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/FakeSpeechRecognizerEngine.cs b/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/FakeSpeechRecognizerEngine.cs new file mode 100644 index 0000000..e3cf0d0 --- /dev/null +++ b/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/FakeSpeechRecognizerEngine.cs @@ -0,0 +1,86 @@ +// Copyright (c) DEMA Consulting +// +// Permission is hereby granted, free of charge, to any person obtaining a copy +// of this software and associated documentation files (the "Software"), to deal +// in the Software without restriction, including without limitation the rights +// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +// copies of the Software, and to permit persons to whom the Software is +// furnished to do so, subject to the following conditions: +// +// The above copyright notice and this permission notice shall be included in all +// copies or substantial portions of the Software. +// +// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +// SOFTWARE. + +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.RecognitionSubsystem; + +namespace DemaConsulting.Speech.Cli.Tests.Commands.RecognitionCommandSubsystem; + +/// +/// Deterministic, in-memory fake used by +/// recognize/ask command unit tests, wrapping a single pre-configured +/// that returns. +/// +internal sealed class FakeSpeechRecognizerEngine : ISpeechRecognizerEngine +{ + /// The session returns. + private readonly FakeRecognitionSession _session; + + /// + /// Initializes a new instance of the class. + /// + /// The session returns. Must not be null. + public FakeSpeechRecognizerEngine(FakeRecognitionSession session) + { + ArgumentNullException.ThrowIfNull(session); + _session = session; + } + + /// Gets the number of times was called. + public int CreateSessionAsyncCallCount { get; private set; } + + /// Gets the number of times was called. + public int DisposeCallCount { get; private set; } + + /// Gets the capture device passed to the most recent call. + public IAudioCaptureDevice? LastDevice { get; private set; } + + /// Gets or sets an exception to throw from , or for none. + public Exception? CreateSessionException { get; set; } + + /// + public bool IsAvailable { get; set; } = true; + + /// + public Task CreateSessionAsync( + IAudioCaptureDevice device, + CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(device); + + CreateSessionAsyncCallCount++; + LastDevice = device; + cancellationToken.ThrowIfCancellationRequested(); + + if (CreateSessionException is not null) + { + throw CreateSessionException; + } + + return Task.FromResult(_session); + } + + /// + public ValueTask DisposeAsync() + { + DisposeCallCount++; + return ValueTask.CompletedTask; + } +} diff --git a/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/RecognizeCommandTests.cs b/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/RecognizeCommandTests.cs index 3cc1b97..ba24756 100644 --- a/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/RecognizeCommandTests.cs +++ b/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/RecognizeCommandTests.cs @@ -32,8 +32,9 @@ namespace DemaConsulting.Speech.Cli.Tests.Commands.RecognitionCommandSubsystem; /// /// Unit tests for , using , -/// , and fake audio device probes so every scenario runs -/// deterministically with no real catalog, network access, native engine, or audio hardware. +/// /, and fake +/// audio device probes so every scenario runs deterministically with no real catalog, network +/// access, native engine, or audio hardware. /// [Collection("Sequential")] public sealed class RecognizeCommandTests @@ -83,6 +84,20 @@ private static string WriteMinimalWavFile(int sampleFrameCount = 0) return path; } + /// + /// Creates a wrapped in a , + /// and wires the catalog's + /// to return the engine. + /// + private static (FakeRecognitionSession Session, FakeSpeechRecognizerEngine Engine) WireRecognizer( + FakeCliModelCatalog catalog) + { + var session = new FakeRecognitionSession(); + var engine = new FakeSpeechRecognizerEngine(session); + catalog.CreateRecognizerEngineOverride = (_, _, _) => Task.FromResult(engine); + return (session, engine); + } + // --- ParseArguments --- /// Test that --stt-model is required. @@ -184,6 +199,53 @@ public void RecognizeCommand_ParseArguments_MalformedStartTimeout_ThrowsArgument ["--stt-model", "model-1", "--mic", "--start-timeout", "not-a-number"])); } + /// + /// Test that 's mic-mode start-timeout default + /// resolves to a fixed 8 seconds when --start-timeout is omitted, independent of + /// whatever value --silence-timeout is given - it never borrows + /// --silence-timeout's own value, proving the requirement's corrected wording. + /// + [Fact] + public void RecognizeCommand_Run_StartTimeoutOmitted_ResolvesToFixedEightSecondDefault() + { + var options = RecognizeCommand.ParseArguments( + ["--stt-model", "model-1", "--mic", "--silence-timeout", "2.5"]); + + var startTimeout = RecognizeCommand.ResolveStartTimeout(options); + + Assert.Equal(TimeSpan.FromSeconds(8), startTimeout); + Assert.NotEqual(RecognizeCommand.ResolveSilenceTimeout(options), startTimeout); + } + + /// + /// Test that 's mic-mode silence-timeout default + /// resolves to a fixed 5 seconds when --silence-timeout is omitted. + /// + [Fact] + public void RecognizeCommand_Run_SilenceTimeoutOmitted_ResolvesToFixedFiveSecondDefault() + { + var options = RecognizeCommand.ParseArguments(["--stt-model", "model-1", "--mic"]); + + var silenceTimeout = RecognizeCommand.ResolveSilenceTimeout(options); + + Assert.Equal(TimeSpan.FromSeconds(5), silenceTimeout); + } + + /// + /// Test that an explicit --start-timeout is resolved verbatim, overriding the fixed + /// default. + /// + [Fact] + public void RecognizeCommand_Run_StartTimeoutGiven_ResolvesToGivenValue() + { + var options = RecognizeCommand.ParseArguments( + ["--stt-model", "model-1", "--mic", "--start-timeout", "2.5"]); + + var startTimeout = RecognizeCommand.ResolveStartTimeout(options); + + Assert.Equal(TimeSpan.FromSeconds(2.5), startTimeout); + } + /// Test that --interim and --final-only parse as flags. [Fact] public void RecognizeCommand_ParseArguments_InterimAndFinalOnlyFlags_ParseTrue() @@ -298,15 +360,14 @@ public void RecognizeCommand_Run_NotDownloadedModel_ThrowsArgumentExceptionWithD // --- --stt-param validation wiring --- - /// Test that a valid --stt-param is forwarded to CreateRecognizer's parameterValues argument. + /// Test that a valid --stt-param is forwarded to CreateRecognizerEngineAsync's parameterValues argument. [Fact] - public void RecognizeCommand_Run_ValidParam_ForwardsToCreateRecognizer() + public void RecognizeCommand_Run_ValidParam_ForwardsToCreateRecognizerEngine() { var beamParameter = new NumericParameter( "beam", "Beam", "Beam width", new NumericParameterBounds(1.0, 10.0, 1.0, 4.0)); var catalog = CreateCatalogWithModel(parameters: [beamParameter]); - var recognizer = new FakeSpeechRecognizer(); - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; + WireRecognizer(catalog); var wavPath = WriteMinimalWavFile(); try { @@ -325,7 +386,7 @@ public void RecognizeCommand_Run_ValidParam_ForwardsToCreateRecognizer() } } - /// Test that an invalid --stt-param value throws before any recognizer is created. + /// Test that an invalid --stt-param value throws before any recognizer engine is created. [Fact] public void RecognizeCommand_Run_InvalidParam_ThrowsArgumentException() { @@ -346,28 +407,19 @@ public void RecognizeCommand_Run_InvalidParam_ThrowsArgumentException() } } - // --- File-input EOF-driven stop flow --- + // --- File-input explicit-drain stop flow --- /// - /// Test that file-input mode drives a real to - /// completion, the recognizer is started and stopped exactly once (via the device's own - /// EndOfFileReached), and disposed exactly once, with no explicit wait needed. + /// Test that file-input mode starts the session once and then explicitly drains it with a + /// single StopAsync call - no longer relying on a separate end-of-file event, since + /// itself now delivers the whole file before + /// returning - and disposes both the engine and the session exactly once. /// [Fact] - public void RecognizeCommand_Run_FileInput_StartsAndStopsRecognizerViaEndOfFile() + public void RecognizeCommand_Run_FileInput_StartsAndStopsRecognizerViaExplicitDrain() { var catalog = CreateCatalogWithModel(); - var recognizer = new FakeSpeechRecognizer(); - // The real recognizer's Start() drives the supplied capture device's own Start() to - // completion before returning (see RecognizeCommand's remarks); simulate that here so - // the device's own EndOfFileReached event - which RecognizeCommand subscribes - // recognizer.Stop() to reentrantly - actually fires within this call, exactly as it - // would with a real SherpaOnnxSpeechRecognizer and WavFileAudioCaptureDevice pair. - catalog.CreateRecognizerOverride = (_, device, _) => - { - recognizer.OnStart = _ => device.Start(); - return recognizer; - }; + var (session, engine) = WireRecognizer(catalog); var wavPath = WriteMinimalWavFile(sampleFrameCount: 160); try { @@ -375,9 +427,10 @@ public void RecognizeCommand_Run_FileInput_StartsAndStopsRecognizerViaEndOfFile( RecognizeCommand.Run(context, catalog, new AudioDeviceFactory()); - Assert.Equal(1, recognizer.StartCallCount); - Assert.Equal(1, recognizer.StopCallCount); - Assert.Equal(1, recognizer.DisposeCallCount); + Assert.Equal(1, session.StartCallCount); + Assert.Equal(1, session.StopCallCount); + Assert.Equal(1, session.DisposeCallCount); + Assert.Equal(1, engine.DisposeCallCount); Assert.Equal(0, context.ExitCode); } finally @@ -388,19 +441,13 @@ public void RecognizeCommand_Run_FileInput_StartsAndStopsRecognizerViaEndOfFile( /// /// Test that file-input mode passes a real over - /// the given path into . + /// the given path into . /// [Fact] - public void RecognizeCommand_Run_FileInput_PassesWavFileCaptureDeviceToCreateRecognizer() + public void RecognizeCommand_Run_FileInput_PassesWavFileCaptureDeviceToCreateSession() { var catalog = CreateCatalogWithModel(); - var recognizer = new FakeSpeechRecognizer(); - IAudioCaptureDevice? capturedDevice = null; - catalog.CreateRecognizerOverride = (_, device, _) => - { - capturedDevice = device; - return recognizer; - }; + var (_, engine) = WireRecognizer(catalog); var wavPath = WriteMinimalWavFile(); try { @@ -408,7 +455,7 @@ public void RecognizeCommand_Run_FileInput_PassesWavFileCaptureDeviceToCreateRec RecognizeCommand.Run(context, catalog, new AudioDeviceFactory()); - Assert.IsType(capturedDevice); + Assert.IsType(engine.LastDevice); } finally { @@ -423,15 +470,12 @@ public void RecognizeCommand_Run_FileInput_PassesWavFileCaptureDeviceToCreateRec public void RecognizeCommand_Run_DefaultVerbosity_PrintsBothInterimAndFinal() { var catalog = CreateCatalogWithModel(); - var recognizer = new FakeSpeechRecognizer + var (session, _) = WireRecognizer(catalog); + session.OnStart = self => { - OnStart = self => - { - self.RaiseResult("hel", isFinal: false); - self.RaiseResult("hello", isFinal: true); - } + self.RaiseResult("hel", isFinal: false); + self.RaiseResult("hello", isFinal: true); }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; var wavPath = WriteMinimalWavFile(); var originalOut = Console.Out; using var writer = new StringWriter { NewLine = "\n" }; @@ -468,16 +512,13 @@ public void RecognizeCommand_Run_DefaultVerbosity_PrintsBothInterimAndFinal() public void RecognizeCommand_Run_ShrinkingInterimSequence_DoesNotLeaveStaleCharacters() { var catalog = CreateCatalogWithModel(); - var recognizer = new FakeSpeechRecognizer + var (session, _) = WireRecognizer(catalog); + session.OnStart = self => { - OnStart = self => - { - self.RaiseResult("hello world", isFinal: false); - self.RaiseResult("hello", isFinal: false); - self.RaiseResult("hello", isFinal: true); - } + self.RaiseResult("hello world", isFinal: false); + self.RaiseResult("hello", isFinal: false); + self.RaiseResult("hello", isFinal: true); }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; var wavPath = WriteMinimalWavFile(); var originalOut = Console.Out; using var writer = new StringWriter { NewLine = "\n" }; @@ -519,15 +560,12 @@ public void RecognizeCommand_Run_ShrinkingInterimSequence_DoesNotLeaveStaleChara public void RecognizeCommand_Run_FinalOnly_SuppressesInterimConsoleOutput() { var catalog = CreateCatalogWithModel(); - var recognizer = new FakeSpeechRecognizer + var (session, _) = WireRecognizer(catalog); + session.OnStart = self => { - OnStart = self => - { - self.RaiseResult("hel", isFinal: false); - self.RaiseResult("hello", isFinal: true); - } + self.RaiseResult("hel", isFinal: false); + self.RaiseResult("hello", isFinal: true); }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; var wavPath = WriteMinimalWavFile(); var originalOut = Console.Out; using var writer = new StringWriter { NewLine = "\n" }; @@ -557,15 +595,12 @@ public void RecognizeCommand_Run_FinalOnly_SuppressesInterimConsoleOutput() public void RecognizeCommand_Run_Interim_SuppressesFinalSettleConsoleOutput() { var catalog = CreateCatalogWithModel(); - var recognizer = new FakeSpeechRecognizer + var (session, _) = WireRecognizer(catalog); + session.OnStart = self => { - OnStart = self => - { - self.RaiseResult("hel", isFinal: false); - self.RaiseResult("hello", isFinal: true); - } + self.RaiseResult("hel", isFinal: false); + self.RaiseResult("hello", isFinal: true); }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; var wavPath = WriteMinimalWavFile(); var originalOut = Console.Out; using var writer = new StringWriter { NewLine = "\n" }; @@ -596,17 +631,14 @@ public void RecognizeCommand_Run_Interim_SuppressesFinalSettleConsoleOutput() public void RecognizeCommand_Run_Output_WritesOnlyFinalResultsOverwritingPriorContent() { var catalog = CreateCatalogWithModel(); - var recognizer = new FakeSpeechRecognizer + var (session, _) = WireRecognizer(catalog); + session.OnStart = self => { - OnStart = self => - { - self.RaiseResult("hel", isFinal: false); - self.RaiseResult("hello", isFinal: true); - self.RaiseResult("wor", isFinal: false); - self.RaiseResult("world", isFinal: true); - } + self.RaiseResult("hel", isFinal: false); + self.RaiseResult("hello", isFinal: true); + self.RaiseResult("wor", isFinal: false); + self.RaiseResult("world", isFinal: true); }; - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; var wavPath = WriteMinimalWavFile(); var outputPath = Path.Join(Path.GetTempPath(), $"recognize-output-{Guid.NewGuid():N}.txt"); File.WriteAllText(outputPath, "stale content that must be overwritten"); @@ -629,7 +661,7 @@ public void RecognizeCommand_Run_Output_WritesOnlyFinalResultsOverwritingPriorCo // --- --capture-device error path --- - /// Test that an unknown --capture-device throws before any recognizer is created. + /// Test that an unknown --capture-device throws before any recognizer engine is created. [Fact] public void RecognizeCommand_Run_UnknownDevice_ThrowsArgumentException() { @@ -655,13 +687,12 @@ public void RecognizeCommand_Run_NoCaptureDeviceAvailable_ThrowsInvalidOperation // --- Disposal ordering --- - /// Test that the recognizer is disposed exactly once after a successful file-input run. + /// Test that both the engine and the session are disposed exactly once after a successful file-input run. [Fact] - public void RecognizeCommand_Run_Success_DisposesRecognizerOnce() + public void RecognizeCommand_Run_Success_DisposesEngineAndSessionOnce() { var catalog = CreateCatalogWithModel(); - var recognizer = new FakeSpeechRecognizer(); - catalog.CreateRecognizerOverride = (_, _, _) => recognizer; + var (session, engine) = WireRecognizer(catalog); var wavPath = WriteMinimalWavFile(); try { @@ -669,7 +700,8 @@ public void RecognizeCommand_Run_Success_DisposesRecognizerOnce() RecognizeCommand.Run(context, catalog, new AudioDeviceFactory()); - Assert.Equal(1, recognizer.DisposeCallCount); + Assert.Equal(1, session.DisposeCallCount); + Assert.Equal(1, engine.DisposeCallCount); } finally { diff --git a/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/SilenceTimeoutRecognizerSessionTests.cs b/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/SilenceTimeoutRecognizerSessionTests.cs index 11291fe..372d9d9 100644 --- a/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/SilenceTimeoutRecognizerSessionTests.cs +++ b/test/DemaConsulting.Speech.Cli.Tests/Commands/RecognitionCommandSubsystem/SilenceTimeoutRecognizerSessionTests.cs @@ -24,406 +24,232 @@ namespace DemaConsulting.Speech.Cli.Tests.Commands.RecognitionCommandSubsystem; /// /// Unit tests for , using -/// and so every scenario +/// and so every scenario /// runs deterministically with no real wall-clock delay. /// +/// +/// is a stateless async-iterator +/// decorator with no background thread and no / +/// surface, so (unlike the previous synchronous, thread-based implementation) there are no +/// thread-race or disposal-ordering scenarios to cover here. Each test drives the decorator by +/// obtaining its directly and calling +/// MoveNextAsync() without immediately awaiting it: per the type's own remarks, the +/// compiler-generated iterator runs synchronously up to its first genuine suspension point +/// (the race between the inner session's next result and the idle-timeout delay), so +/// is already armed by the time the unawaited +/// ValueTask is returned. +/// public sealed class SilenceTimeoutRecognizerSessionTests { - /// Test that construction arms the idle timer once, with the given timeout. + /// Test that a null session is rejected. [Fact] - public void SilenceTimeoutRecognizerSession_Construct_ArmsTimerWithGivenTimeout() + public void SilenceTimeoutRecognizerSession_Construct_NullSession_ThrowsArgumentNullException() { - var recognizer = new FakeSpeechRecognizer(); - var timeProvider = new FakeTimeProvider(); - - using var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromSeconds(5), timeProvider); - - Assert.NotNull(timeProvider.LastTimer); - Assert.Equal(1, timeProvider.LastTimer.ChangeCallCount); - Assert.Equal(TimeSpan.FromSeconds(5), timeProvider.LastTimer.LastDueTime); - } - - /// Test that a partial (non-final) result re-arms the idle timer. - [Fact] - public void SilenceTimeoutRecognizerSession_PartialResultReceived_ResetsIdleTimer() - { - var recognizer = new FakeSpeechRecognizer(); - var timeProvider = new FakeTimeProvider(); - using var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromSeconds(5), timeProvider); - - recognizer.RaiseResult("hel", isFinal: false); - - Assert.Equal(2, timeProvider.LastTimer!.ChangeCallCount); + Assert.Throws( + () => new SilenceTimeoutRecognizerSession(null!, TimeSpan.FromSeconds(5), new FakeTimeProvider())); } - /// Test that a final result re-arms the idle timer. + /// Test that a non-positive idle timeout is rejected. [Fact] - public void SilenceTimeoutRecognizerSession_FinalResultReceived_ResetsIdleTimer() + public void SilenceTimeoutRecognizerSession_Construct_NonPositiveIdleTimeout_ThrowsArgumentOutOfRangeException() { - var recognizer = new FakeSpeechRecognizer(); - var timeProvider = new FakeTimeProvider(); - using var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromSeconds(5), timeProvider); - - recognizer.RaiseResult("hello", isFinal: true); + var session = new FakeRecognitionSession(); - Assert.Equal(2, timeProvider.LastTimer!.ChangeCallCount); + Assert.Throws( + () => new SilenceTimeoutRecognizerSession(session, TimeSpan.Zero, new FakeTimeProvider())); } - /// Test that the idle timer firing with no reset stops the recognizer and raises TimedOut. + /// Test that a non-positive start timeout is rejected. [Fact] - public void SilenceTimeoutRecognizerSession_IdleTimerFires_StopsRecognizerAndRaisesTimedOut() + public void SilenceTimeoutRecognizerSession_Construct_NonPositiveStartTimeout_ThrowsArgumentOutOfRangeException() { - var recognizer = new FakeSpeechRecognizer(); - var timeProvider = new FakeTimeProvider(); - using var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromSeconds(5), timeProvider); - var timedOutRaised = false; - session.TimedOut += (_, _) => timedOutRaised = true; + var session = new FakeRecognitionSession(); - timeProvider.LastTimer!.Fire(); - - Assert.Equal(1, recognizer.StopCallCount); - Assert.True(timedOutRaised); + Assert.Throws( + () => new SilenceTimeoutRecognizerSession( + session, + TimeSpan.FromSeconds(5), + new FakeTimeProvider(), + startTimeout: TimeSpan.Zero)); } - /// Test that resetting before the timer fires prevents a stale timeout from acting (defensive: a later Fire still only calls Stop once here since the fake never auto-cancels a prior "due" state, so this proves the reset call count increased and Stop still reflects a genuine Fire call). + /// Test that a null time provider is accepted, defaulting to the system clock. [Fact] - public void SilenceTimeoutRecognizerSession_ResetThenFire_StopsOnlyOnActualFire() + public void SilenceTimeoutRecognizerSession_Construct_NullTimeProvider_DoesNotThrow() { - var recognizer = new FakeSpeechRecognizer(); - var timeProvider = new FakeTimeProvider(); - using var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromSeconds(5), timeProvider); + var session = new FakeRecognitionSession(); - recognizer.RaiseResult("still talking", isFinal: false); - Assert.Equal(0, recognizer.StopCallCount); + var exception = Record.Exception(() => new SilenceTimeoutRecognizerSession(session, TimeSpan.FromMinutes(10))); - timeProvider.LastTimer!.Fire(); - Assert.Equal(1, recognizer.StopCallCount); + Assert.Null(exception); } - /// Test that Dispose unsubscribes and disposes the timer, and is idempotent. + /// Test that enumeration, with no start timeout given, arms the idle timer with the idle timeout. [Fact] - public void SilenceTimeoutRecognizerSession_Dispose_UnsubscribesAndDisposesTimer() + public async Task SilenceTimeoutRecognizerSession_GetResultsAsync_StartTimeoutOmitted_ArmsTimerWithIdleTimeout() { - var recognizer = new FakeSpeechRecognizer(); + var session = new FakeRecognitionSession(); var timeProvider = new FakeTimeProvider(); - var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromSeconds(5), timeProvider); + var wrapper = new SilenceTimeoutRecognizerSession(session, TimeSpan.FromSeconds(7), timeProvider); - session.Dispose(); - session.Dispose(); + using var cts = new CancellationTokenSource(); + await using var enumerator = wrapper.GetResultsAsync(cts.Token).GetAsyncEnumerator(cts.Token); + var moveNextTask = enumerator.MoveNextAsync(); - Assert.True(timeProvider.LastTimer!.IsDisposed); + Assert.NotNull(timeProvider.LastTimer); + Assert.Equal(TimeSpan.FromSeconds(7), timeProvider.LastTimer.LastDueTime); - // A result raised after disposal must not re-arm the (now disposed) timer. - var changeCallCountAfterDispose = timeProvider.LastTimer.ChangeCallCount; - recognizer.RaiseResult("late", isFinal: true); - Assert.Equal(changeCallCountAfterDispose, timeProvider.LastTimer.ChangeCallCount); + // Drain cleanly so the pending move-next settles before the test ends. + await session.StopAsync(TestContext.Current.CancellationToken); + Assert.False(await moveNextTask); } - /// - /// Test that a timer fire racing a concurrent - /// call (simulated deterministically: dispose first, then invoke the callback the fake - /// timer would otherwise have fired) does not call Stop() or raise TimedOut - /// on the already-disposed session. - /// + /// Test that enumeration, with a start timeout given, arms the idle timer with the start timeout. [Fact] - public void SilenceTimeoutRecognizerSession_FireAfterDispose_DoesNotCallStopOrRaiseTimedOut() + public async Task SilenceTimeoutRecognizerSession_GetResultsAsync_StartTimeoutGiven_ArmsTimerWithStartTimeout() { - var recognizer = new FakeSpeechRecognizer(); + var session = new FakeRecognitionSession(); var timeProvider = new FakeTimeProvider(); - var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromSeconds(5), timeProvider); - var timedOutRaised = false; - session.TimedOut += (_, _) => timedOutRaised = true; - - session.Dispose(); - timeProvider.LastTimer!.Fire(); - - Assert.Equal(0, recognizer.StopCallCount); - Assert.False(timedOutRaised); - } + var wrapper = new SilenceTimeoutRecognizerSession( + session, + TimeSpan.FromSeconds(5), + timeProvider, + startTimeout: TimeSpan.FromSeconds(2)); - /// - /// Test that many rounds of a concurrent recognizer result event racing a concurrent - /// call - on two genuine background - /// threads, released simultaneously via a - never throws (in - /// particular, never lets ObjectDisposedException escape from the disposed timer), - /// regardless of which thread wins the race. - /// - [Fact] - public void SilenceTimeoutRecognizerSession_ConcurrentResultReceivedAndDispose_DoesNotThrow() - { - var iterationsCompleted = 0; - for (var i = 0; i < 200; i++) - { - var recognizer = new FakeSpeechRecognizer(); - var timeProvider = new FakeTimeProvider(); - var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromSeconds(5), timeProvider); - - using var barrier = new Barrier(2); - var disposeThread = new Thread(() => - { - barrier.SignalAndWait(); - session.Dispose(); - }); - var resultThread = new Thread(() => - { - barrier.SignalAndWait(); - recognizer.RaiseResult("still talking", isFinal: false); - }); - - disposeThread.Start(); - resultThread.Start(); - disposeThread.Join(); - resultThread.Join(); - iterationsCompleted++; - } - - Assert.Equal(200, iterationsCompleted); - } + using var cts = new CancellationTokenSource(); + await using var enumerator = wrapper.GetResultsAsync(cts.Token).GetAsyncEnumerator(cts.Token); + var moveNextTask = enumerator.MoveNextAsync(); - /// - /// Test that many rounds of a concurrent idle-timer fire racing a concurrent - /// call - on two genuine background - /// threads, released simultaneously via a - never throws, and the - /// recognizer's Stop() is called at most once regardless of which thread wins the - /// race (proving the idle-timer callback never acts after disposal has already - /// completed, and never races a half-torn-down session). - /// - [Fact] - public void SilenceTimeoutRecognizerSession_ConcurrentTimerFireAndDispose_DoesNotThrowOrActTwice() - { - for (var i = 0; i < 200; i++) - { - var recognizer = new FakeSpeechRecognizer(); - var timeProvider = new FakeTimeProvider(); - var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromSeconds(5), timeProvider); - var timer = timeProvider.LastTimer!; - - using var barrier = new Barrier(2); - var disposeThread = new Thread(() => - { - barrier.SignalAndWait(); - session.Dispose(); - }); - var fireThread = new Thread(() => - { - barrier.SignalAndWait(); - timer.Fire(); - }); - - disposeThread.Start(); - fireThread.Start(); - disposeThread.Join(); - fireThread.Join(); - - Assert.InRange(recognizer.StopCallCount, 0, 1); - } - } + Assert.NotNull(timeProvider.LastTimer); + Assert.Equal(TimeSpan.FromSeconds(2), timeProvider.LastTimer.LastDueTime); - /// - /// Test that the idle timer firing does not deadlock when Stop() blocks waiting - /// for a background decode thread that itself raises ResultReceived (mirroring the - /// real recognizer's own "Stop() drains in-flight audio" contract) - proving the session's - /// idle-timer callback no longer holds a lock across Stop() that the reentrant - /// ResultReceived handler also needs. - /// - [Fact] - public void SilenceTimeoutRecognizerSession_StopBlocksOnReentrantResultReceived_DoesNotDeadlock() - { - var recognizer = new FakeSpeechRecognizer(); - var timeProvider = new FakeTimeProvider(); - using var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromSeconds(5), timeProvider); - - recognizer.OnStop = _ => - { - // Simulate the real recognizer's background decode thread draining already-captured - // audio during Stop() and raising a result from that separate thread, blocking Stop() - // until it has been fully handled - the exact interleaving that deadlocks if the idle - // callback still holds its lock while calling Stop(). - var decodeThread = new Thread(() => recognizer.RaiseResult("draining", isFinal: true)); - decodeThread.Start(); - decodeThread.Join(); - }; - - var idleThread = new Thread(() => timeProvider.LastTimer!.Fire()); - idleThread.Start(); - - Assert.True(idleThread.Join(TimeSpan.FromSeconds(10)), "OnIdle deadlocked against the reentrant ResultReceived raised from Stop()."); - Assert.Equal(1, recognizer.StopCallCount); + await session.StopAsync(TestContext.Current.CancellationToken); + Assert.False(await moveNextTask); } - /// - /// Test that waits for an in-flight - /// idle-timer callback's Stop() call and TimedOut raise to finish before - /// returning, even though that callback no longer holds its lock while making them - so a - /// caller can safely tear down state a TimedOut handler depends on (for example a - /// synchronization primitive) immediately after Dispose() returns. - /// + /// Test that the first yielded result re-arms the timer with the idle timeout, not the start timeout. [Fact] - public void SilenceTimeoutRecognizerSession_DisposeDuringInFlightTimedOut_WaitsForTimedOutToComplete() + public async Task SilenceTimeoutRecognizerSession_GetResultsAsync_FirstResultReceived_ReArmsWithIdleTimeoutNotStartTimeout() { - var recognizer = new FakeSpeechRecognizer(); + var session = new FakeRecognitionSession(); var timeProvider = new FakeTimeProvider(); - var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromSeconds(5), timeProvider); - - using var stopCallStarted = new ManualResetEventSlim(initialState: false); - using var releaseStopCall = new ManualResetEventSlim(initialState: false); - var timedOutHandlerRan = false; - recognizer.OnStop = _ => - { - stopCallStarted.Set(); - releaseStopCall.Wait(); - }; - session.TimedOut += (_, _) => timedOutHandlerRan = true; - - // Fire the idle timer on a background thread; it will block inside Stop() until this - // test thread releases it below. - var idleThread = new Thread(() => timeProvider.LastTimer!.Fire()); - idleThread.Start(); - Assert.True(stopCallStarted.Wait(TimeSpan.FromSeconds(10), TestContext.Current.CancellationToken), "OnIdle never reached Stop()."); - - // Start Dispose() concurrently while OnIdle is still blocked inside Stop(); it must not - // return until Stop() unblocks and TimedOut has been raised. - var disposeThread = new Thread(session.Dispose); - disposeThread.Start(); - Assert.False(disposeThread.Join(TimeSpan.FromMilliseconds(200)), "Dispose() returned while TimedOut was still in flight."); - - releaseStopCall.Set(); - - Assert.True(disposeThread.Join(TimeSpan.FromSeconds(10)), "Dispose() never returned after Stop() was unblocked."); - Assert.True(idleThread.Join(TimeSpan.FromSeconds(10))); - Assert.True(timedOutHandlerRan); - } - - /// - /// Test that a null recognizer is rejected. - /// - [Fact] - public void SilenceTimeoutRecognizerSession_Construct_NullRecognizer_ThrowsArgumentNullException() - { - Assert.Throws( - () => new SilenceTimeoutRecognizerSession(null!, TimeSpan.FromSeconds(5), new FakeTimeProvider())); - } - - /// Test that a non-positive idle timeout is rejected. - [Fact] - public void SilenceTimeoutRecognizerSession_Construct_NonPositiveTimeout_ThrowsArgumentOutOfRangeException() - { - var recognizer = new FakeSpeechRecognizer(); - Assert.Throws( - () => new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.Zero, new FakeTimeProvider())); - } - - /// Test that a null timeProvider defaults to TimeProvider.System without throwing. - [Fact] - public void SilenceTimeoutRecognizerSession_Construct_NullTimeProvider_UsesSystemTimeProvider() - { - var recognizer = new FakeSpeechRecognizer(); - - using var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromMinutes(10)); + var wrapper = new SilenceTimeoutRecognizerSession( + session, + TimeSpan.FromSeconds(5), + timeProvider, + startTimeout: TimeSpan.FromSeconds(2)); - // Constructing does not throw and does not itself call Stop/dispose prematurely. - Assert.Equal(0, recognizer.StopCallCount); - } + using var cts = new CancellationTokenSource(); + await using var enumerator = wrapper.GetResultsAsync(cts.Token).GetAsyncEnumerator(cts.Token); + var firstMoveNextTask = enumerator.MoveNextAsync(); + Assert.Equal(TimeSpan.FromSeconds(2), timeProvider.LastTimer!.LastDueTime); - /// Test that construction arms the idle timer with idleTimeout when startTimeout is omitted. - [Fact] - public void SilenceTimeoutRecognizerSession_Construct_StartTimeoutOmitted_ArmsTimerWithSilenceTimeout() - { - var recognizer = new FakeSpeechRecognizer(); - var timeProvider = new FakeTimeProvider(); + session.RaiseResult("hel", isFinal: false); + Assert.True(await firstMoveNextTask); + Assert.Equal("hel", enumerator.Current.Result.Text); - using var session = new SilenceTimeoutRecognizerSession(recognizer, TimeSpan.FromSeconds(7), timeProvider); + var secondMoveNextTask = enumerator.MoveNextAsync(); + Assert.Equal(TimeSpan.FromSeconds(5), timeProvider.LastTimer.LastDueTime); - Assert.NotNull(timeProvider.LastTimer); - Assert.Equal(1, timeProvider.LastTimer.ChangeCallCount); - Assert.Equal(TimeSpan.FromSeconds(7), timeProvider.LastTimer.LastDueTime); + await session.StopAsync(TestContext.Current.CancellationToken); + Assert.False(await secondMoveNextTask); } - /// Test that construction arms the idle timer with the distinct startTimeout value when given. + /// Test that a second yielded result keeps the timer re-armed with the idle timeout. [Fact] - public void SilenceTimeoutRecognizerSession_Construct_StartTimeoutGiven_ArmsTimerWithStartTimeout() + public async Task SilenceTimeoutRecognizerSession_GetResultsAsync_SecondResultReceived_StaysOnIdleTimeout() { - var recognizer = new FakeSpeechRecognizer(); + var session = new FakeRecognitionSession(); var timeProvider = new FakeTimeProvider(); - - using var session = new SilenceTimeoutRecognizerSession( - recognizer, + var wrapper = new SilenceTimeoutRecognizerSession( + session, TimeSpan.FromSeconds(5), timeProvider, startTimeout: TimeSpan.FromSeconds(2)); - Assert.NotNull(timeProvider.LastTimer); - Assert.Equal(TimeSpan.FromSeconds(2), timeProvider.LastTimer.LastDueTime); - } - - /// Test that the first result received re-arms the timer with idleTimeout, not startTimeout. - [Fact] - public void SilenceTimeoutRecognizerSession_FirstResultReceived_ReArmsWithSilenceTimeoutNotStartTimeout() - { - var recognizer = new FakeSpeechRecognizer(); - var timeProvider = new FakeTimeProvider(); - using var session = new SilenceTimeoutRecognizerSession( - recognizer, - TimeSpan.FromSeconds(5), - timeProvider, - startTimeout: TimeSpan.FromSeconds(2)); + using var cts = new CancellationTokenSource(); + await using var enumerator = wrapper.GetResultsAsync(cts.Token).GetAsyncEnumerator(cts.Token); + var firstMoveNextTask = enumerator.MoveNextAsync(); - recognizer.RaiseResult("hel", isFinal: false); + session.RaiseResult("hel", isFinal: false); + Assert.True(await firstMoveNextTask); + var secondMoveNextTask = enumerator.MoveNextAsync(); + session.RaiseResult("hello", isFinal: true); + Assert.True(await secondMoveNextTask); Assert.Equal(TimeSpan.FromSeconds(5), timeProvider.LastTimer!.LastDueTime); + + var thirdMoveNextTask = enumerator.MoveNextAsync(); + Assert.Equal(TimeSpan.FromSeconds(5), timeProvider.LastTimer.LastDueTime); + + await session.StopAsync(TestContext.Current.CancellationToken); + Assert.False(await thirdMoveNextTask); } - /// Test that a second result received stays re-armed with idleTimeout. + /// Test that a timeout before any result stops the session and raises TimedOut exactly once. [Fact] - public void SilenceTimeoutRecognizerSession_SecondResultReceived_StaysOnSilenceTimeout() + public async Task SilenceTimeoutRecognizerSession_GetResultsAsync_TimeoutBeforeAnyResult_StopsSessionAndRaisesTimedOutOnce() { - var recognizer = new FakeSpeechRecognizer(); + var session = new FakeRecognitionSession(); var timeProvider = new FakeTimeProvider(); - using var session = new SilenceTimeoutRecognizerSession( - recognizer, + var wrapper = new SilenceTimeoutRecognizerSession( + session, TimeSpan.FromSeconds(5), timeProvider, startTimeout: TimeSpan.FromSeconds(2)); + var timedOutCount = 0; + wrapper.TimedOut += (_, _) => timedOutCount++; - recognizer.RaiseResult("hel", isFinal: false); - recognizer.RaiseResult("hello", isFinal: true); + using var cts = new CancellationTokenSource(); + await using var enumerator = wrapper.GetResultsAsync(cts.Token).GetAsyncEnumerator(cts.Token); + var moveNextTask = enumerator.MoveNextAsync(); - Assert.Equal(TimeSpan.FromSeconds(5), timeProvider.LastTimer!.LastDueTime); - Assert.Equal(3, timeProvider.LastTimer.ChangeCallCount); + timeProvider.LastTimer!.Fire(); + + Assert.False(await moveNextTask); + Assert.Equal(1, session.StopCallCount); + Assert.Equal(1, timedOutCount); } - /// Test that the idle timer firing before any result uses startTimeout to stop the recognizer. + /// Test that a timeout after a result stops the session and raises TimedOut exactly once. [Fact] - public void SilenceTimeoutRecognizerSession_IdleTimerFiresBeforeFirstResult_StopsRecognizerAndRaisesTimedOut() + public async Task SilenceTimeoutRecognizerSession_GetResultsAsync_TimeoutAfterResult_StopsSessionAndRaisesTimedOutOnce() { - var recognizer = new FakeSpeechRecognizer(); + var session = new FakeRecognitionSession(); var timeProvider = new FakeTimeProvider(); - using var session = new SilenceTimeoutRecognizerSession( - recognizer, - TimeSpan.FromSeconds(5), - timeProvider, - startTimeout: TimeSpan.FromSeconds(2)); - var timedOutRaised = false; - session.TimedOut += (_, _) => timedOutRaised = true; + var wrapper = new SilenceTimeoutRecognizerSession(session, TimeSpan.FromSeconds(5), timeProvider); + var timedOutCount = 0; + wrapper.TimedOut += (_, _) => timedOutCount++; + using var cts = new CancellationTokenSource(); + await using var enumerator = wrapper.GetResultsAsync(cts.Token).GetAsyncEnumerator(cts.Token); + var firstMoveNextTask = enumerator.MoveNextAsync(); + + session.RaiseResult("hel", isFinal: false); + Assert.True(await firstMoveNextTask); + + var secondMoveNextTask = enumerator.MoveNextAsync(); timeProvider.LastTimer!.Fire(); - Assert.Equal(1, recognizer.StopCallCount); - Assert.True(timedOutRaised); + Assert.False(await secondMoveNextTask); + Assert.Equal(1, session.StopCallCount); + Assert.Equal(1, timedOutCount); } - /// Test that a non-positive startTimeout is rejected. + /// Test that cancelling the token passed to GetResultsAsync propagates a cancellation from the enumeration. [Fact] - public void SilenceTimeoutRecognizerSession_Construct_NonPositiveStartTimeout_ThrowsArgumentOutOfRangeException() + public async Task SilenceTimeoutRecognizerSession_GetResultsAsync_CancellationRequested_PropagatesOperationCanceledException() { - var recognizer = new FakeSpeechRecognizer(); + var session = new FakeRecognitionSession(); var timeProvider = new FakeTimeProvider(); + var wrapper = new SilenceTimeoutRecognizerSession(session, TimeSpan.FromSeconds(5), timeProvider); - Assert.Throws( - () => new SilenceTimeoutRecognizerSession( - recognizer, - TimeSpan.FromSeconds(5), - timeProvider, - startTimeout: TimeSpan.Zero)); + using var cts = new CancellationTokenSource(); + await using var enumerator = wrapper.GetResultsAsync(cts.Token).GetAsyncEnumerator(cts.Token); + var moveNextTask = enumerator.MoveNextAsync(); + + await cts.CancelAsync(); + + await Assert.ThrowsAsync(async () => await moveNextTask); } } diff --git a/test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/FakeSpeechSynthesizerEngine.cs b/test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/FakeSpeechSynthesizerEngine.cs new file mode 100644 index 0000000..e8dfd10 --- /dev/null +++ b/test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/FakeSpeechSynthesizerEngine.cs @@ -0,0 +1,94 @@ +// Copyright (c) DEMA Consulting +// +// Permission is hereby granted, free of charge, to any person obtaining a copy +// of this software and associated documentation files (the "Software"), to deal +// in the Software without restriction, including without limitation the rights +// to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +// copies of the Software, and to permit persons to whom the Software is +// furnished to do so, subject to the following conditions: +// +// The above copyright notice and this permission notice shall be included in all +// copies or substantial portions of the Software. +// +// THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +// IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +// FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +// AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +// LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +// OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +// SOFTWARE. + +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.SynthesisSubsystem; + +namespace DemaConsulting.Speech.Cli.Tests.Commands.SynthesisCommandSubsystem; + +/// +/// Deterministic, in-memory fake used by speak/ +/// ask command unit tests, wrapping a single pre-configured +/// that returns. +/// +internal sealed class FakeSpeechSynthesizerEngine : ISpeechSynthesizerEngine +{ + /// The session returns. + private readonly FakeSynthesisSession _session; + + /// + /// Initializes a new instance of the class. + /// + /// The session returns. Must not be null. + public FakeSpeechSynthesizerEngine(FakeSynthesisSession session) + { + ArgumentNullException.ThrowIfNull(session); + _session = session; + } + + /// Gets the number of times was called. + public int CreateSessionAsyncCallCount { get; private set; } + + /// Gets the number of times was called. + public int DisposeCallCount { get; private set; } + + /// Gets the playback device passed to the most recent call. + public IAudioPlaybackDevice? LastDevice { get; private set; } + + /// Gets or sets an exception to throw from , or for none. + public Exception? CreateSessionException { get; set; } + + /// + public bool IsAvailable { get; set; } = true; + + /// + public Task CreateSessionAsync( + IAudioPlaybackDevice device, + CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(device); + + CreateSessionAsyncCallCount++; + LastDevice = device; + cancellationToken.ThrowIfCancellationRequested(); + + if (CreateSessionException is not null) + { + throw CreateSessionException; + } + + return Task.FromResult(_session); + } + + /// + public Task SpeakAsync(IAudioPlaybackDevice device, string text, CancellationToken cancellationToken = default) => + throw new NotSupportedException($"{nameof(FakeSpeechSynthesizerEngine)} only supports {nameof(CreateSessionAsync)}."); + + /// + public Task> SynthesizeAsync(string text, CancellationToken cancellationToken = default) => + throw new NotSupportedException($"{nameof(FakeSpeechSynthesizerEngine)} only supports {nameof(CreateSessionAsync)}."); + + /// + public ValueTask DisposeAsync() + { + DisposeCallCount++; + return ValueTask.CompletedTask; + } +} diff --git a/test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/FakeSpeechSynthesizer.cs b/test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/FakeSynthesisSession.cs similarity index 64% rename from test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/FakeSpeechSynthesizer.cs rename to test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/FakeSynthesisSession.cs index 7db724a..5756415 100644 --- a/test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/FakeSpeechSynthesizer.cs +++ b/test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/FakeSynthesisSession.cs @@ -23,18 +23,18 @@ namespace DemaConsulting.Speech.Cli.Tests.Commands.SynthesisCommandSubsystem; /// -/// Deterministic, in-memory fake used by speak -/// command unit tests, so no test depends on a real, native sherpa-onnx engine. +/// Deterministic, in-memory fake used by speak/ +/// ask command unit tests, so no test depends on a real, native sherpa-onnx engine. /// -internal sealed class FakeSpeechSynthesizer : ISpeechSynthesizer +internal sealed class FakeSynthesisSession : ISynthesisSession { /// Gets the ordered list of text values passed to . public List SpeakAsyncCalls { get; } = []; - /// Gets the number of times was called. + /// Gets the number of times was called. public int StopCallCount { get; private set; } - /// Gets the number of times was called. + /// Gets the number of times was called. public int DisposeCallCount { get; private set; } /// Gets or sets an exception to throw from , or for none. @@ -53,12 +53,14 @@ internal sealed class FakeSpeechSynthesizer : ISpeechSynthesizer public bool IsAvailable { get; set; } = true; /// - public IAsyncEnumerable SynthesizeStreamAsync(string text, CancellationToken cancellationToken = default) => - throw new NotSupportedException($"{nameof(FakeSpeechSynthesizer)} only supports {nameof(SpeakAsync)}."); + public SynthesisSessionState State { get; private set; } = SynthesisSessionState.Created; /// - public Task PlayStreamAsync(IAsyncEnumerable stream, CancellationToken cancellationToken = default) => - throw new NotSupportedException($"{nameof(FakeSpeechSynthesizer)} only supports {nameof(SpeakAsync)}."); + public event EventHandler? StateChanged; + + /// + public Task> SynthesizeAsync(string text, CancellationToken cancellationToken = default) => + throw new NotSupportedException($"{nameof(FakeSynthesisSession)} only supports {nameof(SpeakAsync)}."); /// public async Task SpeakAsync(string text, CancellationToken cancellationToken = default) @@ -66,8 +68,11 @@ public async Task SpeakAsync(string text, CancellationToken cancellationToken = SpeakAsyncCalls.Add(text); cancellationToken.ThrowIfCancellationRequested(); + SetState(SynthesisSessionState.Running); + if (SpeakAsyncException is not null) { + SetState(SynthesisSessionState.Faulted); throw SpeakAsyncException; } @@ -75,11 +80,30 @@ public async Task SpeakAsync(string text, CancellationToken cancellationToken = { await SpeakAsyncAwaiter().ConfigureAwait(false); } + + SetState(SynthesisSessionState.Stopped); } /// - public void Stop() => StopCallCount++; + public Task StopAsync(CancellationToken cancellationToken = default) + { + StopCallCount++; + SetState(SynthesisSessionState.Stopped); + return Task.CompletedTask; + } /// - public void Dispose() => DisposeCallCount++; + public ValueTask DisposeAsync() + { + DisposeCallCount++; + return ValueTask.CompletedTask; + } + + /// Updates and raises . + private void SetState(SynthesisSessionState newState) + { + var previous = State; + State = newState; + StateChanged?.Invoke(this, new SessionStateChangedEventArgs(previous, newState)); + } } diff --git a/test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/SpeakCommandTests.cs b/test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/SpeakCommandTests.cs index 3803ec9..bdb67a4 100644 --- a/test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/SpeakCommandTests.cs +++ b/test/DemaConsulting.Speech.Cli.Tests/Commands/SynthesisCommandSubsystem/SpeakCommandTests.cs @@ -26,13 +26,15 @@ using DemaConsulting.Speech.Cli.Tests.Commands.DeviceCommandsSubsystem; using DemaConsulting.Speech.Cli.Tests.Commands.ModelCommandsSubsystem; using DemaConsulting.Speech.ModelManagementSubsystem; +using DemaConsulting.Speech.SynthesisSubsystem; namespace DemaConsulting.Speech.Cli.Tests.Commands.SynthesisCommandSubsystem; /// /// Unit tests for , using , -/// , and fake audio device probes so every scenario runs -/// deterministically with no real catalog, network access, native engine, or audio hardware. +/// /, and fake +/// audio device probes so every scenario runs deterministically with no real catalog, network +/// access, native engine, or audio hardware. /// [Collection("Sequential")] public sealed class SpeakCommandTests @@ -51,6 +53,20 @@ private static FakeCliModelCatalog CreateCatalogWithModel( return catalog; } + /// + /// Creates a wrapped in a , + /// and wires the catalog's + /// to return the engine. + /// + private static (FakeSynthesisSession Session, FakeSpeechSynthesizerEngine Engine) WireSynthesizer( + FakeCliModelCatalog catalog) + { + var session = new FakeSynthesisSession(); + var engine = new FakeSpeechSynthesizerEngine(session); + catalog.CreateSynthesizerEngineOverride = (_, _, _) => Task.FromResult(engine); + return (session, engine); + } + // --- ParseArguments --- /// Test that --tts-model is required. @@ -179,15 +195,14 @@ public async Task SpeakCommand_RunAsync_NotDownloadedModel_ThrowsArgumentExcepti // --- --tts-param validation wiring --- - /// Test that a valid --tts-param is forwarded to CreateSynthesizer's parameterValues argument. + /// Test that a valid --tts-param is forwarded to CreateSynthesizerEngineAsync's parameterValues argument. [Fact] - public async Task SpeakCommand_RunAsync_ValidParam_ForwardsToCreateSynthesizer() + public async Task SpeakCommand_RunAsync_ValidParam_ForwardsToCreateSynthesizerEngine() { var rateParameter = new NumericParameter( "rate", "Rate", "Speaking rate", new NumericParameterBounds(0.5, 2.0, 0.1, 1.0)); var catalog = CreateCatalogWithModel(parameters: [rateParameter]); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + WireSynthesizer(catalog); var factory = new FakePlaybackDeviceSource(new FakeAudioPlaybackDeviceProbe([OutputDevice])); using var context = Context.Create(["speak", "--tts-model", "model-1", "--text", "hi", "--tts-param", "rate=1.5"]); @@ -198,7 +213,7 @@ public async Task SpeakCommand_RunAsync_ValidParam_ForwardsToCreateSynthesizer() Assert.Equal(1.5, Assert.IsType(parameterValues["rate"])); } - /// Test that an invalid --tts-param value throws before any synthesizer is created. + /// Test that an invalid --tts-param value throws before any synthesizer engine is created. [Fact] public async Task SpeakCommand_RunAsync_InvalidParam_ThrowsArgumentException() { @@ -219,15 +234,14 @@ await Assert.ThrowsAsync( public async Task SpeakCommand_RunAsync_NoTags_StripsRecognizedTagExactly() { var catalog = CreateCatalogWithModel(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + var (session, _) = WireSynthesizer(catalog); var factory = new FakePlaybackDeviceSource(new FakeAudioPlaybackDeviceProbe([OutputDevice])); using var context = Context.Create( ["speak", "--tts-model", "model-1", "--text", "Hello [laughs] there", "--no-tags"]); await SpeakCommand.RunAsync(context, catalog, factory, CancellationToken.None); - var spokenText = Assert.Single(synthesizer.SpeakAsyncCalls); + var spokenText = Assert.Single(session.SpeakAsyncCalls); Assert.Equal("Hello there", spokenText); } @@ -236,14 +250,13 @@ public async Task SpeakCommand_RunAsync_NoTags_StripsRecognizedTagExactly() public async Task SpeakCommand_RunAsync_WithoutNoTags_PassesOriginalTextUnchanged() { var catalog = CreateCatalogWithModel(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + var (session, _) = WireSynthesizer(catalog); var factory = new FakePlaybackDeviceSource(new FakeAudioPlaybackDeviceProbe([OutputDevice])); using var context = Context.Create(["speak", "--tts-model", "model-1", "--text", "Hello [laughs] there"]); await SpeakCommand.RunAsync(context, catalog, factory, CancellationToken.None); - var spokenText = Assert.Single(synthesizer.SpeakAsyncCalls); + var spokenText = Assert.Single(session.SpeakAsyncCalls); Assert.Equal("Hello [laughs] there", spokenText); } @@ -258,13 +271,7 @@ public async Task SpeakCommand_RunAsync_Output_ConstructsWavFileDeviceFromPrefer { var catalog = CreateCatalogWithModel(); catalog.GetPreferredAudioFormatOverride = _ => AudioFormat.Mono(22050); - var synthesizer = new FakeSpeechSynthesizer(); - IAudioPlaybackDevice? capturedDevice = null; - catalog.CreateSynthesizerOverride = (_, device, _) => - { - capturedDevice = device; - return synthesizer; - }; + var (_, engine) = WireSynthesizer(catalog); // A probe that throws if enumerated, proving --output-audio never touches real device probes. var factory = new FakePlaybackDeviceSource(new ThrowingAudioPlaybackDeviceProbe()); var outputPath = Path.Join(Path.GetTempPath(), $"speak-test-{Guid.NewGuid():N}.wav"); @@ -275,9 +282,9 @@ public async Task SpeakCommand_RunAsync_Output_ConstructsWavFileDeviceFromPrefer await SpeakCommand.RunAsync(context, catalog, factory, CancellationToken.None); - Assert.NotNull(capturedDevice); - Assert.Equal(22050, capturedDevice.SampleRate); - Assert.Equal(1, capturedDevice.ChannelCount); + Assert.NotNull(engine.LastDevice); + Assert.Equal(22050, engine.LastDevice.SampleRate); + Assert.Equal(1, engine.LastDevice.ChannelCount); Assert.True(File.Exists(outputPath)); } finally @@ -289,7 +296,7 @@ public async Task SpeakCommand_RunAsync_Output_ConstructsWavFileDeviceFromPrefer } } - /// Test that an unknown --playback-device throws before any synthesizer is created. + /// Test that an unknown --playback-device throws before any synthesizer engine is created. [Fact] public async Task SpeakCommand_RunAsync_UnknownDevice_ThrowsArgumentException() { @@ -322,32 +329,33 @@ await Assert.ThrowsAsync( public async Task SpeakCommand_RunAsync_Canceled_ReportsErrorCleanly() { var catalog = CreateCatalogWithModel(); - var synthesizer = new FakeSpeechSynthesizer { SpeakAsyncException = new OperationCanceledException() }; - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + var (session, engine) = WireSynthesizer(catalog); + session.SpeakAsyncException = new OperationCanceledException(); var factory = new FakePlaybackDeviceSource(new FakeAudioPlaybackDeviceProbe([OutputDevice])); using var context = Context.Create(["speak", "--tts-model", "model-1", "--text", "hi"]); await SpeakCommand.RunAsync(context, catalog, factory, CancellationToken.None); Assert.Equal(1, context.ExitCode); - Assert.Equal(1, synthesizer.DisposeCallCount); + Assert.Equal(1, session.DisposeCallCount); + Assert.Equal(1, engine.DisposeCallCount); } // --- Disposal ordering --- - /// Test that the synthesizer is disposed exactly once after a successful speak. + /// Test that both the engine and the session are disposed exactly once after a successful speak. [Fact] - public async Task SpeakCommand_RunAsync_Success_DisposesSynthesizerOnce() + public async Task SpeakCommand_RunAsync_Success_DisposesEngineAndSessionOnce() { var catalog = CreateCatalogWithModel(); - var synthesizer = new FakeSpeechSynthesizer(); - catalog.CreateSynthesizerOverride = (_, _, _) => synthesizer; + var (session, engine) = WireSynthesizer(catalog); var factory = new FakePlaybackDeviceSource(new FakeAudioPlaybackDeviceProbe([OutputDevice])); using var context = Context.Create(["speak", "--tts-model", "model-1", "--text", "hi"]); await SpeakCommand.RunAsync(context, catalog, factory, CancellationToken.None); - Assert.Equal(1, synthesizer.DisposeCallCount); + Assert.Equal(1, session.DisposeCallCount); + Assert.Equal(1, engine.DisposeCallCount); Assert.Equal(0, context.ExitCode); } diff --git a/test/DemaConsulting.Speech.Cli.Tests/IntegrationTests.cs b/test/DemaConsulting.Speech.Cli.Tests/IntegrationTests.cs index dfd9e21..2cd6ca3 100644 --- a/test/DemaConsulting.Speech.Cli.Tests/IntegrationTests.cs +++ b/test/DemaConsulting.Speech.Cli.Tests/IntegrationTests.cs @@ -104,6 +104,7 @@ public void SpeechCli_HelpFlag_Provided_ListsAllSubcommands() Assert.Contains("doctor", output); Assert.Contains("speak", output); Assert.Contains("recognize", output); + Assert.Contains("ask", output); } /// diff --git a/test/DemaConsulting.Speech.Cli.Tests/ProgramTests.cs b/test/DemaConsulting.Speech.Cli.Tests/ProgramTests.cs index f6b3929..1255209 100644 --- a/test/DemaConsulting.Speech.Cli.Tests/ProgramTests.cs +++ b/test/DemaConsulting.Speech.Cli.Tests/ProgramTests.cs @@ -417,6 +417,32 @@ public void Program_Run_WithSpeakCommand_DoesNotThrowNotImplemented() } } + /// + /// Test that dispatching to ask without --tts-model reports a clean, + /// missing-model rather than + /// , proving the dispatch table wiring for the + /// eleventh, later-added ask command. Full ask coverage lives in + /// AskCommandTests. + /// + [Fact] + public void Program_Run_WithAskCommand_DoesNotThrowNotImplemented() + { + var originalError = Console.Error; + try + { + using var errWriter = new StringWriter(); + Console.SetError(errWriter); + using var context = Context.Create(["ask"]); + + var exception = Record.Exception(() => Program.Run(context)); + Assert.IsNotType(exception); + } + finally + { + Console.SetError(originalError); + } + } + /// /// Test that Run with short version flag -v displays version. /// diff --git a/test/DemaConsulting.Speech.Demo.Tests/Fakes/FakeRecognitionSession.cs b/test/DemaConsulting.Speech.Demo.Tests/Fakes/FakeRecognitionSession.cs new file mode 100644 index 0000000..b4ed507 --- /dev/null +++ b/test/DemaConsulting.Speech.Demo.Tests/Fakes/FakeRecognitionSession.cs @@ -0,0 +1,202 @@ +using System.Runtime.CompilerServices; +using DemaConsulting.Speech.RecognitionSubsystem; + +namespace DemaConsulting.Speech.Demo.Tests.Fakes; + +/// +/// Hand-written test double for , used instead of a mocking +/// library because the interface's streaming contract needs +/// real, test-controlled asynchronous suspension/resumption rather than a canned return +/// value. +/// +/// +/// Every signaling member below (, , +/// ) completes synchronously on the calling thread: 's +/// enumerator suspends on an unconfigured , whose +/// default continuation behavior runs inline on whichever thread signals it - so a test can +/// push a result and immediately assert the ViewModel observed it, exactly like the old +/// Raise.Event pattern this double replaces. +/// +public sealed class FakeRecognitionSession : IRecognitionSession +{ + /// The gate guarding every mutable field below. + private readonly object _gate = new(); + + /// Results pushed by a test but not yet consumed by an active enumeration. + private readonly List _pending = []; + + /// The signal an idle enumerator is awaiting, if one is suspended. + private TaskCompletionSource? _signal; + + /// Set once has been called. + private bool _completed; + + /// Set once has been called. + private Exception? _fault; + + /// + public bool IsAvailable { get; set; } = true; + + /// + public RecognitionSessionState State { get; private set; } = RecognitionSessionState.Created; + + /// + public event EventHandler? StateChanged; + + /// Gets the number of times was called. + public int StartCallCount { get; private set; } + + /// Gets the number of times was called. + public int StopCallCount { get; private set; } + + /// Gets the number of times was called. + public int DisposeCallCount { get; private set; } + + /// Gets or sets the exception throws, if any. + public Exception? StartException { get; set; } + + /// + /// Gets or sets a callback invoked at the start of , letting a + /// test observe exactly when a Stop was requested (for example, to flip a "still + /// listening" flag a fake device service checks). + /// + public Action? OnStopRequested { get; set; } + + /// + /// Raises , moving to . + /// + /// The state to transition to. + public void RaiseStateChanged(RecognitionSessionState current) + { + var previous = State; + State = current; + StateChanged?.Invoke(this, new SessionStateChangedEventArgs(previous, current)); + } + + /// + /// Pushes one result for a suspended or future enumeration + /// to observe. + /// + /// The result to push. + public void PushResult(SpeechRecognitionResult result) + { + lock (_gate) + { + _pending.Add(new SpeechRecognitionEvent(result)); + var signal = _signal; + _signal = null; + signal?.TrySetResult(true); + } + } + + /// Ends a enumeration cleanly, as a real Stop would. + public void Complete() + { + lock (_gate) + { + _completed = true; + var signal = _signal; + _signal = null; + signal?.TrySetResult(true); + } + } + + /// Faults a enumeration with the given exception. + /// The exception the enumerator should throw. + public void Fault(Exception exception) + { + lock (_gate) + { + _fault = exception; + var signal = _signal; + _signal = null; + signal?.TrySetResult(true); + } + } + + /// + public Task StartAsync(CancellationToken cancellationToken = default) + { + StartCallCount++; + + if (StartException is not null) + { + throw StartException; + } + + RaiseStateChanged(RecognitionSessionState.Running); + return Task.CompletedTask; + } + + /// + public Task StopAsync(CancellationToken cancellationToken = default) + { + StopCallCount++; + OnStopRequested?.Invoke(); + RaiseStateChanged(RecognitionSessionState.Stopped); + Complete(); + return Task.CompletedTask; + } + + /// + public async IAsyncEnumerable GetResultsAsync( + [EnumeratorCancellation] CancellationToken cancellationToken = default) + { + if (!IsAvailable) + { + throw new SpeechRecognizerUnavailableException("The fake recognition session is unavailable."); + } + + while (true) + { + SpeechRecognitionEvent? next; + Exception? fault; + bool completed; + Task? wait; + + lock (_gate) + { + next = _pending.Count > 0 ? _pending[0] : null; + if (next is not null) + { + _pending.RemoveAt(0); + } + + fault = _fault; + completed = _completed; + wait = null; + + if (next is null && fault is null && !completed) + { + _signal = new TaskCompletionSource(); + wait = _signal.Task; + } + } + + if (next is not null) + { + yield return next; + continue; + } + + if (fault is not null) + { + throw fault; + } + + if (completed) + { + yield break; + } + + await wait!.WaitAsync(cancellationToken).ConfigureAwait(false); + } + } + + /// + public ValueTask DisposeAsync() + { + DisposeCallCount++; + return ValueTask.CompletedTask; + } +} diff --git a/test/DemaConsulting.Speech.Demo.Tests/Fakes/FakeSpeechRecognizerEngine.cs b/test/DemaConsulting.Speech.Demo.Tests/Fakes/FakeSpeechRecognizerEngine.cs new file mode 100644 index 0000000..71e837f --- /dev/null +++ b/test/DemaConsulting.Speech.Demo.Tests/Fakes/FakeSpeechRecognizerEngine.cs @@ -0,0 +1,60 @@ +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.RecognitionSubsystem; + +namespace DemaConsulting.Speech.Demo.Tests.Fakes; + +/// +/// Hand-written test double for , used instead of a +/// mocking library so each created session can be a fully test-controlled +/// rather than a canned substitute return value. +/// +public sealed class FakeSpeechRecognizerEngine : ISpeechRecognizerEngine +{ + /// Gets the sessions created by every call to , in order. + public List CreatedSessions { get; } = []; + + /// + public bool IsAvailable { get; set; } = true; + + /// Gets the number of times was called. + public int CreateSessionCallCount { get; private set; } + + /// Gets the number of times was called. + public int DisposeCallCount { get; private set; } + + /// Gets or sets the exception throws, if any. + public Exception? CreateSessionException { get; set; } + + /// + /// Gets or sets a factory producing the session returned by each + /// call; defaults to a fresh + /// per call when unset. + /// + public Func? SessionFactory { get; set; } + + /// + public Task CreateSessionAsync( + IAudioCaptureDevice device, + CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(device); + + CreateSessionCallCount++; + + if (CreateSessionException is not null) + { + throw CreateSessionException; + } + + var session = SessionFactory?.Invoke(device) ?? new FakeRecognitionSession(); + CreatedSessions.Add(session); + return Task.FromResult(session); + } + + /// + public ValueTask DisposeAsync() + { + DisposeCallCount++; + return ValueTask.CompletedTask; + } +} diff --git a/test/DemaConsulting.Speech.Demo.Tests/Fakes/FakeSpeechSynthesizerEngine.cs b/test/DemaConsulting.Speech.Demo.Tests/Fakes/FakeSpeechSynthesizerEngine.cs new file mode 100644 index 0000000..cd7ed0e --- /dev/null +++ b/test/DemaConsulting.Speech.Demo.Tests/Fakes/FakeSpeechSynthesizerEngine.cs @@ -0,0 +1,68 @@ +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.SynthesisSubsystem; + +namespace DemaConsulting.Speech.Demo.Tests.Fakes; + +/// +/// Hand-written test double for , used instead of a +/// mocking library so each created session can be a fully test-controlled +/// rather than a canned substitute return value. +/// +public sealed class FakeSpeechSynthesizerEngine : ISpeechSynthesizerEngine +{ + /// Gets the sessions created by every call to , in order. + public List CreatedSessions { get; } = []; + + /// + public bool IsAvailable { get; set; } = true; + + /// Gets the number of times was called. + public int CreateSessionCallCount { get; private set; } + + /// Gets the number of times was called. + public int DisposeCallCount { get; private set; } + + /// Gets or sets the exception throws, if any. + public Exception? CreateSessionException { get; set; } + + /// + /// Gets or sets a factory producing the session returned by each + /// call; defaults to a fresh + /// per call when unset. + /// + public Func? SessionFactory { get; set; } + + /// + public Task CreateSessionAsync( + IAudioPlaybackDevice device, + CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(device); + + CreateSessionCallCount++; + + if (CreateSessionException is not null) + { + throw CreateSessionException; + } + + var session = SessionFactory?.Invoke(device) ?? new FakeSynthesisSession(); + CreatedSessions.Add(session); + return Task.FromResult(session); + } + + /// + public Task SpeakAsync(IAudioPlaybackDevice device, string text, CancellationToken cancellationToken = default) => + throw new NotSupportedException("Not used by SynthesisPanelViewModel, which only calls CreateSessionAsync."); + + /// + public Task> SynthesizeAsync(string text, CancellationToken cancellationToken = default) => + throw new NotSupportedException("Not used by SynthesisPanelViewModel, which only calls CreateSessionAsync."); + + /// + public ValueTask DisposeAsync() + { + DisposeCallCount++; + return ValueTask.CompletedTask; + } +} diff --git a/test/DemaConsulting.Speech.Demo.Tests/Fakes/FakeSynthesisSession.cs b/test/DemaConsulting.Speech.Demo.Tests/Fakes/FakeSynthesisSession.cs new file mode 100644 index 0000000..4650acc --- /dev/null +++ b/test/DemaConsulting.Speech.Demo.Tests/Fakes/FakeSynthesisSession.cs @@ -0,0 +1,135 @@ +using DemaConsulting.Speech.SynthesisSubsystem; + +namespace DemaConsulting.Speech.Demo.Tests.Fakes; + +/// +/// Hand-written test double for , used instead of a mocking +/// library so a test can fully control the lifecycle transitions a real session would raise +/// around , including cooperative cancellation driven by +/// rather than only the caller's own token. +/// +public sealed class FakeSynthesisSession : ISynthesisSession +{ + /// The source canceled by while a call is in flight. + private CancellationTokenSource? _speakCts; + + /// + public bool IsAvailable { get; set; } = true; + + /// + public SynthesisSessionState State { get; private set; } = SynthesisSessionState.Created; + + /// + public event EventHandler? StateChanged; + + /// Gets the texts passed to every call, in order. + public List SpeakTexts { get; } = []; + + /// Gets the number of times was called. + public int SpeakCallCount => SpeakTexts.Count; + + /// Gets the number of times was called. + public int StopCallCount { get; private set; } + + /// Gets the number of times was called. + public int DisposeCallCount { get; private set; } + + /// + /// Gets or sets the exception throws, if any, after its normal + /// Starting/Running transitions. A also + /// raises to + /// first, exactly like a real session reporting an unrecoverable fault. + /// + public Exception? SpeakException { get; set; } + + /// + /// Gets or sets the in-flight body awaits after its Running + /// transition; defaults to an immediately-completing no-op. Set to a delegate awaiting + /// on its token to simulate an indefinitely long utterance + /// that only unwinds when cancels it. + /// + public Func? SpeakImplementation { get; set; } + + /// + /// Gets or sets a callback invoked at the start of , letting a + /// test observe exactly when a Stop was requested. + /// + public Action? OnStopRequested { get; set; } + + /// + /// Raises , moving to . + /// + /// The state to transition to. + public void RaiseStateChanged(SynthesisSessionState current) + { + var previous = State; + State = current; + StateChanged?.Invoke(this, new SessionStateChangedEventArgs(previous, current)); + } + + /// + public async Task SpeakAsync(string text, CancellationToken cancellationToken = default) + { + ArgumentNullException.ThrowIfNull(text); + + SpeakTexts.Add(text); + + if (SpeakException is not null) + { + if (SpeakException is SynthesisSessionFaultedException) + { + RaiseStateChanged(SynthesisSessionState.Faulted); + } + + throw SpeakException; + } + + using var cts = CancellationTokenSource.CreateLinkedTokenSource(cancellationToken); + _speakCts = cts; + + RaiseStateChanged(SynthesisSessionState.Starting); + RaiseStateChanged(SynthesisSessionState.Running); + + try + { + if (SpeakImplementation is not null) + { + await SpeakImplementation(text, cts.Token).ConfigureAwait(false); + } + else + { + await Task.Yield(); + } + } + catch (OperationCanceledException) + { + RaiseStateChanged(SynthesisSessionState.Stopping); + RaiseStateChanged(SynthesisSessionState.Stopped); + _speakCts = null; + throw; + } + + _speakCts = null; + RaiseStateChanged(SynthesisSessionState.Stopped); + } + + /// + public Task> SynthesizeAsync(string text, CancellationToken cancellationToken = default) => + throw new NotSupportedException("Not used by SynthesisPanelViewModel, which only calls SpeakAsync."); + + /// + public Task StopAsync(CancellationToken cancellationToken = default) + { + StopCallCount++; + OnStopRequested?.Invoke(); + _speakCts?.Cancel(); + return Task.CompletedTask; + } + + /// + public ValueTask DisposeAsync() + { + DisposeCallCount++; + return ValueTask.CompletedTask; + } +} diff --git a/test/DemaConsulting.Speech.Demo.Tests/RecognitionPanelSubsystem/RecognitionPanelViewModelTests.cs b/test/DemaConsulting.Speech.Demo.Tests/RecognitionPanelSubsystem/RecognitionPanelViewModelTests.cs index c17a456..9e09459 100644 --- a/test/DemaConsulting.Speech.Demo.Tests/RecognitionPanelSubsystem/RecognitionPanelViewModelTests.cs +++ b/test/DemaConsulting.Speech.Demo.Tests/RecognitionPanelSubsystem/RecognitionPanelViewModelTests.cs @@ -51,6 +51,21 @@ private static IAudioCaptureDevice CaptureDevice(bool isAvailable = true) return device; } + /// + /// Builds a session-factory substitute whose LoadAsync returns each of the given + /// engines in order (the last is returned for any further call), for any model. + /// + /// The engine(s) to return from successive calls. + /// The composed substitute. + private static IRecognizerSessionFactory SessionFactory(params FakeSpeechRecognizerEngine[] engines) + { + var factory = Substitute.For(); + var tasks = engines.Select(engine => Task.FromResult(engine)).ToArray(); + factory.LoadAsync(Arg.Any(), Arg.Any()) + .Returns(tasks[0], tasks[1..]); + return factory; + } + /// /// Proves that the panel rejects any missing constructor dependency. /// @@ -150,7 +165,7 @@ public void RecognitionPanelViewModel_Refresh_ModelStillInstalled_PreservesSelec /// when invoked with nothing selected. /// [Fact] - public void RecognitionPanelViewModel_Start_NoModelSelected_ReportsErrorState() + public async Task RecognitionPanelViewModel_Start_NoModelSelected_ReportsErrorState() { // Arrange: a panel with no installed models, so nothing can be selected var viewModel = new RecognitionPanelViewModel( @@ -160,7 +175,7 @@ public void RecognitionPanelViewModel_Start_NoModelSelected_ReportsErrorState() // Act: invoke Start directly (bypassing the command's own CanExecute gate, which // already disables the button for this state) to prove the defensive guard behaves // honestly too - viewModel.StartCommand.Execute(null); + await viewModel.StartCommand.ExecuteAsync(null); // Assert: an honest, explanatory error - not an exception Assert.Equal(RecognitionPanelViewModel.NoModelSelectedMessage, viewModel.StatusMessage); @@ -172,7 +187,7 @@ public void RecognitionPanelViewModel_Start_NoModelSelected_ReportsErrorState() /// reports none available. /// [Fact] - public void RecognitionPanelViewModel_Start_NoCaptureDevice_ReportsErrorState() + public async Task RecognitionPanelViewModel_Start_NoCaptureDevice_ReportsErrorState() { // Arrange: an installed model but a machine with no usable capture device var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); @@ -183,7 +198,7 @@ public void RecognitionPanelViewModel_Start_NoCaptureDevice_ReportsErrorState() Catalog(descriptor), deviceService, DeviceSelection(), Substitute.For()); // Act: attempt to start - viewModel.StartCommand.Execute(null); + await viewModel.StartCommand.ExecuteAsync(null); // Assert: the honest device-unavailable outcome Assert.Equal(RecognitionPanelViewModel.NoCaptureDeviceMessage, viewModel.StatusMessage); @@ -192,85 +207,85 @@ public void RecognitionPanelViewModel_Start_NoCaptureDevice_ReportsErrorState() /// /// Proves that Start reports the honest "recognizer unavailable" outcome, and disposes - /// the unavailable recognizer, when the session seam cannot compose a working one. + /// the unavailable engine, when the session seam cannot compose a working one. /// [Fact] - public void RecognitionPanelViewModel_Start_RecognizerUnavailable_ReportsErrorStateAndDisposes() + public async Task RecognitionPanelViewModel_Start_RecognizerUnavailable_ReportsErrorStateAndDisposes() { - // Arrange: an installed model and available device, but a session factory that honestly - // reports it cannot compose a working recognizer + // Arrange: an installed model and available device, but an engine that honestly reports + // it cannot compose a working recognizer var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceService = Substitute.For(); var availableDevice = CaptureDevice(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(false); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); - var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory); + var engine = new FakeSpeechRecognizerEngine { IsAvailable = false }; + var viewModel = new RecognitionPanelViewModel( + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)); // Act: attempt to start - viewModel.StartCommand.Execute(null); + await viewModel.StartCommand.ExecuteAsync(null); - // Assert: the honest unavailable outcome, and the unusable recognizer was released + // Assert: the honest unavailable outcome, and the unusable engine was released Assert.Equal(RecognitionPanelViewModel.RecognizerUnavailableMessage, viewModel.StatusMessage); Assert.Equal(RecognitionStreamingState.Error, viewModel.State); - recognizer.Received(1).Dispose(); + Assert.Equal(1, engine.DisposeCallCount); } /// - /// Proves that Start reports the honest outcome, and releases the recognizer, when the + /// Proves that Start reports the honest outcome, and releases the session, when the /// library reports an unavailable capture device only after composition (an honest /// "reported available but the device failed to start" outcome). /// [Fact] - public void RecognitionPanelViewModel_Start_RecognizerStartThrows_ReportsErrorStateAndDisposes() + public async Task RecognitionPanelViewModel_Start_RecognizerStartThrows_ReportsErrorStateAndDisposes() { - // Arrange: a recognizer that reports itself available but faults when actually started + // Arrange: a session that reports itself available but faults when actually started var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceService = Substitute.For(); var availableDevice = CaptureDevice(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - recognizer.When(r => r.Start()).Throw(new SpeechRecognizerUnavailableException("Capture device failed to start.")); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); - var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory); + var session = new FakeRecognitionSession + { + StartException = new SpeechRecognizerUnavailableException("Capture device failed to start."), + }; + var engine = new FakeSpeechRecognizerEngine { SessionFactory = _ => session }; + var viewModel = new RecognitionPanelViewModel( + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)); // Act: attempt to start - viewModel.StartCommand.Execute(null); + await viewModel.StartCommand.ExecuteAsync(null); - // Assert: the fault is reported honestly, and the recognizer released + // Assert: the fault is reported honestly, and the session released (the engine remains + // cached, since only the session failed to start) Assert.Equal("Capture device failed to start.", viewModel.StatusMessage); Assert.Equal(RecognitionStreamingState.Error, viewModel.State); - recognizer.Received(1).Dispose(); + Assert.Equal(1, session.DisposeCallCount); + Assert.Equal(0, engine.DisposeCallCount); } /// /// Proves that a successful Start enters the Listening state and begins streaming. /// [Fact] - public void RecognitionPanelViewModel_Start_SuccessfulSession_EntersListeningState() + public async Task RecognitionPanelViewModel_Start_SuccessfulSession_EntersListeningState() { - // Arrange: an installed model, an available device, and a recognizer that starts cleanly + // Arrange: an installed model, an available device, and a session that starts cleanly var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceService = Substitute.For(); var availableDevice = CaptureDevice(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); - var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory); + var session = new FakeRecognitionSession(); + var engine = new FakeSpeechRecognizerEngine { SessionFactory = _ => session }; + var viewModel = new RecognitionPanelViewModel( + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)); // Act: start listening - viewModel.StartCommand.Execute(null); + await viewModel.StartCommand.ExecuteAsync(null); // Assert: streaming began and the panel reports it Assert.Equal(RecognitionStreamingState.Listening, viewModel.State); Assert.Null(viewModel.StatusMessage); - recognizer.Received(1).Start(); + Assert.Equal(1, session.StartCallCount); } /// @@ -280,29 +295,28 @@ public void RecognitionPanelViewModel_Start_SuccessfulSession_EntersListeningSta /// depends on. /// [Fact] - public void RecognitionPanelViewModel_ResultReceived_PartialThenFinal_UpdatesTranscriptInOrder() + public async Task RecognitionPanelViewModel_ResultReceived_PartialThenFinal_UpdatesTranscriptInOrder() { // Arrange: a listening session var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceService = Substitute.For(); var availableDevice = CaptureDevice(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); - var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory); - viewModel.StartCommand.Execute(null); + var session = new FakeRecognitionSession(); + var engine = new FakeSpeechRecognizerEngine { SessionFactory = _ => session }; + var viewModel = new RecognitionPanelViewModel( + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)); + await viewModel.StartCommand.ExecuteAsync(null); - // Act: raise a provisional result, then a final one for the same utterance - recognizer.ResultReceived += Raise.Event>(recognizer, new SpeechRecognitionEvent(new SpeechRecognitionResult("hel", false))); + // Act: push a provisional result, then a final one for the same utterance + session.PushResult(new SpeechRecognitionResult("hel", false)); // Assert: the partial line reflects the provisional result; nothing is final yet Assert.Equal("hel", viewModel.Partial); Assert.Empty(viewModel.Finals); // Act: the recognizer decides the utterance is complete - recognizer.ResultReceived += Raise.Event>(recognizer, new SpeechRecognitionEvent(new SpeechRecognitionResult("hello", true))); + session.PushResult(new SpeechRecognitionResult("hello", true)); // Assert: the line is committed and the partial is cleared Assert.Equal(["hello"], viewModel.Finals); @@ -315,29 +329,58 @@ public void RecognitionPanelViewModel_ResultReceived_PartialThenFinal_UpdatesTra /// line followed by any in-progress partial. /// [Fact] - public void RecognitionPanelViewModel_BuildTranscriptText_FinalsAndPartial_RendersInOrder() + public async Task RecognitionPanelViewModel_BuildTranscriptText_FinalsAndPartial_RendersInOrder() { // Arrange: a listening session with two committed lines and one in-progress partial var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceService = Substitute.For(); var availableDevice = CaptureDevice(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); - var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory); - viewModel.StartCommand.Execute(null); + var session = new FakeRecognitionSession(); + var engine = new FakeSpeechRecognizerEngine { SessionFactory = _ => session }; + var viewModel = new RecognitionPanelViewModel( + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)); + await viewModel.StartCommand.ExecuteAsync(null); // Act: commit two lines and leave a third in progress - recognizer.ResultReceived += Raise.Event>(recognizer, new SpeechRecognitionEvent(new SpeechRecognitionResult("one", true))); - recognizer.ResultReceived += Raise.Event>(recognizer, new SpeechRecognitionEvent(new SpeechRecognitionResult("two", true))); - recognizer.ResultReceived += Raise.Event>(recognizer, new SpeechRecognitionEvent(new SpeechRecognitionResult("thr", false))); + session.PushResult(new SpeechRecognitionResult("one", true)); + session.PushResult(new SpeechRecognitionResult("two", true)); + session.PushResult(new SpeechRecognitionResult("thr", false)); // Assert: the transcript renders both committed lines then the trailing partial Assert.Equal($"one{Environment.NewLine}two{Environment.NewLine}thr", viewModel.BuildTranscriptText()); } + /// + /// Proves that a session transitioning to + /// (for example, the bound capture device being lost mid-session) is reported honestly + /// through and + /// , driven entirely by the + /// mapping rather than ad hoc assignment. + /// + [Fact] + public async Task RecognitionPanelViewModel_StateChanged_SessionTransitionsToFaulted_ReportsErrorState() + { + // Arrange: a listening session + var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); + var deviceService = Substitute.For(); + var availableDevice = CaptureDevice(); + deviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); + var session = new FakeRecognitionSession(); + var engine = new FakeSpeechRecognizerEngine { SessionFactory = _ => session }; + var viewModel = new RecognitionPanelViewModel( + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)); + await viewModel.StartCommand.ExecuteAsync(null); + + // Act: the session itself reports an unrecoverable fault (not caused by this panel + // calling Stop or Start) + session.RaiseStateChanged(RecognitionSessionState.Faulted); + + // Assert: the panel reports the honest error state, driven by the StateChanged mapping + Assert.Equal(RecognitionStreamingState.Error, viewModel.State); + Assert.Equal(RecognitionPanelViewModel.SessionFaultedMessage, viewModel.StatusMessage); + } + /// /// Proves that clicking the shared device-selection panel's Refresh while this panel is /// actively listening stops the session first - deterministically, via the registered @@ -348,7 +391,7 @@ public void RecognitionPanelViewModel_BuildTranscriptText_FinalsAndPartial_Rende public async Task RecognitionPanelViewModel_PreRefreshHook_WhileListening_StopsSessionBeforeDeviceRefreshSucceeds() { // Arrange: a device service whose RefreshDevices() refuses while a "still listening" flag - // is true, and a recognizer whose Stop() flips that flag false - standing in for the real + // is true, and a session whose Stop flips that flag false - standing in for the real // library's "refuses a refresh while a stream is active" contract var stillListening = false; var deviceService = Substitute.For(); @@ -367,24 +410,21 @@ public async Task RecognitionPanelViewModel_PreRefreshHook_WhileListening_StopsS var availableDevice = CaptureDevice(); var captureDeviceService = Substitute.For(); captureDeviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - recognizer.When(r => r.Stop()).Do(_ => stillListening = false); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); + var session = new FakeRecognitionSession { OnStopRequested = () => stillListening = false }; + var engine = new FakeSpeechRecognizerEngine { SessionFactory = _ => session }; var viewModel = new RecognitionPanelViewModel( - Catalog(descriptor), captureDeviceService, deviceSelection, sessionFactory); + Catalog(descriptor), captureDeviceService, deviceSelection, SessionFactory(engine)); // Act: start listening, mark the session as actively streaming, then refresh the shared // device-selection panel (as the "Refresh devices" button would) - viewModel.StartCommand.Execute(null); + await viewModel.StartCommand.ExecuteAsync(null); stillListening = true; var exception = await Record.ExceptionAsync(() => deviceSelection.Refresh()); // Assert: the session was stopped by the hook, the refresh completed without throwing, // and the panel returned to idle Assert.Null(exception); - recognizer.Received(1).Stop(); + Assert.Equal(1, session.StopCallCount); Assert.Equal(RecognitionStreamingState.Idle, viewModel.State); Assert.False(viewModel.CanStop); } @@ -392,7 +432,7 @@ public async Task RecognitionPanelViewModel_PreRefreshHook_WhileListening_StopsS /// /// Proves that the registered pre-refresh hook is a safe no-op when no listening session /// is in flight, so a "Refresh devices" click while the panel is idle never calls Stop on - /// a recognizer that was never even created. + /// a session that was never even created. /// [Fact] public async Task RecognitionPanelViewModel_PreRefreshHook_WhileIdle_IsNoOpAndDeviceRefreshSucceeds() @@ -403,198 +443,187 @@ public async Task RecognitionPanelViewModel_PreRefreshHook_WhileIdle_IsNoOpAndDe var deviceService = Substitute.For(); var availableDevice = CaptureDevice(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); + var engine = new FakeSpeechRecognizerEngine(); var deviceSelection = DeviceSelection(); var viewModel = new RecognitionPanelViewModel( - Catalog(descriptor), deviceService, deviceSelection, sessionFactory); + Catalog(descriptor), deviceService, deviceSelection, SessionFactory(engine)); // Act: refresh the shared device-selection panel while idle, with listening never started var exception = await Record.ExceptionAsync(() => deviceSelection.Refresh()); - // Assert: the refresh completes without throwing, Stop is never requested on a - // recognizer that was never even created, and the panel remains idle + // Assert: the refresh completes without throwing, no session was ever created, and the + // panel remains idle Assert.Null(exception); - recognizer.DidNotReceive().Stop(); + Assert.Equal(0, engine.CreateSessionCallCount); Assert.Equal(RecognitionStreamingState.Idle, viewModel.State); } /// /// Proves that Stop ends an in-flight session deterministically without disposing the - /// recognizer - it is cached and reused across Start/Stop cycles (see - /// 's "Recognizer reuse" remarks) - and reports the - /// stop rather than an error. + /// cached engine - it is reused across Start/Stop cycles (see + /// 's "Engine/session reuse" remarks) - and reports + /// the stop rather than an error. /// [Fact] - public void RecognitionPanelViewModel_Stop_DuringListening_StopsWithoutDisposingSession() + public async Task RecognitionPanelViewModel_Stop_DuringListening_StopsWithoutDisposingSession() { // Arrange: a listening session var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceService = Substitute.For(); var availableDevice = CaptureDevice(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); - var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory); - viewModel.StartCommand.Execute(null); + var session = new FakeRecognitionSession(); + var engine = new FakeSpeechRecognizerEngine { SessionFactory = _ => session }; + var viewModel = new RecognitionPanelViewModel( + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)); + await viewModel.StartCommand.ExecuteAsync(null); // Act: stop - viewModel.StopCommand.Execute(null); + await viewModel.StopCommand.ExecuteAsync(null); - // Assert: the session was stopped but the recognizer itself is kept alive (model stays - // loaded), and the panel reports the stop - recognizer.Received(1).Stop(); - recognizer.DidNotReceive().Dispose(); + // Assert: the session was stopped, is released (single-use), but the engine itself is + // kept alive (model stays loaded), and the panel reports the stop + Assert.Equal(1, session.StopCallCount); + Assert.Equal(0, engine.DisposeCallCount); Assert.Equal(RecognitionStreamingState.Idle, viewModel.State); Assert.Equal(RecognitionPanelViewModel.StoppedMessage, viewModel.StatusMessage); } /// - /// Proves the central "Recognizer reuse" guarantee: starting, stopping, and starting - /// again for the same model/capture-device selection builds the recognizer only once, - /// instead of reloading its model on every Start click. + /// Proves the central "Engine reuse" guarantee: starting, stopping, and starting again for + /// the same model/capture-device selection loads the engine only once, instead of + /// reloading its model on every Start click, even though - because a session is + /// single-use - a fresh session is created for each run. /// [Fact] - public void RecognitionPanelViewModel_StartStopStart_SameSelection_ReusesRecognizer() + public async Task RecognitionPanelViewModel_StartStopStart_SameSelection_ReusesEngine() { // Arrange: a panel ready to listen var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceService = Substitute.For(); var availableDevice = CaptureDevice(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); + var engine = new FakeSpeechRecognizerEngine(); + var sessionFactory = SessionFactory(engine); var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory); // Act: start, stop, start again - the same selection throughout - viewModel.StartCommand.Execute(null); - viewModel.StopCommand.Execute(null); - viewModel.StartCommand.Execute(null); - - // Assert: the expensive composition step ran exactly once; Start/Stop ran on the same - // cached instance twice each - sessionFactory.Received(1).Create(Arg.Any(), Arg.Any()); - recognizer.Received(2).Start(); - recognizer.Received(1).Stop(); + await viewModel.StartCommand.ExecuteAsync(null); + await viewModel.StopCommand.ExecuteAsync(null); + await viewModel.StartCommand.ExecuteAsync(null); + + // Assert: the expensive engine-load step ran exactly once; a fresh single-use session + // was created for each of the two runs + await sessionFactory.Received(1).LoadAsync(Arg.Any(), Arg.Any()); + Assert.Equal(2, engine.CreateSessionCallCount); + Assert.All(engine.CreatedSessions, createdSession => Assert.Equal(1, createdSession.StartCallCount)); } /// - /// Proves that a device refresh invalidates the cached recognizer: a Start after the - /// refresh builds a fresh recognizer/device pair rather than reusing one bound to a now- - /// stale device. + /// Proves that a device refresh invalidates only the cached session (not the engine): a + /// Start after the refresh builds a fresh session/device pair without reloading the + /// engine. /// [Fact] - public async Task RecognitionPanelViewModel_DeviceRefresh_InvalidatesCachedRecognizer() + public async Task RecognitionPanelViewModel_DeviceRefresh_InvalidatesCachedSessionButNotEngine() { - // Arrange: a panel with a cached (idle) recognizer from a prior Start/Stop cycle + // Arrange: a panel with a cached (idle) session from a prior Start/Stop cycle var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceService = Substitute.For(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(_ => CaptureDevice()); - var firstRecognizer = Substitute.For(); - firstRecognizer.IsAvailable.Returns(true); - var secondRecognizer = Substitute.For(); - secondRecognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()) - .Returns(firstRecognizer, secondRecognizer); + var engine = new FakeSpeechRecognizerEngine(); + var sessionFactory = SessionFactory(engine); var deviceSelection = DeviceSelection(); var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, deviceSelection, sessionFactory); - viewModel.StartCommand.Execute(null); - viewModel.StopCommand.Execute(null); + await viewModel.StartCommand.ExecuteAsync(null); + await viewModel.StopCommand.ExecuteAsync(null); + var firstSession = engine.CreatedSessions.Single(); // Act: refresh the shared device-selection panel while idle, then start again await deviceSelection.Refresh(); - viewModel.StartCommand.Execute(null); - - // Assert: the stale recognizer was disposed by the refresh, and the second Start - // composed an entirely new recognizer rather than reusing the stale one - firstRecognizer.Received(1).Dispose(); - sessionFactory.Received(2).Create(Arg.Any(), Arg.Any()); - secondRecognizer.Received(1).Start(); + await viewModel.StartCommand.ExecuteAsync(null); + + // Assert: the stale session was released by the refresh, a second session was created + // from the same cached engine, and the engine was never reloaded + Assert.Equal(1, firstSession.DisposeCallCount); + Assert.Equal(2, engine.CreateSessionCallCount); + await sessionFactory.Received(1).LoadAsync(Arg.Any(), Arg.Any()); + Assert.Equal(1, engine.CreatedSessions[1].StartCallCount); } /// /// Proves that changing the selected recognition model invalidates a cached (idle) - /// recognizer, so the next Start builds a recognizer for the newly selected model instead - /// of reusing one loaded for the previous model. + /// engine, so the next Start loads a new engine for the newly selected model instead of + /// reusing one loaded for the previous model. /// [Fact] - public void RecognitionPanelViewModel_SelectedModelChanged_InvalidatesCachedRecognizer() + public async Task RecognitionPanelViewModel_SelectedModelChanged_InvalidatesCachedEngine() { - // Arrange: a panel with two installed models and a cached (idle) recognizer for the first + // Arrange: a panel with two installed models and a cached (idle) engine for the first var firstDescriptor = FakeSpeechModel.Descriptor("stt-a", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var secondDescriptor = FakeSpeechModel.Descriptor("stt-b", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceService = Substitute.For(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(_ => CaptureDevice()); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); + var engine = new FakeSpeechRecognizerEngine(); var viewModel = new RecognitionPanelViewModel( - Catalog(firstDescriptor, secondDescriptor), deviceService, DeviceSelection(), sessionFactory); - viewModel.StartCommand.Execute(null); - viewModel.StopCommand.Execute(null); + Catalog(firstDescriptor, secondDescriptor), deviceService, DeviceSelection(), SessionFactory(engine)); + await viewModel.StartCommand.ExecuteAsync(null); + await viewModel.StopCommand.ExecuteAsync(null); // Act: pick the other installed model while idle viewModel.SelectedModel = viewModel.AvailableModels.Single(model => model.Id == "stt-b"); + await Task.Yield(); - // Assert: the recognizer cached for the previous model was disposed - recognizer.Received(1).Dispose(); + // Assert: the engine cached for the previous model was disposed + Assert.Equal(1, engine.DisposeCallCount); } /// /// Proves that changing the selected recognition model while a session is actively - /// listening stops that session and invalidates the cached recognizer, rather than - /// disposing it without stopping first - which would leave State stuck at - /// Listening forever, since a later Stop would see no cached recognizer and - /// no-op. SelectedModel has a public setter and is not guarded against this at the + /// listening stops that session and invalidates the cached engine, rather than disposing + /// it without stopping first - which would leave State stuck at Listening + /// forever, since a later Stop would see no cached session and no-op. + /// SelectedModel has a public setter and is not guarded against this at the /// property level (only the view disables the model picker while listening), so the /// change can arrive mid-session. /// [Fact] - public void RecognitionPanelViewModel_SelectedModelChanged_WhileListening_StopsAndInvalidatesRecognizer() + public async Task RecognitionPanelViewModel_SelectedModelChanged_WhileListening_StopsAndInvalidatesEngine() { - // Arrange: a panel with two installed models and an actively listening recognizer for - // the first + // Arrange: a panel with two installed models and an actively listening session for the + // first var firstDescriptor = FakeSpeechModel.Descriptor("stt-a", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var secondDescriptor = FakeSpeechModel.Descriptor("stt-b", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceService = Substitute.For(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(_ => CaptureDevice()); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); + var session = new FakeRecognitionSession(); + var engine = new FakeSpeechRecognizerEngine { SessionFactory = _ => session }; var viewModel = new RecognitionPanelViewModel( - Catalog(firstDescriptor, secondDescriptor), deviceService, DeviceSelection(), sessionFactory); - viewModel.StartCommand.Execute(null); + Catalog(firstDescriptor, secondDescriptor), deviceService, DeviceSelection(), SessionFactory(engine)); + await viewModel.StartCommand.ExecuteAsync(null); // Act: pick the other installed model while still listening (bypassing the view's // disabled picker, e.g. a direct property set) viewModel.SelectedModel = viewModel.AvailableModels.Single(model => model.Id == "stt-b"); + await Task.Yield(); - // Assert: the active session was stopped and the recognizer was disposed, and the panel - // is not stuck in Listening - recognizer.Received(1).Stop(); - recognizer.Received(1).Dispose(); + // Assert: the active session was stopped and the engine was disposed, and the panel is + // not stuck in Listening + Assert.Equal(1, session.StopCallCount); + Assert.Equal(1, engine.DisposeCallCount); Assert.Equal(RecognitionStreamingState.Idle, viewModel.State); } /// - /// Proves that changing the selected capture device invalidates a cached (idle) - /// recognizer, so the next Start builds a recognizer bound to the newly selected device - /// instead of reusing one bound to the previous device. + /// Proves that changing the selected capture device invalidates only a cached (idle) + /// session - never the cached engine - so the next Start builds a session bound to the + /// newly selected device while reusing the already-loaded engine. /// [Fact] - public void RecognitionPanelViewModel_SelectedCaptureDeviceChanged_InvalidatesCachedRecognizer() + public async Task RecognitionPanelViewModel_SelectedCaptureDeviceChanged_InvalidatesSessionButNotEngine() { // Arrange: a panel sharing a device-selection panel reporting two capture devices, with a - // cached (idle) recognizer bound to the first + // cached (idle) session bound to the first var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceA = new AudioDeviceDescription("Mic A", AudioDeviceDirection.Capture, 1, 48_000); var deviceB = new AudioDeviceDescription("Mic B", AudioDeviceDirection.Capture, 1, 48_000); @@ -602,34 +631,36 @@ public void RecognitionPanelViewModel_SelectedCaptureDeviceChanged_InvalidatesCa deviceService.EnumerateCaptureDevices().Returns([deviceA, deviceB]); deviceService.EnumeratePlaybackDevices().Returns([]); deviceService.CreateCaptureDevice(Arg.Any()).Returns(_ => CaptureDevice()); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); + var engine = new FakeSpeechRecognizerEngine(); var deviceSelection = new DeviceSelectionViewModel(deviceService); - var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, deviceSelection, sessionFactory); - viewModel.StartCommand.Execute(null); - viewModel.StopCommand.Execute(null); + var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, deviceSelection, SessionFactory(engine)); + await viewModel.StartCommand.ExecuteAsync(null); + await viewModel.StopCommand.ExecuteAsync(null); + var firstSession = engine.CreatedSessions.Single(); // Act: pick the other capture device while idle deviceSelection.SelectedCaptureDevice = deviceB; + await Task.Yield(); - // Assert: the recognizer cached for the previous device was disposed - recognizer.Received(1).Dispose(); + // Assert: the session cached for the previous device was disposed, but the engine was + // never reloaded + Assert.Equal(1, firstSession.DisposeCallCount); + Assert.Equal(0, engine.DisposeCallCount); } /// /// Proves that changing the selected capture device while a session is actively listening - /// stops that session and invalidates the cached recognizer, rather than silently - /// leaving it bound to the now-abandoned device: the capture picker is not disabled while - /// listening (unlike the model picker; see ), - /// so this change can arrive mid-session. + /// stops that session and invalidates the cached session (not the engine), rather than + /// silently leaving it bound to the now-abandoned device: the capture picker is not + /// disabled while listening (unlike the model picker; see + /// ), so this change can arrive + /// mid-session. /// [Fact] - public void RecognitionPanelViewModel_SelectedCaptureDeviceChanged_WhileListening_StopsAndInvalidatesRecognizer() + public async Task RecognitionPanelViewModel_SelectedCaptureDeviceChanged_WhileListening_StopsAndInvalidatesSession() { // Arrange: a panel sharing a device-selection panel reporting two capture devices, with - // an actively listening recognizer bound to the first + // an actively listening session bound to the first var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceA = new AudioDeviceDescription("Mic A", AudioDeviceDirection.Capture, 1, 48_000); var deviceB = new AudioDeviceDescription("Mic B", AudioDeviceDirection.Capture, 1, 48_000); @@ -637,21 +668,20 @@ public void RecognitionPanelViewModel_SelectedCaptureDeviceChanged_WhileListenin deviceService.EnumerateCaptureDevices().Returns([deviceA, deviceB]); deviceService.EnumeratePlaybackDevices().Returns([]); deviceService.CreateCaptureDevice(Arg.Any()).Returns(_ => CaptureDevice()); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); + var session = new FakeRecognitionSession(); + var engine = new FakeSpeechRecognizerEngine { SessionFactory = _ => session }; var deviceSelection = new DeviceSelectionViewModel(deviceService); - var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, deviceSelection, sessionFactory); - viewModel.StartCommand.Execute(null); + var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, deviceSelection, SessionFactory(engine)); + await viewModel.StartCommand.ExecuteAsync(null); // Act: pick the other capture device while still listening deviceSelection.SelectedCaptureDevice = deviceB; + await Task.Yield(); - // Assert: the active session was stopped and the recognizer bound to the old device was - // disposed, rather than being silently left cached and bound to the abandoned device - recognizer.Received(1).Stop(); - recognizer.Received(1).Dispose(); + // Assert: the active session was stopped and released, but the engine persists + Assert.Equal(1, session.StopCallCount); + Assert.Equal(1, session.DisposeCallCount); + Assert.Equal(0, engine.DisposeCallCount); Assert.Equal(RecognitionStreamingState.Idle, viewModel.State); } @@ -660,7 +690,7 @@ public void RecognitionPanelViewModel_SelectedCaptureDeviceChanged_WhileListenin /// "stop when not running" contract. /// [Fact] - public void RecognitionPanelViewModel_Stop_NothingListening_IsSafeNoOp() + public async Task RecognitionPanelViewModel_Stop_NothingListening_IsSafeNoOp() { // Arrange: a freshly composed panel that never started var viewModel = new RecognitionPanelViewModel( @@ -668,7 +698,7 @@ public void RecognitionPanelViewModel_Stop_NothingListening_IsSafeNoOp() Substitute.For()); // Act: stop with nothing running - var exception = Record.Exception(() => viewModel.StopCommand.Execute(null)); + var exception = await Record.ExceptionAsync(() => viewModel.StopCommand.ExecuteAsync(null)); // Assert: no fault, and the panel remains idle with no status to report Assert.Null(exception); @@ -677,35 +707,64 @@ public void RecognitionPanelViewModel_Stop_NothingListening_IsSafeNoOp() } /// - /// Proves that a result received after Stop is no longer applied to the transcript, - /// since the handler was unsubscribed when the session ended. + /// Proves that calling Stop twice concurrently - as 's + /// AllowConcurrentExecutions permits - both complete without throwing, rather than + /// racing to double-dispose the same session. /// [Fact] - public void RecognitionPanelViewModel_Dispose_ReleasesActiveSessionWithoutThrowing() + public async Task RecognitionPanelViewModel_Stop_CalledTwiceConcurrently_BothCompleteWithoutThrowing() { // Arrange: a listening session var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceService = Substitute.For(); var availableDevice = CaptureDevice(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); - var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory); - viewModel.StartCommand.Execute(null); + var session = new FakeRecognitionSession(); + var engine = new FakeSpeechRecognizerEngine { SessionFactory = _ => session }; + var viewModel = new RecognitionPanelViewModel( + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)); + await viewModel.StartCommand.ExecuteAsync(null); + + // Act: invoke Stop twice without awaiting the first before starting the second + var first = viewModel.StopCommand.ExecuteAsync(null); + var second = viewModel.StopCommand.ExecuteAsync(null); + var exception = await Record.ExceptionAsync(() => Task.WhenAll(first, second)); + + // Assert: both complete without throwing, and the panel settles at Idle + Assert.Null(exception); + Assert.Equal(RecognitionStreamingState.Idle, viewModel.State); + } + + /// + /// Proves that releases an active + /// session cleanly without throwing, idempotently. + /// + [Fact] + public async Task RecognitionPanelViewModel_Dispose_ReleasesActiveSessionWithoutThrowing() + { + // Arrange: a listening session + var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); + var deviceService = Substitute.For(); + var availableDevice = CaptureDevice(); + deviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); + var session = new FakeRecognitionSession(); + var engine = new FakeSpeechRecognizerEngine { SessionFactory = _ => session }; + var viewModel = new RecognitionPanelViewModel( + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)); + await viewModel.StartCommand.ExecuteAsync(null); // Act: dispose the panel directly (as the shell would on shutdown) and again for // idempotency - var exception = Record.Exception(() => + var exception = await Record.ExceptionAsync(async () => { - viewModel.Dispose(); - viewModel.Dispose(); + await viewModel.DisposeAsync(); + await viewModel.DisposeAsync(); }); - // Assert: no fault, and the recognizer was released exactly once + // Assert: no fault, and the session/engine were each released exactly once Assert.Null(exception); - recognizer.Received(1).Dispose(); + Assert.Equal(1, session.DisposeCallCount); + Assert.Equal(1, engine.DisposeCallCount); } /// @@ -762,12 +821,12 @@ public void RecognitionPanelViewModel_ModelInstalled_NonMatchingRole_DoesNotTrig } /// - /// Proves that unsubscribes from + /// Proves that unsubscribes from /// , so a later install completing after /// disposal is never applied. /// [Fact] - public void RecognitionPanelViewModel_Dispose_UnsubscribesFromModelInstalled_NoRefreshAfterDispose() + public async Task RecognitionPanelViewModel_Dispose_UnsubscribesFromModelInstalled_NoRefreshAfterDispose() { // Arrange: a panel composed over a catalog reporting no installed models yet var catalog = Substitute.For(); @@ -778,7 +837,7 @@ public void RecognitionPanelViewModel_Dispose_UnsubscribesFromModelInstalled_NoR Substitute.For()); // Act: dispose the panel, then simulate a later install completing - viewModel.Dispose(); + await viewModel.DisposeAsync(); descriptors = [FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition)]; var exception = Record.Exception(() => catalog.ModelInstalled += Raise.Event>( catalog, new ModelInstalledEventArgs("stt", SpeechModelRole.Recognition))); @@ -797,31 +856,29 @@ public void RecognitionPanelViewModel_Dispose_UnsubscribesFromModelInstalled_NoR /// so the model-selection control is disabled only while a session is actively streaming. /// [Fact] - public void RecognitionPanelViewModel_CanChangeModel_TogglesAcrossStateTransitions() + public async Task RecognitionPanelViewModel_CanChangeModel_TogglesAcrossStateTransitions() { - // Arrange: an installed model, an available device, and a recognizer that starts cleanly + // Arrange: an installed model, an available device, and a session that starts cleanly var descriptor = FakeSpeechModel.Descriptor("stt", SpeechModelState.Downloaded, role: SpeechModelRole.Recognition); var deviceService = Substitute.For(); var availableDevice = CaptureDevice(); deviceService.CreateCaptureDevice(Arg.Any()).Returns(availableDevice); - var recognizer = Substitute.For(); - recognizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any()).Returns(recognizer); - var viewModel = new RecognitionPanelViewModel(Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory); + var engine = new FakeSpeechRecognizerEngine(); + var viewModel = new RecognitionPanelViewModel( + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)); // Assert: true at the initial Idle state Assert.True(viewModel.CanChangeModel); // Act: start listening - viewModel.StartCommand.Execute(null); + await viewModel.StartCommand.ExecuteAsync(null); // Assert: false while actively listening Assert.Equal(RecognitionStreamingState.Listening, viewModel.State); Assert.False(viewModel.CanChangeModel); // Act: stop - viewModel.StopCommand.Execute(null); + await viewModel.StopCommand.ExecuteAsync(null); // Assert: reverts to true after Stop Assert.Equal(RecognitionStreamingState.Idle, viewModel.State); @@ -834,13 +891,13 @@ public void RecognitionPanelViewModel_CanChangeModel_TogglesAcrossStateTransitio /// different model after a failed Start. /// [Fact] - public void RecognitionPanelViewModel_CanChangeModel_ErrorState_IsTrue() + public async Task RecognitionPanelViewModel_CanChangeModel_ErrorState_IsTrue() { // Arrange: a panel with no installed models, so Start reports the Error state var viewModel = new RecognitionPanelViewModel( Catalog(), Substitute.For(), DeviceSelection(), Substitute.For()); - viewModel.StartCommand.Execute(null); + await viewModel.StartCommand.ExecuteAsync(null); Assert.Equal(RecognitionStreamingState.Error, viewModel.State); // Act & Assert: the model-selection control remains enabled at the Error state diff --git a/test/DemaConsulting.Speech.Demo.Tests/RecognitionPanelSubsystem/RecognizerSessionFactoryTests.cs b/test/DemaConsulting.Speech.Demo.Tests/RecognitionPanelSubsystem/RecognizerSessionFactoryTests.cs index 6567ae5..656c094 100644 --- a/test/DemaConsulting.Speech.Demo.Tests/RecognitionPanelSubsystem/RecognizerSessionFactoryTests.cs +++ b/test/DemaConsulting.Speech.Demo.Tests/RecognitionPanelSubsystem/RecognizerSessionFactoryTests.cs @@ -1,9 +1,7 @@ -using DemaConsulting.Speech.AudioSubsystem; using DemaConsulting.Speech.Demo.RecognitionPanelSubsystem; using DemaConsulting.Speech.Demo.Tests.Fakes; using DemaConsulting.Speech.ModelManagementSubsystem; using DemaConsulting.Speech.RecognitionSubsystem; -using NSubstitute; namespace DemaConsulting.Speech.Demo.Tests.RecognitionPanelSubsystem; @@ -11,7 +9,7 @@ namespace DemaConsulting.Speech.Demo.Tests.RecognitionPanelSubsystem; /// Unit tests for . /// /// -/// The "correct role composes a working recognizer" path delegates to the library's own +/// The "correct role composes a working engine" path delegates to the library's own /// , which requires an - /// an interface only the library's own assemblies can implement (see the type's remarks). /// That path is therefore outside this test project's reach and remains covered by the @@ -41,50 +39,36 @@ public void RecognizerSessionFactory_Constructor_NullStore_ThrowsArgumentNullExc } /// - /// Proves that Create rejects a missing model. + /// Proves that LoadAsync rejects a missing model. /// [Fact] - public void RecognizerSessionFactory_Create_NullModel_ThrowsArgumentNullException() + public async Task RecognizerSessionFactory_LoadAsync_NullModel_ThrowsArgumentNullException() { // Arrange var factory = new RecognizerSessionFactory(IsolatedStore()); - var device = Substitute.For(); // Act & Assert - Assert.Throws(() => factory.Create(null!, device)); + await Assert.ThrowsAsync( + () => factory.LoadAsync(null!, TestContext.Current.CancellationToken)); } /// - /// Proves that Create rejects a missing capture device. + /// Proves that LoadAsync honestly reports a model that does not implement the library's + /// recognition role as an unavailable engine, exactly like a model that is not installed, + /// rather than throwing. /// [Fact] - public void RecognizerSessionFactory_Create_NullDevice_ThrowsArgumentNullException() - { - // Arrange - var factory = new RecognizerSessionFactory(IsolatedStore()); - - // Act & Assert - Assert.Throws(() => factory.Create(new FakeSpeechModel(), null!)); - } - - /// - /// Proves that Create honestly reports a model that does not implement the library's - /// recognition role as an unavailable recognizer, exactly like a model that is not - /// installed, rather than throwing. - /// - [Fact] - public void RecognizerSessionFactory_Create_ModelNotRecognitionRole_ReturnsUnavailableRecognizer() + public async Task RecognizerSessionFactory_LoadAsync_ModelNotRecognitionRole_ReturnsUnavailableEngine() { // Arrange: a fake model that only ever implements the public ISpeechModel contract var factory = new RecognizerSessionFactory(IsolatedStore()); - var device = Substitute.For(); var model = new FakeSpeechModel(role: SpeechModelRole.Recognition); // Act - var recognizer = factory.Create(model, device); + var engine = await factory.LoadAsync(model, TestContext.Current.CancellationToken); - // Assert: the honest unavailable fallback, not a real recognizer or an exception - Assert.Same(UnavailableSpeechRecognizer.Instance, recognizer); - Assert.False(recognizer.IsAvailable); + // Assert: the honest unavailable fallback, not a real engine or an exception + Assert.Same(UnavailableSpeechRecognizerEngine.Instance, engine); + Assert.False(engine.IsAvailable); } } diff --git a/test/DemaConsulting.Speech.Demo.Tests/SynthesisPanelSubsystem/SynthesisPanelViewModelTests.cs b/test/DemaConsulting.Speech.Demo.Tests/SynthesisPanelSubsystem/SynthesisPanelViewModelTests.cs index 325ba6d..f372e78 100644 --- a/test/DemaConsulting.Speech.Demo.Tests/SynthesisPanelSubsystem/SynthesisPanelViewModelTests.cs +++ b/test/DemaConsulting.Speech.Demo.Tests/SynthesisPanelSubsystem/SynthesisPanelViewModelTests.cs @@ -1,6 +1,7 @@ using DemaConsulting.Speech.AudioSubsystem; using DemaConsulting.Speech.Demo.DeviceSelectionSubsystem; using DemaConsulting.Speech.Demo.ModelCatalogSubsystem; +using DemaConsulting.Speech.Demo.ModelSettingsSubsystem; using DemaConsulting.Speech.Demo.SynthesisPanelSubsystem; using DemaConsulting.Speech.Demo.Tests.Fakes; using DemaConsulting.Speech.ModelManagementSubsystem; @@ -14,6 +15,12 @@ namespace DemaConsulting.Speech.Demo.Tests.SynthesisPanelSubsystem; /// public class SynthesisPanelViewModelTests { + /// A reusable fake playback device description. + private static readonly AudioDeviceDescription SpeakerA = new("Speaker A", AudioDeviceDirection.Playback, 2, 44100); + + /// A second, distinct fake playback device description. + private static readonly AudioDeviceDescription SpeakerB = new("Speaker B", AudioDeviceDirection.Playback, 2, 48000); + /// /// Builds a device-selection panel over a fake reporting no devices, for the constructor /// parameter every panel requires. @@ -27,6 +34,20 @@ private static DeviceSelectionViewModel DeviceSelection() return new DeviceSelectionViewModel(service); } + /// + /// Builds a device-selection panel over the supplied playback devices, with the first + /// selected. + /// + /// The playback devices to offer. + /// The composed panel. + private static DeviceSelectionViewModel DeviceSelection(params AudioDeviceDescription[] devices) + { + var service = Substitute.For(); + service.EnumerateCaptureDevices().Returns([]); + service.EnumeratePlaybackDevices().Returns(devices); + return new DeviceSelectionViewModel(service); + } + /// /// Builds a catalog service fake returning the supplied descriptors. /// @@ -51,6 +72,25 @@ private static IAudioPlaybackDevice PlaybackDevice(bool isAvailable = true) return device; } + /// + /// Builds a session-factory substitute whose LoadAsync returns each of the given + /// engines in order (the last is returned for any further call), for any model/parameter + /// combination. + /// + /// The engine(s) to return from successive calls. + /// The composed substitute. + private static ISynthesizerSessionFactory SessionFactory(params FakeSpeechSynthesizerEngine[] engines) + { + var factory = Substitute.For(); + var tasks = engines.Select(engine => Task.FromResult(engine)).ToArray(); + factory.LoadAsync( + Arg.Any(), + Arg.Any?>(), + Arg.Any()) + .Returns(tasks[0], tasks[1..]); + return factory; + } + /// /// Proves that the panel rejects any missing constructor dependency. /// @@ -65,17 +105,13 @@ public void SynthesisPanelViewModel_Constructor_NullDependency_ThrowsArgumentNul // Act & Assert: each missing dependency is a programming error Assert.Throws( - () => new SynthesisPanelViewModel( - null!, deviceService, deviceSelection, sessionFactory)); + () => new SynthesisPanelViewModel(null!, deviceService, deviceSelection, sessionFactory)); Assert.Throws( - () => new SynthesisPanelViewModel( - catalog, null!, deviceSelection, sessionFactory)); + () => new SynthesisPanelViewModel(catalog, null!, deviceSelection, sessionFactory)); Assert.Throws( - () => new SynthesisPanelViewModel( - catalog, deviceService, null!, sessionFactory)); + () => new SynthesisPanelViewModel(catalog, deviceService, null!, sessionFactory)); Assert.Throws( - () => new SynthesisPanelViewModel( - catalog, deviceService, deviceSelection, null!)); + () => new SynthesisPanelViewModel(catalog, deviceService, deviceSelection, null!)); } /// @@ -192,38 +228,35 @@ public async Task SynthesisPanelViewModel_Play_NoPlaybackDevice_ReportsErrorStat /// /// Proves that Play reports the honest "synthesizer unavailable" outcome, and disposes - /// the unavailable synthesizer, when the session seam cannot compose a working one. + /// the unavailable engine, when the session seam cannot compose a working one. /// [Fact] public async Task SynthesisPanelViewModel_Play_SynthesizerUnavailable_ReportsErrorStateAndDisposes() { - // Arrange: an installed model and available device, but a session factory that honestly - // reports it cannot compose a working synthesizer + // Arrange: an installed model and available device, but an engine that honestly reports + // it cannot compose a working synthesizer var descriptor = FakeSpeechModel.Descriptor("tts", SpeechModelState.Downloaded, role: SpeechModelRole.Synthesis); var deviceService = Substitute.For(); var availableDevice = PlaybackDevice(); deviceService.CreatePlaybackDevice(Arg.Any()).Returns(availableDevice); - var synthesizer = Substitute.For(); - synthesizer.IsAvailable.Returns(false); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any(), Arg.Any>()).Returns(synthesizer); + var engine = new FakeSpeechSynthesizerEngine { IsAvailable = false }; var viewModel = new SynthesisPanelViewModel( - Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory); + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)); // Act: attempt to play await viewModel.PlayCommand.ExecuteAsync(null); - // Assert: the honest unavailable outcome, and the unusable synthesizer was released + // Assert: the honest unavailable outcome, and the unusable engine was released Assert.Equal(SynthesisPanelViewModel.SynthesizerUnavailableMessage, viewModel.StatusMessage); Assert.Equal(SynthesisPlaybackState.Error, viewModel.State); - synthesizer.Received(1).Dispose(); + Assert.Equal(1, engine.DisposeCallCount); } /// - /// Proves that the embedded ModelSettingsViewModel.BuildValueBag - /// content genuinely reaches the injected - /// call during Play, closing the "settings bag has no real consumer" gap the demo - /// previously documented. + /// Proves that the embedded ModelSettingsViewModel.BuildValueBag content genuinely + /// reaches the injected call during + /// Play, closing the "settings bag has no real consumer" gap the demo previously + /// documented. /// [Fact] public async Task SynthesisPanelViewModel_Play_ModelDeclaresChoiceParameter_ForwardsValueBagToSessionFactory() @@ -243,11 +276,8 @@ [new ChoiceParameterOption("af_bella", "Bella")], var deviceService = Substitute.For(); var availableDevice = PlaybackDevice(); deviceService.CreatePlaybackDevice(Arg.Any()).Returns(availableDevice); - var synthesizer = Substitute.For(); - synthesizer.IsAvailable.Returns(true); - synthesizer.SpeakAsync(Arg.Any(), Arg.Any()).Returns(Task.CompletedTask); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any(), Arg.Any>()).Returns(synthesizer); + var engine = new FakeSpeechSynthesizerEngine(); + var sessionFactory = SessionFactory(engine); var viewModel = new SynthesisPanelViewModel( Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory) { @@ -257,37 +287,34 @@ [new ChoiceParameterOption("af_bella", "Bella")], // Act await viewModel.PlayCommand.ExecuteAsync(null); - // Assert: the value bag built from the embedded settings panel reached Create - sessionFactory.Received(1).Create( + // Assert: the value bag built from the embedded settings panel reached LoadAsync + await sessionFactory.Received(1).LoadAsync( Arg.Any(), - Arg.Any(), - Arg.Is>(bag => IsBellaVoiceBag(bag))); + Arg.Is?>(bag => IsBellaVoiceBag(bag)), + Arg.Any()); } /// Matches a value bag containing exactly the expected "voice" -> "af_bella" entry. - private static bool IsBellaVoiceBag(IReadOnlyDictionary bag) => - bag.Count == 1 && bag.TryGetValue("voice", out var value) && Equals(value, "af_bella"); + private static bool IsBellaVoiceBag(IReadOnlyDictionary? bag) => + bag is not null && bag.Count == 1 && bag.TryGetValue("voice", out var value) && Equals(value, "af_bella"); /// /// Proves that a full, successful Play lifecycle transitions through Synthesizing then - /// Playing before settling on Idle, and disposes the synthesizer afterward. + /// Playing before settling on Idle, and disposes the session afterward via the + /// mapping. /// [Fact] public async Task SynthesisPanelViewModel_Play_SuccessfulSession_TransitionsThroughLifecycleToIdle() { - // Arrange: an installed model, an available device, and a synthesizer that speaks + // Arrange: an installed model, an available device, and a session that speaks // successfully var descriptor = FakeSpeechModel.Descriptor("tts", SpeechModelState.Downloaded, role: SpeechModelRole.Synthesis); var deviceService = Substitute.For(); var availableDevice = PlaybackDevice(); deviceService.CreatePlaybackDevice(Arg.Any()).Returns(availableDevice); - var synthesizer = Substitute.For(); - synthesizer.IsAvailable.Returns(true); - synthesizer.SpeakAsync(Arg.Any(), Arg.Any()).Returns(Task.CompletedTask); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any(), Arg.Any>()).Returns(synthesizer); + var engine = new FakeSpeechSynthesizerEngine(); var viewModel = new SynthesisPanelViewModel( - Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory) + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)) { Text = "Hello [whispers] world", }; @@ -304,38 +331,32 @@ public async Task SynthesisPanelViewModel_Play_SuccessfulSession_TransitionsThro await viewModel.PlayCommand.ExecuteAsync(null); // Assert: the lifecycle passed through synthesizing and playing before settling idle, - // the exact text was forwarded, and the synthesizer was released + // and the exact text was forwarded Assert.Equal( [SynthesisPlaybackState.Synthesizing, SynthesisPlaybackState.Playing, SynthesisPlaybackState.Idle], states); Assert.Null(viewModel.StatusMessage); - await synthesizer.Received(1).SpeakAsync("Hello [whispers] world", Arg.Any()); - synthesizer.Received(1).Dispose(); + var session = engine.CreatedSessions.Single(); + Assert.Equal(["Hello [whispers] world"], session.SpeakTexts); } /// /// Proves that is /// at , during both /// and , - /// and reverts to once playback settles back to Idle - so the - /// model-selection control and its embedded settings panel are disabled for the entire - /// active session, not just while audio is actually playing. + /// and reverts to once playback settles back to Idle. /// [Fact] public async Task SynthesisPanelViewModel_Play_SuccessfulSession_CanChangeModelTogglesAcrossLifecycle() { - // Arrange: an installed model, an available device, and a synthesizer that speaks + // Arrange: an installed model, an available device, and a session that speaks // successfully var descriptor = FakeSpeechModel.Descriptor("tts", SpeechModelState.Downloaded, role: SpeechModelRole.Synthesis); var deviceService = Substitute.For(); var availableDevice = PlaybackDevice(); deviceService.CreatePlaybackDevice(Arg.Any()).Returns(availableDevice); - var synthesizer = Substitute.For(); - synthesizer.IsAvailable.Returns(true); - synthesizer.SpeakAsync(Arg.Any(), Arg.Any()).Returns(Task.CompletedTask); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any(), Arg.Any>()).Returns(synthesizer); - var viewModel = new SynthesisPanelViewModel(Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory) + var engine = new FakeSpeechSynthesizerEngine(); + var viewModel = new SynthesisPanelViewModel(Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)) { Text = "Hello world", }; @@ -387,70 +408,69 @@ public async Task SynthesisPanelViewModel_CanChangeModel_ErrorState_IsTrue() [Fact] public async Task SynthesisPanelViewModel_Stop_DuringPlayback_CancelsSessionAndReportsStopped() { - // Arrange: a synthesizer whose SpeakAsync only completes when its token is canceled, + // Arrange: a session whose SpeakAsync only completes when its token is canceled, // simulating an in-flight, indefinitely long utterance var descriptor = FakeSpeechModel.Descriptor("tts", SpeechModelState.Downloaded, role: SpeechModelRole.Synthesis); var deviceService = Substitute.For(); var availableDevice = PlaybackDevice(); deviceService.CreatePlaybackDevice(Arg.Any()).Returns(availableDevice); - var synthesizer = Substitute.For(); - synthesizer.IsAvailable.Returns(true); - synthesizer.SpeakAsync(Arg.Any(), Arg.Any()) - .Returns(callInfo => Task.Delay(Timeout.Infinite, callInfo.ArgAt(1))); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any(), Arg.Any>()).Returns(synthesizer); + var session = new FakeSynthesisSession + { + SpeakImplementation = (_, token) => Task.Delay(Timeout.Infinite, token), + }; + var engine = new FakeSpeechSynthesizerEngine { SessionFactory = _ => session }; var viewModel = new SynthesisPanelViewModel( - Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory); + Catalog(descriptor), deviceService, DeviceSelection(), SessionFactory(engine)); // Act: start playback, then stop it before it would ever complete on its own var playTask = viewModel.PlayCommand.ExecuteAsync(null); - viewModel.StopCommand.Execute(null); + await viewModel.StopCommand.ExecuteAsync(null); await playTask; - // Assert: the session was stopped on the synthesizer itself, cancellation unwound the - // task cleanly, and the panel reports the stop rather than an error - synthesizer.Received(1).Stop(); + // Assert: the session was stopped, cancellation unwound the task cleanly, and the panel + // reports the stop rather than an error + Assert.Equal(1, session.StopCallCount); Assert.Equal(SynthesisPlaybackState.Idle, viewModel.State); Assert.Equal(SynthesisPanelViewModel.StoppedMessage, viewModel.StatusMessage); - synthesizer.Received(1).Dispose(); } /// /// Proves that clicking the shared device-selection panel's Refresh while this panel's - /// Play is in flight requests Stop and awaits its actual completion (not just its - /// cancellation request) before the device refresh proceeds, letting the refresh succeed - /// deterministically rather than racing the still-open playback device. + /// Play is in flight requests Stop and awaits its actual completion before releasing the + /// cached session, letting the refresh succeed deterministically rather than racing the + /// still-open playback device. /// [Fact] public async Task SynthesisPanelViewModel_PreRefreshHook_WhilePlaying_StopsAndAwaitsExecutionTaskBeforeDeviceRefreshSucceeds() { - // Arrange: a synthesizer whose SpeakAsync only completes when its token is canceled, + // Arrange: a session whose SpeakAsync only completes when its token is canceled, // simulating an in-flight, indefinitely long utterance, sharing the panel's own // device-selection panel so Refresh() exercises the real registered hook var descriptor = FakeSpeechModel.Descriptor("tts", SpeechModelState.Downloaded, role: SpeechModelRole.Synthesis); var deviceService = Substitute.For(); var availableDevice = PlaybackDevice(); deviceService.CreatePlaybackDevice(Arg.Any()).Returns(availableDevice); - var synthesizer = Substitute.For(); - synthesizer.IsAvailable.Returns(true); - synthesizer.SpeakAsync(Arg.Any(), Arg.Any()) - .Returns(callInfo => Task.Delay(Timeout.Infinite, callInfo.ArgAt(1))); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any(), Arg.Any>()).Returns(synthesizer); + var session = new FakeSynthesisSession + { + SpeakImplementation = (_, token) => Task.Delay(Timeout.Infinite, token), + }; + var engine = new FakeSpeechSynthesizerEngine { SessionFactory = _ => session }; var deviceSelection = DeviceSelection(); var viewModel = new SynthesisPanelViewModel( - Catalog(descriptor), deviceService, deviceSelection, sessionFactory); + Catalog(descriptor), deviceService, deviceSelection, SessionFactory(engine)); // Act: start playback without awaiting it, then refresh the shared device-selection // panel (as the "Refresh devices" button would) var playTask = viewModel.PlayCommand.ExecuteAsync(null); var exception = await Record.ExceptionAsync(() => deviceSelection.Refresh()); - // Assert: Stop was requested on the synthesizer, the device refresh completed without + // Assert: Stop was requested on the session, the device refresh completed without // throwing (proving the hook genuinely awaited PlayAsync's own completion rather than - // just requesting cancellation and returning immediately), and the panel is idle + // just requesting cancellation and returning immediately), the session was released, and + // the panel is idle Assert.Null(exception); - synthesizer.Received(1).Stop(); + Assert.Equal(1, session.StopCallCount); + Assert.Equal(1, session.DisposeCallCount); Assert.Equal(SynthesisPlaybackState.Idle, viewModel.State); // Cleanup: the in-flight Play task has already completed by the time Refresh() returned @@ -472,24 +492,133 @@ public async Task SynthesisPanelViewModel_PreRefreshHook_WhileIdle_IsNoOpAndDevi var deviceService = Substitute.For(); var availableDevice = PlaybackDevice(); deviceService.CreatePlaybackDevice(Arg.Any()).Returns(availableDevice); - var synthesizer = Substitute.For(); - synthesizer.IsAvailable.Returns(true); - var sessionFactory = Substitute.For(); - sessionFactory.Create(Arg.Any(), Arg.Any(), Arg.Any>()).Returns(synthesizer); + var engine = new FakeSpeechSynthesizerEngine(); var deviceSelection = DeviceSelection(); var viewModel = new SynthesisPanelViewModel( - Catalog(descriptor), deviceService, deviceSelection, sessionFactory); + Catalog(descriptor), deviceService, deviceSelection, SessionFactory(engine)); // Act: refresh the shared device-selection panel while idle, with no Play ever started var exception = await Record.ExceptionAsync(() => deviceSelection.Refresh()); - // Assert: the refresh completes without throwing, Stop is never requested on a - // synthesizer that was never even created, and the panel remains idle + // Assert: the refresh completes without throwing, no session was ever created, and the + // panel remains idle Assert.Null(exception); - synthesizer.DidNotReceive().Stop(); + Assert.Equal(0, engine.CreateSessionCallCount); Assert.Equal(SynthesisPlaybackState.Idle, viewModel.State); } + /// + /// Proves the central bugfix this redesign exists for: calling Play twice in a row with + /// the selected model and settings parameter values unchanged reuses the exact same + /// cached engine and session, rather than reloading the model and recreating the session + /// on every click. + /// + [Fact] + public async Task SynthesisPanelViewModel_Play_CalledTwiceWithUnchangedModelAndParameters_ReusesSameSessionWithoutReload() + { + // Arrange: an installed model, an available device, and a session that speaks + // successfully + var descriptor = FakeSpeechModel.Descriptor("tts", SpeechModelState.Downloaded, role: SpeechModelRole.Synthesis); + var deviceService = Substitute.For(); + var availableDevice = PlaybackDevice(); + deviceService.CreatePlaybackDevice(Arg.Any()).Returns(availableDevice); + var engine = new FakeSpeechSynthesizerEngine(); + var sessionFactory = SessionFactory(engine); + var viewModel = new SynthesisPanelViewModel( + Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory) + { + Text = "Hello world", + }; + + // Act: play twice in a row with nothing changed + await viewModel.PlayCommand.ExecuteAsync(null); + await viewModel.PlayCommand.ExecuteAsync(null); + + // Assert: the engine was loaded exactly once, exactly one session was created, and both + // Play calls spoke through that same session + await sessionFactory.Received(1).LoadAsync( + Arg.Any(), Arg.Any?>(), Arg.Any()); + Assert.Equal(1, engine.CreateSessionCallCount); + var session = engine.CreatedSessions.Single(); + Assert.Equal(["Hello world", "Hello world"], session.SpeakTexts); + } + + /// + /// Proves that changing a settings parameter value between two Play calls reloads the + /// cached engine (since the engine was composed with the stale parameter value) and, as a + /// direct consequence, recreates the session built from it. + /// + [Fact] + public async Task SynthesisPanelViewModel_Play_ParameterValueChanged_ReloadsEngineAndRecreatesSession() + { + // Arrange: a model declaring one boolean parameter, two engines to be loaded in sequence + var parameter = new BooleanParameter("denoise", "Denoise", "Removes noise.", true); + var descriptor = FakeSpeechModel.Descriptor( + "tts", SpeechModelState.Downloaded, role: SpeechModelRole.Synthesis, parameters: [parameter]); + var deviceService = Substitute.For(); + var availableDevice = PlaybackDevice(); + deviceService.CreatePlaybackDevice(Arg.Any()).Returns(availableDevice); + var firstEngine = new FakeSpeechSynthesizerEngine(); + var secondEngine = new FakeSpeechSynthesizerEngine(); + var sessionFactory = SessionFactory(firstEngine, secondEngine); + var viewModel = new SynthesisPanelViewModel( + Catalog(descriptor), deviceService, DeviceSelection(), sessionFactory) + { + Text = "Hello world", + }; + + // Act: play once, change the declared parameter's value, then play again + await viewModel.PlayCommand.ExecuteAsync(null); + var denoiseSetting = viewModel.Settings.Parameters.OfType().Single(); + denoiseSetting.Value = !denoiseSetting.Value; + await viewModel.PlayCommand.ExecuteAsync(null); + + // Assert: the engine was reloaded for the second Play (the stale first engine was + // disposed), and a fresh session was created from the new engine + await sessionFactory.Received(2).LoadAsync( + Arg.Any(), Arg.Any?>(), Arg.Any()); + Assert.Equal(1, firstEngine.DisposeCallCount); + Assert.Equal(1, firstEngine.CreateSessionCallCount); + Assert.Equal(1, secondEngine.CreateSessionCallCount); + } + + /// + /// Proves that changing the selected playback device between two Play calls recreates + /// only the cached session - rebound to the newly selected device - without reloading the + /// unrelated cached engine. + /// + [Fact] + public async Task SynthesisPanelViewModel_Play_PlaybackDeviceChanged_RecreatesSessionButNotEngine() + { + // Arrange: an installed model, two distinct playback devices, and an engine shared across + // both Play calls + var descriptor = FakeSpeechModel.Descriptor("tts", SpeechModelState.Downloaded, role: SpeechModelRole.Synthesis); + var deviceService = Substitute.For(); + deviceService.CreatePlaybackDevice(Arg.Any()).Returns(_ => PlaybackDevice()); + var engine = new FakeSpeechSynthesizerEngine(); + var sessionFactory = SessionFactory(engine); + var deviceSelection = DeviceSelection(SpeakerA, SpeakerB); + deviceSelection.SelectedPlaybackDevice = SpeakerA; + var viewModel = new SynthesisPanelViewModel( + Catalog(descriptor), deviceService, deviceSelection, sessionFactory) + { + Text = "Hello world", + }; + + // Act: play once, change the selected playback device, then play again + await viewModel.PlayCommand.ExecuteAsync(null); + var firstSession = engine.CreatedSessions.Single(); + deviceSelection.SelectedPlaybackDevice = SpeakerB; + await viewModel.PlayCommand.ExecuteAsync(null); + + // Assert: the engine was loaded only once, but a second, distinct session was created + await sessionFactory.Received(1).LoadAsync( + Arg.Any(), Arg.Any?>(), Arg.Any()); + Assert.Equal(2, engine.CreateSessionCallCount); + Assert.Equal(1, firstSession.DisposeCallCount); + Assert.NotSame(firstSession, engine.CreatedSessions[1]); + } + /// /// Proves that the example tag hints are drawn from the library's closed Natural Language /// Audio Tag vocabulary and include the tags this phase's task explicitly calls out. @@ -557,12 +686,12 @@ public void SynthesisPanelViewModel_ModelInstalled_NonMatchingRole_DoesNotTrigge } /// - /// Proves that unsubscribes from + /// Proves that unsubscribes from /// , so a later install completing after /// disposal is never applied. /// [Fact] - public void SynthesisPanelViewModel_Dispose_UnsubscribesFromModelInstalled_NoRefreshAfterDispose() + public async Task SynthesisPanelViewModel_DisposeAsync_UnsubscribesFromModelInstalled_NoRefreshAfterDispose() { // Arrange: a panel composed over a catalog reporting no installed models yet var catalog = Substitute.For(); @@ -573,7 +702,7 @@ public void SynthesisPanelViewModel_Dispose_UnsubscribesFromModelInstalled_NoRef Substitute.For()); // Act: dispose the panel, then simulate a later install completing - viewModel.Dispose(); + await viewModel.DisposeAsync(); descriptors = [FakeSpeechModel.Descriptor("tts", SpeechModelState.Downloaded, role: SpeechModelRole.Synthesis)]; var exception = Record.Exception(() => catalog.ModelInstalled += Raise.Event>( catalog, new ModelInstalledEventArgs("tts", SpeechModelRole.Synthesis))); @@ -585,13 +714,11 @@ public void SynthesisPanelViewModel_Dispose_UnsubscribesFromModelInstalled_NoRef } /// - /// Proves that is safe to call with no - /// active synthesizer, and idempotent when called more than once - this is a new - /// capability on this class, so unlike RecognitionPanelViewModel it has no prior - /// coverage to rely on. + /// Proves that is safe to call with no + /// active session, and idempotent when called more than once. /// [Fact] - public void SynthesisPanelViewModel_Dispose_NoActiveSynthesizer_IsSafeAndIdempotent() + public async Task SynthesisPanelViewModel_DisposeAsync_NoActiveSession_IsSafeAndIdempotent() { // Arrange: a freshly composed panel that never played anything var viewModel = new SynthesisPanelViewModel( @@ -599,10 +726,10 @@ public void SynthesisPanelViewModel_Dispose_NoActiveSynthesizer_IsSafeAndIdempot Substitute.For()); // Act: dispose twice - var exception = Record.Exception(() => + var exception = await Record.ExceptionAsync(async () => { - viewModel.Dispose(); - viewModel.Dispose(); + await viewModel.DisposeAsync(); + await viewModel.DisposeAsync(); }); // Assert: no fault diff --git a/test/DemaConsulting.Speech.Demo.Tests/SynthesisPanelSubsystem/SynthesizerSessionFactoryTests.cs b/test/DemaConsulting.Speech.Demo.Tests/SynthesisPanelSubsystem/SynthesizerSessionFactoryTests.cs index 179ea1e..e1ba1a1 100644 --- a/test/DemaConsulting.Speech.Demo.Tests/SynthesisPanelSubsystem/SynthesizerSessionFactoryTests.cs +++ b/test/DemaConsulting.Speech.Demo.Tests/SynthesisPanelSubsystem/SynthesizerSessionFactoryTests.cs @@ -1,9 +1,7 @@ -using DemaConsulting.Speech.AudioSubsystem; using DemaConsulting.Speech.Demo.SynthesisPanelSubsystem; using DemaConsulting.Speech.Demo.Tests.Fakes; using DemaConsulting.Speech.ModelManagementSubsystem; using DemaConsulting.Speech.SynthesisSubsystem; -using NSubstitute; namespace DemaConsulting.Speech.Demo.Tests.SynthesisPanelSubsystem; @@ -11,7 +9,7 @@ namespace DemaConsulting.Speech.Demo.Tests.SynthesisPanelSubsystem; /// Unit tests for . /// /// -/// The "correct role composes a working synthesizer" path delegates to the library's own +/// The "correct role composes a working engine" path delegates to the library's own /// , which requires an - /// an interface only the library's own assemblies can implement (see the type's remarks). /// That path is therefore outside this test project's reach and remains covered by the @@ -41,72 +39,57 @@ public void SynthesizerSessionFactory_Constructor_NullStore_ThrowsArgumentNullEx } /// - /// Proves that Create rejects a missing model. + /// Proves that LoadAsync rejects a missing model. /// [Fact] - public void SynthesizerSessionFactory_Create_NullModel_ThrowsArgumentNullException() + public async Task SynthesizerSessionFactory_LoadAsync_NullModel_ThrowsArgumentNullException() { // Arrange var factory = new SynthesizerSessionFactory(IsolatedStore()); - var device = Substitute.For(); // Act & Assert - Assert.Throws(() => factory.Create(null!, device)); + await Assert.ThrowsAsync( + () => factory.LoadAsync(null!, null, TestContext.Current.CancellationToken)); } /// - /// Proves that Create rejects a missing playback device. + /// Proves that LoadAsync honestly reports a model that does not implement the library's + /// synthesis role as an unavailable engine, exactly like a model that is not installed, + /// rather than throwing. /// [Fact] - public void SynthesizerSessionFactory_Create_NullDevice_ThrowsArgumentNullException() - { - // Arrange - var factory = new SynthesizerSessionFactory(IsolatedStore()); - - // Act & Assert - Assert.Throws(() => factory.Create(new FakeSpeechModel(), null!)); - } - - /// - /// Proves that Create honestly reports a model that does not implement the library's - /// synthesis role as an unavailable synthesizer, exactly like a model that is not - /// installed, rather than throwing. - /// - [Fact] - public void SynthesizerSessionFactory_Create_ModelNotSynthesisRole_ReturnsUnavailableSynthesizer() + public async Task SynthesizerSessionFactory_LoadAsync_ModelNotSynthesisRole_ReturnsUnavailableEngine() { // Arrange: a fake model that only ever implements the public ISpeechModel contract var factory = new SynthesizerSessionFactory(IsolatedStore()); - var device = Substitute.For(); var model = new FakeSpeechModel(role: SpeechModelRole.Synthesis); // Act - var synthesizer = factory.Create(model, device); + var engine = await factory.LoadAsync(model, null, TestContext.Current.CancellationToken); - // Assert: the honest unavailable fallback, not a real synthesizer or an exception - Assert.Same(UnavailableSpeechSynthesizer.Instance, synthesizer); - Assert.False(synthesizer.IsAvailable); + // Assert: the honest unavailable fallback, not a real engine or an exception + Assert.Same(UnavailableSpeechSynthesizerEngine.Instance, engine); + Assert.False(engine.IsAvailable); } /// - /// Proves that Create honestly reports a wrong-role model as unavailable even when an + /// Proves that LoadAsync honestly reports a wrong-role model as unavailable even when an /// optional parameterValues bag is supplied, since the role check happens before /// any parameter is consulted. /// [Fact] - public void SynthesizerSessionFactory_Create_ModelNotSynthesisRoleWithParameterValues_ReturnsUnavailableSynthesizer() + public async Task SynthesizerSessionFactory_LoadAsync_ModelNotSynthesisRoleWithParameterValues_ReturnsUnavailableEngine() { // Arrange var factory = new SynthesizerSessionFactory(IsolatedStore()); - var device = Substitute.For(); var model = new FakeSpeechModel(role: SpeechModelRole.Synthesis); IReadOnlyDictionary parameterValues = new Dictionary { ["voice"] = "af" }; // Act - var synthesizer = factory.Create(model, device, parameterValues); + var engine = await factory.LoadAsync(model, parameterValues, TestContext.Current.CancellationToken); // Assert - Assert.Same(UnavailableSpeechSynthesizer.Instance, synthesizer); - Assert.False(synthesizer.IsAvailable); + Assert.Same(UnavailableSpeechSynthesizerEngine.Instance, engine); + Assert.False(engine.IsAvailable); } } diff --git a/test/DemaConsulting.Speech.Tests/ModelManagementSubsystem/Fakes/FakeSynthesisModel.cs b/test/DemaConsulting.Speech.Tests/ModelManagementSubsystem/Fakes/FakeSynthesisModel.cs index 821c74a..cc73bbc 100644 --- a/test/DemaConsulting.Speech.Tests/ModelManagementSubsystem/Fakes/FakeSynthesisModel.cs +++ b/test/DemaConsulting.Speech.Tests/ModelManagementSubsystem/Fakes/FakeSynthesisModel.cs @@ -25,7 +25,7 @@ public sealed class FakeSynthesisModel : ISynthesisModel /// The function delegates to, or /// to keep the default hook's behavior (always 0). Lets /// tests prove a resolved, non-zero speaker id reaches - /// ISynthesisEngine.Generate without a real multi-speaker + /// ISynthesisBackend.Generate without a real multi-speaker /// model. /// public FakeSynthesisModel( diff --git a/test/DemaConsulting.Speech.Tests/ModelManagementSubsystem/SherpaOnnxVitsLibriTtsEnglishSynthesisModelTests.cs b/test/DemaConsulting.Speech.Tests/ModelManagementSubsystem/SherpaOnnxVitsLibriTtsEnglishSynthesisModelTests.cs index ecf99f2..134531c 100644 --- a/test/DemaConsulting.Speech.Tests/ModelManagementSubsystem/SherpaOnnxVitsLibriTtsEnglishSynthesisModelTests.cs +++ b/test/DemaConsulting.Speech.Tests/ModelManagementSubsystem/SherpaOnnxVitsLibriTtsEnglishSynthesisModelTests.cs @@ -355,4 +355,23 @@ public async Task SherpaOnnxVitsLibriTtsEnglishSynthesisModel_InstallAsync_Synth await File.ReadAllBytesAsync(extractedTokensPath, TestContext.Current.CancellationToken)); Assert.False(File.Exists(archivePath)); } + + /// + /// Proves that throws + /// for a null or empty staged-files directory, rather than attempting to resolve an + /// archive path against it. + /// + [Theory] + [InlineData(null)] + [InlineData("")] + public async Task SherpaOnnxVitsLibriTtsEnglishSynthesisModel_InstallAsync_NullOrEmptyDirectory_ThrowsArgumentException(string? stagedFilesDirectory) + { + // Arrange + ISpeechModel model = new SherpaOnnxVitsLibriTtsEnglishSynthesisModel(); + + // Act & Assert: ArgumentException.ThrowIfNullOrEmpty throws ArgumentNullException for + // null and ArgumentException for empty, so accept either via the shared base type. + await Assert.ThrowsAnyAsync( + () => model.InstallAsync(stagedFilesDirectory!, TestContext.Current.CancellationToken)); + } } diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/DedicatedWorkerTests.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/DedicatedWorkerTests.cs new file mode 100644 index 0000000..00d2ab0 --- /dev/null +++ b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/DedicatedWorkerTests.cs @@ -0,0 +1,100 @@ +using DemaConsulting.Speech.Diagnostics; +using DemaConsulting.Speech.RecognitionSubsystem; +using NSubstitute; + +namespace DemaConsulting.Speech.Tests.RecognitionSubsystem; + +/// +/// Unit tests for , exercising the cooperative-cancel-then-abandon +/// policy (Decision #4) without any real recognition backend. +/// +public class DedicatedWorkerTests +{ + /// + /// Proves that a delegate which observes the cancellation token cooperatively and returns + /// promptly lets the returned task complete normally, without waiting out the abandon + /// timeout. + /// + [Fact] + public async Task DedicatedWorker_Run_CooperativeCancellation_CompletesPromptly() + { + // Arrange: a worker with a generous abandon timeout that should never be reached, and a + // delegate that waits on the token and returns the instant cancellation is requested + var worker = new DedicatedWorker(abandonTimeout: TimeSpan.FromSeconds(30)); + using var cts = new CancellationTokenSource(); + using var started = new ManualResetEventSlim(false); + + var task = worker.RunAsync( + token => + { + started.Set(); + var signal = new ManualResetEventSlim(false); + while (!token.IsCancellationRequested) + { + signal.Wait(TimeSpan.FromMilliseconds(5)); + } + }, + cts.Token); + + // Act: wait for the delegate to actually start, then cancel + started.Wait(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + await cts.CancelAsync(); + + // Assert: the task completes normally (not cancelled), well within the generous timeout + await task.WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + Assert.True(task.IsCompletedSuccessfully); + } + + /// + /// Proves that a delegate which never observes cancellation is abandoned once + /// elapses, the returned task completes as + /// cancelled, and the abandonment is reported through diagnostics at + /// . + /// + [Fact] + public async Task DedicatedWorker_Run_NonCooperativeDelegate_AbandonsAfterTimeoutAndReportsDiagnostics() + { + // Arrange: a worker with a near-zero abandon timeout and a diagnostics substitute, and a + // delegate that blocks forever regardless of the supplied token + var diagnostics = Substitute.For(); + var worker = new DedicatedWorker( + abandonTimeout: TimeSpan.FromMilliseconds(1), + diagnostics: diagnostics, + diagnosticsCategory: "RecognitionSubsystem"); + using var cts = new CancellationTokenSource(); + using var neverSignaled = new ManualResetEventSlim(false); + + var task = worker.RunAsync(_ => neverSignaled.Wait(), cts.Token); + + // Act: cancel immediately so the delegate is given its (near-zero) abandon window + await cts.CancelAsync(); + + // Assert: the task completes as cancelled rather than hanging forever + await Assert.ThrowsAnyAsync(() => task); + diagnostics.Received().Report( + SpeechDiagnosticLevel.Warning, + "RecognitionSubsystem", + Arg.Is(message => message.Contains("abandon", StringComparison.OrdinalIgnoreCase))); + + // Cleanup: release the abandoned background thread so it can exit + neverSignaled.Set(); + } + + /// + /// Proves that the delegate runs on a dedicated, non-pooled thread + /// (), not an ordinary thread-pool thread. + /// + [Fact] + public async Task DedicatedWorker_Run_UsesLongRunningTaskCreationOption() + { + // Arrange + var worker = new DedicatedWorker(); + var isThreadPoolThread = true; + + // Act: record whether the delegate's thread is a thread-pool thread + await worker.RunAsync(_ => isThreadPoolThread = Thread.CurrentThread.IsThreadPoolThread, TestContext.Current.CancellationToken); + + // Assert: a LongRunning task is scheduled onto a dedicated thread, not the thread pool + Assert.False(isThreadPoolThread); + } +} diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/Fakes/FakeRecognitionEngine.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/Fakes/FakeRecognitionEngine.cs index 33c3d4f..607cff0 100644 --- a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/Fakes/FakeRecognitionEngine.cs +++ b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/Fakes/FakeRecognitionEngine.cs @@ -3,12 +3,12 @@ namespace DemaConsulting.Speech.Tests.RecognitionSubsystem.Fakes; /// -/// Deterministic test double that records the mono samples +/// Deterministic test double that records the mono samples /// it was fed and yields a scripted sequence of recognition results, so the recognizer's /// threading, resampling, and event-emission behavior can be verified with no native /// sherpa-onnx runtime and no downloaded model. /// -internal sealed class FakeRecognitionEngine : IRecognitionEngine +internal sealed class FakeRecognitionEngine : IRecognitionBackend { /// The results this engine yields, in order, one per call. private readonly Queue _scriptedResults; @@ -25,6 +25,17 @@ internal sealed class FakeRecognitionEngine : IRecognitionEngine /// The exception to throw from , when one was scripted. private readonly Exception? _flushException; + /// + /// A handle waits on before returning, when one was supplied, + /// standing in for a native call that never observes cancellation and blocks until it is + /// genuinely done - used to prove the dedicated worker's abandon-timeout policy, not the + /// recognizer, is what keeps teardown bounded. + /// + private readonly WaitHandle? _acceptSamplesBlock; + + /// A callback invoked synchronously from , or for none. + private readonly Action? _onDispose; + /// Every mono sample this engine has been fed, in the order it arrived. private readonly List _acceptedSamples = []; @@ -53,18 +64,33 @@ internal sealed class FakeRecognitionEngine : IRecognitionEngine /// normally. Used to prove the recognizer's teardown still completes and reports the /// fault when the engine's flush fails. /// + /// + /// A handle for to wait on indefinitely before returning, or + /// to accept samples without blocking. Used to simulate a native + /// call that never returns and never observes cancellation. + /// + /// + /// A callback invoked synchronously from , or + /// for none. Used to observe exactly when this shared backend is disposed relative to + /// other state (for example, an owning session's lifecycle), which a simple call count + /// cannot capture. + /// public FakeRecognitionEngine( IEnumerable? scriptedResults = null, Exception? acceptSamplesException = null, Exception? resetException = null, SpeechRecognitionResult? scriptedFlushResult = null, - Exception? flushException = null) + Exception? flushException = null, + WaitHandle? acceptSamplesBlock = null, + Action? onDispose = null) { _scriptedResults = new Queue(scriptedResults ?? []); _acceptSamplesException = acceptSamplesException; _resetException = resetException; _scriptedFlushResult = scriptedFlushResult; _flushException = flushException; + _acceptSamplesBlock = acceptSamplesBlock; + _onDispose = onDispose; } /// Gets every mono sample this engine has been fed, in arrival order. @@ -86,13 +112,16 @@ public FakeRecognitionEngine( public void AcceptSamples(ReadOnlySpan monoSamples) { AcceptSamplesCallCount++; + _acceptedSamples.AddRange(monoSamples.ToArray()); + + // Deliberately blocks after recording the samples and before observing any fault, mirroring + // a genuinely stuck native call that ignores cancellation entirely. + _acceptSamplesBlock?.WaitOne(); if (_acceptSamplesException is not null) { throw _acceptSamplesException; } - - _acceptedSamples.AddRange(monoSamples.ToArray()); } /// @@ -134,5 +163,9 @@ public void Reset() } /// - public void Dispose() => DisposeCallCount++; + public void Dispose() + { + DisposeCallCount++; + _onDispose?.Invoke(); + } } diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/Fakes/FakeRecognitionEngineFactory.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/Fakes/FakeRecognitionEngineFactory.cs index d555094..15e0797 100644 --- a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/Fakes/FakeRecognitionEngineFactory.cs +++ b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/Fakes/FakeRecognitionEngineFactory.cs @@ -4,11 +4,11 @@ namespace DemaConsulting.Speech.Tests.RecognitionSubsystem.Fakes; /// -/// Deterministic test double that hands out a +/// Deterministic test double that hands out a /// pre-configured and records the arguments it was asked /// to load, so composition can be verified without a model directory or a native runtime. /// -internal sealed class FakeRecognitionEngineFactory : IRecognitionEngineFactory +internal sealed class FakeRecognitionEngineFactory : IRecognitionBackendFactory { /// The exception to throw from , when one was scripted. private readonly Exception? _createException; @@ -58,7 +58,7 @@ public FakeRecognitionEngineFactory( public int CreateCallCount { get; private set; } /// - public IRecognitionEngine Create( + public IRecognitionBackend Create( IRecognitionModel model, string installedModelDirectory, IReadOnlyDictionary? parameterValues = null) diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/RecognitionEngineBusyExceptionTests.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/RecognitionEngineBusyExceptionTests.cs new file mode 100644 index 0000000..c999948 --- /dev/null +++ b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/RecognitionEngineBusyExceptionTests.cs @@ -0,0 +1,59 @@ +using DemaConsulting.Speech.RecognitionSubsystem; + +namespace DemaConsulting.Speech.Tests.RecognitionSubsystem; + +/// +/// Unit tests for . +/// +public class RecognitionEngineBusyExceptionTests +{ + /// + /// Proves that the default (parameterless) constructor of + /// produces a non-empty default message. + /// + [Fact] + public void RecognitionEngineBusyException_Constructor_Default_HasNonEmptyMessage() + { + // Act: construct with no arguments + var exception = new RecognitionEngineBusyException(); + + // Assert: a default, non-empty message is provided + Assert.False(string.IsNullOrEmpty(exception.Message)); + } + + /// + /// Proves that exposes the message supplied + /// to its single-argument constructor, confirming standard exception conformance. + /// + [Fact] + public void RecognitionEngineBusyException_Constructor_WithMessage_ExposesMessage() + { + // Arrange: a specific message + const string message = "Cannot create a recognition session: this engine's backend is already leased."; + + // Act: construct the exception with the message + var exception = new RecognitionEngineBusyException(message); + + // Assert: the message is exposed unchanged + Assert.Equal(message, exception.Message); + } + + /// + /// Proves that exposes both the message and + /// inner exception supplied to its two-argument constructor. + /// + [Fact] + public void RecognitionEngineBusyException_Constructor_WithInnerException_ExposesBoth() + { + // Arrange: a message and an inner exception + const string message = "Cannot create a recognition session: this engine's backend is already leased."; + var inner = new InvalidOperationException("lease semaphore faulted"); + + // Act: construct the exception with both + var exception = new RecognitionEngineBusyException(message, inner); + + // Assert: both the message and inner exception are exposed unchanged + Assert.Equal(message, exception.Message); + Assert.Same(inner, exception.InnerException); + } +} diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/RecognitionResultBufferTests.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/RecognitionResultBufferTests.cs new file mode 100644 index 0000000..a6ef2e3 --- /dev/null +++ b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/RecognitionResultBufferTests.cs @@ -0,0 +1,82 @@ +using DemaConsulting.Speech.Diagnostics; +using DemaConsulting.Speech.RecognitionSubsystem; + +namespace DemaConsulting.Speech.Tests.RecognitionSubsystem; + +/// +/// Unit tests for , exercising its provisional-coalescing +/// and final-result-queueing backpressure policy (Decision #5) directly, without a session. +/// +public class RecognitionResultBufferTests +{ + /// + /// Proves that a provisional result buffered before its own final is superseded (never + /// delivered) once that final is added, rather than being delivered after it as a stale + /// partial transcript: the buffer always drains queued finals before the provisional + /// slot, so an un-cleared provisional would otherwise surface after its already-superseded + /// final. + /// + [Fact] + public async Task RecognitionResultBuffer_AddResult_FinalAfterProvisional_SupersedesProvisional() + { + // Arrange + var buffer = new RecognitionResultBuffer(NullSpeechDiagnostics.Instance, "RecognitionSubsystem"); + + // Act: a provisional arrives, then its own final - both before anything reads the buffer + buffer.AddResult(new SpeechRecognitionEvent(new SpeechRecognitionResult("hello", IsFinal: false))); + buffer.AddResult(new SpeechRecognitionEvent(new SpeechRecognitionResult("hello world", IsFinal: true))); + buffer.Complete(); + + // Assert: only the final survived - the superseded provisional was never delivered + var results = await CollectAsync(buffer); + var result = Assert.Single(results); + Assert.True(result.Result.IsFinal); + Assert.Equal("hello world", result.Result.Text); + } + + /// + /// Proves that a provisional result read before its own final arrives is still delivered + /// normally - superseding only ever discards a not-yet-consumed provisional, never one + /// already handed to the consumer. + /// + [Fact] + public async Task RecognitionResultBuffer_AddResult_ProvisionalThenFinalForDifferentUtterances_DeliversBoth() + { + // Arrange + var buffer = new RecognitionResultBuffer(NullSpeechDiagnostics.Instance, "RecognitionSubsystem"); + + // Act: a provisional for one utterance, then (after it would have been read) a final for + // a later, unrelated utterance + buffer.AddResult(new SpeechRecognitionEvent(new SpeechRecognitionResult("partial", IsFinal: false))); + var afterFirstRead = await CollectOneAsync(buffer); + buffer.AddResult(new SpeechRecognitionEvent(new SpeechRecognitionResult("done", IsFinal: true))); + buffer.Complete(); + var afterSecondRead = await CollectAsync(buffer); + + // Assert: both results were delivered, each exactly once + Assert.Equal("partial", afterFirstRead.Text); + Assert.False(afterFirstRead.IsFinal); + var second = Assert.Single(afterSecondRead); + Assert.Equal("done", second.Result.Text); + Assert.True(second.Result.IsFinal); + } + + private static async Task> CollectAsync(RecognitionResultBuffer buffer) + { + var results = new List(); + await foreach (var result in buffer.ReadAllAsync(TestContext.Current.CancellationToken)) + { + results.Add(result); + } + + return results; + } + + private static async Task CollectOneAsync(RecognitionResultBuffer buffer) + { + await using var enumerator = buffer.ReadAllAsync(TestContext.Current.CancellationToken).GetAsyncEnumerator(); + var hasResult = await enumerator.MoveNextAsync(); + Assert.True(hasResult, "Expected at least one buffered result."); + return enumerator.Current.Result; + } +} diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionEngineAccuracyTests.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionEngineAccuracyTests.cs index aabfd55..777044b 100644 --- a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionEngineAccuracyTests.cs +++ b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionEngineAccuracyTests.cs @@ -214,7 +214,7 @@ private static float[] ReadMonoPcm16Wav(string path) /// is fed once the real audio is exhausted so the last in-flight utterance's endpoint /// fires and its final result is not silently lost. /// - private static string Transcribe(IRecognitionEngine engine, float[] samples, int sampleRate) + private static string Transcribe(IRecognitionBackend engine, float[] samples, int sampleRate) { var chunkSize = sampleRate / 10; // 0.1-second chunks, matching this project's existing test convention var transcript = new StringBuilder(); @@ -240,10 +240,10 @@ private static string Transcribe(IRecognitionEngine engine, float[] samples, int } /// - /// Repeatedly calls until it reports nothing + /// Repeatedly calls until it reports nothing /// new, appending every finalized result's text to . /// - private static void DrainFinalResults(IRecognitionEngine engine, StringBuilder transcript) + private static void DrainFinalResults(IRecognitionBackend engine, StringBuilder transcript) { while (engine.TryDecode(out var result)) { @@ -357,7 +357,7 @@ public void SherpaOnnxRecognitionEngine_Transcribe_RealCrossingTheBarRecording_N /// The real engine to feed. Must not be . /// The utterance's normalized mono samples, at the engine's declared rate. /// The sample rate is at, in Hz. - private static void FeedWithoutTrailingSilence(IRecognitionEngine engine, float[] samples, int sampleRate) + private static void FeedWithoutTrailingSilence(IRecognitionBackend engine, float[] samples, int sampleRate) { var chunkSize = sampleRate / 10; // 0.1-second chunks, matching this project's existing test convention @@ -388,7 +388,7 @@ private static void FeedWithoutTrailingSilence(IRecognitionEngine engine, float[ /// finalized result; either would prove the bug this test guards against. /// private static void FeedSilenceCollectingAnyText( - IRecognitionEngine engine, + IRecognitionBackend engine, int sampleRate, double seconds, StringBuilder leakedText) @@ -416,7 +416,7 @@ private static void FeedSilenceCollectingAnyText( /// not discard audio already accepted via AcceptWaveform but not yet decoded, so an /// abandoned utterance's tail (for example a push-to-talk release with no trailing /// silence) decoded into the next session instead of being discarded, even when only - /// silence was fed after . The fix makes + /// silence was fed after . The fix makes /// session-end Reset() create a replacement stream and dispose the existing one /// rather than resetting the existing stream's hypothesis in place, discarding any /// buffered, not-yet-decoded audio along with it. A fake-engine test cannot exercise this: @@ -482,7 +482,7 @@ public void SherpaOnnxRecognitionEngine_Reset_AbandonedUtteranceWithNoTrailingSi } /// - /// Proves that recovers the abandoned + /// Proves that recovers the abandoned /// utterance's trailing words as a final result - rather than merely proving they are /// not lost into the next session, as /// diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionSessionTests.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionSessionTests.cs new file mode 100644 index 0000000..c548332 --- /dev/null +++ b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxRecognitionSessionTests.cs @@ -0,0 +1,1029 @@ +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.Diagnostics; +using DemaConsulting.Speech.ModelManagementSubsystem; +using DemaConsulting.Speech.RecognitionSubsystem; +using DemaConsulting.Speech.Tests.ModelManagementSubsystem.Fakes; +using DemaConsulting.Speech.Tests.RecognitionSubsystem.Fakes; +using NSubstitute; +using NSubstitute.Core; + +namespace DemaConsulting.Speech.Tests.RecognitionSubsystem; + +/// +/// Unit tests for , exercising the full capture → +/// resample → backend → buffered-result pipeline through a substitute capture device and a +/// fake backend, with no microphone and no native sherpa-onnx runtime. +/// +/// +/// Every test constructs its own session directly (bypassing , +/// whose lease behavior is covered by SherpaOnnxSpeechRecognizerEngineTests) with a no-op +/// releaseLease callback, since this session's own lifecycle is what is under test. +/// +public class SherpaOnnxRecognitionSessionTests +{ + /// + /// Proves that starting a freshly created session subscribes to and starts the capture + /// device and transitions it to . + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StartAsync_FromCreated_TransitionsToRunning() + { + // Arrange + var device = CreateCaptureDevice(); + await using var session = CreateSession(new FakeRecognitionEngine(), device); + + // Act + await session.StartAsync(TestContext.Current.CancellationToken); + + // Assert + Assert.Equal(RecognitionSessionState.Running, session.State); + device.Received(1).Start(); + } + + /// + /// Proves that returns control to + /// its caller well before a slow, synchronous + /// call returns - closing the review finding that the native device-start call used to + /// run inline on the calling thread, blocking it (for example a UI thread) for the whole + /// capture window. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StartAsync_SlowDeviceStart_DoesNotBlockCaller() + { + // Arrange: a device whose Start() blocks until this test explicitly releases it + using var startEntered = new ManualResetEventSlim(false); + using var startRelease = new ManualResetEventSlim(false); + var device = CreateCaptureDevice(); + device.When(d => d.Start()).Do(_ => + { + startEntered.Set(); + startRelease.Wait(TestContext.Current.CancellationToken); + }); + await using var session = CreateSession(new FakeRecognitionEngine(), device); + + // Act: call StartAsync and prove it returns a task - without blocking this thread - well + // before the device's own blocking Start() call has returned + var startTask = session.StartAsync(TestContext.Current.CancellationToken); + var enteredInTime = startEntered.Wait(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + + // Assert: the device call was genuinely entered, but StartAsync has not yet completed and + // this thread was never blocked waiting for it - the session is still Starting, not yet + // Running + Assert.True(enteredInTime, "The capture device's Start() was never entered."); + Assert.False(startTask.IsCompleted); + Assert.Equal(RecognitionSessionState.Starting, session.State); + + // Act: release the blocked device call and let the start genuinely finish + startRelease.Set(); + await startTask.WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + + // Assert: the session has now converged to Running, and the device was started exactly + // once + Assert.Equal(RecognitionSessionState.Running, session.State); + device.Received(1).Start(); + } + + /// + /// Proves that starting a session that has already reached + /// throws + /// rather than permitting a restart (Decision #1): a session is single-use. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StartAsync_FromStopped_ThrowsInvalidOperationException() + { + // Arrange: a session that was started and stopped once already + await using var session = CreateSession(new FakeRecognitionEngine(), CreateCaptureDevice()); + await session.StartAsync(TestContext.Current.CancellationToken); + await session.StopAsync(TestContext.Current.CancellationToken); + + // Act & Assert + await Assert.ThrowsAsync(() => session.StartAsync(TestContext.Current.CancellationToken)); + } + + /// + /// Proves that two concurrent calls + /// both complete once the session has converged on , + /// rather than one of them hanging or throwing. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StopAsync_CalledConcurrentlyTwice_BothCompleteOnceStopped() + { + // Arrange + var engine = new FakeRecognitionEngine(); + var device = CreateCaptureDevice(); + await using var session = CreateSession(engine, device); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act: stop concurrently from two callers + var first = session.StopAsync(TestContext.Current.CancellationToken); + var second = session.StopAsync(TestContext.Current.CancellationToken); + await Task.WhenAll(first, second).WaitAsync(TimeSpan.FromSeconds(10), TestContext.Current.CancellationToken); + + // Assert: both callers converged, and the non-idempotent teardown steps ran exactly once - + // the second caller must not have repeated them after observing the Stopping state + Assert.Equal(RecognitionSessionState.Stopped, session.State); + Assert.Equal(1, engine.ResetCallCount); + device.Received(1).Stop(); + } + + /// + /// Proves that stopping flushes any trailing audio the backend had accepted but not yet + /// decoded, so the flushed final result is enumerable via + /// once the stop completes. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StopAsync_FlushesTrailingResultsBeforeCompleting() + { + // Arrange: a backend that yields one flushed final result when the stream ends + var engine = new FakeRecognitionEngine(scriptedFlushResult: new SpeechRecognitionResult("cut off", IsFinal: true)); + await using var session = CreateSession(engine, CreateCaptureDevice()); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act + await session.StopAsync(TestContext.Current.CancellationToken); + + // Assert: the flushed final result is enumerable, and the enumeration ends cleanly since + // the buffer was completed by the same stop + var results = await CollectAsync(session.GetResultsAsync(TestContext.Current.CancellationToken)); + var result = Assert.Single(results); + Assert.Equal("cut off", result.Result.Text); + Assert.True(result.Result.IsFinal); + } + + /// + /// Proves that is single-consumer: + /// a second concurrent enumeration attempt, made while one enumeration is still active, + /// throws immediately. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_GetResultsAsync_CalledConcurrently_ThrowsInvalidOperationException() + { + // Arrange: a session with no data and no completion yet, so the first enumeration blocks + // waiting for more + await using var session = CreateSession(new FakeRecognitionEngine(), CreateCaptureDevice()); + using var firstEnumerationCts = new CancellationTokenSource(); + var firstEnumerator = session.GetResultsAsync(firstEnumerationCts.Token).GetAsyncEnumerator(TestContext.Current.CancellationToken); + var firstMoveNext = firstEnumerator.MoveNextAsync(); + + // Act & Assert: a second, concurrent enumeration attempt fails fast + var secondEnumerator = session.GetResultsAsync(TestContext.Current.CancellationToken).GetAsyncEnumerator(TestContext.Current.CancellationToken); + await Assert.ThrowsAsync(async () => await secondEnumerator.MoveNextAsync()); + + // Cleanup: release the first, still-pending enumeration + await firstEnumerationCts.CancelAsync(); + await Assert.ThrowsAnyAsync(async () => await firstMoveNext); + } + + /// + /// Proves that cancelling the token passed to + /// ends only that enumeration - the session itself keeps running and is not stopped. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_GetResultsAsync_CancelledToken_EndsEnumerationWithoutStoppingSession() + { + // Arrange: a running session with an active enumeration + await using var session = CreateSession(new FakeRecognitionEngine(), CreateCaptureDevice()); + await session.StartAsync(TestContext.Current.CancellationToken); + using var cts = new CancellationTokenSource(); + var enumerationTask = CollectAsync(session.GetResultsAsync(cts.Token)); + + // Act: cancel only the enumeration's token + await cts.CancelAsync(); + + // Assert: the enumeration ends via cancellation, but the session is still running + await Assert.ThrowsAnyAsync(() => enumerationTask); + Assert.Equal(RecognitionSessionState.Running, session.State); + + // Cleanup + await session.StopAsync(TestContext.Current.CancellationToken); + } + + /// + /// Proves that a slow consumer of + /// never sees every provisional result produced while it was not reading: later + /// provisional results coalesce into the latest one (Decision #5). + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_GetResultsAsync_SlowConsumer_CoalescesProvisionalResults() + { + // Arrange: a backend that yields three successive provisional results before anything + // ever reads them + var engine = new FakeRecognitionEngine( + [ + new SpeechRecognitionResult("a", IsFinal: false), + new SpeechRecognitionResult("ab", IsFinal: false), + new SpeechRecognitionResult("abc", IsFinal: false) + ]); + var device = CreateCaptureDevice(); + await using var session = CreateSession(engine, device); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act: feed one block (decoded eagerly into all three scripted provisional results) with + // no consumer reading yet, then stop and read back what survived + RaiseFrameCaptured(device, [0.1f]); + await session.StopAsync(TestContext.Current.CancellationToken); + var results = await CollectAsync(session.GetResultsAsync(TestContext.Current.CancellationToken)); + + // Assert: only the latest provisional result survived the overwrite + var result = Assert.Single(results); + Assert.Equal("abc", result.Result.Text); + Assert.False(result.Result.IsFinal); + } + + /// + /// Proves that a slow consumer of + /// never loses a final result while the buffered backlog remains under the byte cap + /// (Decision #5): every final result is still delivered, in order. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_GetResultsAsync_SlowConsumer_NeverDropsFinalResultsUnderByteCap() + { + // Arrange: a backend that yields several short final results before anything reads them + var engine = new FakeRecognitionEngine( + [ + new SpeechRecognitionResult("one", IsFinal: true), + new SpeechRecognitionResult("two", IsFinal: true), + new SpeechRecognitionResult("three", IsFinal: true) + ]); + var device = CreateCaptureDevice(); + await using var session = CreateSession(engine, device); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act + RaiseFrameCaptured(device, [0.1f]); + await session.StopAsync(TestContext.Current.CancellationToken); + var results = await CollectAsync(session.GetResultsAsync(TestContext.Current.CancellationToken)); + + // Assert: every final result survived, in order + Assert.Collection( + results, + r => Assert.Equal("one", r.Result.Text), + r => Assert.Equal("two", r.Result.Text), + r => Assert.Equal("three", r.Result.Text)); + } + + /// + /// Proves that once a session has transitioned to , + /// enumerating throws + /// from the enumerator. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_GetResultsAsync_SessionFaulted_ThrowsRecognitionSessionFaultedException() + { + // Arrange: a running session whose device goes unavailable mid-session + var device = CreateCaptureDevice(); + await using var session = CreateSession(new FakeRecognitionEngine(), device); + await session.StartAsync(TestContext.Current.CancellationToken); + device.IsAvailable.Returns(false); + RaiseFrameCaptured(device, [0.1f]); + + // Act & Assert + await Assert.ThrowsAsync(async () => + { + await foreach (var _ in session.GetResultsAsync(TestContext.Current.CancellationToken)) + { + // No iterations are expected to survive: the fault is observed before/at this point. + } + }); + } + + /// + /// Proves that a dedicated worker's delegate which never observes cancellation (standing + /// in for a stuck native call) is bounded by the abandon-timeout policy rather than + /// hanging forever, and that the + /// abandonment is reported through diagnostics (Decision #4). + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_NativeCallExceedsAbandonTimeout_TaskCompletesAndDiagnosticsReportsWarning() + { + // Arrange: a backend whose AcceptSamples blocks forever, and a worker with a near-zero + // abandon timeout so the test stays fast + using var neverSignaled = new ManualResetEvent(false); + var engine = new FakeRecognitionEngine(acceptSamplesBlock: neverSignaled); + var diagnostics = Substitute.For(); + var worker = new DedicatedWorker( + abandonTimeout: TimeSpan.FromMilliseconds(1), + diagnostics: diagnostics, + diagnosticsCategory: "RecognitionSubsystem"); + var device = CreateCaptureDevice(); + await using var session = CreateSession(engine, device, worker: worker); + await session.StartAsync(TestContext.Current.CancellationToken); + RaiseFrameCaptured(device, [0.1f]); + + // Give the pump thread a moment to actually enter the blocking call before stopping + await Task.Delay(TimeSpan.FromMilliseconds(50), TestContext.Current.CancellationToken); + + // Act: stop must complete within a bounded time, not hang on the stuck backend call + await session.StopAsync(TestContext.Current.CancellationToken).WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + + // Assert: teardown still converged, and the abandonment was reported + Assert.Equal(RecognitionSessionState.Stopped, session.State); + diagnostics.Received().Report( + SpeechDiagnosticLevel.Warning, + "RecognitionSubsystem", + Arg.Is(message => message.Contains("abandon", StringComparison.OrdinalIgnoreCase))); + + // Cleanup: release the abandoned background thread so it can exit + neverSignaled.Set(); + } + + /// + /// Proves that cancelling 's own + /// while is still + /// blocking is actually honored - closing the review finding that the device-start worker + /// call used , so a caller's cancellation request was + /// ignored indefinitely rather than bounded by the abandon-timeout policy every other + /// dedicated-worker call in this session already uses. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StartAsync_CancelledWhileDeviceStarting_AbandonsWithinTimeout() + { + // Arrange: a device whose Start() blocks forever (standing in for a stuck native call, + // consistent with IAudioCaptureDevice.Start() having no cancellation token of its own), + // and a worker with a near-zero abandon timeout so the test stays fast + using var startEntered = new ManualResetEventSlim(false); + using var neverReturns = new ManualResetEventSlim(false); + var device = CreateCaptureDevice(); + device.When(d => d.Start()).Do(_ => + { + startEntered.Set(); + neverReturns.Wait(); + }); + var diagnostics = Substitute.For(); + var worker = new DedicatedWorker( + abandonTimeout: TimeSpan.FromMilliseconds(1), + diagnostics: diagnostics, + diagnosticsCategory: "RecognitionSubsystem"); + await using var session = CreateSession(new FakeRecognitionEngine(), device, worker: worker, diagnostics: diagnostics); + + using var startCts = new CancellationTokenSource(); + var startTask = session.StartAsync(startCts.Token); + var enteredInTime = startEntered.Wait(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + Assert.True(enteredInTime, "The capture device's Start() was never entered."); + + // Act: cancel the caller's own token while Start() is still blocked; the call must + // complete within a bounded time rather than waiting forever for a device call that never + // honors cancellation + await startCts.CancelAsync(); + + // Assert: the abandoned start surfaces as a failure to start, not a silent hang, and the + // abandonment was reported + await Assert.ThrowsAsync( + () => startTask.WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken)); + Assert.Equal(RecognitionSessionState.Faulted, session.State); + diagnostics.Received().Report( + SpeechDiagnosticLevel.Warning, + "RecognitionSubsystem", + Arg.Is(message => message.Contains("abandon", StringComparison.OrdinalIgnoreCase))); + + // Cleanup: release the abandoned background thread so it can exit + neverReturns.Set(); + } + + /// + /// Proves that a capture device going unavailable mid-session transitions the session to + /// . + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_DeviceLostMidSession_TransitionsToFaulted() + { + // Arrange: a running session whose device later reports itself unavailable + var device = CreateCaptureDevice(); + await using var session = CreateSession(new FakeRecognitionEngine(), device); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act + device.IsAvailable.Returns(false); + RaiseFrameCaptured(device, [0.1f]); + + // Assert + Assert.Equal(RecognitionSessionState.Faulted, session.State); + } + + /// + /// Proves that raises every state + /// transition, in order, across a full Start/Stop lifecycle. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StateChanged_EmitsEveryTransitionInOrder() + { + // Arrange + await using var session = CreateSession(new FakeRecognitionEngine(), CreateCaptureDevice()); + var transitions = new List<(RecognitionSessionState Previous, RecognitionSessionState Current)>(); + session.StateChanged += (_, args) => transitions.Add((args.Previous, args.Current)); + + // Act + await session.StartAsync(TestContext.Current.CancellationToken); + await session.StopAsync(TestContext.Current.CancellationToken); + + // Assert + Assert.Equal( + [ + (RecognitionSessionState.Created, RecognitionSessionState.Starting), + (RecognitionSessionState.Starting, RecognitionSessionState.Running), + (RecognitionSessionState.Running, RecognitionSessionState.Stopping), + (RecognitionSessionState.Stopping, RecognitionSessionState.Stopped) + ], transitions); + } + + /// + /// Proves, across many iterations, that the Stopping transition is always raised before + /// the Stopped transition it logically precedes (findings 32/33): with every dependency + /// trivially fast to complete (as here), the teardown that produces the Stopped transition + /// can genuinely run to completion synchronously the instant it is started, so only + /// starting it strictly after Stopping has already been raised - never racing that + /// ordering on whether the underlying tasks happen to complete synchronously - keeps every + /// iteration below in order. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StateChanged_StoppingAlwaysPrecedesStoppedAcrossManyIterations() + { + for (var iteration = 0; iteration < 200; iteration++) + { + // Arrange + await using var session = CreateSession(new FakeRecognitionEngine(), CreateCaptureDevice()); + var transitions = new List(); + session.StateChanged += (_, args) => transitions.Add(args.Current); + + // Act + await session.StartAsync(TestContext.Current.CancellationToken); + await session.StopAsync(TestContext.Current.CancellationToken); + + // Assert: Stopping must appear, and strictly before Stopped + var stoppingIndex = transitions.IndexOf(RecognitionSessionState.Stopping); + var stoppedIndex = transitions.IndexOf(RecognitionSessionState.Stopped); + Assert.True(stoppingIndex >= 0, $"Iteration {iteration}: Stopping transition was never raised."); + Assert.True(stoppedIndex >= 0, $"Iteration {iteration}: Stopped transition was never raised."); + Assert.True( + stoppingIndex < stoppedIndex, + $"Iteration {iteration}: Stopped (index {stoppedIndex}) was raised before or alongside Stopping (index {stoppingIndex}): [{string.Join(", ", transitions)}]"); + } + } + + /// + /// Proves that a captured frame still flows through downmixing into the backend - the + /// same pipeline SherpaOnnxSpeechRecognizer used before the Engine/Session split. + /// Uses a device sample rate equal to the model's so the resampling step is an identity + /// pass-through, keeping the downmix arithmetic exactly predictable. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_FrameCaptured_StereoAtModelRate_FeedsDownmixedMonoToBackend() + { + // Arrange: a stereo device already at the model's declared rate + var device = CreateCaptureDevice(sampleRate: 16000, channelCount: 2); + var engine = new FakeRecognitionEngine(); + await using var session = CreateSession(engine, device); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act: raise one stereo block of four frames, then stop to drain the pipeline + RaiseFrameCaptured(device, [0.0f, 0.0f, 1.0f, 1.0f, 2.0f, 2.0f, 3.0f, 3.0f]); + await session.StopAsync(TestContext.Current.CancellationToken); + + // Assert: each stereo frame's two identical channels averaged to that same value + Assert.Equal(1, engine.AcceptSamplesCallCount); + Assert.Equal([0.0f, 1.0f, 2.0f, 3.0f], engine.AcceptedSamples); + } + + /// + /// Proves that a final result's text is passed through the owning model's + /// hook, with isFinal + /// set to , before it is buffered for . + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_FrameCaptured_FinalResult_AppliesModelNormalizeTextWithIsFinalTrue() + { + // Arrange: a model whose NormalizeText is a distinguishable, recorded transform + var device = CreateCaptureDevice(); + var engine = new FakeRecognitionEngine([new SpeechRecognitionResult("HELLO WORLD", IsFinal: true)]); + var calls = new List<(string Text, bool IsFinal)>(); + var model = new FakeRecognitionModel(normalizeText: (text, isFinal) => + { + calls.Add((text, isFinal)); + return $"normalized:{text}"; + }); + await using var session = CreateSession(engine, device, model); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act + RaiseFrameCaptured(device, [0.1f, 0.2f]); + await session.StopAsync(TestContext.Current.CancellationToken); + var results = await CollectAsync(session.GetResultsAsync(TestContext.Current.CancellationToken)); + + // Assert: the model saw the raw text with isFinal true, and the buffered result carries + // the model's normalized text while preserving IsFinal + var result = Assert.Single(results); + Assert.Equal("normalized:HELLO WORLD", result.Result.Text); + Assert.True(result.Result.IsFinal); + var call = Assert.Single(calls); + Assert.Equal("HELLO WORLD", call.Text); + Assert.True(call.IsFinal); + } + + /// + /// Proves that a fault in the backend's during + /// teardown is contained and reported rather than propagated, matching the best-effort + /// reset guarantee documented on . + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StopAsync_BackendResetFails_CompletesAndReportsFault() + { + // Arrange: a running session whose backend always faults on Reset() + var diagnostics = Substitute.For(); + var engine = new FakeRecognitionEngine(resetException: new InvalidOperationException("reset failed")); + await using var session = CreateSession(engine, CreateCaptureDevice(), diagnostics: diagnostics); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act + var exception = await Record.ExceptionAsync(() => session.StopAsync(TestContext.Current.CancellationToken)); + + // Assert: nothing escaped, the fault was reported, and teardown still converged + Assert.Null(exception); + Assert.Equal(RecognitionSessionState.Stopped, session.State); + diagnostics.Received().Report( + SpeechDiagnosticLevel.Error, + "RecognitionSubsystem", + Arg.Is(message => message.Contains("Failed to reset the recognition backend", StringComparison.Ordinal))); + } + + /// + /// Proves that a concurrent cannot + /// observe a call's intermediate + /// state, converge the session to + /// , and return while the device is still + /// being started - closing the review finding that this race could leave capture running + /// against a session the caller believes is stopped. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StopAsync_ConcurrentWithStartAsync_DeviceEndsGenuinelyStopped() + { + // Arrange: a device whose Start() blocks until this test explicitly releases it, + // standing in for a slow synchronous device start that a concurrent StopAsync could + // otherwise race past. + using var startEntered = new ManualResetEventSlim(false); + using var startRelease = new ManualResetEventSlim(false); + var device = CreateCaptureDevice(); + device.When(d => d.Start()).Do(_ => + { + startEntered.Set(); + startRelease.Wait(TestContext.Current.CancellationToken); + }); + var engine = new FakeRecognitionEngine(); + await using var session = CreateSession(engine, device); + + // Act: begin starting, wait until Start() has genuinely been entered, then begin + // stopping concurrently before Start() returns. Both calls are dispatched through + // Task.Run: StartAsync's body (and therefore the lock it holds) runs synchronously on + // its own thread, and StopAsync - a synchronous method that itself blocks acquiring the + // same lock before it can even return a Task - must run on a thread other than this + // test's own, or this test's own thread would deadlock waiting on the very lock whose + // release depends on this test later calling startRelease.Set(). + var startTask = Task.Run(() => session.StartAsync(TestContext.Current.CancellationToken), TestContext.Current.CancellationToken); + startEntered.Wait(TestContext.Current.CancellationToken); + var stopTask = Task.Run( + async () => await session.StopAsync(TestContext.Current.CancellationToken), + TestContext.Current.CancellationToken); + + // Assert: StopAsync cannot race ahead of the still-in-flight StartAsync - both calls + // serialize on the same lock, so StopAsync has not completed while Start() is blocked + await Task.Delay(TimeSpan.FromMilliseconds(50), TestContext.Current.CancellationToken); + Assert.False(stopTask.IsCompleted); + + // Act: let the device start complete + startRelease.Set(); + await startTask; + await stopTask.WaitAsync(TimeSpan.FromSeconds(10), TestContext.Current.CancellationToken); + + // Assert: the session converged to genuinely Stopped, and the device was genuinely + // started then genuinely stopped exactly once each - never left running + Assert.Equal(RecognitionSessionState.Stopped, session.State); + device.Received(1).Start(); + device.Received(1).Stop(); + } + + /// + /// Proves that the pump is draining the pending-frame channel before a synchronous-replay + /// capture device (for example, a file-backed device) emits its blocks from within + /// Start(), so a file longer than the bounded channel's capacity does not silently + /// drop its earliest blocks before anything exists to read them. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StartAsync_DeviceEmitsManyBlocksSynchronouslyFromStart_NoneAreDropped() + { + // Arrange: a device whose Start() synchronously raises far more FrameCaptured blocks than + // the pump's bounded channel can hold at once, exactly like a file-backed device replaying + // an entire file before Start() returns. Each block is only raised once the previous one + // has genuinely been accepted by the backend - deterministically proving the pump is + // already draining the channel concurrently with Start(), rather than racing an + // unsynchronized flood against however fast the pump thread happens to be scheduled. + const int blockCount = 200; + var device = CreateCaptureDevice(); + IAudioCaptureDevice? capturedDevice = null; + var engine = new FakeRecognitionEngine(); + device.When(d => d.Start()).Do(_ => + { + for (var i = 0; i < blockCount; i++) + { + RaiseFrameCaptured(capturedDevice!, [0.1f]); + + var expected = i + 1; + Assert.True( + SpinWait.SpinUntil(() => engine.AcceptSamplesCallCount >= expected, TimeSpan.FromSeconds(5)), + $"The pump never accepted block {expected} of {blockCount}; it was not draining the channel concurrently with Start()."); + } + }); + capturedDevice = device; + await using var session = CreateSession(engine, device); + + // Act: start (synchronously emitting every block before returning, each one drained + // before the next is raised), then stop to converge the pump + await session.StartAsync(TestContext.Current.CancellationToken); + await session.StopAsync(TestContext.Current.CancellationToken).WaitAsync(TimeSpan.FromSeconds(10), TestContext.Current.CancellationToken); + + // Assert: every block emitted from within Start() was still accepted by the backend - + // none were dropped because the pump was already draining the channel before Start() ran + Assert.Equal(blockCount, engine.AcceptSamplesCallCount); + } + + /// + /// Proves that a session faulted mid-stream (for example, by the capture device becoming + /// unavailable) still runs the full teardown - resetting the shared backend and stopping + /// the device - once disposed, rather than releasing the engine's lease while the capture + /// stream and pump could still be active. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_DeviceLostMidSession_DisposeAsyncStillTearsDownBackendAndDevice() + { + // Arrange: a running session whose device later reports itself unavailable + var device = CreateCaptureDevice(); + var engine = new FakeRecognitionEngine(); + var releaseCount = 0; + await using var session = CreateSession(engine, device, releaseLease: () => releaseCount++); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act: fault the session, then dispose it + device.IsAvailable.Returns(false); + RaiseFrameCaptured(device, [0.1f]); + Assert.Equal(RecognitionSessionState.Faulted, session.State); + await session.DisposeAsync(); + + // Assert: teardown genuinely ran - the backend was reset and the device stopped - and + // the fault was preserved through to Disposed, with the lease released exactly once + Assert.Equal(RecognitionSessionState.Disposed, session.State); + Assert.Equal(1, engine.ResetCallCount); + device.Received(1).Stop(); + Assert.Equal(1, releaseCount); + } + + /// + /// Proves that 's + /// parameter is genuinely observed (finding 19): a caller + /// who cancels it stops waiting for that call's own completion promptly, without being + /// stuck behind a slow drain - but the shared teardown itself is never aborted by that + /// cancellation, since it is shared with every other concurrent/overlapping caller (and + /// ), all of whom still require the + /// drain to genuinely happen. + /// + [Fact(Timeout = 10000)] + public async Task SherpaOnnxRecognitionSession_StopAsync_CallerTokenCanceled_ReturnsEarlyWithoutAbortingSharedTeardown() + { + // Arrange: a backend whose AcceptSamples blocks until this test releases it + using var block = new ManualResetEvent(false); + var engine = new FakeRecognitionEngine(acceptSamplesBlock: block); + var device = CreateCaptureDevice(); + await using var session = CreateSession(engine, device); + await session.StartAsync(TestContext.Current.CancellationToken); + RaiseFrameCaptured(device, [0.1f]); + + // Wait until the pump has genuinely entered (and is blocked inside) AcceptSamples + SpinWait.SpinUntil(() => engine.AcceptSamplesCallCount >= 1, TimeSpan.FromSeconds(5)); + Assert.Equal(1, engine.AcceptSamplesCallCount); + + // Act: call StopAsync with a token that is canceled immediately after the call begins + using var cts = new CancellationTokenSource(); + var stopTask = session.StopAsync(cts.Token); + await cts.CancelAsync(); + + // Assert: this caller's own wait is canceled promptly, well before the still-blocked + // drain could ever converge on its own + Exception? stopException = null; + try + { + await stopTask; + } + catch (Exception ex) + { + stopException = ex; + } + + Assert.IsType(stopException, exactMatch: false); + + // Assert: the shared teardown itself was not aborted by that cancellation - a second, + // uncancelled StopAsync call still observes the same in-flight teardown, which only + // converges once the backend genuinely unblocks + var secondStopTask = session.StopAsync(TestContext.Current.CancellationToken); + await Task.Delay(TimeSpan.FromMilliseconds(100), TestContext.Current.CancellationToken); + Assert.False(secondStopTask.IsCompleted); + + block.Set(); + await secondStopTask.WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + Assert.Equal(RecognitionSessionState.Stopped, session.State); + } + + /// + /// Proves that the engine's exclusivity lease is not released until the dedicated pump + /// worker has genuinely exited - not merely been abandoned after its timeout - so a new + /// session (or engine disposal) can never touch or dispose the shared backend while an + /// abandoned pump thread is still inside a blocking backend call. + /// + [Fact(Timeout = 10000)] + public async Task SherpaOnnxRecognitionSession_DisposeAsync_AbandonedPumpWorker_DoesNotReleaseLeaseUntilWorkerExits() + { + // Arrange: a backend whose AcceptSamples blocks forever (until this test releases it), + // and a worker with a near-zero abandon timeout so the test stays fast + using var neverSignaled = new ManualResetEvent(false); + var engine = new FakeRecognitionEngine(acceptSamplesBlock: neverSignaled); + var worker = new DedicatedWorker( + abandonTimeout: TimeSpan.FromMilliseconds(1), + diagnostics: NullSpeechDiagnostics.Instance, + diagnosticsCategory: "RecognitionSubsystem"); + var device = CreateCaptureDevice(); + var releaseCount = 0; + var session = CreateSession(engine, device, worker: worker, releaseLease: () => releaseCount++); + await session.StartAsync(TestContext.Current.CancellationToken); + RaiseFrameCaptured(device, [0.1f]); + await Task.Delay(TimeSpan.FromMilliseconds(50), TestContext.Current.CancellationToken); + + // Act: stop (completes quickly via the abandon policy) then begin disposing + await session.StopAsync(TestContext.Current.CancellationToken).WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + var disposeTask = session.DisposeAsync().AsTask(); + + // Assert: the lease must not be released while the abandoned pump thread is still stuck + // inside the backend's blocking call + await Task.Delay(TimeSpan.FromMilliseconds(100), TestContext.Current.CancellationToken); + Assert.Equal(0, releaseCount); + Assert.False(disposeTask.IsCompleted); + + // Act: release the abandoned background thread so it can genuinely exit + neverSignaled.Set(); + await disposeTask.WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + + // Assert: only once the worker genuinely exited was the lease released + Assert.Equal(1, releaseCount); + } + + /// + /// Proves that two concurrent + /// calls share the exact same in-flight teardown, rather than the second call returning + /// the instant the first merely begins - both complete only once the real teardown (and + /// the lease release it gates) is genuinely done. + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_DisposeAsync_CalledConcurrentlyTwice_BothCompleteAfterSingleTeardown() + { + // Arrange + var engine = new FakeRecognitionEngine(); + var device = CreateCaptureDevice(); + var releaseCount = 0; + var session = CreateSession(engine, device, releaseLease: () => releaseCount++); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act: dispose concurrently from two callers + async Task DisposeOnceAsync() => await session.DisposeAsync(); + var first = DisposeOnceAsync(); + var second = DisposeOnceAsync(); + await Task.WhenAll(first, second).WaitAsync(TimeSpan.FromSeconds(10), TestContext.Current.CancellationToken); + + // Assert: both callers converged, the destructive teardown steps ran exactly once, and + // the lease was released exactly once + Assert.Equal(RecognitionSessionState.Disposed, session.State); + Assert.Equal(1, engine.ResetCallCount); + device.Received(1).Stop(); + Assert.Equal(1, releaseCount); + } + + /// + /// Proves that a handler calling + /// back into this session (for example ) + /// and then synchronously blocking on the result does not deadlock (finding 21): the event + /// must be raised only after _syncRoot has been released. + /// + [Fact(Timeout = 5000)] + public async Task SherpaOnnxRecognitionSession_StateChangedHandlerBlocksOnStopAsync_DoesNotDeadlock() + { + // Arrange + var session = CreateSession(new FakeRecognitionEngine(), CreateCaptureDevice()); + var handlerCompleted = false; + session.StateChanged += (_, args) => + { + if (args.Current == RecognitionSessionState.Running) + { + // A host handler that synchronously blocks on the result of calling back into + // this very session must not deadlock against the state lock. + session.StopAsync(CancellationToken.None).GetAwaiter().GetResult(); + handlerCompleted = true; + } + }; + + try + { + // Act + await session.StartAsync(TestContext.Current.CancellationToken); + + // Assert + Assert.True(handlerCompleted); + Assert.Equal(RecognitionSessionState.Stopped, session.State); + } + finally + { + await session.DisposeAsync(); + } + } + + /// + /// Proves that a backend exception thrown from + /// faults the session and completes + /// with , rather than leaving the session + /// with its result buffer open forever + /// (finding 22). + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_BackendThrowsFromAcceptSamples_FaultsSessionAndCompletesResultBuffer() + { + // Arrange + var engine = new FakeRecognitionEngine(acceptSamplesException: new InvalidOperationException("Backend failure.")); + var device = CreateCaptureDevice(); + await using var session = CreateSession(engine, device); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act: a captured block reaches the pump thread and the backend throws while accepting it + RaiseFrameCaptured(device, [0.1f]); + SpinWait.SpinUntil(() => session.State == RecognitionSessionState.Faulted, TimeSpan.FromSeconds(5)); + + // Assert: the session faulted, and a consumer of GetResultsAsync is unblocked with the fault + // rather than hanging forever + Assert.Equal(RecognitionSessionState.Faulted, session.State); + await Assert.ThrowsAsync(async () => + { + await foreach (var _ in session.GetResultsAsync(TestContext.Current.CancellationToken)) + { + // No iterations are expected to survive the fault. + } + }); + } + + /// + /// Proves that calling on a session + /// that was never started still completes its result buffer, so + /// returns instead of hanging + /// forever (finding 23). + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StopAsync_BeforeStartAsync_CompletesResultBuffer() + { + // Arrange + await using var session = CreateSession(new FakeRecognitionEngine(), CreateCaptureDevice()); + + // Act + await session.StopAsync(TestContext.Current.CancellationToken); + + // Assert + Assert.Equal(RecognitionSessionState.Stopped, session.State); + using var bounded = new CancellationTokenSource(TimeSpan.FromSeconds(5)); + await foreach (var _ in session.GetResultsAsync(bounded.Token)) + { + // No results are expected: the buffer should complete immediately and empty. + } + } + + /// + /// Proves that when the dedicated pump worker is abandoned after its timeout, the shared + /// recognition backend is not reset (nor the device stopped) until the pump thread has + /// genuinely exited, closing the race in which the raw pump thread could still be inside + /// while teardown concurrently reset the + /// same backend (finding 24). + /// + [Fact(Timeout = 10000)] + public async Task SherpaOnnxRecognitionSession_AbandonedPumpWorker_DoesNotResetBackendUntilWorkerExits() + { + // Arrange: a backend whose AcceptSamples blocks forever, and a worker with a near-zero + // abandon timeout so the test stays fast + using var neverSignaled = new ManualResetEvent(false); + var engine = new FakeRecognitionEngine(acceptSamplesBlock: neverSignaled); + var worker = new DedicatedWorker( + abandonTimeout: TimeSpan.FromMilliseconds(1), + diagnostics: NullSpeechDiagnostics.Instance, + diagnosticsCategory: "RecognitionSubsystem"); + var device = CreateCaptureDevice(); + await using var session = CreateSession(engine, device, worker: worker); + await session.StartAsync(TestContext.Current.CancellationToken); + RaiseFrameCaptured(device, [0.1f]); + await Task.Delay(TimeSpan.FromMilliseconds(50), TestContext.Current.CancellationToken); + + // Act: stop completes quickly via the abandon policy + await session.StopAsync(TestContext.Current.CancellationToken).WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + + // Assert: the backend must not be reset, nor the device stopped, while the abandoned pump + // thread may still be inside the blocking backend call + await Task.Delay(TimeSpan.FromMilliseconds(100), TestContext.Current.CancellationToken); + Assert.Equal(0, engine.ResetCallCount); + device.DidNotReceive().Stop(); + + // Act: release the abandoned background thread so it can genuinely exit + neverSignaled.Set(); + + // Waits for both the backend reset and the device stop (not merely the first of the two + // sequential calls ResetBackendAndStopDeviceCore makes) so this assertion cannot flake on + // the brief window between them. + SpinWait.SpinUntil( + () => engine.ResetCallCount >= 1 && device.ReceivedCalls().Any(call => call.GetMethodInfo().Name == nameof(IAudioCaptureDevice.Stop)), + TimeSpan.FromSeconds(5)); + + // Assert: only once the worker genuinely exited was the backend reset and device stopped + Assert.Equal(1, engine.ResetCallCount); + device.Received(1).Stop(); + } + + /// + /// Proves that passing an already-canceled token to + /// only bounds this caller's own wait - the call throws + /// immediately rather than waiting for teardown - while the shared teardown itself still + /// converges for every other observer (findings 19/25). + /// + [Fact] + public async Task SherpaOnnxRecognitionSession_StopAsync_PreCanceledToken_ThrowsButTeardownStillConverges() + { + // Arrange + var engine = new FakeRecognitionEngine(); + var device = CreateCaptureDevice(); + await using var session = CreateSession(engine, device); + await session.StartAsync(TestContext.Current.CancellationToken); + + using var preCanceled = new CancellationTokenSource(); + await preCanceled.CancelAsync(); + + // Act & Assert: this caller's own wait is bounded by its already-canceled token + await Assert.ThrowsAsync(() => session.StopAsync(preCanceled.Token)); + + // Assert: the shared teardown was never aborted by that caller's canceled wait - it still + // converges for every other observer + SpinWait.SpinUntil(() => session.State == RecognitionSessionState.Stopped, TimeSpan.FromSeconds(5)); + Assert.Equal(RecognitionSessionState.Stopped, session.State); + } + + /// + /// Builds a substitute capture device reporting itself available with the given capture + /// format. + /// + private static IAudioCaptureDevice CreateCaptureDevice(int sampleRate = 16000, int channelCount = 1) + { + var device = Substitute.For(); + device.IsAvailable.Returns(true); + device.SampleRate.Returns(sampleRate); + device.ChannelCount.Returns(channelCount); + return device; + } + + /// + /// Raises the substitute capture device's + /// event with one block of interleaved samples, standing in for a real audio callback. + /// + private static void RaiseFrameCaptured(IAudioCaptureDevice captureDevice, float[] samples) + { + captureDevice.FrameCaptured += Raise.Event>( + captureDevice, + new AudioCaptureFrameEventArgs(samples)); + } + + /// + /// Builds a directly over the supplied backend + /// and device, bypassing 's lease since this + /// session's own lifecycle, not the engine's, is under test here. + /// + private static SherpaOnnxRecognitionSession CreateSession( + IRecognitionBackend engine, + IAudioCaptureDevice device, + IRecognitionModel? model = null, + ISpeechDiagnostics? diagnostics = null, + DedicatedWorker? worker = null, + Action? releaseLease = null) => + new( + engine, + device, + targetSampleRate: 16000, + model ?? new FakeRecognitionModel(), + releaseLease: releaseLease ?? (static () => { }), + diagnostics ?? NullSpeechDiagnostics.Instance, + worker ?? new DedicatedWorker(diagnostics: diagnostics ?? NullSpeechDiagnostics.Instance)); + + /// Drains an asynchronous sequence of recognition events into a list. + private static async Task> CollectAsync(IAsyncEnumerable source) + { + var results = new List(); + await foreach (var item in source) + { + results.Add(item); + } + + return results; + } +} diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerEngineTests.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerEngineTests.cs new file mode 100644 index 0000000..dc0b5f0 --- /dev/null +++ b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerEngineTests.cs @@ -0,0 +1,257 @@ +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.Diagnostics; +using DemaConsulting.Speech.RecognitionSubsystem; +using DemaConsulting.Speech.Tests.ModelManagementSubsystem.Fakes; +using DemaConsulting.Speech.Tests.RecognitionSubsystem.Fakes; +using NSubstitute; + +namespace DemaConsulting.Speech.Tests.RecognitionSubsystem; + +/// +/// Unit tests for , proving the single-session +/// exclusivity lease (Decision #2) over a fake backend and a substitute capture device, with +/// no native sherpa-onnx runtime. +/// +public class SherpaOnnxSpeechRecognizerEngineTests +{ + /// + /// Proves that creating a session with no prior session active succeeds and returns a + /// real, available session bound to the supplied device. + /// + [Fact] + public async Task SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_NoActiveSession_ReturnsSession() + { + // Arrange: an engine over a fake backend and an available capture device + var engine = new SherpaOnnxSpeechRecognizerEngine( + new FakeRecognitionEngine(), new FakeRecognitionModel(), NullSpeechDiagnostics.Instance); + var device = CreateCaptureDevice(); + + // Act + var session = await engine.CreateSessionAsync(device, TestContext.Current.CancellationToken); + + // Assert: a real, available session was returned + Assert.True(session.IsAvailable); + Assert.IsType(session); + + // Cleanup + await session.DisposeAsync(); + await engine.DisposeAsync(); + } + + /// + /// Proves that a concurrent + /// call, made while a previously created session still holds the engine's exclusivity + /// lease, fails fast with rather than queuing. + /// + [Fact] + public async Task SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_SessionAlreadyLeased_ThrowsRecognitionEngineBusyException() + { + // Arrange: an engine with one already-created session + var engine = new SherpaOnnxSpeechRecognizerEngine( + new FakeRecognitionEngine(), new FakeRecognitionModel(), NullSpeechDiagnostics.Instance); + var firstSession = await engine.CreateSessionAsync(CreateCaptureDevice(), TestContext.Current.CancellationToken); + + // Act & Assert: a second concurrent session request fails fast + await Assert.ThrowsAsync(() => engine.CreateSessionAsync(CreateCaptureDevice(), TestContext.Current.CancellationToken)); + + // Cleanup + await firstSession.DisposeAsync(); + await engine.DisposeAsync(); + } + + /// + /// Proves that the exclusivity lease is still held - and a concurrent + /// still fails fast - + /// while a prior session's own is still in + /// flight, not yet complete (Decision #2). + /// + [Fact] + public async Task SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_PriorSessionDisposing_ThrowsRecognitionEngineBusyException() + { + // Arrange: a running session whose capture device blocks when stopped, standing in for + // teardown work that has not yet completed. The block is lifted only by this test's own + // explicit stopGate.Set() (in the finally below), never by a timeout - a bounded wait here + // would race against this test's own continuation under heavy parallel test-run CPU + // contention: if that continuation were delayed past the bound, Stop() would return (and + // the lease would be released) before the Act below ever ran, intermittently passing for + // the wrong reason. The capture device's own TestContext cancellation token is observed + // purely as a safety net so a failed/aborted test run cannot leave this thread blocked + // forever, not as part of the behavior under test. + using var stopGate = new ManualResetEventSlim(false); + var device = CreateCaptureDevice(); + device.When(d => d.Stop()).Do(_ => stopGate.Wait(TestContext.Current.CancellationToken)); + var engine = new SherpaOnnxSpeechRecognizerEngine( + new FakeRecognitionEngine(), new FakeRecognitionModel(), NullSpeechDiagnostics.Instance); + var session = await engine.CreateSessionAsync(device, TestContext.Current.CancellationToken); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act: begin disposing the session but do not await completion yet + var disposeTask = session.DisposeAsync().AsTask(); + try + { + // Assert: a concurrent create fails fast while the lease is still held + await Assert.ThrowsAsync(() => engine.CreateSessionAsync(CreateCaptureDevice(), TestContext.Current.CancellationToken)); + } + finally + { + // Cleanup: release the blocked teardown and let disposal complete + stopGate.Set(); + await disposeTask; + } + + await engine.DisposeAsync(); + } + + /// + /// Proves that once a prior session has fully disposed - releasing the exclusivity lease + /// - a fresh call + /// succeeds and returns a new session over the same backend. + /// + [Fact] + public async Task SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_AfterPriorSessionFullyDisposed_ReturnsNewSession() + { + // Arrange: an engine whose first session has already fully disposed + var engine = new SherpaOnnxSpeechRecognizerEngine( + new FakeRecognitionEngine(), new FakeRecognitionModel(), NullSpeechDiagnostics.Instance); + var firstSession = await engine.CreateSessionAsync(CreateCaptureDevice(), TestContext.Current.CancellationToken); + await firstSession.StartAsync(TestContext.Current.CancellationToken); + await firstSession.DisposeAsync(); + + // Act: create a new session now that the lease has been released + var secondSession = await engine.CreateSessionAsync(CreateCaptureDevice(), TestContext.Current.CancellationToken); + + // Assert: a new, independent, available session was returned + Assert.True(secondSession.IsAvailable); + Assert.NotSame(firstSession, secondSession); + + // Cleanup + await secondSession.DisposeAsync(); + await engine.DisposeAsync(); + } + + /// + /// Proves that rejects + /// a null device, since this is a genuine caller error rather than an ordinary + /// unavailable state. + /// + [Fact] + public async Task SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException() + { + // Arrange + var engine = new SherpaOnnxSpeechRecognizerEngine( + new FakeRecognitionEngine(), new FakeRecognitionModel(), NullSpeechDiagnostics.Instance); + + // Act & Assert + await Assert.ThrowsAsync(() => engine.CreateSessionAsync(null!, TestContext.Current.CancellationToken)); + + // Cleanup + await engine.DisposeAsync(); + } + + /// + /// Proves that disposing an engine with an active session disposes that session first - + /// so its own teardown (and the lease release it performs) completes cleanly - before the + /// shared backend itself is disposed. + /// + [Fact] + public async Task SherpaOnnxSpeechRecognizerEngine_DisposeAsync_WithActiveSession_DisposesSessionFirst() + { + // Arrange: an engine with one active, started session, and a backend that records + // whether the session had already reached Disposed by the time the backend itself was + // disposed + IRecognitionSession? session = null; + var sessionDisposedBeforeBackend = false; + var backend = new FakeRecognitionEngine(onDispose: () => + sessionDisposedBeforeBackend = session?.State == RecognitionSessionState.Disposed); + var engine = new SherpaOnnxSpeechRecognizerEngine(backend, new FakeRecognitionModel(), NullSpeechDiagnostics.Instance); + session = await engine.CreateSessionAsync(CreateCaptureDevice(), TestContext.Current.CancellationToken); + await session.StartAsync(TestContext.Current.CancellationToken); + + // Act: dispose the engine directly, without disposing the session first + await engine.DisposeAsync(); + + // Assert: the session was already Disposed by the time the backend was disposed, and the + // backend was disposed exactly once + Assert.True(sessionDisposedBeforeBackend); + Assert.Equal(1, backend.DisposeCallCount); + } + + /// + /// Proves that returns + /// the honest fallback, reporting a + /// Warning diagnostic rather than attempting to lease the backend, when the supplied + /// capture device itself reports unavailable. + /// + [Fact] + public async Task SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_DeviceUnavailable_ReturnsFallbackAndReportsWarning() + { + // Arrange: an engine over a fake backend, an unavailable capture device, and a substitute diagnostics sink + var diagnostics = Substitute.For(); + var engine = new SherpaOnnxSpeechRecognizerEngine( + new FakeRecognitionEngine(), new FakeRecognitionModel(), diagnostics); + var device = Substitute.For(); + device.IsAvailable.Returns(false); + + // Act + var session = await engine.CreateSessionAsync(device, TestContext.Current.CancellationToken); + + // Assert: the honest unavailable fallback was returned, and a Warning was reported + Assert.Same(UnavailableRecognitionSession.Instance, session); + diagnostics.Received().Report( + SpeechDiagnosticLevel.Warning, + "RecognitionSubsystem", + Arg.Is(message => message.Contains("unavailable", StringComparison.OrdinalIgnoreCase))); + + // Cleanup + await engine.DisposeAsync(); + } + + /// + /// Proves that and + /// racing with no synchronization between them + /// never tears the engine: every CreateSessionAsync attempt either succeeds with a + /// genuinely usable session or fails with or + /// , never with some other exception that + /// would indicate it observed a half-disposed lease or backend. + /// + [Fact] + public async Task SherpaOnnxSpeechRecognizerEngine_CreateSessionAsync_RacingDisposeAsync_NeverObservesTornState() + { + for (var i = 0; i < 50; i++) + { + // Arrange + var engine = new SherpaOnnxSpeechRecognizerEngine( + new FakeRecognitionEngine(), new FakeRecognitionModel(), NullSpeechDiagnostics.Instance); + var device = CreateCaptureDevice(); + + // Act: race session creation against engine disposal with no synchronization + var createTask = engine.CreateSessionAsync(device, TestContext.Current.CancellationToken); + var disposeTask = engine.DisposeAsync().AsTask(); + + var createException = await Record.ExceptionAsync(async () => + { + var session = await createTask; + await session.DisposeAsync(); + }); + await disposeTask; + + // Assert: a create that lost the race observes disposal cleanly, never a torn state + Assert.True( + createException is null or ObjectDisposedException or RecognitionEngineBusyException, + $"Unexpected exception from a racing CreateSessionAsync: {createException}"); + } + } + + /// + /// Builds a substitute capture device reporting itself available at a plain mono 16 kHz + /// format. + /// + private static IAudioCaptureDevice CreateCaptureDevice() + { + var device = Substitute.For(); + device.IsAvailable.Returns(true); + device.SampleRate.Returns(16000); + device.ChannelCount.Returns(1); + return device; + } +} diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerTests.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerTests.cs deleted file mode 100644 index 3b99f52..0000000 --- a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SherpaOnnxSpeechRecognizerTests.cs +++ /dev/null @@ -1,748 +0,0 @@ -using DemaConsulting.Speech.AudioSubsystem; -using DemaConsulting.Speech.Diagnostics; -using DemaConsulting.Speech.ModelManagementSubsystem; -using DemaConsulting.Speech.RecognitionSubsystem; -using DemaConsulting.Speech.Tests.ModelManagementSubsystem.Fakes; -using DemaConsulting.Speech.Tests.RecognitionSubsystem.Fakes; -using NSubstitute; - -namespace DemaConsulting.Speech.Tests.RecognitionSubsystem; - -/// -/// Unit tests for , exercising the full capture → -/// resample → engine → event pipeline through a substitute capture device and a fake engine, -/// with no microphone and no native sherpa-onnx runtime. -/// -/// -/// Every test is deterministic without timing assumptions: Stop() completes the -/// recognizer's internal queue and joins its consumer task, so all results derived from -/// frames raised before the call have been delivered by the time it returns. -/// -public class SherpaOnnxSpeechRecognizerTests -{ - /// - /// Proves that starting subscribes to the capture device and starts it, so audio begins - /// flowing only once the recognizer is ready to receive it. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Start_Always_SubscribesAndStartsCaptureDevice() - { - // Arrange: a substitute mono 16 kHz capture device and a fake engine - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - using var recognizer = new SherpaOnnxSpeechRecognizer( - new FakeRecognitionEngine(), captureDevice, 16000, new FakeRecognitionModel()); - - // Act: start recognition - recognizer.Start(); - - // Assert: the device was started and the recognizer reports itself available - captureDevice.Received(1).Start(); - Assert.True(recognizer.IsAvailable); - - // Cleanup: stop so the consumer task is joined before the test ends - recognizer.Stop(); - } - - /// - /// Proves that starting an already-running recognizer is a safe no-op rather than opening - /// a second capture session. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Start_AlreadyRunning_IsNoOp() - { - // Arrange: a running recognizer over a substitute device - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - using var recognizer = new SherpaOnnxSpeechRecognizer( - new FakeRecognitionEngine(), captureDevice, 16000, new FakeRecognitionModel()); - recognizer.Start(); - - // Act: start again - recognizer.Start(); - - // Assert: the device was started exactly once - captureDevice.Received(1).Start(); - - // Cleanup - recognizer.Stop(); - } - - /// - /// Proves that a recognizer supports multiple / - /// cycles on the same instance without - /// reconstruction, so a host may construct one recognizer once and reuse it across many - /// conversation turns for low-latency, repeated recognition, per 's - /// "hot recognition" reuse guidance. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_MultipleStartStopCycles_ReusesSameInstanceWithoutReconstruction() - { - // Arrange: a single recognizer instance over a substitute device - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - using var recognizer = new SherpaOnnxSpeechRecognizer( - new FakeRecognitionEngine(), captureDevice, 16000, new FakeRecognitionModel()); - - // Act: run three independent Start/Stop cycles on the same instance - recognizer.Start(); - recognizer.Stop(); - recognizer.Start(); - recognizer.Stop(); - recognizer.Start(); - recognizer.Stop(); - - // Assert: every cycle genuinely started and stopped the capture device, proving the - // recognizer remains usable across repeated cycles without being disposed and recreated - captureDevice.Received(3).Start(); - captureDevice.Received(3).Stop(); - Assert.True(recognizer.IsAvailable); - } - - /// - /// Proves that every Start/Stop cycle resets the owned engine's decoder state, so a - /// "hot" engine reused across repeated cycles never carries a partially decoded - /// utterance from one cycle into the next. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_MultipleStartStopCycles_ResetsEngineEachCycle() - { - // Arrange: a single recognizer instance over a substitute device - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine(); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel()); - - // Act: run three independent Start/Stop cycles on the same instance - recognizer.Start(); - recognizer.Stop(); - recognizer.Start(); - recognizer.Stop(); - recognizer.Start(); - recognizer.Stop(); - - // Assert: the engine was reset exactly once per cycle - Assert.Equal(3, engine.ResetCallCount); - } - - /// - /// Proves that a captured frame flows through downmixing and resampling into the engine, - /// converted to the mono rate the model declared. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_FrameCaptured_StereoAtHigherRate_FeedsResampledMonoToEngine() - { - // Arrange: a stereo 32 kHz device feeding a 16 kHz model - var captureDevice = CreateCaptureDevice(sampleRate: 32000, channelCount: 2); - var engine = new FakeRecognitionEngine(); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel()); - recognizer.Start(); - - // Act: raise one stereo block of four frames, then stop to drain the pipeline - RaiseFrameCaptured(captureDevice, [0.0f, 0.0f, 1.0f, 1.0f, 2.0f, 2.0f, 3.0f, 3.0f]); - recognizer.Stop(); - - // Assert: four stereo frames became four mono samples, then anti-aliased and halved by - // the resampler - Assert.Equal(1, engine.AcceptSamplesCallCount); - Assert.Collection( - engine.AcceptedSamples, - sample => Assert.InRange(sample, 0.11705f, 0.11706f), - sample => Assert.InRange(sample, 2.06629f, 2.06630f)); - } - - /// - /// Proves that each result the engine decodes is raised through - /// , preserving both provisional and final - /// results in order. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_FrameCaptured_EngineDecodesResults_RaisesResultReceivedInOrder() - { - // Arrange: an engine scripted to yield one partial then one final result - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine( - [ - new SpeechRecognitionResult("hello", IsFinal: false), - new SpeechRecognitionResult("hello world", IsFinal: true) - ]); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel()); - var received = new List(); - recognizer.ResultReceived += (_, args) => received.Add(args.Result); - recognizer.Start(); - - // Act: raise one frame, then stop to drain the pipeline - RaiseFrameCaptured(captureDevice, [0.1f, 0.2f, 0.3f]); - recognizer.Stop(); - - // Assert: both results arrived, in order, with their finality preserved - Assert.Equal(2, received.Count); - Assert.Equal(new SpeechRecognitionResult("hello", IsFinal: false), received[0]); - Assert.Equal(new SpeechRecognitionResult("hello world", IsFinal: true), received[1]); - } - - /// - /// Proves that stopping unsubscribes from the capture device and stops it, so no further - /// audio is consumed after the call returns. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Stop_WhileRunning_UnsubscribesAndStopsCaptureDevice() - { - // Arrange: a running recognizer with a scripted engine - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine([new SpeechRecognitionResult("ignored", IsFinal: true)]); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel()); - var received = new List(); - recognizer.ResultReceived += (_, args) => received.Add(args.Result); - recognizer.Start(); - - // Act: stop, then raise a frame that must no longer be consumed - recognizer.Stop(); - RaiseFrameCaptured(captureDevice, [0.5f, 0.5f]); - - // Assert: the device was stopped and the post-stop frame never reached the engine - captureDevice.Received(1).Stop(); - Assert.Equal(0, engine.AcceptSamplesCallCount); - Assert.Empty(received); - } - - /// - /// Proves that stopping a recognizer that was never started is a safe no-op. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Stop_NotRunning_IsNoOp() - { - // Arrange: a recognizer that has never been started - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - using var recognizer = new SherpaOnnxSpeechRecognizer( - new FakeRecognitionEngine(), captureDevice, 16000, new FakeRecognitionModel()); - - // Act: stop without ever starting - var exception = Record.Exception(recognizer.Stop); - - // Assert: nothing threw and the device was never touched - Assert.Null(exception); - captureDevice.DidNotReceive().Stop(); - } - - /// - /// Proves that stopping a running recognizer resets the owned engine's decoder state, so - /// a subsequent on the same "hot" engine - /// never inherits a partially decoded utterance from before the stop. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Stop_Running_ResetsEngine() - { - // Arrange: a running recognizer with a fake engine - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine(); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel()); - recognizer.Start(); - - // Act: stop the recognizer - recognizer.Stop(); - - // Assert: the engine was reset exactly once - Assert.Equal(1, engine.ResetCallCount); - } - - /// - /// Proves that stopping a recognizer that is not running does not reset the engine, - /// matching the existing no-op behavior of that case. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Stop_NotRunning_DoesNotResetEngine() - { - // Arrange: a recognizer that has never been started - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine(); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel()); - - // Act: stop without ever starting - recognizer.Stop(); - - // Assert: the engine was never reset - Assert.Equal(0, engine.ResetCallCount); - } - - /// - /// Proves that disposal stops the pipeline, disposes the owned engine, and is idempotent. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Dispose_CalledTwice_StopsAndDisposesEngineOnce() - { - // Arrange: a running recognizer with a fake engine - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine(); - var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel()); - recognizer.Start(); - - // Act: dispose twice - recognizer.Dispose(); - recognizer.Dispose(); - - // Assert: the device was stopped once and the engine disposed exactly once - captureDevice.Received(1).Stop(); - Assert.Equal(1, engine.DisposeCallCount); - } - - /// - /// Proves that disposing a running recognizer resets the owned engine's decoder state - /// exactly once, as part of the shared stop sequence - /// performs before disposing the engine. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Dispose_Running_ResetsEngineOnce() - { - // Arrange: a running recognizer with a fake engine - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine(); - var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel()); - recognizer.Start(); - - // Act: dispose the recognizer - recognizer.Dispose(); - - // Assert: the engine was reset exactly once - Assert.Equal(1, engine.ResetCallCount); - } - - /// - /// Proves that disposing a running recognizer flushes trailing audio the same way - /// does - - /// shares the same flush-then-reset teardown path, so it must not skip the flush just - /// because it is terminal. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Dispose_EngineHasFlushableTrailingAudio_RaisesFlushedFinalResult() - { - // Arrange: a running recognizer whose engine yields a flushed final result on TryFlush - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine( - scriptedFlushResult: new SpeechRecognitionResult("cut off", IsFinal: true)); - var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel()); - var received = new List(); - recognizer.ResultReceived += (_, args) => received.Add(args.Result); - recognizer.Start(); - - // Act: dispose - recognizer.Dispose(); - - // Assert: the flush was invoked once and its result was raised as a final result, before - // the engine was reset - Assert.Equal(1, engine.FlushCallCount); - Assert.Equal([new SpeechRecognitionResult("cut off", IsFinal: true)], received); - Assert.Equal(1, engine.ResetCallCount); - } - - /// - /// Proves that a fault in the engine's during - /// teardown is contained and reported rather than propagated, and that - /// still completes and still permits a later - /// - matching the best-effort reset guarantee documented on the class and in the design - /// doc. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Stop_EngineResetFails_CompletesReportsFaultAndPermitsRestart() - { - // Arrange: a running recognizer whose engine always faults on Reset() - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var diagnostics = Substitute.For(); - var engine = new FakeRecognitionEngine(resetException: new InvalidOperationException("reset failed")); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel(), diagnostics); - recognizer.Start(); - - // Act: stop despite the faulting reset, then start again - var exception = Record.Exception(() => - { - recognizer.Stop(); - recognizer.Start(); - }); - - // Assert: nothing escaped, the fault was reported, and the capture device was still - // stopped and restarted as part of the (best-effort) teardown and later restart - Assert.Null(exception); - Assert.Equal(1, engine.ResetCallCount); - captureDevice.Received(1).Stop(); - captureDevice.Received(2).Start(); - diagnostics.Received(1).Report( - SpeechDiagnosticLevel.Error, - "RecognitionSubsystem", - Arg.Is(message => message.Contains("Failed to reset the recognition engine", StringComparison.Ordinal))); - } - - /// - /// Proves that stopping delivers a final result flushed from trailing audio the engine - /// had accepted but not yet decoded - for example the tail of an utterance released with - /// no trailing silence - rather than silently discarding it when the engine is reset. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Stop_EngineHasFlushableTrailingAudio_RaisesFlushedFinalResult() - { - // Arrange: a running recognizer whose engine yields a flushed final result on TryFlush - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine( - scriptedFlushResult: new SpeechRecognitionResult("cut off", IsFinal: true)); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel()); - var received = new List(); - recognizer.ResultReceived += (_, args) => received.Add(args.Result); - recognizer.Start(); - - // Act: stop - recognizer.Stop(); - - // Assert: the flush was invoked once and its result was raised as a final result, before - // the engine was reset for the next session - Assert.Equal(1, engine.FlushCallCount); - Assert.Equal([new SpeechRecognitionResult("cut off", IsFinal: true)], received); - Assert.Equal(1, engine.ResetCallCount); - } - - /// - /// Proves that stopping raises nothing extra when the engine has nothing to flush, - /// matching the existing "no new result" behavior for a quiet session. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Stop_EngineHasNothingToFlush_RaisesNoExtraResult() - { - // Arrange: a running recognizer whose engine yields no flushed result - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine(); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel()); - var received = new List(); - recognizer.ResultReceived += (_, args) => received.Add(args.Result); - recognizer.Start(); - - // Act: stop - recognizer.Stop(); - - // Assert: the flush was still attempted, but nothing was raised - Assert.Equal(1, engine.FlushCallCount); - Assert.Empty(received); - } - - /// - /// Proves that a fault in the engine's during - /// teardown is contained and reported rather than propagated, and that - /// still completes and still resets the - /// engine afterward, matching the same best-effort containment already proven for - /// faults. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Stop_EngineFlushFails_CompletesResetsEngineAndReportsFault() - { - // Arrange: a running recognizer whose engine always faults on TryFlush() - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var diagnostics = Substitute.For(); - var engine = new FakeRecognitionEngine(flushException: new InvalidOperationException("flush failed")); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel(), diagnostics); - var received = new List(); - recognizer.ResultReceived += (_, args) => received.Add(args.Result); - recognizer.Start(); - - // Act: stop despite the faulting flush - var exception = Record.Exception(recognizer.Stop); - - // Assert: nothing escaped, nothing was raised, the engine was still reset, and the fault - // was reported - Assert.Null(exception); - Assert.Empty(received); - Assert.Equal(1, engine.FlushCallCount); - Assert.Equal(1, engine.ResetCallCount); - diagnostics.Received(1).Report( - SpeechDiagnosticLevel.Error, - "RecognitionSubsystem", - Arg.Is(message => message.Contains("Failed to flush the recognition engine", StringComparison.Ordinal))); - } - - /// - /// Proves that starting a disposed recognizer throws - /// rather than silently doing nothing. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Start_AfterDispose_ThrowsObjectDisposedException() - { - // Arrange: a disposed recognizer - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var recognizer = new SherpaOnnxSpeechRecognizer( - new FakeRecognitionEngine(), captureDevice, 16000, new FakeRecognitionModel()); - recognizer.Dispose(); - - // Act & Assert: starting after disposal is rejected - Assert.Throws(recognizer.Start); - } - - /// - /// Proves that a capture device which claimed to be available but fails on start surfaces - /// as and leaves the recognizer stopped. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Start_CaptureDeviceFails_ThrowsSpeechRecognizerUnavailableException() - { - // Arrange: a device whose Start throws - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - captureDevice - .When(device => device.Start()) - .Do(_ => throw new AudioDeviceUnavailableException("native stream open failed")); - var diagnostics = Substitute.For(); - using var recognizer = new SherpaOnnxSpeechRecognizer( - new FakeRecognitionEngine(), captureDevice, 16000, new FakeRecognitionModel(), diagnostics); - - // Act & Assert: the first-use failure is surfaced as the documented exception - var exception = Assert.Throws(recognizer.Start); - Assert.IsType(exception.InnerException); - diagnostics.Received().Report( - SpeechDiagnosticLevel.Error, - "RecognitionSubsystem", - Arg.Is(message => message.Contains("capture device could not start", StringComparison.Ordinal))); - } - - /// - /// Proves that a failed resets the engine - /// as part of its rollback, exactly like a normal - /// would - the consumer's unconditional trailing flush would otherwise mark the real - /// engine's stream finished even though no audio was ever accepted (the capture device - /// never actually started), leaving a later retry unable to decode. This test proves only - /// that is invoked during the rollback; the fake - /// engine does not model the real engine's "finished stream" semantics, so it cannot - /// itself prove a subsequent retry can decode. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Start_CaptureDeviceFails_ResetsEngineDuringRollback() - { - // Arrange: a device whose Start throws - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - captureDevice - .When(device => device.Start()) - .Do(_ => throw new AudioDeviceUnavailableException("native stream open failed")); - var engine = new FakeRecognitionEngine(); - using var recognizer = new SherpaOnnxSpeechRecognizer( - engine, captureDevice, 16000, new FakeRecognitionModel()); - - // Act - Assert.Throws(recognizer.Start); - - // Assert: the rollback reset the engine, exactly like a normal Stop() would. - Assert.Equal(1, engine.ResetCallCount); - } - - /// - /// Proves that a host result handler which throws is contained and reported through the - /// diagnostics sink, never rethrown into the capture callback and never stopping the - /// recognizer. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_ResultReceived_HandlerThrows_ReportsFaultAndDoesNotRethrow() - { - // Arrange: a running recognizer whose result handler always throws - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var diagnostics = Substitute.For(); - var engine = new FakeRecognitionEngine([new SpeechRecognitionResult("boom", IsFinal: true)]); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel(), diagnostics); - recognizer.ResultReceived += (_, _) => throw new InvalidOperationException("handler failed"); - recognizer.Start(); - - // Act: raise a frame that produces a result, then drain - var exception = Record.Exception(() => - { - RaiseFrameCaptured(captureDevice, [0.1f, 0.2f]); - recognizer.Stop(); - }); - - // Assert: nothing escaped, and the fault was reported as a structural fact - Assert.Null(exception); - diagnostics.Received().Report( - SpeechDiagnosticLevel.Error, - "RecognitionSubsystem", - Arg.Is(message => message.Contains("could not be recognized", StringComparison.Ordinal))); - } - - /// - /// Proves that an engine fault while consuming a frame is contained and reported, so one - /// bad block never ends the session. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_FrameCaptured_EngineThrows_ReportsFaultAndKeepsRunning() - { - // Arrange: a running recognizer over an engine that always faults on accept - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var diagnostics = Substitute.For(); - var engine = new FakeRecognitionEngine( - acceptSamplesException: new InvalidOperationException("engine failed")); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel(), diagnostics); - recognizer.Start(); - - // Act: raise two frames, then drain - RaiseFrameCaptured(captureDevice, [0.1f]); - RaiseFrameCaptured(captureDevice, [0.2f]); - recognizer.Stop(); - - // Assert: both frames were still delivered to the engine and both faults were reported - Assert.Equal(2, engine.AcceptSamplesCallCount); - diagnostics.Received(2).Report( - SpeechDiagnosticLevel.Error, - "RecognitionSubsystem", - Arg.Is(message => message.Contains("could not be recognized", StringComparison.Ordinal))); - } - - /// - /// Proves that a capture device reporting an unusable audio format degrades to a - /// pass-through conversion and reports the fact, rather than throwing at composition. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Constructor_DeviceReportsUnusableFormat_FallsBackToPassThrough() - { - // Arrange: a device reporting a zero sample rate and zero channels - var captureDevice = CreateCaptureDevice(sampleRate: 0, channelCount: 0); - var diagnostics = Substitute.For(); - var engine = new FakeRecognitionEngine(); - - // Act: compose, then push one frame through the pipeline - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel(), diagnostics); - recognizer.Start(); - RaiseFrameCaptured(captureDevice, [0.1f, 0.2f]); - recognizer.Stop(); - - // Assert: the format warning was reported and the samples passed through untouched - diagnostics.Received().Report( - SpeechDiagnosticLevel.Warning, - "RecognitionSubsystem", - Arg.Is(message => message.Contains("unusable audio format", StringComparison.Ordinal))); - Assert.Equal([0.1f, 0.2f], engine.AcceptedSamples); - } - - /// - /// Proves that an empty capture block is ignored rather than being pushed into the engine. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_FrameCaptured_EmptyBlock_IsIgnored() - { - // Arrange: a running recognizer over a fake engine - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine(); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, new FakeRecognitionModel()); - recognizer.Start(); - - // Act: raise an empty block, then drain - RaiseFrameCaptured(captureDevice, []); - recognizer.Stop(); - - // Assert: the engine was never asked to accept anything - Assert.Equal(0, engine.AcceptSamplesCallCount); - } - - /// - /// Proves that the constructor rejects a null model, since every recognizer must have an - /// owning model to apply its - /// hook to results. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_Constructor_NullModel_ThrowsArgumentNullException() - { - // Arrange - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - - // Act & Assert - Assert.Throws(() => - new SherpaOnnxSpeechRecognizer(new FakeRecognitionEngine(), captureDevice, 16000, null!)); - } - - /// - /// Proves that a final result's text is passed through the owning model's - /// hook, with isFinal - /// set to , before - /// raises it - proving the wiring itself, not just the hook's own logic. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_FrameCaptured_FinalResult_AppliesModelNormalizeTextWithIsFinalTrue() - { - // Arrange: a model whose NormalizeText is a distinguishable, recorded transform - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine([new SpeechRecognitionResult("HELLO WORLD", IsFinal: true)]); - var calls = new List<(string Text, bool IsFinal)>(); - var model = new FakeRecognitionModel(normalizeText: (text, isFinal) => - { - calls.Add((text, isFinal)); - return $"normalized:{text}"; - }); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, model); - var received = new List(); - recognizer.ResultReceived += (_, args) => received.Add(args.Result); - recognizer.Start(); - - // Act: raise one frame that produces the scripted final result, then drain - RaiseFrameCaptured(captureDevice, [0.1f, 0.2f]); - recognizer.Stop(); - - // Assert: the model saw the raw text with isFinal true, and the raised result carries - // the model's normalized text while preserving IsFinal - var result = Assert.Single(received); - Assert.Equal("normalized:HELLO WORLD", result.Text); - Assert.True(result.IsFinal); - var call = Assert.Single(calls); - Assert.Equal("HELLO WORLD", call.Text); - Assert.True(call.IsFinal); - } - - /// - /// Proves that a provisional result's text is passed through the owning model's - /// hook, with isFinal - /// set to , before - /// raises it. - /// - [Fact] - public void SherpaOnnxSpeechRecognizer_FrameCaptured_ProvisionalResult_AppliesModelNormalizeTextWithIsFinalFalse() - { - // Arrange: a model whose NormalizeText is a distinguishable, recorded transform - var captureDevice = CreateCaptureDevice(sampleRate: 16000, channelCount: 1); - var engine = new FakeRecognitionEngine([new SpeechRecognitionResult("HELLO", IsFinal: false)]); - var calls = new List<(string Text, bool IsFinal)>(); - var model = new FakeRecognitionModel(normalizeText: (text, isFinal) => - { - calls.Add((text, isFinal)); - return $"normalized:{text}"; - }); - using var recognizer = new SherpaOnnxSpeechRecognizer(engine, captureDevice, 16000, model); - var received = new List(); - recognizer.ResultReceived += (_, args) => received.Add(args.Result); - recognizer.Start(); - - // Act: raise one frame that produces the scripted provisional result, then drain - RaiseFrameCaptured(captureDevice, [0.1f, 0.2f]); - recognizer.Stop(); - - // Assert: the model saw the raw text with isFinal false, and the raised result carries - // the model's normalized text while preserving IsFinal - var result = Assert.Single(received); - Assert.Equal("normalized:HELLO", result.Text); - Assert.False(result.IsFinal); - var call = Assert.Single(calls); - Assert.Equal("HELLO", call.Text); - Assert.False(call.IsFinal); - } - - /// - /// Builds a substitute capture device reporting itself available with the given capture - /// format. - /// - /// The rate the device reports. - /// The channel count the device reports. - /// The configured substitute capture device. - private static IAudioCaptureDevice CreateCaptureDevice(int sampleRate, int channelCount) - { - var captureDevice = Substitute.For(); - captureDevice.IsAvailable.Returns(true); - captureDevice.SampleRate.Returns(sampleRate); - captureDevice.ChannelCount.Returns(channelCount); - return captureDevice; - } - - /// - /// Raises the substitute capture device's - /// event with one block of interleaved samples, standing in for a real audio callback. - /// - /// The substitute capture device to raise the event on. - /// The interleaved samples the block carries. - private static void RaiseFrameCaptured(IAudioCaptureDevice captureDevice, float[] samples) - { - captureDevice.FrameCaptured += Raise.Event>( - captureDevice, - new AudioCaptureFrameEventArgs(samples)); - } -} diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SpeechRecognizerFactoryTests.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SpeechRecognizerFactoryTests.cs index ff97e85..bb59e16 100644 --- a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SpeechRecognizerFactoryTests.cs +++ b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/SpeechRecognizerFactoryTests.cs @@ -11,7 +11,7 @@ namespace DemaConsulting.Speech.Tests.RecognitionSubsystem; /// /// Unit tests for , proving that composition never /// throws for an ordinary machine state and honestly degrades to -/// instead. +/// instead. /// public sealed class SpeechRecognizerFactoryTests : IDisposable { @@ -81,89 +81,60 @@ public void Dispose() /// /// Proves that a model whose files are not installed composes to the honest unavailable - /// recognizer, and that the engine is never loaded. + /// engine, and that the backend is never loaded. /// [Fact] - public void SpeechRecognizerFactory_Create_ModelNotInstalled_ReturnsUnavailableRecognizer() + public async Task SpeechRecognizerFactory_LoadAsync_ModelNotInstalled_ReturnsUnavailableEngine() { - // Arrange: an available capture device but a directory that does not exist - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + // Arrange: a directory that does not exist + var backendFactory = new FakeRecognitionEngineFactory(); var missingDirectory = Path.Join(_installedModelDirectory, "not-installed"); // Act: compose against the missing model directory - var recognizer = SpeechRecognizerFactory.Create( - new FakeRecognitionModel(), missingDirectory, captureDevice, null, engineFactory); + await using var engine = await SpeechRecognizerFactory.LoadAsync(new FakeRecognitionModel(), missingDirectory, null, backendFactory, cancellationToken: TestContext.Current.CancellationToken); - // Assert: the honest fallback is returned and no engine was loaded - Assert.Same(UnavailableSpeechRecognizer.Instance, recognizer); - Assert.Equal(0, engineFactory.CreateCallCount); - } - - /// - /// Proves that a machine with no usable capture device composes to the honest unavailable - /// recognizer rather than loading a model that could never be fed. - /// - [Fact] - public void SpeechRecognizerFactory_Create_CaptureDeviceUnavailable_ReturnsUnavailableRecognizer() - { - // Arrange: an installed model but the shared unavailable capture device - var engineFactory = new FakeRecognitionEngineFactory(); - - // Act: compose against the unavailable device - var recognizer = SpeechRecognizerFactory.Create( - new FakeRecognitionModel(), - _installedModelDirectory, - UnavailableAudioCaptureDevice.Instance, - null, - engineFactory); - - // Assert: the honest fallback is returned and no engine was loaded - Assert.Same(UnavailableSpeechRecognizer.Instance, recognizer); - Assert.Equal(0, engineFactory.CreateCallCount); + // Assert: the honest fallback is returned and no backend was loaded + Assert.Same(UnavailableSpeechRecognizerEngine.Instance, engine); + Assert.Equal(0, backendFactory.CreateCallCount); } /// /// Proves that a model declaring a non-recognition role composes to the honest unavailable - /// recognizer. + /// engine. /// [Fact] - public void SpeechRecognizerFactory_Create_ModelRoleIsNotRecognition_ReturnsUnavailableRecognizer() + public async Task SpeechRecognizerFactory_LoadAsync_ModelRoleIsNotRecognition_ReturnsUnavailableEngine() { - // Arrange: an installed model that declares the synthesis role, and an available device - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + // Arrange: an installed model that declares the synthesis role + var backendFactory = new FakeRecognitionEngineFactory(); // Act: compose against the wrong-role model - var recognizer = SpeechRecognizerFactory.Create( - new WrongRoleRecognitionModel(), _installedModelDirectory, captureDevice, null, engineFactory); + await using var engine = await SpeechRecognizerFactory.LoadAsync(new WrongRoleRecognitionModel(), _installedModelDirectory, null, backendFactory, cancellationToken: TestContext.Current.CancellationToken); - // Assert: the honest fallback is returned and no engine was loaded - Assert.Same(UnavailableSpeechRecognizer.Instance, recognizer); - Assert.Equal(0, engineFactory.CreateCallCount); + // Assert: the honest fallback is returned and no backend was loaded + Assert.Same(UnavailableSpeechRecognizerEngine.Instance, engine); + Assert.Equal(0, backendFactory.CreateCallCount); } /// /// Proves that a native-runtime or model-file load failure degrades to the honest - /// unavailable recognizer instead of propagating out of composition. + /// unavailable engine instead of faulting the returned task. /// [Fact] - public void SpeechRecognizerFactory_Create_EngineLoadFails_ReturnsUnavailableRecognizerAndDoesNotThrow() + public async Task SpeechRecognizerFactory_LoadAsync_EngineLoadFails_ReturnsUnavailableEngineAndDoesNotFaultTask() { - // Arrange: an installed model, an available device, and an engine factory that faults - var captureDevice = CreateAvailableCaptureDevice(); + // Arrange: an installed model and a backend factory that faults var diagnostics = Substitute.For(); - var engineFactory = new FakeRecognitionEngineFactory( + var backendFactory = new FakeRecognitionEngineFactory( createException: new DllNotFoundException("sherpa-onnx-c-api")); // Act: compose, capturing any exception that escapes - ISpeechRecognizer? recognizer = null; - var exception = Record.Exception(() => recognizer = SpeechRecognizerFactory.Create( - new FakeRecognitionModel(), _installedModelDirectory, captureDevice, diagnostics, engineFactory)); + ISpeechRecognizerEngine? engine = null; + var exception = await Record.ExceptionAsync(async () => engine = await SpeechRecognizerFactory.LoadAsync(new FakeRecognitionModel(), _installedModelDirectory, diagnostics, backendFactory, cancellationToken: TestContext.Current.CancellationToken)); // Assert: composition succeeded honestly and reported the fault as a structural fact Assert.Null(exception); - Assert.Same(UnavailableSpeechRecognizer.Instance, recognizer); + Assert.Same(UnavailableSpeechRecognizerEngine.Instance, engine); diagnostics.Received().Report( SpeechDiagnosticLevel.Error, "RecognitionSubsystem", @@ -171,53 +142,48 @@ public void SpeechRecognizerFactory_Create_EngineLoadFails_ReturnsUnavailableRec } /// - /// Proves that an installed recognition model plus an available capture device composes a - /// real recognizer wired to the injected engine factory, with the installed-model - /// directory passed through unchanged. + /// Proves that an installed recognition model composes a real engine wired to the + /// injected backend factory, with the installed-model directory passed through unchanged. /// [Fact] - public void SpeechRecognizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealRecognizer() + public async Task SpeechRecognizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine() { - // Arrange: an installed model, an available device, and a fake engine factory - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + // Arrange: an installed model and a fake backend factory + var backendFactory = new FakeRecognitionEngineFactory(); var model = new FakeRecognitionModel(); - // Act: compose a recognizer - using var recognizer = SpeechRecognizerFactory.Create( - model, _installedModelDirectory, captureDevice, null, engineFactory); + // Act: compose an engine + await using var engine = await SpeechRecognizerFactory.LoadAsync(model, _installedModelDirectory, null, backendFactory, cancellationToken: TestContext.Current.CancellationToken); - // Assert: a real recognizer was built from the injected engine, for the right model - Assert.IsType(recognizer); - Assert.True(recognizer.IsAvailable); - Assert.Equal(1, engineFactory.CreateCallCount); - Assert.Same(model, engineFactory.RequestedModel); - Assert.Equal(_installedModelDirectory, engineFactory.RequestedInstalledModelDirectory); + // Assert: a real engine was built from the injected backend, for the right model + Assert.IsType(engine); + Assert.True(engine.IsAvailable); + Assert.Equal(1, backendFactory.CreateCallCount); + Assert.Same(model, backendFactory.RequestedModel); + Assert.Equal(_installedModelDirectory, backendFactory.RequestedInstalledModelDirectory); } /// - /// Proves that an unrecognized parameter id is silently ignored (never throws) and is + /// Proves that an unrecognized parameter id is silently ignored (never faults) and is /// reported at when a diagnostics sink is /// supplied - preserving this library's deliberate cross-model settings-dictionary-reuse /// contract. /// [Fact] - public void SpeechRecognizerFactory_Create_UnrecognizedParameterId_ComposesAndReportsInfo() + public async Task SpeechRecognizerFactory_LoadAsync_UnrecognizedParameterId_ComposesAndReportsInfo() { // Arrange - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + var backendFactory = new FakeRecognitionEngineFactory(); var model = new FakeRecognitionModel(); var diagnostics = Substitute.For(); IReadOnlyDictionary parameterValues = new Dictionary { ["typo-id"] = 1 }; // Act - using var recognizer = SpeechRecognizerFactory.Create( - model, _installedModelDirectory, captureDevice, diagnostics, engineFactory, parameterValues); + await using var engine = await SpeechRecognizerFactory.LoadAsync(model, _installedModelDirectory, diagnostics, backendFactory, parameterValues, TestContext.Current.CancellationToken); - // Assert: still a real, working recognizer, plus the observability diagnostic - Assert.IsType(recognizer); - Assert.True(recognizer.IsAvailable); + // Assert: still a real, working engine, plus the observability diagnostic + Assert.IsType(engine); + Assert.True(engine.IsAvailable); diagnostics.Received(1).Report( SpeechDiagnosticLevel.Info, "RecognitionSubsystem", @@ -225,376 +191,377 @@ public void SpeechRecognizerFactory_Create_UnrecognizedParameterId_ComposesAndRe } /// - /// Proves that an out-of-range value for a recognized numeric parameter throws - /// synchronously from Create, rather than - /// silently clamping. + /// Proves that an out-of-range value for a recognized numeric parameter faults the + /// returned task with , rather than silently clamping. /// [Fact] - public void SpeechRecognizerFactory_Create_RecognizedNumericParameterOutOfRange_Throws() + public async Task SpeechRecognizerFactory_LoadAsync_RecognizedNumericParameterOutOfRange_FaultsWithArgumentException() { // Arrange - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + var backendFactory = new FakeRecognitionEngineFactory(); var model = new FakeRecognitionModel(); IReadOnlyDictionary parameterValues = new Dictionary { ["sensitivity"] = 5.0 }; // Act / Assert - var exception = Assert.Throws(() => SpeechRecognizerFactory.Create( - model, _installedModelDirectory, captureDevice, null, engineFactory, parameterValues)); + var exception = await Assert.ThrowsAsync(() => SpeechRecognizerFactory.LoadAsync(model, _installedModelDirectory, null, backendFactory, parameterValues, TestContext.Current.CancellationToken)); Assert.Contains("sensitivity", exception.Message, StringComparison.Ordinal); - Assert.Equal(0, engineFactory.CreateCallCount); + Assert.Equal(0, backendFactory.CreateCallCount); } /// - /// Proves that a value not matching any declared choice option throws - /// synchronously from Create. + /// Proves that a value not matching any declared choice option faults the returned task + /// with . /// [Fact] - public void SpeechRecognizerFactory_Create_RecognizedChoiceParameterInvalidOption_Throws() + public async Task SpeechRecognizerFactory_LoadAsync_RecognizedChoiceParameterInvalidOption_FaultsWithArgumentException() { // Arrange - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + var backendFactory = new FakeRecognitionEngineFactory(); var model = new FakeRecognitionModel(); IReadOnlyDictionary parameterValues = new Dictionary { ["language"] = "klingon" }; // Act / Assert - var exception = Assert.Throws(() => SpeechRecognizerFactory.Create( - model, _installedModelDirectory, captureDevice, null, engineFactory, parameterValues)); + var exception = await Assert.ThrowsAsync(() => SpeechRecognizerFactory.LoadAsync(model, _installedModelDirectory, null, backendFactory, parameterValues, TestContext.Current.CancellationToken)); Assert.Contains("language", exception.Message, StringComparison.Ordinal); Assert.Contains("klingon", exception.Message, StringComparison.Ordinal); - Assert.Equal(0, engineFactory.CreateCallCount); + Assert.Equal(0, backendFactory.CreateCallCount); } /// - /// Proves that a non-boolean value for a recognized throws - /// synchronously from Create. + /// Proves that a non-boolean value for a recognized faults + /// the returned task with . /// [Fact] - public void SpeechRecognizerFactory_Create_RecognizedBooleanParameterWrongType_Throws() + public async Task SpeechRecognizerFactory_LoadAsync_RecognizedBooleanParameterWrongType_FaultsWithArgumentException() { // Arrange - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + var backendFactory = new FakeRecognitionEngineFactory(); var model = new FakeRecognitionModel(); IReadOnlyDictionary parameterValues = new Dictionary { ["denoise"] = "yes" }; // Act / Assert - var exception = Assert.Throws(() => SpeechRecognizerFactory.Create( - model, _installedModelDirectory, captureDevice, null, engineFactory, parameterValues)); + var exception = await Assert.ThrowsAsync(() => SpeechRecognizerFactory.LoadAsync(model, _installedModelDirectory, null, backendFactory, parameterValues, TestContext.Current.CancellationToken)); Assert.Contains("denoise", exception.Message, StringComparison.Ordinal); - Assert.Equal(0, engineFactory.CreateCallCount); + Assert.Equal(0, backendFactory.CreateCallCount); } /// - /// Proves that a wrong CLR type for a recognized numeric parameter throws - /// synchronously from Create. + /// Proves that a wrong CLR type for a recognized numeric parameter faults the returned + /// task with . /// [Fact] - public void SpeechRecognizerFactory_Create_RecognizedNumericParameterWrongType_Throws() + public async Task SpeechRecognizerFactory_LoadAsync_RecognizedNumericParameterWrongType_FaultsWithArgumentException() { // Arrange - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + var backendFactory = new FakeRecognitionEngineFactory(); var model = new FakeRecognitionModel(); IReadOnlyDictionary parameterValues = new Dictionary { ["sensitivity"] = "high" }; // Act / Assert - var exception = Assert.Throws(() => SpeechRecognizerFactory.Create( - model, _installedModelDirectory, captureDevice, null, engineFactory, parameterValues)); + var exception = await Assert.ThrowsAsync(() => SpeechRecognizerFactory.LoadAsync(model, _installedModelDirectory, null, backendFactory, parameterValues, TestContext.Current.CancellationToken)); Assert.Contains("sensitivity", exception.Message, StringComparison.Ordinal); - Assert.Equal(0, engineFactory.CreateCallCount); + Assert.Equal(0, backendFactory.CreateCallCount); } /// /// Proves that a supplied parameterValues bag genuinely reaches the model's own - /// two-argument CreateEngineConfig override, not merely the engine factory. + /// two-argument CreateEngineConfig override, not merely the backend factory. /// [Fact] - public void SpeechRecognizerFactory_Create_ParameterValuesSupplied_ReachesModelCreateEngineConfig() + public async Task SpeechRecognizerFactory_LoadAsync_ParameterValuesSupplied_ReachesModelCreateEngineConfig() { // Arrange: an installed model whose CreateEngineConfig override encodes the language value - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + var backendFactory = new FakeRecognitionEngineFactory(); var model = new ParameterCapturingRecognitionModel(); IReadOnlyDictionary parameterValues = new Dictionary { ["language"] = "en-gb" }; - // Act: compose a recognizer, supplying parameterValues - using var recognizer = SpeechRecognizerFactory.Create( - model, _installedModelDirectory, captureDevice, null, engineFactory, parameterValues); + // Act: compose an engine, supplying parameterValues + await using var engine = await SpeechRecognizerFactory.LoadAsync(model, _installedModelDirectory, null, backendFactory, parameterValues, TestContext.Current.CancellationToken); // Assert: the value reached the model's own CreateEngineConfig override - Assert.IsType(recognizer); - Assert.Equal("en-gb", engineFactory.RequestedConfig?.ModelConfig.ModelType); + Assert.IsType(engine); + Assert.Equal("en-gb", backendFactory.RequestedConfig?.ModelConfig.ModelType); } /// - /// Proves that the public composition overload rejects a null model, since a null - /// argument is a programming error rather than an ordinary machine state. + /// Proves that the public composition overload faults the returned task with + /// for a null model, since a null argument is a + /// programming error rather than an ordinary machine state. /// [Fact] - public void SpeechRecognizerFactory_Create_NullModel_ThrowsArgumentNullException() + public async Task SpeechRecognizerFactory_LoadAsync_NullModel_FaultsWithArgumentNullException() { - // Arrange: an available capture device - var captureDevice = CreateAvailableCaptureDevice(); - // Act & Assert: a null model is rejected - Assert.Throws( - () => SpeechRecognizerFactory.Create(null!, _installedModelDirectory, captureDevice)); + await Assert.ThrowsAsync( + () => SpeechRecognizerFactory.LoadAsync(null!, _installedModelDirectory, cancellationToken: TestContext.Current.CancellationToken)); } /// - /// Proves that the public composition overload rejects a null capture device. + /// Proves that a cancelled token faults the returned task with + /// rather than letting composition proceed. /// [Fact] - public void SpeechRecognizerFactory_Create_NullCaptureDevice_ThrowsArgumentNullException() + public async Task SpeechRecognizerFactory_LoadAsync_CancelledToken_FaultsWithOperationCanceledException() { - // Act & Assert: a null capture device is rejected - Assert.Throws( - () => SpeechRecognizerFactory.Create(new FakeRecognitionModel(), _installedModelDirectory, null!)); + // Arrange: a token already cancelled before the call + using var cts = new CancellationTokenSource(); + await cts.CancelAsync(); + + // Act & Assert: the cancellation is honored rather than silently ignored + await Assert.ThrowsAsync( + () => SpeechRecognizerFactory.LoadAsync( + new FakeRecognitionModel(), _installedModelDirectory, cancellationToken: cts.Token)); } /// /// Proves that a model not yet installed in the store composes to the honest unavailable - /// recognizer, and that the engine is never loaded. + /// engine, and that the backend is never loaded. /// [Fact] - public void SpeechRecognizerFactory_Create_WithStoreModelNotInstalled_ReturnsUnavailableRecognizer() + public async Task SpeechRecognizerFactory_LoadAsync_WithStoreModelNotInstalled_ReturnsUnavailableEngine() { - // Arrange: an available capture device and a store with no installed model directory - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + // Arrange: a store with no installed model directory + var backendFactory = new FakeRecognitionEngineFactory(); var model = new FakeRecognitionModel(); // Act: compose against the store, which resolves to a directory that does not exist - var recognizer = SpeechRecognizerFactory.Create(model, _store, captureDevice, null, engineFactory); + await using var engine = await SpeechRecognizerFactory.LoadAsync(model, _store, null, backendFactory, cancellationToken: TestContext.Current.CancellationToken); - // Assert: the honest fallback is returned and no engine was loaded - Assert.Same(UnavailableSpeechRecognizer.Instance, recognizer); - Assert.Equal(0, engineFactory.CreateCallCount); + // Assert: the honest fallback is returned and no backend was loaded + Assert.Same(UnavailableSpeechRecognizerEngine.Instance, engine); + Assert.Equal(0, backendFactory.CreateCallCount); } /// - /// Proves that an installed recognition model plus an available capture device composes a - /// real recognizer wired to the injected engine factory, with the directory resolved - /// through the store rather than hard-coded. + /// Proves that an installed recognition model composes a real engine wired to the + /// injected backend factory, with the directory resolved through the store rather than + /// hard-coded. /// [Fact] - public void SpeechRecognizerFactory_Create_WithStoreModelInstalledAndDeviceAvailable_ReturnsRealRecognizer() + public async Task SpeechRecognizerFactory_LoadAsync_WithStoreModelInstalled_ReturnsRealEngine() { - // Arrange: a model installed via the store, an available device, and a fake engine factory - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + // Arrange: a model installed via the store, and a fake backend factory + var backendFactory = new FakeRecognitionEngineFactory(); var model = new FakeRecognitionModel(); Directory.CreateDirectory(_store.GetCurrentDirectory(model.Id)); - // Act: compose a recognizer through the store overload - using var recognizer = SpeechRecognizerFactory.Create(model, _store, captureDevice, null, engineFactory); + // Act: compose an engine through the store overload + await using var engine = await SpeechRecognizerFactory.LoadAsync(model, _store, null, backendFactory, cancellationToken: TestContext.Current.CancellationToken); - // Assert: a real recognizer was built, and the directory was resolved through the store - Assert.IsType(recognizer); - Assert.Equal(1, engineFactory.CreateCallCount); - Assert.Equal(_store.GetCurrentDirectory(model.Id), engineFactory.RequestedInstalledModelDirectory); + // Assert: a real engine was built, and the directory was resolved through the store + Assert.IsType(engine); + Assert.Equal(1, backendFactory.CreateCallCount); + Assert.Equal(_store.GetCurrentDirectory(model.Id), backendFactory.RequestedInstalledModelDirectory); } /// /// Proves that a supplied parameterValues bag genuinely reaches the model's own /// two-argument CreateEngineConfig override when composed through the PUBLIC - /// store-based composition overload (with no injected engineFactory), so the real + /// store-based composition overload (with no injected backendFactory), so the real /// production delegation chain (public store overload → public string overload → - /// internal string+engineFactory overload → real + /// internal string+backendFactory overload → real /// -free SherpaOnnxRecognitionEngineFactory) /// is exercised end to end. The model throws from within its own /// CreateEngineConfig override, after recording the received parameter bag, so the /// test never reaches a real native sherpa-onnx engine construction. /// [Fact] - public void SpeechRecognizerFactory_Create_WithStoreParameterValuesSupplied_ReachesModelCreateEngineConfig() + public async Task SpeechRecognizerFactory_LoadAsync_WithStoreParameterValuesSupplied_ReachesModelCreateEngineConfig() { - // Arrange: a throwing parameter-capturing model installed via the store, an available device, and a parameter bag - var captureDevice = CreateAvailableCaptureDevice(); + // Arrange: a throwing parameter-capturing model installed via the store, and a parameter bag var model = new ThrowingParameterCapturingRecognitionModel(); Directory.CreateDirectory(_store.GetCurrentDirectory(model.Id)); IReadOnlyDictionary parameterValues = new Dictionary { ["language"] = "en-gb" }; - // Act: compose a recognizer through the genuine public store overload, supplying - // parameterValues, with no engineFactory argument so overload resolution can only bind + // Act: compose an engine through the genuine public store overload, supplying + // parameterValues, with no backendFactory argument so overload resolution can only bind // to the real public overload - var recognizer = SpeechRecognizerFactory.Create(model, _store, captureDevice, null, parameterValues); + await using var engine = await SpeechRecognizerFactory.LoadAsync(model, _store, null, parameterValues, TestContext.Current.CancellationToken); // Assert: the value reached the model's own CreateEngineConfig override via the real // production chain, and the model's throw was honestly swallowed into the unavailable - // fallback rather than propagating - Assert.Same(UnavailableSpeechRecognizer.Instance, recognizer); + // fallback rather than faulting the task + Assert.Same(UnavailableSpeechRecognizerEngine.Instance, engine); Assert.Equal(parameterValues, model.RequestedParameterValues); } /// - /// Proves that the public store-based composition overload rejects a null model. + /// Proves that the public store-based composition overload faults the returned task with + /// for a null model. /// [Fact] - public void SpeechRecognizerFactory_Create_WithStoreNullModel_ThrowsArgumentNullException() + public async Task SpeechRecognizerFactory_LoadAsync_WithStoreNullModel_FaultsWithArgumentNullException() { - // Arrange: an available capture device - var captureDevice = CreateAvailableCaptureDevice(); - // Act & Assert: a null model is rejected - Assert.Throws( - () => SpeechRecognizerFactory.Create(null!, _store, captureDevice)); + await Assert.ThrowsAsync( + () => SpeechRecognizerFactory.LoadAsync(null!, _store, cancellationToken: TestContext.Current.CancellationToken)); } /// - /// Proves that the public store-based composition overload rejects a null store. + /// Proves that the public store-based composition overload faults the returned task with + /// for a null store. /// [Fact] - public void SpeechRecognizerFactory_Create_WithStoreNullStore_ThrowsArgumentNullException() + public async Task SpeechRecognizerFactory_LoadAsync_WithStoreNullStore_FaultsWithArgumentNullException() { - // Arrange: an available capture device - var captureDevice = CreateAvailableCaptureDevice(); - // Act & Assert: a null store is rejected - Assert.Throws( - () => SpeechRecognizerFactory.Create(new FakeRecognitionModel(), (SpeechModelStore)null!, captureDevice)); - } - - /// - /// Proves that the public store-based composition overload rejects a null capture device, - /// confirming the delegation still reaches the string-overload's own null check. - /// - [Fact] - public void SpeechRecognizerFactory_Create_WithStoreNullCaptureDevice_ThrowsArgumentNullException() - { - // Act & Assert: a null capture device is rejected - Assert.Throws( - () => SpeechRecognizerFactory.Create(new FakeRecognitionModel(), _store, null!)); + await Assert.ThrowsAsync( + () => SpeechRecognizerFactory.LoadAsync(new FakeRecognitionModel(), (SpeechModelStore)null!, cancellationToken: TestContext.Current.CancellationToken)); } /// /// Proves that a model not yet installed in the catalog's store composes to the honest - /// unavailable recognizer, and that the engine is never loaded. + /// unavailable engine, and that the backend is never loaded. /// [Fact] - public void SpeechRecognizerFactory_Create_WithCatalogModelNotInstalled_ReturnsUnavailableRecognizer() + public async Task SpeechRecognizerFactory_LoadAsync_WithCatalogModelNotInstalled_ReturnsUnavailableEngine() { - // Arrange: an available capture device and a catalog whose store has no installed model directory - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + // Arrange: a catalog whose store has no installed model directory + var backendFactory = new FakeRecognitionEngineFactory(); var model = new FakeRecognitionModel(); // Act: compose against the catalog, which resolves to a directory that does not exist - var recognizer = SpeechRecognizerFactory.Create(model, _catalog, captureDevice, null, engineFactory); + await using var engine = await SpeechRecognizerFactory.LoadAsync(model, _catalog, null, backendFactory, cancellationToken: TestContext.Current.CancellationToken); - // Assert: the honest fallback is returned and no engine was loaded - Assert.Same(UnavailableSpeechRecognizer.Instance, recognizer); - Assert.Equal(0, engineFactory.CreateCallCount); + // Assert: the honest fallback is returned and no backend was loaded + Assert.Same(UnavailableSpeechRecognizerEngine.Instance, engine); + Assert.Equal(0, backendFactory.CreateCallCount); } /// - /// Proves that an installed recognition model plus an available capture device composes a - /// real recognizer wired to the injected engine factory, with the directory resolved - /// through the catalog's own store rather than a second, disconnected store. + /// Proves that an installed recognition model composes a real engine wired to the + /// injected backend factory, with the directory resolved through the catalog's own store + /// rather than a second, disconnected store. /// [Fact] - public void SpeechRecognizerFactory_Create_WithCatalogModelInstalledAndDeviceAvailable_ReturnsRealRecognizer() + public async Task SpeechRecognizerFactory_LoadAsync_WithCatalogModelInstalled_ReturnsRealEngine() { - // Arrange: a model installed via the catalog's store, an available device, and a fake engine factory - var captureDevice = CreateAvailableCaptureDevice(); - var engineFactory = new FakeRecognitionEngineFactory(); + // Arrange: a model installed via the catalog's store, and a fake backend factory + var backendFactory = new FakeRecognitionEngineFactory(); var model = new FakeRecognitionModel(); Directory.CreateDirectory(_store.GetCurrentDirectory(model.Id)); - // Act: compose a recognizer through the catalog overload - using var recognizer = SpeechRecognizerFactory.Create(model, _catalog, captureDevice, null, engineFactory); + // Act: compose an engine through the catalog overload + await using var engine = await SpeechRecognizerFactory.LoadAsync(model, _catalog, null, backendFactory, cancellationToken: TestContext.Current.CancellationToken); - // Assert: a real recognizer was built, and the directory was resolved through the catalog's store - Assert.IsType(recognizer); - Assert.Equal(1, engineFactory.CreateCallCount); - Assert.Equal(_catalog.Store.GetCurrentDirectory(model.Id), engineFactory.RequestedInstalledModelDirectory); + // Assert: a real engine was built, and the directory was resolved through the catalog's store + Assert.IsType(engine); + Assert.Equal(1, backendFactory.CreateCallCount); + Assert.Equal(_catalog.Store.GetCurrentDirectory(model.Id), backendFactory.RequestedInstalledModelDirectory); } /// /// Proves that a supplied parameterValues bag genuinely reaches the model's own /// two-argument CreateEngineConfig override when composed through the PUBLIC - /// catalog-based composition overload (with no injected engineFactory), so the real + /// catalog-based composition overload (with no injected backendFactory), so the real /// production delegation chain (public catalog overload → public store overload → public - /// string overload → internal string+engineFactory overload → real + /// string overload → internal string+backendFactory overload → real /// SherpaOnnxRecognitionEngineFactory) is exercised end to end. The model throws /// from within its own CreateEngineConfig override, after recording the received /// parameter bag, so the test never reaches a real native sherpa-onnx engine construction. /// [Fact] - public void SpeechRecognizerFactory_Create_WithCatalogParameterValuesSupplied_ReachesModelCreateEngineConfig() + public async Task SpeechRecognizerFactory_LoadAsync_WithCatalogParameterValuesSupplied_ReachesModelCreateEngineConfig() { - // Arrange: a throwing parameter-capturing model installed via the catalog's store, an available device, and a parameter bag - var captureDevice = CreateAvailableCaptureDevice(); + // Arrange: a throwing parameter-capturing model installed via the catalog's store, and a parameter bag var model = new ThrowingParameterCapturingRecognitionModel(); Directory.CreateDirectory(_store.GetCurrentDirectory(model.Id)); IReadOnlyDictionary parameterValues = new Dictionary { ["language"] = "en-gb" }; - // Act: compose a recognizer through the genuine public catalog overload, supplying - // parameterValues, with no engineFactory argument so overload resolution can only bind + // Act: compose an engine through the genuine public catalog overload, supplying + // parameterValues, with no backendFactory argument so overload resolution can only bind // to the real public overload - var recognizer = SpeechRecognizerFactory.Create(model, _catalog, captureDevice, null, parameterValues); + await using var engine = await SpeechRecognizerFactory.LoadAsync(model, _catalog, null, parameterValues, TestContext.Current.CancellationToken); // Assert: the value reached the model's own CreateEngineConfig override via the real // production chain, and the model's throw was honestly swallowed into the unavailable - // fallback rather than propagating - Assert.Same(UnavailableSpeechRecognizer.Instance, recognizer); + // fallback rather than faulting the task + Assert.Same(UnavailableSpeechRecognizerEngine.Instance, engine); Assert.Equal(parameterValues, model.RequestedParameterValues); } /// - /// Proves that the public catalog-based composition overload rejects a null model. + /// Proves that the public catalog-based composition overload faults the returned task + /// with for a null model. /// [Fact] - public void SpeechRecognizerFactory_Create_WithCatalogNullModel_ThrowsArgumentNullException() + public async Task SpeechRecognizerFactory_LoadAsync_WithCatalogNullModel_FaultsWithArgumentNullException() { - // Arrange: an available capture device - var captureDevice = CreateAvailableCaptureDevice(); - // Act & Assert: a null model is rejected - Assert.Throws( - () => SpeechRecognizerFactory.Create(null!, _catalog, captureDevice)); + await Assert.ThrowsAsync( + () => SpeechRecognizerFactory.LoadAsync(null!, _catalog, cancellationToken: TestContext.Current.CancellationToken)); } /// - /// Proves that the public catalog-based composition overload rejects a null catalog. + /// Proves that the public catalog-based composition overload faults the returned task + /// with for a null catalog. /// [Fact] - public void SpeechRecognizerFactory_Create_WithCatalogNullCatalog_ThrowsArgumentNullException() + public async Task SpeechRecognizerFactory_LoadAsync_WithCatalogNullCatalog_FaultsWithArgumentNullException() { - // Arrange: an available capture device - var captureDevice = CreateAvailableCaptureDevice(); - // Act & Assert: a null catalog is rejected - Assert.Throws( - () => SpeechRecognizerFactory.Create(new FakeRecognitionModel(), (SpeechModelCatalog)null!, captureDevice)); + await Assert.ThrowsAsync( + () => SpeechRecognizerFactory.LoadAsync(new FakeRecognitionModel(), (SpeechModelCatalog)null!, cancellationToken: TestContext.Current.CancellationToken)); } /// - /// Proves that the public catalog-based composition overload rejects a null capture - /// device, confirming the delegation still reaches the store-overload's own null check. + /// Proves that when ignores cancellation + /// but still finishes within the dedicated worker's abandon grace period - so the worker + /// genuinely completes rather than being abandoned - + /// still honors the cancellation request (finding 30) rather than reporting a loaded + /// engine, and disposes the backend that was created so it does not leak. /// - [Fact] - public void SpeechRecognizerFactory_Create_WithCatalogNullCaptureDevice_ThrowsArgumentNullException() + [Fact(Timeout = 10000)] + public async Task SpeechRecognizerFactory_LoadAsync_CancelledDuringGraceWindowCreateSucceeds_ThrowsAndDisposesBackend() { - // Act & Assert: a null capture device is rejected - Assert.Throws( - () => SpeechRecognizerFactory.Create(new FakeRecognitionModel(), _catalog, null!)); + // Arrange: a backend factory whose Create blocks until released, then succeeds, ignoring + // cancellation entirely - exactly like the real native binding, which has no in-flight + // cancellation primitive of its own. + using var createStarted = new SemaphoreSlim(0, 1); + using var createRelease = new SemaphoreSlim(0, 1); + var engine = new FakeRecognitionEngine(); + var backendFactory = new BlockingRecognitionEngineFactory(createStarted, createRelease, engine); + using var cts = new CancellationTokenSource(); + + // Act: start loading, wait until Create has begun, cancel, then let Create finish quickly + // (well within the worker's default abandon timeout) so the worker is not abandoned + var loadTask = SpeechRecognizerFactory.LoadAsync( + new FakeRecognitionModel(), _installedModelDirectory, null, backendFactory, cancellationToken: cts.Token); + await createStarted.WaitAsync(TestContext.Current.CancellationToken); + await cts.CancelAsync(); + createRelease.Release(); + + // Assert: the cancellation is still honored even though backend creation genuinely + // succeeded, and the backend that was created is disposed rather than leaked + await Assert.ThrowsAsync(() => loadTask); + Assert.Equal(1, engine.DisposeCallCount); } /// - /// Builds a substitute capture device that reports itself available with a realistic - /// stereo 48 kHz capture format. + /// Test-only whose signals a + /// semaphore once called, then blocks until the test explicitly releases a second + /// semaphore before returning the pre-configured engine - simulating a native load call + /// that ignores cancellation but still finishes within the dedicated worker's abandon + /// grace period. /// - /// The configured substitute capture device. - private static IAudioCaptureDevice CreateAvailableCaptureDevice() + private sealed class BlockingRecognitionEngineFactory( + SemaphoreSlim createStarted, + SemaphoreSlim createRelease, + FakeRecognitionEngine engine) : IRecognitionBackendFactory { - var captureDevice = Substitute.For(); - captureDevice.IsAvailable.Returns(true); - captureDevice.SampleRate.Returns(48000); - captureDevice.ChannelCount.Returns(2); - return captureDevice; + public IRecognitionBackend Create( + IRecognitionModel model, + string installedModelDirectory, + IReadOnlyDictionary? parameterValues = null) + { + createStarted.Release(); + + // Bounded wait purely as a safety net so a failed/timed-out test cannot leave this + // dedicated worker thread blocked forever; the behavior under test only relies on the + // release happening promptly. + createRelease.Wait(TimeSpan.FromSeconds(30)); + return engine; + } } /// @@ -634,8 +601,8 @@ SherpaOnnx.OnlineRecognizerConfig IRecognitionModel.CreateEngineConfig(string in /// Test-only recognition model whose two-argument CreateEngineConfig override /// encodes a supplied "language" parameter value into the returned config's /// ModelConfig.ModelType, used to prove a parameterValues bag supplied to - /// - /// genuinely reaches the model, not merely the engine factory. + /// 's LoadAsync overloads genuinely reaches + /// the model, not merely the backend factory. /// private sealed class ParameterCapturingRecognitionModel : IRecognitionModel { @@ -668,7 +635,7 @@ SherpaOnnx.OnlineRecognizerConfig IRecognitionModel.CreateEngineConfig(string in /// /// Encodes a supplied "language" string value into the returned config's /// ModelConfig.ModelType field, so a test can assert the value it passed as - /// parameterValues reached this method, not merely the engine factory that + /// parameterValues reached this method, not merely the backend factory that /// called it. /// /// @@ -694,7 +661,7 @@ SherpaOnnx.OnlineRecognizerConfig IRecognitionModel.CreateEngineConfig( /// and then throws, short-circuiting before any real native sherpa-onnx engine /// construction could occur. Used to prove that a parameterValues bag supplied to /// the genuine PUBLIC store/catalog composition overloads (with no injected - /// engineFactory test seam) reaches this model through the real production + /// backendFactory test seam) reaches this model through the real production /// delegation chain, without requiring a working native engine. /// private sealed class ThrowingParameterCapturingRecognitionModel : IRecognitionModel diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableRecognitionSessionTests.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableRecognitionSessionTests.cs new file mode 100644 index 0000000..d8f1126 --- /dev/null +++ b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableRecognitionSessionTests.cs @@ -0,0 +1,102 @@ +using DemaConsulting.Speech.RecognitionSubsystem; + +namespace DemaConsulting.Speech.Tests.RecognitionSubsystem; + +/// +/// Unit tests for . +/// +public class UnavailableRecognitionSessionTests +{ + /// + /// Proves that reports itself as + /// unavailable. + /// + [Fact] + public void UnavailableRecognitionSession_IsAvailable_Read_ReturnsFalse() + { + // Arrange & Act: read the availability flag from the shared instance + var isAvailable = UnavailableRecognitionSession.Instance.IsAvailable; + + // Assert: the session honestly reports itself as unavailable + Assert.False(isAvailable); + } + + /// + /// Proves that calling throws + /// , since this session has no real + /// engine or capture device to start. + /// + [Fact] + public async Task UnavailableRecognitionSession_StartAsync_Always_ThrowsSpeechRecognizerUnavailableException() + { + // Arrange: the shared unavailable session + var session = UnavailableRecognitionSession.Instance; + + // Act & Assert + await Assert.ThrowsAsync(() => session.StartAsync(TestContext.Current.CancellationToken)); + } + + /// + /// Proves that enumerating + /// throws . + /// + [Fact] + public async Task UnavailableRecognitionSession_GetResultsAsync_Always_ThrowsSpeechRecognizerUnavailableException() + { + // Arrange: the shared unavailable session + var session = UnavailableRecognitionSession.Instance; + + // Act & Assert: enumerating the (lazily-evaluated) sequence surfaces the exception + await Assert.ThrowsAsync(async () => + { + await foreach (var _ in session.GetResultsAsync(TestContext.Current.CancellationToken)) + { + // No iterations are expected: the exception is thrown before the first result. + } + }); + } + + /// + /// Proves that calling is a safe + /// no-op that never throws, since this session was never running. + /// + [Fact] + public async Task UnavailableRecognitionSession_StopAsync_Always_IsSafeNoOp() + { + // Arrange: the shared unavailable session + var session = UnavailableRecognitionSession.Instance; + + // Act + var exception = await Record.ExceptionAsync(() => session.StopAsync(TestContext.Current.CancellationToken)); + + // Assert: nothing threw + Assert.Null(exception); + } + + /// + /// Proves that calling twice, + /// along with subscribing to and unsubscribing from , + /// are all safe no-ops that never invalidate the shared instance. + /// + [Fact] + public async Task UnavailableRecognitionSession_DisposeAsync_CalledTwice_DoesNotThrow() + { + // Arrange: the shared unavailable session and a handler to (un)subscribe + var session = UnavailableRecognitionSession.Instance; + EventHandler handler = (_, _) => { }; + + // Act + var exception = await Record.ExceptionAsync(async () => + { + session.StateChanged += handler; + session.StateChanged -= handler; + await session.DisposeAsync(); + await session.DisposeAsync(); + }); + + // Assert: nothing throws and the shared instance is still usable + Assert.Null(exception); + Assert.False(UnavailableRecognitionSession.Instance.IsAvailable); + Assert.Equal(RecognitionSessionState.Created, UnavailableRecognitionSession.Instance.State); + } +} diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableSpeechRecognizerEngineTests.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableSpeechRecognizerEngineTests.cs new file mode 100644 index 0000000..cd8b056 --- /dev/null +++ b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableSpeechRecognizerEngineTests.cs @@ -0,0 +1,147 @@ +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.RecognitionSubsystem; +using NSubstitute; + +namespace DemaConsulting.Speech.Tests.RecognitionSubsystem; + +/// +/// Unit tests for and +/// . +/// +public class UnavailableSpeechRecognizerEngineTests +{ + /// + /// Proves that reports itself as + /// unavailable. + /// + [Fact] + public void UnavailableSpeechRecognizerEngine_IsAvailable_Read_ReturnsFalse() + { + // Arrange & Act: read the availability flag from the shared instance + var isAvailable = UnavailableSpeechRecognizerEngine.Instance.IsAvailable; + + // Assert: the engine honestly reports itself as unavailable + Assert.False(isAvailable); + } + + /// + /// Proves that never + /// throws for ordinary unavailability, instead always returning the shared + /// . + /// + [Fact] + public async Task UnavailableSpeechRecognizerEngine_CreateSessionAsync_Always_ReturnsUnavailableSession() + { + // Arrange: the shared unavailable engine and an arbitrary device + var engine = UnavailableSpeechRecognizerEngine.Instance; + var device = Substitute.For(); + + // Act: create a session + var session = await engine.CreateSessionAsync(device, TestContext.Current.CancellationToken); + + // Assert: the shared unavailable session was returned + Assert.Same(UnavailableRecognitionSession.Instance, session); + } + + /// + /// Proves that still + /// rejects a null device, since that is a genuine caller error rather than an ordinary + /// unavailable state. + /// + [Fact] + public async Task UnavailableSpeechRecognizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException() + { + // Arrange: the shared unavailable engine + var engine = UnavailableSpeechRecognizerEngine.Instance; + + // Act & Assert + await Assert.ThrowsAsync(() => engine.CreateSessionAsync(null!, TestContext.Current.CancellationToken)); + } + + /// + /// Proves that still + /// honors an already-cancelled token, since that is a genuine caller error rather than + /// an ordinary unavailable state - matching the + /// contract every implementation must honor. + /// + [Fact] + public async Task UnavailableSpeechRecognizerEngine_CreateSessionAsync_CancelledToken_ThrowsOperationCanceledException() + { + // Arrange: the shared unavailable engine, an arbitrary device, and an already-cancelled token + var engine = UnavailableSpeechRecognizerEngine.Instance; + var device = Substitute.For(); + using var cts = new CancellationTokenSource(); + await cts.CancelAsync(); + + // Act & Assert + await Assert.ThrowsAsync(() => engine.CreateSessionAsync(device, cts.Token)); + } + + /// + /// Proves that calling twice + /// is a safe no-op that never invalidates the shared instance. + /// + [Fact] + public async Task UnavailableSpeechRecognizerEngine_DisposeAsync_CalledTwice_DoesNotThrow() + { + // Arrange: the shared unavailable engine + var engine = UnavailableSpeechRecognizerEngine.Instance; + + // Act + await engine.DisposeAsync(); + await engine.DisposeAsync(); + + // Assert: the shared instance is still usable + Assert.False(UnavailableSpeechRecognizerEngine.Instance.IsAvailable); + } + + /// + /// Proves that exposes the message + /// supplied to its single-argument constructor, confirming standard exception conformance. + /// + [Fact] + public void SpeechRecognizerUnavailableException_Constructor_WithMessage_ExposesMessage() + { + // Arrange: a specific message + const string message = "No recognition model is installed."; + + // Act: construct the exception with the message + var exception = new SpeechRecognizerUnavailableException(message); + + // Assert: the message is exposed unchanged + Assert.Equal(message, exception.Message); + } + + /// + /// Proves that exposes both the message + /// and inner exception supplied to its two-argument constructor. + /// + [Fact] + public void SpeechRecognizerUnavailableException_Constructor_WithInnerException_ExposesBoth() + { + // Arrange: a message and an inner exception + const string message = "Recognition engine failed to load."; + var inner = new InvalidOperationException("native runtime missing"); + + // Act: construct the exception with both + var exception = new SpeechRecognizerUnavailableException(message, inner); + + // Assert: both the message and inner exception are exposed unchanged + Assert.Equal(message, exception.Message); + Assert.Same(inner, exception.InnerException); + } + + /// + /// Proves that the default (parameterless) constructor of + /// produces a non-empty default message. + /// + [Fact] + public void SpeechRecognizerUnavailableException_Constructor_Default_HasNonEmptyMessage() + { + // Act: construct with no arguments + var exception = new SpeechRecognizerUnavailableException(); + + // Assert: a default, non-empty message is provided + Assert.False(string.IsNullOrEmpty(exception.Message)); + } +} diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableSpeechRecognizerTests.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableSpeechRecognizerTests.cs deleted file mode 100644 index 7c45a2c..0000000 --- a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/UnavailableSpeechRecognizerTests.cs +++ /dev/null @@ -1,128 +0,0 @@ -using DemaConsulting.Speech.RecognitionSubsystem; - -namespace DemaConsulting.Speech.Tests.RecognitionSubsystem; - -/// -/// Unit tests for and -/// . -/// -public class UnavailableSpeechRecognizerTests -{ - /// - /// Proves that reports itself as - /// unavailable. - /// - [Fact] - public void UnavailableSpeechRecognizer_IsAvailable_Read_ReturnsFalse() - { - // Arrange & Act: read the availability flag from the shared instance - var isAvailable = UnavailableSpeechRecognizer.Instance.IsAvailable; - - // Assert: the recognizer honestly reports itself as unavailable - Assert.False(isAvailable); - } - - /// - /// Proves that calling on the unavailable recognizer - /// throws . - /// - [Fact] - public void UnavailableSpeechRecognizer_Start_Always_ThrowsSpeechRecognizerUnavailableException() - { - // Arrange: the shared unavailable recognizer - var recognizer = UnavailableSpeechRecognizer.Instance; - - // Act & Assert: starting an unavailable recognizer throws the documented exception - Assert.Throws(recognizer.Start); - } - - /// - /// Proves that calling on the unavailable recognizer - /// throws . - /// - [Fact] - public void UnavailableSpeechRecognizer_Stop_Always_ThrowsSpeechRecognizerUnavailableException() - { - // Arrange: the shared unavailable recognizer - var recognizer = UnavailableSpeechRecognizer.Instance; - - // Act & Assert: stopping an unavailable recognizer throws the documented exception - Assert.Throws(recognizer.Stop); - } - - /// - /// Proves that subscribing to and unsubscribing from - /// , and disposing the shared instance - /// repeatedly, are all safe no-ops that never throw or invalidate the instance. - /// - [Fact] - public void UnavailableSpeechRecognizer_SubscriptionAndDispose_Always_AreSafeNoOps() - { - // Arrange: the shared unavailable recognizer and a handler to (un)subscribe - var recognizer = UnavailableSpeechRecognizer.Instance; - EventHandler handler = (_, _) => { }; - - // Act: subscribe, unsubscribe, and dispose twice - var exception = Record.Exception(() => - { - recognizer.ResultReceived += handler; - recognizer.ResultReceived -= handler; - recognizer.Dispose(); - recognizer.Dispose(); - }); - - // Assert: nothing throws and the shared instance is still usable - Assert.Null(exception); - Assert.False(UnavailableSpeechRecognizer.Instance.IsAvailable); - } - - /// - /// Proves that exposes the message - /// supplied to its single-argument constructor, confirming standard exception conformance. - /// - [Fact] - public void SpeechRecognizerUnavailableException_Constructor_WithMessage_ExposesMessage() - { - // Arrange: a specific message - const string message = "No recognition model is installed."; - - // Act: construct the exception with the message - var exception = new SpeechRecognizerUnavailableException(message); - - // Assert: the message is exposed unchanged - Assert.Equal(message, exception.Message); - } - - /// - /// Proves that exposes both the message - /// and inner exception supplied to its two-argument constructor. - /// - [Fact] - public void SpeechRecognizerUnavailableException_Constructor_WithInnerException_ExposesBoth() - { - // Arrange: a message and an inner exception - const string message = "Recognition engine failed to load."; - var inner = new InvalidOperationException("native runtime missing"); - - // Act: construct the exception with both - var exception = new SpeechRecognizerUnavailableException(message, inner); - - // Assert: both the message and inner exception are exposed unchanged - Assert.Equal(message, exception.Message); - Assert.Same(inner, exception.InnerException); - } - - /// - /// Proves that the default (parameterless) constructor of - /// produces a non-empty default message. - /// - [Fact] - public void SpeechRecognizerUnavailableException_Constructor_Default_HasNonEmptyMessage() - { - // Act: construct with no arguments - var exception = new SpeechRecognizerUnavailableException(); - - // Assert: a default, non-empty message is provided - Assert.False(string.IsNullOrEmpty(exception.Message)); - } -} diff --git a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/WordErrorRateCalculator.cs b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/WordErrorRateCalculator.cs index ca0622e..543f63c 100644 --- a/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/WordErrorRateCalculator.cs +++ b/test/DemaConsulting.Speech.Tests/RecognitionSubsystem/WordErrorRateCalculator.cs @@ -16,7 +16,7 @@ namespace DemaConsulting.Speech.Tests.RecognitionSubsystem; /// with silence and a synthesized tone (see that test class's own remarks). Because nothing /// in src/ depends on this type, it carries no production requirement of its own; the /// requirement it supports -/// (Speech-Recognition-RecognitionEngine-RealSpeechAccuracy) is satisfied by the +/// (Speech-Recognition-RecognitionBackend-RealSpeechAccuracy) is satisfied by the /// accuracy test that consumes it, not by this calculation helper itself. /// /// Both transcripts are independently normalized before comparison (lower-cased, punctuation diff --git a/test/DemaConsulting.Speech.Tests/SpeechTests.cs b/test/DemaConsulting.Speech.Tests/SpeechTests.cs index 5cdc1c7..3d3e7e3 100644 --- a/test/DemaConsulting.Speech.Tests/SpeechTests.cs +++ b/test/DemaConsulting.Speech.Tests/SpeechTests.cs @@ -230,16 +230,16 @@ public async Task Speech_SystemIntegration_ModelCatalog_EnumeratesAndTracksDownl /// /// Proves that the system composes the full streaming recognition pipeline end to end: - /// builds a real recognizer over an installed model - /// and an available capture device, a captured audio block flows through resampling into - /// the engine, and the resulting recognition text is surfaced through - /// . + /// loads a real engine over an installed model, an + /// is bound to an available capture device, a captured + /// audio block flows through resampling into the backend, and the resulting recognition + /// text is surfaced through . /// [Fact] - public void Speech_SystemIntegration_StreamingRecognition_CapturedAudioProducesRecognitionResults() + public async Task Speech_SystemIntegration_StreamingRecognition_CapturedAudioProducesRecognitionResults() { // Arrange: a scratch installed-model directory, an available capture device, and a - // deterministic engine standing in for the native sherpa-onnx runtime + // deterministic backend standing in for the native sherpa-onnx runtime var installedModelDirectory = Path.Join( Path.GetTempPath(), "DemaConsulting.Speech.Tests", Guid.NewGuid().ToString("N")); Directory.CreateDirectory(installedModelDirectory); @@ -249,29 +249,44 @@ public void Speech_SystemIntegration_StreamingRecognition_CapturedAudioProducesR captureDevice.IsAvailable.Returns(true); captureDevice.SampleRate.Returns(16000); captureDevice.ChannelCount.Returns(1); - var engineFactory = new FakeRecognitionEngineFactory(new FakeRecognitionEngine( + var backendFactory = new FakeRecognitionEngineFactory(new FakeRecognitionEngine( [ new SpeechRecognitionResult("hello", IsFinal: false), new SpeechRecognitionResult("hello world", IsFinal: true) ])); - // Act: compose the recognizer, stream one captured block, and drain the pipeline - using var recognizer = SpeechRecognizerFactory.Create( - new FakeRecognitionModel(), installedModelDirectory, captureDevice, null, engineFactory); + // Act: load the engine, create a session, stream one captured block, and drain the pipeline + var cancellationToken = TestContext.Current.CancellationToken; + await using var engine = await SpeechRecognizerFactory.LoadAsync( + new FakeRecognitionModel(), installedModelDirectory, null, backendFactory, + cancellationToken: cancellationToken); + await using var session = await engine.CreateSessionAsync(captureDevice, cancellationToken); + var received = new List(); - recognizer.ResultReceived += (_, args) => received.Add(args.Result); - recognizer.Start(); + var pump = Task.Run(async () => + { + await foreach (var evt in session.GetResultsAsync(cancellationToken)) + { + received.Add(evt.Result); + } + }, cancellationToken); + + await session.StartAsync(cancellationToken); captureDevice.FrameCaptured += Raise.Event>( captureDevice, new AudioCaptureFrameEventArgs([0.1f, 0.2f, 0.3f])); - recognizer.Stop(); - - // Assert: the system produced a provisional and a final result from the captured audio - Assert.True(recognizer.IsAvailable); - Assert.Equal(2, received.Count); - Assert.False(received[0].IsFinal); - Assert.Equal("hello world", received[1].Text); - Assert.True(received[1].IsFinal); + await session.StopAsync(cancellationToken); + await pump; + + // Assert: the system produced the final recognition result from the captured audio. + // The scripted provisional result ("hello") arrived before the consumer started + // draining, and its own final ("hello world") was buffered before it was ever read, + // so the provisional was superseded rather than delivered as a stale partial + // transcript trailing its own final (Decision #5). + Assert.True(session.IsAvailable); + var result = Assert.Single(received); + Assert.True(result.IsFinal); + Assert.Equal("hello world", result.Text); } finally { @@ -281,16 +296,16 @@ public void Speech_SystemIntegration_StreamingRecognition_CapturedAudioProducesR /// /// Proves that the system composes the full chunked streaming synthesis pipeline end to - /// end: builds a real synthesizer over an - /// installed model and an available playback device, text flows through Layer 2 - /// rendering and the synthesis engine, and the resulting audio is written to the playback - /// device in order. + /// end: loads a real engine over an installed + /// model, an is bound to an available playback device, + /// text flows through Layer 2 rendering and the synthesis backend, and the resulting + /// audio is written to the playback device in order. /// [Fact] public async Task Speech_SystemIntegration_StreamingSynthesis_TextProducesPlayedAudio() { // Arrange: a scratch installed-model directory, an available playback device, and a - // deterministic engine standing in for the native sherpa-onnx runtime + // deterministic backend standing in for the native sherpa-onnx runtime var installedModelDirectory = Path.Join( Path.GetTempPath(), "DemaConsulting.Speech.Tests", Guid.NewGuid().ToString("N")); Directory.CreateDirectory(installedModelDirectory); @@ -300,13 +315,16 @@ public async Task Speech_SystemIntegration_StreamingSynthesis_TextProducesPlayed playbackDevice.IsAvailable.Returns(true); playbackDevice.SampleRate.Returns(16000); playbackDevice.ChannelCount.Returns(1); - var engineFactory = new FakeSynthesisEngineFactory(new FakeSynthesisEngine(sampleRate: 16000)); - - // Act: compose the synthesizer and speak one plain-text utterance end to end - using var synthesizer = SpeechSynthesizerFactory.Create( - new FakeSynthesisModel(), installedModelDirectory, playbackDevice, null, engineFactory); - Assert.True(synthesizer.IsAvailable); - await synthesizer.SpeakAsync("hello world", TestContext.Current.CancellationToken); + var backendFactory = new FakeSynthesisEngineFactory(new FakeSynthesisEngine(sampleRate: 16000)); + + // Act: load the engine, create a session, and speak one plain-text utterance end to end + var cancellationToken = TestContext.Current.CancellationToken; + await using var engine = await SpeechSynthesizerFactory.LoadAsync( + new FakeSynthesisModel(), installedModelDirectory, null, backendFactory, + cancellationToken: cancellationToken); + Assert.True(engine.IsAvailable); + await using var session = await engine.CreateSessionAsync(playbackDevice, cancellationToken); + await session.SpeakAsync("hello world", cancellationToken); // Assert: the system started the device, wrote at least one block of audio, and stopped it playbackDevice.Received(1).Start(); diff --git a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/DedicatedWorkerTests.cs b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/DedicatedWorkerTests.cs new file mode 100644 index 0000000..0d60e94 --- /dev/null +++ b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/DedicatedWorkerTests.cs @@ -0,0 +1,100 @@ +using DemaConsulting.Speech.Diagnostics; +using DemaConsulting.Speech.SynthesisSubsystem; +using NSubstitute; + +namespace DemaConsulting.Speech.Tests.SynthesisSubsystem; + +/// +/// Unit tests for , proving the cooperative-cancel-then-abandon +/// policy deterministically, without relying on the real two-second default timeout. +/// +public sealed class DedicatedWorkerTests +{ + /// + /// Proves that a delegate honoring cancellation completes promptly, without waiting for + /// the abandon timeout to elapse. + /// + [Fact] + public async Task DedicatedWorker_Run_CooperativeCancellation_CompletesPromptly() + { + // Arrange: a delegate that cooperatively observes cancellation and stops immediately + using var cts = new CancellationTokenSource(); + + // Act + var task = DedicatedWorker.Run( + token => + { + while (!token.IsCancellationRequested) + { + SpinWait.SpinUntil(() => token.IsCancellationRequested, TimeSpan.FromMilliseconds(5)); + } + + token.ThrowIfCancellationRequested(); + return 0; + }, + cts.Token, + null, + "SynthesisSubsystem", + TimeSpan.FromSeconds(10)); + await cts.CancelAsync(); + + // Assert: the call completes (cancelled) well before the 10-second abandon timeout + await Assert.ThrowsAnyAsync(() => task); + } + + /// + /// Proves that a delegate which never observes cancellation is abandoned once the + /// (short, injected) abandon timeout elapses, that the returned task completes as + /// cancelled, and that a warning is reported through diagnostics. + /// + [Fact] + public async Task DedicatedWorker_Run_NonCooperativeDelegate_AbandonsAfterTimeoutAndReportsDiagnostics() + { + // Arrange: a delegate that never checks its token, and a short injected abandon timeout + using var cts = new CancellationTokenSource(); + using var release = new SemaphoreSlim(0, 1); + var diagnostics = Substitute.For(); + + // Act + var task = DedicatedWorker.Run( + _ => + { + release.Wait(TestContext.Current.CancellationToken); + return 0; + }, + cts.Token, + diagnostics, + "SynthesisSubsystem", + TimeSpan.FromMilliseconds(50)); + await cts.CancelAsync(); + + // Assert: the call is abandoned and reported as cancelled, rather than hanging + await Assert.ThrowsAnyAsync(() => task); + diagnostics.Received().Report( + SpeechDiagnosticLevel.Warning, + "SynthesisSubsystem", + Arg.Is(message => message.Contains("abandoned", StringComparison.Ordinal))); + + // Release the abandoned worker thread so it does not outlive the test. + release.Release(); + } + + /// + /// Proves that runs its delegate on a dedicated + /// long-running thread rather than a pooled thread-pool thread, by observing + /// from inside the delegate. + /// + [Fact] + public async Task DedicatedWorker_Run_UsesLongRunningTaskCreationOption() + { + // Act + var isThreadPoolThread = await DedicatedWorker.Run( + _ => Thread.CurrentThread.IsThreadPoolThread, + CancellationToken.None, + null, + "SynthesisSubsystem"); + + // Assert + Assert.False(isThreadPoolThread); + } +} diff --git a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/Fakes/FakeSynthesisEngine.cs b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/Fakes/FakeSynthesisEngine.cs index a6c67af..7bb6b00 100644 --- a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/Fakes/FakeSynthesisEngine.cs +++ b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/Fakes/FakeSynthesisEngine.cs @@ -3,12 +3,12 @@ namespace DemaConsulting.Speech.Tests.SynthesisSubsystem.Fakes; /// -/// Deterministic test double that records every call it +/// Deterministic test double that records every call it /// receives and returns scripted (or generated) audio, so the synthesizer's chunking, /// pipelining, and playback behavior can be verified with no native sherpa-onnx runtime and /// no downloaded model. /// -internal sealed class FakeSynthesisEngine : ISynthesisEngine +internal sealed class FakeSynthesisEngine : ISynthesisBackend { /// The exception to throw from , when one was scripted. private readonly Exception? _generateException; diff --git a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/Fakes/FakeSynthesisEngineFactory.cs b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/Fakes/FakeSynthesisEngineFactory.cs index e633edd..cb3c3ac 100644 --- a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/Fakes/FakeSynthesisEngineFactory.cs +++ b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/Fakes/FakeSynthesisEngineFactory.cs @@ -4,11 +4,11 @@ namespace DemaConsulting.Speech.Tests.SynthesisSubsystem.Fakes; /// -/// Deterministic test double that hands out a +/// Deterministic test double that hands out a /// pre-configured and records the arguments it was asked /// to load, so composition can be verified without a model directory or a native runtime. /// -internal sealed class FakeSynthesisEngineFactory : ISynthesisEngineFactory +internal sealed class FakeSynthesisEngineFactory : ISynthesisBackendFactory { /// The exception to throw from , when one was scripted. private readonly Exception? _createException; @@ -46,7 +46,7 @@ public FakeSynthesisEngineFactory( public int CreateCallCount { get; private set; } /// - public ISynthesisEngine Create(ISynthesisModel model, string installedModelDirectory) + public ISynthesisBackend Create(ISynthesisModel model, string installedModelDirectory) { CreateCallCount++; RequestedModel = model; diff --git a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerEngineTests.cs b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerEngineTests.cs new file mode 100644 index 0000000..c909a2c --- /dev/null +++ b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerEngineTests.cs @@ -0,0 +1,233 @@ +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.SynthesisSubsystem; +using DemaConsulting.Speech.Tests.ModelManagementSubsystem.Fakes; +using DemaConsulting.Speech.Tests.SynthesisSubsystem.Fakes; +using NSubstitute; + +namespace DemaConsulting.Speech.Tests.SynthesisSubsystem; + +/// +/// Unit tests for , exercising session +/// creation, its exclusivity lease, the one-shot / +/// convenience methods, and disposal. +/// +public sealed class SherpaOnnxSpeechSynthesizerEngineTests +{ + /// + /// Proves that a freshly constructed engine always reports itself available, since it is + /// only ever constructed after a successful backend load. + /// + [Fact] + public void SherpaOnnxSpeechSynthesizerEngine_IsAvailable_Always_ReturnsTrue() + { + // Arrange + var engine = new SherpaOnnxSpeechSynthesizerEngine(new FakeSynthesisEngine(), new FakeSynthesisModel()); + + // Act & Assert + Assert.True(engine.IsAvailable); + } + + /// + /// Proves that succeeds when no + /// lease is held, returning a real, usable session bound to the supplied device. + /// + [Fact] + public async Task SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_NoLeaseHeld_ReturnsRealSession() + { + // Arrange + var engine = new SherpaOnnxSpeechSynthesizerEngine(new FakeSynthesisEngine(), new FakeSynthesisModel()); + var device = CreateAvailablePlaybackDevice(); + + // Act + await using var session = await engine.CreateSessionAsync(device, TestContext.Current.CancellationToken); + + // Assert + Assert.True(session.IsAvailable); + Assert.Equal(SynthesisSessionState.Created, session.State); + } + + /// + /// Proves that a concurrent call + /// while a lease is already held fails fast with , + /// rather than queueing or waiting. + /// + [Fact] + public async Task SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_LeaseAlreadyHeld_ThrowsSynthesisEngineBusyException() + { + // Arrange: one session already leased + var engine = new SherpaOnnxSpeechSynthesizerEngine(new FakeSynthesisEngine(), new FakeSynthesisModel()); + var device = CreateAvailablePlaybackDevice(); + await using var firstSession = await engine.CreateSessionAsync(device, TestContext.Current.CancellationToken); + + // Act & Assert: a second, concurrent lease attempt fails fast + await Assert.ThrowsAsync( + () => engine.CreateSessionAsync(device, TestContext.Current.CancellationToken)); + } + + /// + /// Proves that once a leased session has been disposed, the lease is released and a new + /// session may be created. + /// + [Fact] + public async Task SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_AfterPriorSessionDisposed_SucceedsAgain() + { + // Arrange: lease and release a first session + var engine = new SherpaOnnxSpeechSynthesizerEngine(new FakeSynthesisEngine(), new FakeSynthesisModel()); + var device = CreateAvailablePlaybackDevice(); + var firstSession = await engine.CreateSessionAsync(device, TestContext.Current.CancellationToken); + await firstSession.DisposeAsync(); + + // Act: a second lease attempt, after the first session's disposal, succeeds + await using var secondSession = await engine.CreateSessionAsync(device, TestContext.Current.CancellationToken); + + // Assert + Assert.True(secondSession.IsAvailable); + } + + /// + /// Proves that rejects a null + /// device. + /// + [Fact] + public async Task SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException() + { + // Arrange + var engine = new SherpaOnnxSpeechSynthesizerEngine(new FakeSynthesisEngine(), new FakeSynthesisModel()); + + // Act & Assert + await Assert.ThrowsAsync( + () => engine.CreateSessionAsync(null!, TestContext.Current.CancellationToken)); + } + + /// + /// Proves that the one-shot convenience + /// method creates a session, speaks through it, and disposes it, releasing the lease so a + /// subsequent call succeeds without the caller managing a session at all. + /// + [Fact] + public async Task SherpaOnnxSpeechSynthesizerEngine_SpeakAsync_CalledTwice_CreatesAndDisposesASessionEachTime() + { + // Arrange + var engine = new SherpaOnnxSpeechSynthesizerEngine(new FakeSynthesisEngine(), new FakeSynthesisModel()); + var device = CreateAvailablePlaybackDevice(); + + // Act: two independent one-shot calls + await engine.SpeakAsync(device, "Hello world.", TestContext.Current.CancellationToken); + await engine.SpeakAsync(device, "Hello again.", TestContext.Current.CancellationToken); + + // Assert: both calls genuinely started and stopped the device, proving the lease was + // released between calls rather than leaking + device.Received(2).Start(); + device.Received(2).Stop(); + } + + /// + /// Proves that the one-shot + /// convenience method creates and disposes a session internally, returning the + /// full-fidelity synthesized segments with no playback device required. + /// + [Fact] + public async Task SherpaOnnxSpeechSynthesizerEngine_SynthesizeAsync_NoDeviceSupplied_ReturnsSegments() + { + // Arrange + var fakeBackend = new FakeSynthesisEngine(); + var engine = new SherpaOnnxSpeechSynthesizerEngine(fakeBackend, new FakeSynthesisModel()); + + // Act + var segments = await engine.SynthesizeAsync("Hello world.", TestContext.Current.CancellationToken); + + // Assert + var segment = Assert.Single(segments); + Assert.NotEmpty(segment.Samples); + Assert.Single(fakeBackend.GenerateCalls); + } + + /// + /// Proves that disposal is idempotent and disposes the owned backend exactly once, even + /// when called more than once. + /// + [Fact] + public async Task SherpaOnnxSpeechSynthesizerEngine_DisposeAsync_CalledTwice_DisposesBackendOnce() + { + // Arrange + var backend = new FakeSynthesisEngine(); + var engine = new SherpaOnnxSpeechSynthesizerEngine(backend, new FakeSynthesisModel()); + + // Act + await engine.DisposeAsync(); + await engine.DisposeAsync(); + + // Assert + Assert.Equal(1, backend.DisposeCallCount); + } + + /// + /// Proves that disposing the engine while a session is still leased disposes that session + /// first (best-effort), releasing its lease, before disposing the owned backend. + /// + [Fact] + public async Task SherpaOnnxSpeechSynthesizerEngine_DisposeAsync_WithActiveLeasedSession_DisposesSessionFirst() + { + // Arrange: lease a session but never dispose it directly + var backend = new FakeSynthesisEngine(); + var engine = new SherpaOnnxSpeechSynthesizerEngine(backend, new FakeSynthesisModel()); + var device = CreateAvailablePlaybackDevice(); + var session = await engine.CreateSessionAsync(device, TestContext.Current.CancellationToken); + + // Act: dispose the engine directly + await engine.DisposeAsync(); + + // Assert: the leased session was disposed (and so transitioned away from Created) and + // the backend was disposed too + Assert.Equal(SynthesisSessionState.Disposed, session.State); + Assert.Equal(1, backend.DisposeCallCount); + } + + /// + /// Proves that and + /// racing with no synchronization between them + /// never tears the engine: every CreateSessionAsync attempt either succeeds with a + /// genuinely usable session or fails with , never + /// with some other exception that would indicate it observed a half-disposed lease or + /// backend. + /// + [Fact] + public async Task SherpaOnnxSpeechSynthesizerEngine_CreateSessionAsync_RacingDisposeAsync_NeverObservesTornState() + { + for (var i = 0; i < 50; i++) + { + // Arrange + var engine = new SherpaOnnxSpeechSynthesizerEngine(new FakeSynthesisEngine(), new FakeSynthesisModel()); + var device = CreateAvailablePlaybackDevice(); + + // Act: race session creation against engine disposal with no synchronization + var createTask = engine.CreateSessionAsync(device, TestContext.Current.CancellationToken); + var disposeTask = engine.DisposeAsync().AsTask(); + + var createException = await Record.ExceptionAsync(async () => + { + var session = await createTask; + await session.DisposeAsync(); + }); + await disposeTask; + + // Assert: a create that lost the race observes disposal cleanly, never a torn state + Assert.True( + createException is null or ObjectDisposedException or SynthesisEngineBusyException, + $"Unexpected exception from a racing CreateSessionAsync: {createException}"); + } + } + + /// + /// Builds a substitute playback device reporting itself available with a realistic + /// format. + /// + private static IAudioPlaybackDevice CreateAvailablePlaybackDevice(int sampleRate = 16000, int channelCount = 1) + { + var device = Substitute.For(); + device.IsAvailable.Returns(true); + device.SampleRate.Returns(sampleRate); + device.ChannelCount.Returns(channelCount); + return device; + } +} diff --git a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerTests.cs b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerTests.cs deleted file mode 100644 index ae37623..0000000 --- a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSpeechSynthesizerTests.cs +++ /dev/null @@ -1,689 +0,0 @@ -using DemaConsulting.Speech.AudioSubsystem; -using DemaConsulting.Speech.Diagnostics; -using DemaConsulting.Speech.ModelManagementSubsystem; -using DemaConsulting.Speech.SynthesisSubsystem; -using DemaConsulting.Speech.Tests.ModelManagementSubsystem.Fakes; -using DemaConsulting.Speech.Tests.SynthesisSubsystem.Fakes; -using NSubstitute; - -namespace DemaConsulting.Speech.Tests.SynthesisSubsystem; - -/// -/// Unit tests for , exercising the full chunk → -/// synthesize → resample → play pipeline through a fake engine and a substitute playback -/// device, with no speakers and no native sherpa-onnx runtime. -/// -/// -/// Every test is deterministic without timing assumptions: the pipeline's bounded channel and -/// awaited tasks mean every produced segment is either fully consumed or the operation has -/// genuinely completed by the time an assertion runs. -/// -public class SherpaOnnxSpeechSynthesizerTests -{ - /// - /// Proves that plain text with no tags synthesizes to at least one segment carrying real - /// audio at the engine's declared sample rate. - /// - [Fact] - public async Task SynthesizeStreamAsync_PlainText_YieldsAudioSegment() - { - // Arrange - var engine = new FakeSynthesisEngine(sampleRate: 22050); - var playbackDevice = CreateAvailablePlaybackDevice(); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - - // Act - var segments = await CollectAsync( - synthesizer.SynthesizeStreamAsync("Hello world.", TestContext.Current.CancellationToken)); - - // Assert: one segment of real synthesized audio at the engine's rate - var segment = Assert.Single(segments); - Assert.NotEmpty(segment.Samples); - Assert.Equal(22050, segment.SampleRate); - Assert.Single(engine.GenerateCalls); - } - - /// - /// Proves that a pause tag produces a pure-silence segment with no engine call, ordered - /// between the flushed text segments surrounding it. - /// - [Fact] - public async Task SynthesizeStreamAsync_TextWithPauseTag_YieldsSilenceSegmentWithNoEngineCall() - { - // Arrange - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - - // Act - var segments = await CollectAsync( - synthesizer.SynthesizeStreamAsync("Hello [short pause] world.", TestContext.Current.CancellationToken)); - - // Assert: a pure-silence segment appears, and the engine was never asked to synthesize it - var silenceSegment = Assert.Single(segments, segment => segment.Samples.Count == 0); - Assert.True(silenceSegment.PostSilence > TimeSpan.Zero); - Assert.DoesNotContain(engine.GenerateCalls, call => call.Text.Length == 0); - } - - /// - /// Proves that a fault while synthesizing a segment is reported through diagnostics and - /// surfaces to the caller enumerating the stream, rather than hanging. - /// - [Fact] - public async Task SynthesizeStreamAsync_EngineThrows_ReportsFaultAndPropagatesToCaller() - { - // Arrange - var engine = new FakeSynthesisEngine(generateException: new InvalidOperationException("engine failed")); - var playbackDevice = CreateAvailablePlaybackDevice(); - var diagnostics = Substitute.For(); - using var synthesizer = new SherpaOnnxSpeechSynthesizer( - engine, playbackDevice, new FakeSynthesisModel(), diagnostics: diagnostics); - - // Act & Assert: enumerating the stream surfaces the engine's fault - await Assert.ThrowsAsync(async () => - { - await foreach (var _ in synthesizer.SynthesizeStreamAsync("Hello world.", TestContext.Current.CancellationToken)) - { - // Draining is enough to trigger the fault. - } - }); - - diagnostics.Received().Report( - SpeechDiagnosticLevel.Error, - "SynthesisSubsystem", - Arg.Is(message => message.Contains("could not be synthesized", StringComparison.Ordinal))); - } - - /// - /// Proves that playing a stream starts the playback device, writes silence and resampled - /// audio in order, and stops the device once every segment has played. - /// - [Fact] - public async Task PlayStreamAsync_OrderedSegments_StartsWritesInOrderAndStops() - { - // Arrange: a mono 16 kHz playback device and two pre-built segments (silence, then audio) - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(sampleRate: 16000, channelCount: 1); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - SynthesizedSpeech[] segments = - [ - new SynthesizedSpeech([], engine.SampleRate, TimeSpan.Zero, TimeSpan.FromMilliseconds(100)), - new SynthesizedSpeech([0.1f, 0.2f], engine.SampleRate, TimeSpan.Zero, TimeSpan.Zero) - ]; - - // Act - await synthesizer.PlayStreamAsync(ToAsyncEnumerable(segments), TestContext.Current.CancellationToken); - - // Assert: the device lifecycle and write order are exactly as expected - Received.InOrder(() => - { - playbackDevice.Start(); - playbackDevice.Write(Arg.Any>()); - playbackDevice.Write(Arg.Any>()); - playbackDevice.Stop(); - }); - } - - /// - /// Proves that a synthesizer supports multiple independent PlayStreamAsync - /// sessions on the same instance without reconstruction, so a host may construct one - /// synthesizer once and reuse it across many conversation turns for low-latency, repeated - /// synthesis, per 's "hot synthesis" reuse guidance. - /// - [Fact] - public async Task PlayStreamAsync_CalledTwiceOnSameInstance_ReusesSameInstanceWithoutReconstruction() - { - // Arrange: a single synthesizer instance over a mono 16 kHz playback device - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(sampleRate: 16000, channelCount: 1); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - SynthesizedSpeech[] segments = - [ - new SynthesizedSpeech([0.1f, 0.2f], engine.SampleRate, TimeSpan.Zero, TimeSpan.Zero) - ]; - - // Act: run two independent play sessions on the same instance - await synthesizer.PlayStreamAsync(ToAsyncEnumerable(segments), TestContext.Current.CancellationToken); - await synthesizer.PlayStreamAsync(ToAsyncEnumerable(segments), TestContext.Current.CancellationToken); - - // Assert: both sessions genuinely started and stopped the playback device, proving the - // synthesizer remains usable across repeated sessions without being disposed and recreated - playbackDevice.Received(2).Start(); - playbackDevice.Received(2).Stop(); - } - - /// - /// Proves that a playback device fault while writing propagates to the caller, and that - /// the device is still stopped in the guaranteeing finally block. - /// - [Fact] - public async Task PlayStreamAsync_PlaybackDeviceWriteThrows_PropagatesAndStillStopsDevice() - { - // Arrange - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(); - playbackDevice - .When(device => device.Write(Arg.Any>())) - .Do(_ => throw new AudioDeviceUnavailableException("speakers disconnected")); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - SynthesizedSpeech[] segments = [new SynthesizedSpeech([0.1f], engine.SampleRate, TimeSpan.Zero, TimeSpan.Zero)]; - - // Act & Assert: the fault propagates, but the device is still stopped - await Assert.ThrowsAsync( - () => synthesizer.PlayStreamAsync(ToAsyncEnumerable(segments), TestContext.Current.CancellationToken)); - playbackDevice.Received(1).Stop(); - } - - /// - /// Proves that a playback device that stops working (reported unavailable) surfaces - /// honestly rather than hanging or crashing. - /// - [Fact] - public async Task PlayStreamAsync_PlaybackDeviceUnavailable_ThrowsRatherThanHanging() - { - // Arrange: an unavailable playback device, whose Start() throws per its contract - var engine = new FakeSynthesisEngine(); - using var synthesizer = new SherpaOnnxSpeechSynthesizer( - engine, UnavailableAudioPlaybackDevice.Instance, new FakeSynthesisModel()); - SynthesizedSpeech[] segments = [new SynthesizedSpeech([0.1f], engine.SampleRate, TimeSpan.Zero, TimeSpan.Zero)]; - - // Act & Assert - await Assert.ThrowsAsync( - () => synthesizer.PlayStreamAsync(ToAsyncEnumerable(segments), TestContext.Current.CancellationToken)); - } - - /// - /// Proves that a playback device whose Start() itself throws still reaches the - /// finally block's Stop() teardown, rather than leaking whatever partial - /// resource the device acquired before faulting. - /// - [Fact] - public async Task PlayStreamAsync_PlaybackDeviceStartThrows_StillCallsStop() - { - // Arrange: a substitute device whose Start() throws, matching an unavailable-device fault - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(); - playbackDevice - .When(device => device.Start()) - .Do(_ => throw new AudioDeviceUnavailableException("speakers disconnected before start")); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - SynthesizedSpeech[] segments = [new SynthesizedSpeech([0.1f], engine.SampleRate, TimeSpan.Zero, TimeSpan.Zero)]; - - // Act & Assert: the Start() fault propagates, but the finally block still calls Stop() - await Assert.ThrowsAsync( - () => synthesizer.PlayStreamAsync(ToAsyncEnumerable(segments), TestContext.Current.CancellationToken)); - playbackDevice.Received(1).Stop(); - } - - /// - /// Proves that cancels an in-flight - /// session deterministically, ending its task - /// without hanging - and, critically, only after the producer's in-flight native-style - /// Generate call has genuinely returned, never orphaning it. This is the regression - /// coverage for the AccessViolationException crash: before the fix, - /// SynthesizeStreamCore could return control to its caller (and the caller could - /// then dispose the engine) while the producer task was still mid-way through a native - /// call on a background thread. - /// - /// - /// A Timeout is set as a safety net: if the fix ever regressed to orphaning the - /// producer task (or to hanging indefinitely), this test would fail fast with a timeout - /// rather than hanging the whole test run forever. - /// - [Fact(Timeout = 5000)] - public async Task Stop_WhileSpeaking_CancelsInFlightSessionOnlyAfterInFlightGenerateReturns() - { - // Arrange: an engine that signals it has started, then blocks until the test explicitly - // releases it - simulating a native call that keeps running for a little while after - // cancellation is requested, exactly like the real sherpa-onnx binding, which has no - // in-flight cancellation primitive of its own. - using var generateStarted = new SemaphoreSlim(0, 1); - using var generateRelease = new SemaphoreSlim(0, 1); - var engine = new BlockingSynthesisEngine(generateStarted, generateRelease, TestContext.Current.CancellationToken); - var playbackDevice = CreateAvailablePlaybackDevice(); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - - // Act: start speaking, wait until synthesis has begun, then stop - var speakTask = synthesizer.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); - await generateStarted.WaitAsync(TestContext.Current.CancellationToken); - synthesizer.Stop(); - - // Assert: the session must not be reported complete while the producer's Generate call - // is still in flight - this is exactly the window in which the caller previously could - // (and, per the bug report, did) dispose the engine out from under it. - Assert.False(speakTask.IsCompleted); - - // Act: only now let the in-flight native-style call finish, as the real engine eventually - // would on its own - generateRelease.Release(); - - // Assert: the session ends via cancellation rather than hanging or faulting some other way - await Assert.ThrowsAnyAsync(() => speakTask); - Assert.True(engine.GenerateReturned); - } - - /// - /// Proves that never - /// returns control to its caller while the producer's Generate call is still in - /// flight, even when cancellation - rather than - - /// is what ends the enumeration, and that disposing the engine only once the enumeration - /// has genuinely finished is safe (never touches a still-executing call). - /// - [Fact(Timeout = 5000)] - public async Task SynthesizeStreamAsync_CancelledMidGenerate_AwaitsProducerBeforeEnumerationCompletesAndDisposalIsSafe() - { - // Arrange - using var generateStarted = new SemaphoreSlim(0, 1); - using var generateRelease = new SemaphoreSlim(0, 1); - var engine = new BlockingSynthesisEngine(generateStarted, generateRelease, TestContext.Current.CancellationToken); - var playbackDevice = CreateAvailablePlaybackDevice(); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - using var cts = new CancellationTokenSource(); - - // Act: begin enumerating on a background task, cancel once Generate has started but - // before it returns - var enumerationTask = Task.Run(async () => - { - await foreach (var _ in synthesizer.SynthesizeStreamAsync("Hello world.", cts.Token)) - { - // Draining is enough; no segment is expected to arrive before cancellation. - } - }, TestContext.Current.CancellationToken); - await generateStarted.WaitAsync(TestContext.Current.CancellationToken); - await cts.CancelAsync(); - - // Assert: the enumeration must not complete while Generate is still in flight - the exact - // window in which the previously-orphaned producer task could race a caller's disposal - Assert.False(enumerationTask.IsCompleted); - - // Act: let the in-flight native-style call finish - generateRelease.Release(); - await Assert.ThrowsAnyAsync(() => enumerationTask); - - // Assert: Generate genuinely returned before the enumeration completed, and only one - // Generate call was ever made - disposing now must be safe, proving the engine was never - // touched again (and, in the real engine, never freed) while still executing - Assert.True(engine.GenerateReturned); - Assert.Equal(1, engine.GenerateCallCount); - var disposeException = Record.Exception(synthesizer.Dispose); - Assert.Null(disposeException); - } - - /// - /// Proves that abandoning enumeration of - /// for a reason other than the stream's own cancellationToken being - /// cancelled - exactly what happens when a consumer's await foreach body throws an - /// unrelated exception, such as observing a - /// playback device fault - still completes promptly instead of hanging. - /// - /// - /// This is the regression coverage for a hang introduced by the crash fix itself: the - /// producer writes into a bounded channel, which only unblocks a full write when - /// either the reader keeps draining or the write's token is cancelled. Before adding the - /// producer's own, always-owned cancellation (independent of the caller's token), disposing - /// the enumerator early - without the caller's token ever being cancelled - awaited the - /// producer task in the finally block while the producer sat blocked forever inside - /// a full channel's WriteAsync, since nothing was left to drain it. A - /// Timeout is set as a safety net so this test fails fast rather than hanging the - /// whole run if that regresses. - /// - [Fact(Timeout = 5000)] - public async Task SynthesizeStreamAsync_EnumerationAbandonedWithoutCancellation_DisposesPromptlyInsteadOfHanging() - { - // Arrange: enough sentences to exceed the producer's bounded look-ahead capacity of 8, so - // the producer is still blocked mid-stream, waiting for channel space that will never free up, - // by the time enumeration is abandoned below. - var manySentences = string.Concat(Enumerable.Range(1, 12).Select(i => $"Sentence number {i}. ")); - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - - // Act: start enumerating (this launches the background producer), consume exactly one - // segment to prove the pipeline is genuinely running, then abandon enumeration by - // disposing the enumerator directly - exactly what the compiler-generated `await foreach` - // cleanup does when a consumer's loop body throws for an unrelated reason - without ever - // cancelling the stream's own cancellationToken. - var enumerator = synthesizer - .SynthesizeStreamAsync(manySentences, TestContext.Current.CancellationToken) - .GetAsyncEnumerator(TestContext.Current.CancellationToken); - var hasFirstSegment = await enumerator.MoveNextAsync(); - Assert.True(hasFirstSegment); - - // Assert: disposal completes promptly (the [Fact(Timeout = ...)] above is the actual - // safety net; this await simply must not hang). - var disposeException = await Record.ExceptionAsync(async () => await enumerator.DisposeAsync()); - Assert.Null(disposeException); - } - - /// - /// Proves that calling with no session in flight is - /// a safe no-op. - /// - [Fact] - public void Stop_NoSessionInFlight_IsNoOp() - { - // Arrange - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - - // Act - var exception = Record.Exception(synthesizer.Stop); - - // Assert - Assert.Null(exception); - } - - /// - /// Proves that disposal is idempotent and disposes the owned engine exactly once, even - /// when called more than once. - /// - [Fact] - public void Dispose_CalledTwice_DisposesEngineOnce() - { - // Arrange - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(); - var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - - // Act - synthesizer.Dispose(); - synthesizer.Dispose(); - - // Assert - Assert.Equal(1, engine.DisposeCallCount); - } - - /// - /// Proves that operating on a disposed synthesizer throws - /// rather than silently doing nothing. - /// - [Fact] - public void SynthesizeStreamAsync_AfterDispose_ThrowsObjectDisposedException() - { - // Arrange - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(); - var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - synthesizer.Dispose(); - - // Act & Assert - Assert.Throws( - () => synthesizer.SynthesizeStreamAsync("hello", TestContext.Current.CancellationToken)); - } - - /// - /// Proves that always reports - /// , since this type is only ever constructed after successful - /// composition. - /// - [Fact] - public void IsAvailable_Always_ReturnsTrue() - { - // Arrange - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - - // Act & Assert - Assert.True(synthesizer.IsAvailable); - } - - /// - /// Proves that genuinely waits for the - /// playback device to report a drained queue before stopping it, rather than treating - /// "every segment enqueued" as "finished playing" - the root cause of the TTS panel - /// cutting audio off almost instantly. Uses a controllable fake (an NSubstitute stub whose - /// PendingSampleCount getter signals a semaphore on every read) so the assertion - /// that the task has not yet completed is driven by an observed poll, not by an arbitrary - /// sleep racing the implementation. - /// - [Fact] - public async Task PlayStreamAsync_PlaybackDeviceReportsPendingSamples_WaitsForDrainBeforeStopping() - { - // Arrange: a playback device that reports samples still pending until the test releases it - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(); - var pendingSampleCount = 1; - using var polled = new SemaphoreSlim(0); - playbackDevice.PendingSampleCount.Returns(_ => - { - var value = Volatile.Read(ref pendingSampleCount); - polled.Release(); - return value; - }); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - SynthesizedSpeech[] segments = [new SynthesizedSpeech([0.1f], engine.SampleRate, TimeSpan.Zero, TimeSpan.Zero)]; - - // Act: start playback; every segment is enqueued almost immediately, but the fake device - // keeps reporting pending samples until the test signals otherwise - var playTask = synthesizer.PlayStreamAsync(ToAsyncEnumerable(segments), TestContext.Current.CancellationToken); - - // Wait until the drain wait has genuinely begun polling the device at least once - await polled.WaitAsync(TestContext.Current.CancellationToken); - - // Assert: playback has not been treated as finished while samples are still reported pending - Assert.False(playTask.IsCompleted); - playbackDevice.DidNotReceive().Stop(); - - // Act: signal the fake device has now genuinely drained - Volatile.Write(ref pendingSampleCount, 0); - await playTask; - - // Assert: only once the device reports a drained queue does playback stop - playbackDevice.Received(1).Stop(); - } - - /// - /// Proves that SherpaOnnxSpeechSynthesizer.GenerateSegment (exercised - /// through ) resolves the speaker id - /// passed to ISynthesisEngine.Generate from - /// rather than a hard-coded 0, for - /// both a default (no parameter bag supplied) case and a non-default (bag supplied) case. - /// - [Fact] - public async Task SynthesizeStreamAsync_NoParameterValues_ResolvesDefaultSpeakerIdFromModel() - { - // Arrange: a model whose ResolveSpeakerId hook always returns a distinctive non-zero id - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(); - var model = new FakeSynthesisModel(resolveSpeakerId: _ => 7); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, model); - - // Act - await CollectAsync(synthesizer.SynthesizeStreamAsync("Hello world.", TestContext.Current.CancellationToken)); - - // Assert - var call = Assert.Single(engine.GenerateCalls); - Assert.Equal(7, call.SpeakerId); - } - - /// - /// Proves that a supplied parameter value bag reaches - /// and, through it, the speaker id passed - /// to . - /// - [Fact] - public async Task SynthesizeStreamAsync_ParameterValuesSupplied_ResolvesSpeakerIdFromBag() - { - // Arrange: a model whose ResolveSpeakerId hook echoes back a bag value - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(); - var model = new FakeSynthesisModel( - resolveSpeakerId: values => values is not null && values.TryGetValue("voice", out var value) && value is int id ? id : 0); - IReadOnlyDictionary parameterValues = new Dictionary { ["voice"] = 3 }; - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, model, parameterValues); - - // Act - await CollectAsync(synthesizer.SynthesizeStreamAsync("Hello world.", TestContext.Current.CancellationToken)); - - // Assert - var call = Assert.Single(engine.GenerateCalls); - Assert.Equal(3, call.SpeakerId); - } - - /// - /// Proves that a segment's own Natural Language Audio Tag speed/volume overrides are - /// unaffected by, and coexist correctly with, a supplied parameter value bag driving - /// speaker-id resolution - the two mechanisms remain independent. - /// - [Fact] - public async Task SynthesizeStreamAsync_ParameterValuesSuppliedAlongsideSpeedTag_BothMechanismsApplyIndependently() - { - // Arrange - var engine = new FakeSynthesisEngine(); - var playbackDevice = CreateAvailablePlaybackDevice(); - var model = new FakeSynthesisModel(resolveSpeakerId: _ => 5); - IReadOnlyDictionary parameterValues = new Dictionary { ["voice"] = 5 }; - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, model, parameterValues); - - // Act: "fast" is a natural language audio tag mapped to the model's own "tempo" parameter - await CollectAsync(synthesizer.SynthesizeStreamAsync("[fast] Hello world.", TestContext.Current.CancellationToken)); - - // Assert: the speaker id still resolves from the bag, and the segment's speed differs from 1.0 - var call = Assert.Single(engine.GenerateCalls); - Assert.Equal(5, call.SpeakerId); - Assert.NotEqual(1.0f, call.Speed); - } - - /// - /// Proves that a long, multi-sentence input still produces every segment, in the exact - /// order the sentences appear in the source text, and that - /// is never called concurrently with itself - the - /// pipeline must remain strictly sequential regardless of how many chunks the input - /// produces. This test's consumer () drains the stream as fast - /// as segments are produced, so it does not exercise or distinguish the pending-segment - /// channel's specific capacity (currently 8); it verifies ordering and single-threaded - /// production hold at a scale well beyond a single chunk. - /// - [Fact] - public async Task SynthesizeStreamAsync_LongMultiSentenceInput_ProducesOrderedSegmentsSequentially() - { - // Arrange: 12 short, uniquely numbered sentences, so ordering across many chunks can be - // cross-checked against the source text. - const int sentenceCount = 12; - var sentences = Enumerable.Range(1, sentenceCount).Select(i => $"Sentence number {i}."); - var text = string.Join(" ", sentences); - var engine = new FakeSynthesisEngine(simulatedGenerateDelay: TimeSpan.FromMilliseconds(5)); - var playbackDevice = CreateAvailablePlaybackDevice(); - using var synthesizer = new SherpaOnnxSpeechSynthesizer(engine, playbackDevice, new FakeSynthesisModel()); - - // Act - var segments = await CollectAsync( - synthesizer.SynthesizeStreamAsync(text, TestContext.Current.CancellationToken)); - - // Assert: every sentence was synthesized exactly once, in source order, and every - // resulting segment reflects that same order (audio length is proportional to text - // length in the fake, so segment order can be cross-checked against call order) - Assert.Equal(sentenceCount, engine.GenerateCalls.Count); - for (var i = 0; i < sentenceCount; i++) - { - Assert.Contains($"number {i + 1}", engine.GenerateCalls[i].Text, StringComparison.Ordinal); - } - - Assert.Equal(sentenceCount, segments.Count); - for (var i = 0; i < sentenceCount; i++) - { - Assert.Equal(engine.GenerateCalls[i].Text.Length * 4, segments[i].Samples.Count); - } - - // Assert: Generate was never entered concurrently with itself - Assert.Equal(1, engine.MaxConcurrentGenerateCalls); - } - - /// - /// Builds a substitute playback device reporting itself available with the given format. - /// - private static IAudioPlaybackDevice CreateAvailablePlaybackDevice(int sampleRate = 16000, int channelCount = 1) - { - var device = Substitute.For(); - device.IsAvailable.Returns(true); - device.SampleRate.Returns(sampleRate); - device.ChannelCount.Returns(channelCount); - return device; - } - - /// Materializes an asynchronous sequence into a list. - private static async Task> CollectAsync(IAsyncEnumerable stream) - { - var results = new List(); - await foreach (var item in stream) - { - results.Add(item); - } - - return results; - } - - /// Wraps an array of pre-built segments as an asynchronous sequence. - private static async IAsyncEnumerable ToAsyncEnumerable(IEnumerable segments) - { - foreach (var segment in segments) - { - yield return segment; - } - - await Task.CompletedTask; - } - - /// - /// Test-only whose signals a - /// semaphore once called, then blocks until the test explicitly releases a second - /// semaphore - simulating a native call that keeps running for a while after the - /// session's cancellation token is cancelled, exactly like the real sherpa-onnx binding, - /// which has no in-flight cancellation primitive of its own and so cannot be interrupted - /// by or an externally cancelled token. - /// - /// - /// Deliberately does not unblock from : - /// the whole point of the regression this fake supports is that - /// must never call (or let - /// a caller do so) while this call is still in flight, so a test relying on - /// to unblock it would either mask that exact bug or deadlock - /// against the fix. Callers must release explicitly to - /// let the in-flight call complete. However, still waits against - /// , so a failed or timed-out test unblocks the - /// background producer thread instead of leaving it blocked for the rest of the test run. - /// - /// Released once is called. - /// Awaited by before it returns. - /// - /// The owning test's own cancellation/timeout token (for example, - /// ), observed only as a failure-mode safety - /// net so the wait never outlives the test, not as part of the behavior under test. - /// - private sealed class BlockingSynthesisEngine( - SemaphoreSlim generateStarted, - SemaphoreSlim generateRelease, - CancellationToken testCancellationToken) : ISynthesisEngine - { - public int SampleRate => 16000; - - /// Gets the number of calls this engine has received. - public int GenerateCallCount { get; private set; } - - /// Gets a value indicating whether the in-flight call has genuinely returned. - public bool GenerateReturned { get; private set; } - - public EngineAudio Generate(string text, float speed, int speakerId) - { - GenerateCallCount++; - generateStarted.Release(); - - // Blocks until the test explicitly allows this in-flight call to complete, rather - // than observing any cancellation token that the synthesizer under test might use - - // matching the real native call this fake stands in for. The test's own - // cancellation/timeout token is still observed here purely as a safety net so a - // failed or timed-out test cannot leave this background thread blocked forever. - generateRelease.Wait(testCancellationToken); - - GenerateReturned = true; - return new EngineAudio([], SampleRate); - } - - public void Dispose() - { - // Intentionally does not unblock Generate(): see remarks above. - } - } -} diff --git a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSynthesisSessionTests.cs b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSynthesisSessionTests.cs new file mode 100644 index 0000000..53fd7d7 --- /dev/null +++ b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SherpaOnnxSynthesisSessionTests.cs @@ -0,0 +1,1222 @@ +using DemaConsulting.Speech.AudioSubsystem; +using DemaConsulting.Speech.Diagnostics; +using DemaConsulting.Speech.ModelManagementSubsystem; +using DemaConsulting.Speech.SynthesisSubsystem; +using DemaConsulting.Speech.Tests.ModelManagementSubsystem.Fakes; +using DemaConsulting.Speech.Tests.SynthesisSubsystem.Fakes; +using NSubstitute; + +namespace DemaConsulting.Speech.Tests.SynthesisSubsystem; + +/// +/// Unit tests for , exercising the full chunk → +/// synthesize → resample → play pipeline through a fake backend and a substitute playback +/// device, with no speakers and no native sherpa-onnx runtime, plus the state machine and +/// overlap-rule behavior unique to the session abstraction. +/// +/// +/// Every test is deterministic without timing assumptions: segments are synthesized and +/// played strictly in order (one at a time), so every assertion runs only once the awaited +/// call has genuinely completed. +/// +public class SherpaOnnxSynthesisSessionTests +{ + /// + /// Proves that plain text with no tags synthesizes to at least one segment carrying real + /// audio at the backend's declared sample rate. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SynthesizeAsync_PlainText_YieldsAudioSegment() + { + // Arrange + var backend = new FakeSynthesisEngine(sampleRate: 22050); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + + // Act + var segments = await session.SynthesizeAsync("Hello world.", TestContext.Current.CancellationToken); + + // Assert: one segment of real synthesized audio at the backend's rate + var segment = Assert.Single(segments); + Assert.NotEmpty(segment.Samples); + Assert.Equal(22050, segment.SampleRate); + Assert.Single(backend.GenerateCalls); + } + + /// + /// Proves that returns the full-fidelity + /// ordered segment list, including a pure-silence segment for a pause tag, with no + /// backend call for the silent segment. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SynthesizeAsync_ReturnsFullFidelitySegmentListIncludingSilence() + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + + // Act + var segments = await session.SynthesizeAsync("Hello [short pause] world.", TestContext.Current.CancellationToken); + + // Assert: a pure-silence segment appears, and the backend was never asked to synthesize it + var silenceSegment = Assert.Single(segments, segment => segment.Samples.Count == 0); + Assert.True(silenceSegment.PostSilence > TimeSpan.Zero); + Assert.DoesNotContain(backend.GenerateCalls, call => call.Text.Length == 0); + } + + /// + /// Proves that a fault while synthesizing a segment is reported through diagnostics, + /// propagates to the caller, and transitions the session to . + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SynthesizeAsync_BackendThrows_ReportsFaultAndTransitionsToFaulted() + { + // Arrange + var backend = new FakeSynthesisEngine(generateException: new InvalidOperationException("backend failed")); + var device = CreateAvailablePlaybackDevice(); + var diagnostics = Substitute.For(); + await using var session = CreateSession(backend, device, diagnostics: diagnostics); + + // Act & Assert: the call surfaces the backend's fault + await Assert.ThrowsAsync( + () => session.SynthesizeAsync("Hello world.", TestContext.Current.CancellationToken)); + + diagnostics.Received().Report( + SpeechDiagnosticLevel.Error, + "SynthesisSubsystem", + Arg.Is(message => message.Contains("faulted", StringComparison.Ordinal))); + Assert.Equal(SynthesisSessionState.Faulted, session.State); + } + + /// + /// Proves that a faulted session rejects any further operation with + /// , rather than attempting to synthesize + /// again. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SynthesizeAsync_AfterFault_ThrowsSynthesisSessionFaultedException() + { + // Arrange: fault the session with one failing call + var backend = new FakeSynthesisEngine(generateException: new InvalidOperationException("backend failed")); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + await Assert.ThrowsAsync( + () => session.SynthesizeAsync("Hello world.", TestContext.Current.CancellationToken)); + + // Act & Assert: a subsequent call fails fast with the faulted-session exception + await Assert.ThrowsAsync( + () => session.SynthesizeAsync("Hello again.", TestContext.Current.CancellationToken)); + } + + /// + /// Proves that an thrown by the backend itself - + /// with no cancellation ever requested by the caller or by StopAsync - is treated + /// as a genuine, unrequested fault: the session transitions to + /// (and rejects further operations), rather + /// than being misclassified as ordinary cancellation and silently allowed back to + /// for reuse. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SynthesizeAsync_BackendThrowsUnrequestedOperationCanceledException_TransitionsToFaultedNotStopped() + { + // Arrange: a backend whose Generate throws OperationCanceledException on its own, + // independent of any cancellation token this session ever observes + var backend = new FakeSynthesisEngine(generateException: new OperationCanceledException("backend cancelled unexpectedly")); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + + // Act: synthesize with no cancellation ever requested + await Assert.ThrowsAsync( + () => session.SynthesizeAsync("Hello world.", TestContext.Current.CancellationToken)); + + // Assert: the session is genuinely Faulted - not Stopped - and rejects reuse + Assert.Equal(SynthesisSessionState.Faulted, session.State); + await Assert.ThrowsAsync( + () => session.SynthesizeAsync("Hello again.", TestContext.Current.CancellationToken)); + } + + /// + /// Proves that starts the playback device, + /// writes resampled audio, and stops the device once synthesis and playback have + /// completed. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SpeakAsync_PlainText_StartsWritesAndStopsDevice() + { + // Arrange: a mono 16 kHz playback device + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(sampleRate: 16000, channelCount: 1); + await using var session = CreateSession(backend, device); + + // Act + await session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + + // Assert: the device lifecycle and write order are exactly as expected + Received.InOrder(() => + { + device.Start(); + device.Write(Arg.Any>()); + device.Stop(); + }); + } + + /// + /// Proves that a session supports multiple independent + /// calls on the same instance without reconstruction, so a host may construct one session + /// per model/device combination and reuse it across many conversation turns for + /// low-latency, repeated synthesis. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SpeakAsync_CalledTwiceOnSameInstance_ReusesSameInstanceWithoutReconstruction() + { + // Arrange: a single session instance over a mono 16 kHz playback device + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(sampleRate: 16000, channelCount: 1); + await using var session = CreateSession(backend, device); + + // Act: run two independent speak calls on the same instance + await session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + await session.SpeakAsync("Hello again.", TestContext.Current.CancellationToken); + + // Assert: both calls genuinely started and stopped the playback device, proving the + // session remains usable across repeated calls without being disposed and recreated + device.Received(2).Start(); + device.Received(2).Stop(); + Assert.Equal(SynthesisSessionState.Stopped, session.State); + } + + /// + /// Proves that a playback device fault while writing propagates to the caller, and that + /// the device is still stopped in the guaranteeing finally block. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SpeakAsync_PlaybackDeviceWriteThrows_PropagatesAndStillStopsDevice() + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + device + .When(d => d.Write(Arg.Any>())) + .Do(_ => throw new AudioDeviceUnavailableException("speakers disconnected")); + await using var session = CreateSession(backend, device); + + // Act & Assert: the fault propagates, but the device is still stopped + await Assert.ThrowsAsync( + () => session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken)); + device.Received(1).Stop(); + } + + /// + /// Proves that a playback device that stops working (reported unavailable) surfaces + /// honestly rather than hanging or crashing. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SpeakAsync_PlaybackDeviceUnavailable_ThrowsRatherThanHanging() + { + // Arrange: an unavailable playback device, whose Start() throws per its contract + var backend = new FakeSynthesisEngine(); + await using var session = CreateSession(backend, UnavailableAudioPlaybackDevice.Instance); + + // Act & Assert + await Assert.ThrowsAsync( + () => session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken)); + } + + /// + /// Proves that a playback device whose Start() itself throws still reaches the + /// finally block's Stop() teardown, rather than leaking whatever partial + /// resource the device acquired before faulting. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SpeakAsync_PlaybackDeviceStartThrows_StillCallsStop() + { + // Arrange: a substitute device whose Start() throws, matching an unavailable-device fault + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + device + .When(d => d.Start()) + .Do(_ => throw new AudioDeviceUnavailableException("speakers disconnected before start")); + await using var session = CreateSession(backend, device); + + // Act & Assert: the Start() fault propagates, but the finally block still calls Stop() + await Assert.ThrowsAsync( + () => session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken)); + device.Received(1).Stop(); + } + + /// + /// Proves that returns control to its caller + /// well before a slow, synchronous call returns - + /// closing the review finding that the native playback device-start call used to run + /// inline on the calling thread, before this already-async method's first await, blocking + /// the caller (for example a UI thread) for however long the native call took. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SpeakAsync_SlowPlaybackDeviceStart_DoesNotBlockCaller() + { + // Arrange: a device whose Start() blocks until this test explicitly releases it + using var startEntered = new ManualResetEventSlim(false); + using var startRelease = new ManualResetEventSlim(false); + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + device.When(d => d.Start()).Do(_ => + { + startEntered.Set(); + startRelease.Wait(TestContext.Current.CancellationToken); + }); + await using var session = CreateSession(backend, device); + + // Act: call SpeakAsync and prove it returns control to this thread - without blocking - + // well before the device's own blocking Start() call has returned + var speakTask = session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + var enteredInTime = startEntered.Wait(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + + // Assert: the device call was genuinely entered, but SpeakAsync has not yet completed and + // this thread was never blocked waiting for it + Assert.True(enteredInTime, "The playback device's Start() was never entered."); + Assert.False(speakTask.IsCompleted); + + // Act: release the blocked device call and let the speak call genuinely finish + startRelease.Set(); + await speakTask.WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + + // Assert: the device was started and stopped exactly once each + device.Received(1).Start(); + device.Received(1).Stop(); + } + + /// + /// Proves that fulfils its documented contract + /// of completing only once the in-flight operation has genuinely stopped: its returned + /// task does not complete while the backend's native-style Generate call is still + /// in flight, so a caller that awaits StopAsync can safely assume the device/backend + /// is quiescent the instant it returns - never orphaning the in-flight call. + /// + /// + /// A Timeout is set as a safety net: if quiescent-awaiting ever regressed to + /// orphaning the worker (or to hanging indefinitely), this test would fail fast with a + /// timeout rather than hanging the whole test run forever. + /// + [Fact(Timeout = 5000)] + public async Task SherpaOnnxSynthesisSession_StopAsync_WhileSpeaking_DoesNotCompleteUntilInFlightGenerateReturns() + { + // Arrange: a backend that signals it has started, then blocks until the test explicitly + // releases it - simulating a native call that keeps running for a little while after + // cancellation is requested, exactly like the real sherpa-onnx binding, which has no + // in-flight cancellation primitive of its own. + using var generateStarted = new SemaphoreSlim(0, 1); + using var generateRelease = new SemaphoreSlim(0, 1); + var backend = new BlockingSynthesisEngine(generateStarted, generateRelease, TestContext.Current.CancellationToken); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + + // Act: start speaking, wait until synthesis has begun, then stop - without awaiting yet + var speakTask = session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + await generateStarted.WaitAsync(TestContext.Current.CancellationToken); + var stopTask = session.StopAsync(TestContext.Current.CancellationToken); + + // Assert: StopAsync's own task must not be reported complete while the backend's + // Generate call is still in flight - this is exactly the window in which a caller + // previously could (and, per the bug report this design supersedes, did) wrongly believe + // the device/backend was quiescent and dispose the engine out from under it. + await Task.Delay(TimeSpan.FromMilliseconds(50), TestContext.Current.CancellationToken); + Assert.False(stopTask.IsCompleted); + Assert.False(speakTask.IsCompleted); + + // Act: only now let the in-flight native-style call finish, as the real backend + // eventually would on its own + generateRelease.Release(); + + // Assert: StopAsync itself completes only once the operation has truly stopped, and the + // operation ends via cancellation rather than hanging or faulting some other way + await stopTask.WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + await Assert.ThrowsAnyAsync(() => speakTask); + Assert.True(backend.GenerateReturned); + Assert.Equal(SynthesisSessionState.Stopped, session.State); + } + + /// + /// Proves that calling with no operation in + /// flight is a safe no-op. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_StopAsync_NoOperationInFlight_IsNoOp() + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + + // Act + var exception = await Record.ExceptionAsync(() => session.StopAsync(TestContext.Current.CancellationToken)); + + // Assert + Assert.Null(exception); + } + + /// + /// Proves that a call racing against the + /// in-flight operation's own completion never surfaces an unhandled + /// - closing the check-then-act race window in which + /// StopAsync/DisposeAsync read the operation's cancellation source outside + /// the lock and call CancelAsync on it just as the operation's own finally + /// block disposes that same source. + /// + /// + /// Repeated many times with a backend that completes near-instantly to maximize the + /// chance of landing inside the narrow race window if the fix ever regressed. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_StopAsync_RacingOperationCompletion_NeverThrowsObjectDisposedException() + { + for (var i = 0; i < 200; i++) + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + + // Act: fire the operation and race a Stop against its completion with no synchronization + var speakTask = session.SpeakAsync("Hi.", TestContext.Current.CancellationToken); + var stopTask = session.StopAsync(TestContext.Current.CancellationToken); + + var speakException = await Record.ExceptionAsync(() => speakTask); + var stopException = await Record.ExceptionAsync(() => stopTask); + + // Assert: neither task ever surfaces an ObjectDisposedException from the race + Assert.IsNotType(speakException); + Assert.IsNotType(stopException); + } + } + + /// + /// Proves that calling while another + /// call is already in flight on the same + /// session throws rather than producing undefined + /// interleaving. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SpeakAsync_CalledWhileAlreadySpeaking_ThrowsInvalidOperationException() + { + // Arrange: a backend that blocks the first call in flight + using var generateStarted = new SemaphoreSlim(0, 1); + using var generateRelease = new SemaphoreSlim(0, 1); + var backend = new BlockingSynthesisEngine(generateStarted, generateRelease, TestContext.Current.CancellationToken); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + var firstCall = session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + await generateStarted.WaitAsync(TestContext.Current.CancellationToken); + + // Act & Assert: a second, overlapping call is rejected + await Assert.ThrowsAsync( + () => session.SpeakAsync("Hello again.", TestContext.Current.CancellationToken)); + + // Cleanup: release the first call so it completes and the session can be disposed cleanly + generateRelease.Release(); + await firstCall; + } + + /// + /// Proves that calling while a + /// call is already in flight on the same + /// session is also rejected, proving the overlap rule applies across both operations, not + /// just between two calls of the same kind. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SynthesizeAsync_CalledWhileAlreadySpeaking_ThrowsInvalidOperationException() + { + // Arrange + using var generateStarted = new SemaphoreSlim(0, 1); + using var generateRelease = new SemaphoreSlim(0, 1); + var backend = new BlockingSynthesisEngine(generateStarted, generateRelease, TestContext.Current.CancellationToken); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + var firstCall = session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + await generateStarted.WaitAsync(TestContext.Current.CancellationToken); + + // Act & Assert + await Assert.ThrowsAsync( + () => session.SynthesizeAsync("Hello again.", TestContext.Current.CancellationToken)); + + // Cleanup + generateRelease.Release(); + await firstCall; + } + + /// + /// Proves that an overlapping call rejected with + /// never overwrites the session's tracked in-flight operation: a concurrent + /// issued immediately after the rejection + /// still awaits the original, still-in-flight operation (and its raw native completion) + /// before disposing the backend/releasing the lease, rather than observing only the + /// rejected call's already-completed (faulted) task and disposing the backend out from + /// under the original call (finding 20). + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_DisposeAsync_AfterRejectedOverlappingCall_StillAwaitsOriginalOperation() + { + // Arrange: a backend that blocks the first (accepted) call in flight + using var generateStarted = new SemaphoreSlim(0, 1); + using var generateRelease = new SemaphoreSlim(0, 1); + var backend = new BlockingSynthesisEngine(generateStarted, generateRelease, TestContext.Current.CancellationToken); + var device = CreateAvailablePlaybackDevice(); + var session = CreateSession(backend, device); + var firstCall = session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + await generateStarted.WaitAsync(TestContext.Current.CancellationToken); + + // Act: a second, overlapping call is rejected without disturbing the first call's + // tracked operation, then DisposeAsync is issued while the first call is still blocked + // inside native Generate + await Assert.ThrowsAsync( + () => session.SynthesizeAsync("Hello again.", TestContext.Current.CancellationToken)); + var disposeTask = session.DisposeAsync().AsTask(); + + // Assert: DisposeAsync does not complete while the original call is still blocked - + // proving it is awaiting the original operation, not the rejected call's faulted task + await Task.Delay(50, TestContext.Current.CancellationToken); + Assert.False(disposeTask.IsCompleted); + Assert.False(backend.GenerateReturned); + + // Release the original call so both it and disposal can complete + generateRelease.Release(); + await Record.ExceptionAsync(() => firstCall); + await disposeTask.WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + + Assert.True(backend.GenerateReturned); + } + + /// + /// Proves that raises every expected + /// transition, in order, for one successful + /// call: Created → Starting → Running → Stopping → Stopped. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_StateChanged_OneSuccessfulOperation_RaisesExpectedTransitionsInOrder() + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + List observedStates = []; + session.StateChanged += (_, args) => observedStates.Add(args.Current); + + // Act + await session.SynthesizeAsync("Hello world.", TestContext.Current.CancellationToken); + + // Assert + Assert.Equal( + [ + SynthesisSessionState.Starting, + SynthesisSessionState.Running, + SynthesisSessionState.Stopping, + SynthesisSessionState.Stopped + ], + observedStates); + } + + /// + /// Proves that a handler that throws does not + /// propagate out of the session and does not destabilize its own lifecycle, mirroring this + /// library's event-exception-isolation convention. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_StateChanged_HandlerThrows_IsIsolatedAndDoesNotPropagate() + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + var diagnostics = Substitute.For(); + await using var session = CreateSession(backend, device, diagnostics: diagnostics); + session.StateChanged += (_, _) => throw new InvalidOperationException("handler faulted"); + + // Act + var exception = await Record.ExceptionAsync( + () => session.SynthesizeAsync("Hello world.", TestContext.Current.CancellationToken)); + + // Assert: the operation itself still completed successfully + Assert.Null(exception); + diagnostics.Received().Report( + SpeechDiagnosticLevel.Warning, + "SynthesisSubsystem", + Arg.Is(message => message.Contains("StateChanged", StringComparison.Ordinal))); + } + + /// + /// Proves that disposal is idempotent and releases the engine's exclusivity lease exactly + /// once, even when called more than once. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_DisposeAsync_CalledTwice_ReleasesLeaseOnce() + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + var releaseCount = 0; + var session = new SherpaOnnxSynthesisSession( + backend, device, new FakeSynthesisModel(), null, NullSpeechDiagnostics.Instance, () => releaseCount++); + + // Act + await session.DisposeAsync(); + await session.DisposeAsync(); + + // Assert + Assert.Equal(1, releaseCount); + Assert.Equal(SynthesisSessionState.Disposed, session.State); + } + + /// + /// Proves that DisposeAsync does not release the engine's exclusivity lease while + /// an in-flight operation's backend call is still demonstrably running - closing the race + /// in which a prior implementation released the lease the instant cancellation was + /// requested, before the operation had genuinely stopped. + /// + /// + /// A Timeout is set as a safety net: if disposal ever regressed to not awaiting the + /// in-flight operation at all, this test would still pass trivially, but if it regressed to + /// awaiting indefinitely (ignoring the abandon policy), this would fail fast instead of + /// hanging the whole test run forever. + /// + [Fact(Timeout = 5000)] + public async Task SherpaOnnxSynthesisSession_DisposeAsync_WhileSpeaking_DoesNotReleaseLeaseBeforeOperationSettles() + { + // Arrange: a backend that blocks inside Generate until explicitly released, simulating a + // native call still genuinely in flight when disposal is requested. + using var generateStarted = new SemaphoreSlim(0, 1); + using var generateRelease = new SemaphoreSlim(0, 1); + var backend = new BlockingSynthesisEngine(generateStarted, generateRelease, TestContext.Current.CancellationToken); + var device = CreateAvailablePlaybackDevice(); + var releaseCount = 0; + var session = new SherpaOnnxSynthesisSession( + backend, device, new FakeSynthesisModel(), null, NullSpeechDiagnostics.Instance, () => releaseCount++); + + // Act: start speaking, wait until the backend call has begun, then start disposing + var speakTask = session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + await generateStarted.WaitAsync(TestContext.Current.CancellationToken); + var disposeTask = session.DisposeAsync().AsTask(); + + // Let the in-flight backend call finish so neither task hangs for the rest of the test + generateRelease.Release(); + await disposeTask; + + // Assert: by the time DisposeAsync returned, the backend call had genuinely finished and + // the operation had settled (whichever way it settled), so the lease was never released + // while the operation was still running + Assert.True(backend.GenerateReturned); + Assert.True(speakTask.IsCompleted); + Assert.Equal(1, releaseCount); + + // Observe the task's outcome so it is never reported as an unobserved exception + await Record.ExceptionAsync(() => speakTask); + } + + /// + /// Proves that starting an operation and recording it so DisposeAsync can await it + /// are atomic: racing against + /// with no synchronization between them never + /// lets disposal observe "no operation registered yet" and proceed to dispose the backend + /// while is already (or about to be) running - + /// which the fake backend used here would otherwise surface as a leaked + /// out of the in-flight operation. + /// + /// + /// Mirrors + /// but races DisposeAsync - which actually disposes the shared backend - instead of + /// StopAsync, which never does. Run many iterations with no synchronization to + /// maximize the chance of landing inside the narrow registration race window if the fix + /// ever regressed. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_DisposeAsync_RacingSpeakAsync_NeverDisposesBackendWhileOperationInFlight() + { + for (var i = 0; i < 200; i++) + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + var session = CreateSession(backend, device); + + // Act: fire the operation and race a Dispose against its completion with no synchronization + var speakTask = session.SpeakAsync("Hi.", TestContext.Current.CancellationToken); + var disposeTask = session.DisposeAsync().AsTask(); + + var speakException = await Record.ExceptionAsync(() => speakTask); + await disposeTask; + + // Assert: the operation never surfaces an ObjectDisposedException from the backend + // having been disposed while Generate was still (or about to be) in flight + Assert.IsNotType(speakException); + } + } + + /// + /// Proves that two concurrent calls share the + /// exact same in-flight teardown, rather than the second call returning the instant the + /// first merely begins - both complete only once the real teardown (and the lease release + /// it gates) is genuinely done. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_DisposeAsync_CalledConcurrentlyTwice_BothCompleteAfterSingleTeardown() + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + var releaseCount = 0; + var session = new SherpaOnnxSynthesisSession( + backend, device, new FakeSynthesisModel(), null, NullSpeechDiagnostics.Instance, () => releaseCount++); + + // Act: dispose concurrently from two callers + async Task DisposeOnceAsync() => await session.DisposeAsync(); + var first = DisposeOnceAsync(); + var second = DisposeOnceAsync(); + await Task.WhenAll(first, second).WaitAsync(TimeSpan.FromSeconds(10), TestContext.Current.CancellationToken); + + // Assert: both callers converged, and the lease was released exactly once + Assert.Equal(SynthesisSessionState.Disposed, session.State); + Assert.Equal(1, releaseCount); + } + + /// + /// Proves that the engine's exclusivity lease is not released until the dedicated + /// operation worker has genuinely exited - not merely been abandoned after its timeout - + /// so a new session (or engine disposal) can never touch or dispose the shared backend + /// while an abandoned worker is still inside a blocking native Generate call. + /// + [Fact(Timeout = 10000)] + public async Task SherpaOnnxSynthesisSession_DisposeAsync_AbandonedWorker_DoesNotReleaseLeaseUntilWorkerExits() + { + // Arrange: a backend whose Generate blocks forever (until this test releases it) + using var generateStarted = new SemaphoreSlim(0, 1); + using var generateRelease = new SemaphoreSlim(0, 1); + var backend = new BlockingSynthesisEngine(generateStarted, generateRelease, TestContext.Current.CancellationToken); + var device = CreateAvailablePlaybackDevice(); + var releaseCount = 0; + var session = new SherpaOnnxSynthesisSession( + backend, device, new FakeSynthesisModel(), null, NullSpeechDiagnostics.Instance, () => releaseCount++); + + // Act: start speaking, wait until the backend call has begun, then begin disposing + // directly (DisposeAsync requests its own cancellation and - per finding 27 - must wait + // for the native call's genuine raw completion, not just the abandon-aware operation + // task, before releasing the lease) + var speakTask = session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + await generateStarted.WaitAsync(TestContext.Current.CancellationToken); + var disposeTask = session.DisposeAsync().AsTask(); + + // Assert: the lease must not be released while the abandoned worker is still stuck inside + // the backend's blocking Generate call + await Task.Delay(TimeSpan.FromMilliseconds(100), TestContext.Current.CancellationToken); + Assert.Equal(0, releaseCount); + Assert.False(disposeTask.IsCompleted); + + // Act: release the abandoned background call so it can genuinely finish + generateRelease.Release(); + await disposeTask.WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + + // Assert: only once the worker genuinely exited was the lease released + Assert.True(backend.GenerateReturned); + Assert.Equal(1, releaseCount); + + // Observe the task's outcome so it is never reported as an unobserved exception + await Record.ExceptionAsync(() => speakTask); + } + + /// + /// Proves that an abandoned native Generate call (finding 26) faults the session + /// rather than returning it to the reusable + /// state, so a subsequent / + /// call can never start a second native + /// call concurrently with the still-running abandoned one - it is instead rejected with + /// immediately, with no need to wait. + /// + [Fact(Timeout = 10000)] + public async Task SherpaOnnxSynthesisSession_AbandonedGenerate_FaultsSessionAndRejectsReuseImmediately() + { + // Arrange: a backend whose Generate blocks well past the default 2s abandon timeout + using var generateStarted = new SemaphoreSlim(0, 1); + using var generateRelease = new SemaphoreSlim(0, 1); + var backend = new BlockingSynthesisEngine(generateStarted, generateRelease, TestContext.Current.CancellationToken); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + + // Act: start speaking, wait until the backend call has begun, then stop - the real + // backend never honors cancellation, so the dedicated worker abandons it after its + // default 2s timeout + var speakTask = session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + await generateStarted.WaitAsync(TestContext.Current.CancellationToken); + _ = session.StopAsync(TestContext.Current.CancellationToken); + + // Assert: once the abandon-aware operation task has settled (bounded by the default 2s + // abandon timeout plus margin), the session is Faulted - not the reusable Stopped state - + // and a new call is rejected immediately rather than needing to wait for the still-running + // abandoned native call + await WaitForStateAsync(session, SynthesisSessionState.Faulted, TimeSpan.FromSeconds(5)); + + var reuseException = await Record.ExceptionAsync( + () => session.SynthesizeAsync("Second call.", TestContext.Current.CancellationToken)); + Assert.IsType(reuseException); + + // Cleanup: release the abandoned background call so disposal does not hang + generateRelease.Release(); + await Record.ExceptionAsync(() => speakTask); + } + + /// + /// Proves that does not complete merely because + /// the abandon-aware operation task has settled (finding 27): while the native + /// Generate call is still genuinely running in the background past the abandon + /// timeout, the task StopAsync returns must remain incomplete, matching its + /// documented "operation has stopped" contract. + /// + [Fact(Timeout = 10000)] + public async Task SherpaOnnxSynthesisSession_StopAsync_AbandonedGenerate_DoesNotCompleteUntilNativeCallGenuinelyReturns() + { + // Arrange + using var generateStarted = new SemaphoreSlim(0, 1); + using var generateRelease = new SemaphoreSlim(0, 1); + var backend = new BlockingSynthesisEngine(generateStarted, generateRelease, TestContext.Current.CancellationToken); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + + // Act: start speaking, wait until synthesis has begun, then stop + var speakTask = session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + await generateStarted.WaitAsync(TestContext.Current.CancellationToken); + var stopTask = session.StopAsync(TestContext.Current.CancellationToken); + + // Assert: even once the default 2s abandon timeout has elapsed and the session has + // faulted, StopAsync's own task must still not be complete, because the native call is + // still genuinely running against the backend + await WaitForStateAsync(session, SynthesisSessionState.Faulted, TimeSpan.FromSeconds(5)); + Assert.False(stopTask.IsCompleted); + + // Act: only now let the in-flight native-style call finish + generateRelease.Release(); + + // Assert: StopAsync completes only once the native call has genuinely returned + await stopTask.WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + Assert.True(backend.GenerateReturned); + + await Record.ExceptionAsync(() => speakTask); + } + + /// + /// Proves that DisposeAsync, called after an abandoned native Generate call + /// has already faulted the session (finding 30), still waits for that native call's + /// genuine raw completion before releasing the engine's exclusivity lease. The session's + /// abandon-fault handling clears the operation's tracked task eagerly, alongside the + /// Faulted transition, so a naive re-read of that cleared task (without also re-checking + /// the raw native-call completion) would incorrectly let a caller that disposes only after + /// observing the fault conclude there is nothing left to await. + /// + [Fact(Timeout = 10000)] + public async Task SherpaOnnxSynthesisSession_DisposeAsync_CalledAfterAbandonmentFault_DoesNotReleaseLeaseUntilNativeCallGenuinelyReturns() + { + // Arrange + using var generateStarted = new SemaphoreSlim(0, 1); + using var generateRelease = new SemaphoreSlim(0, 1); + var backend = new BlockingSynthesisEngine(generateStarted, generateRelease, TestContext.Current.CancellationToken); + var device = CreateAvailablePlaybackDevice(); + var releaseCount = 0; + var session = new SherpaOnnxSynthesisSession( + backend, device, new FakeSynthesisModel(), null, NullSpeechDiagnostics.Instance, () => releaseCount++); + + // Act: start speaking, wait until the backend call has begun, then stop - the real + // backend never honors cancellation, so the dedicated worker abandons it after its + // default 2s timeout, faulting the session and clearing its tracked operation task + var speakTask = session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + await generateStarted.WaitAsync(TestContext.Current.CancellationToken); + _ = session.StopAsync(TestContext.Current.CancellationToken); + await WaitForStateAsync(session, SynthesisSessionState.Faulted, TimeSpan.FromSeconds(5)); + + // Act: only now - after the fault, with the tracked operation task already cleared - + // begin disposing + var disposeTask = session.DisposeAsync().AsTask(); + + // Assert: the lease must not be released while the abandoned native call is still + // genuinely running, even though there is no tracked operation task left to await + await Task.Delay(TimeSpan.FromMilliseconds(100), TestContext.Current.CancellationToken); + Assert.Equal(0, releaseCount); + Assert.False(disposeTask.IsCompleted); + + // Act: release the abandoned background call so it can genuinely finish + generateRelease.Release(); + await disposeTask.WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + + // Assert: only once the native call genuinely returned was the lease released + Assert.True(backend.GenerateReturned); + Assert.Equal(1, releaseCount); + + await Record.ExceptionAsync(() => speakTask); + } + + /// + /// Proves that an abandoned playback call (the + /// device-start review finding mirroring finding 27, but for the one-shot playback-device + /// startup rather than the per-segment native Generate call) is routed through the + /// same cooperative-cancel-then-abandon policy: cancelling + /// while is still genuinely blocked must not hang + /// forever, and + /// must never be called while the abandoned call + /// is still genuinely running. + /// + [Fact(Timeout = 10000)] + public async Task SherpaOnnxSynthesisSession_StopAsync_AbandonedDeviceStart_DoesNotCallDeviceStopUntilStartGenuinelyReturns() + { + // Arrange: a playback device whose Start() blocks until explicitly released, standing in + // for a stuck native device-open call (IAudioPlaybackDevice.Start() has no cancellation + // token of its own) + using var startEntered = new SemaphoreSlim(0, 1); + using var startRelease = new SemaphoreSlim(0, 1); + var device = CreateAvailablePlaybackDevice(); + var deviceStopCalled = false; + device.When(d => d.Start()).Do(_ => + { + startEntered.Release(); + startRelease.Wait(TestContext.Current.CancellationToken); + }); + device.When(d => d.Stop()).Do(_ => deviceStopCalled = true); + var backend = new FakeSynthesisEngine(); + await using var session = CreateSession(backend, device); + + // Act: start speaking, wait until the device-start call has begun, then stop - the fake + // device never honors cancellation, so the dedicated worker abandons it after its default + // 2s timeout + var speakTask = session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + await startEntered.WaitAsync(TestContext.Current.CancellationToken); + var stopTask = session.StopAsync(TestContext.Current.CancellationToken); + + // Assert: even once the default 2s abandon timeout has elapsed and the session has + // faulted, StopAsync's own task must still not be complete, and the playback device must + // never have been stopped, because the abandoned Start() call is still genuinely running + await WaitForStateAsync(session, SynthesisSessionState.Faulted, TimeSpan.FromSeconds(5)); + Assert.False(stopTask.IsCompleted); + Assert.False(deviceStopCalled); + + // Act: only now let the abandoned device-start call genuinely finish + startRelease.Release(); + + // Assert: StopAsync completes, and the device was stopped, only once Start() genuinely + // returned + await stopTask.WaitAsync(TimeSpan.FromSeconds(5), TestContext.Current.CancellationToken); + Assert.True(deviceStopCalled); + + await Record.ExceptionAsync(() => speakTask); + } + + /// + /// Proves that a canceled caller token passed to + /// only bounds that caller's own wait (finding 28) rather than aborting the shared + /// cancel-and-await teardown: the caller observes + /// promptly, but the in-flight operation is still genuinely requested to cancel and the + /// session still converges normally afterward. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_StopAsync_CallerTokenPreCanceled_ThrowsButOperationStillStops() + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + + var speakTask = session.SpeakAsync("Hello world.", TestContext.Current.CancellationToken); + + using var preCanceled = new CancellationTokenSource(); + await preCanceled.CancelAsync(); + + // Act & Assert: the caller's own wait is bounded by its own token and throws promptly + await Assert.ThrowsAnyAsync(() => session.StopAsync(preCanceled.Token)); + + // Assert: the operation itself was still genuinely requested to stop and settles normally + await Record.ExceptionAsync(() => speakTask); + Assert.NotEqual(SynthesisSessionState.Running, session.State); + } + + /// + /// Polls 's until it + /// equals or elapses, used only to + /// bound a wait for an asynchronous state transition driven by + /// 's own abandon-timeout delay, not as a substitute for + /// awaiting a task directly. + /// + private static async Task WaitForStateAsync(ISynthesisSession session, SynthesisSessionState expected, TimeSpan timeout) + { + var deadline = DateTime.UtcNow + timeout; + while (session.State != expected && DateTime.UtcNow < deadline) + { + await Task.Delay(TimeSpan.FromMilliseconds(20), TestContext.Current.CancellationToken); + } + + Assert.Equal(expected, session.State); + } + + /// + /// Proves that operating on a disposed session throws + /// rather than silently doing nothing. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SynthesizeAsync_AfterDispose_ThrowsObjectDisposedException() + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + var session = CreateSession(backend, device); + await session.DisposeAsync(); + + // Act & Assert + await Assert.ThrowsAsync( + () => session.SynthesizeAsync("hello", TestContext.Current.CancellationToken)); + } + + /// + /// Proves that reports + /// for a freshly created session, but once disposed. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_IsAvailable_BeforeAndAfterDispose_ReflectsLifecycle() + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + var session = CreateSession(backend, device); + + // Act & Assert: available before disposal + Assert.True(session.IsAvailable); + + // Act + await session.DisposeAsync(); + + // Assert: unavailable after disposal + Assert.False(session.IsAvailable); + } + + /// + /// Proves that genuinely waits for the playback + /// device to report a drained queue before stopping it, rather than treating "every + /// segment enqueued" as "finished playing". Uses a controllable fake (an NSubstitute stub + /// whose PendingSampleCount getter signals a semaphore on every read) so the + /// assertion that the task has not yet completed is driven by an observed poll, not by an + /// arbitrary sleep racing the implementation. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SpeakAsync_PlaybackDeviceReportsPendingSamples_WaitsForDrainBeforeStopping() + { + // Arrange: a playback device that reports samples still pending until the test releases it + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + var pendingSampleCount = 1; + using var polled = new SemaphoreSlim(0); + device.PendingSampleCount.Returns(_ => + { + var value = Volatile.Read(ref pendingSampleCount); + polled.Release(); + return value; + }); + await using var session = CreateSession(backend, device); + + // Act: start speaking; every segment is enqueued almost immediately, but the fake device + // keeps reporting pending samples until the test signals otherwise + var speakTask = session.SpeakAsync("Hi.", TestContext.Current.CancellationToken); + + // Wait until the drain wait has genuinely begun polling the device at least once + await polled.WaitAsync(TestContext.Current.CancellationToken); + + // Assert: playback has not been treated as finished while samples are still reported pending + Assert.False(speakTask.IsCompleted); + device.DidNotReceive().Stop(); + + // Act: signal the fake device has now genuinely drained + Volatile.Write(ref pendingSampleCount, 0); + await speakTask; + + // Assert: only once the device reports a drained queue does playback stop + device.Received(1).Stop(); + } + + /// + /// Proves that the speaker id resolved from + /// (rather than a hard-coded 0) reaches the backend, for both a default (no + /// parameter bag supplied) case and a non-default (bag supplied) case. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SynthesizeAsync_NoParameterValues_ResolvesDefaultSpeakerIdFromModel() + { + // Arrange: a model whose ResolveSpeakerId hook always returns a distinctive non-zero id + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + var model = new FakeSynthesisModel(resolveSpeakerId: _ => 7); + await using var session = CreateSession(backend, device, model); + + // Act + await session.SynthesizeAsync("Hello world.", TestContext.Current.CancellationToken); + + // Assert + var call = Assert.Single(backend.GenerateCalls); + Assert.Equal(7, call.SpeakerId); + } + + /// + /// Proves that a supplied parameter value bag reaches + /// and, through it, the speaker id passed + /// to . + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SynthesizeAsync_ParameterValuesSupplied_ResolvesSpeakerIdFromBag() + { + // Arrange: a model whose ResolveSpeakerId hook echoes back a bag value + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + var model = new FakeSynthesisModel( + resolveSpeakerId: values => values is not null && values.TryGetValue("voice", out var value) && value is int id ? id : 0); + IReadOnlyDictionary parameterValues = new Dictionary { ["voice"] = 3 }; + await using var session = CreateSession(backend, device, model, parameterValues); + + // Act + await session.SynthesizeAsync("Hello world.", TestContext.Current.CancellationToken); + + // Assert + var call = Assert.Single(backend.GenerateCalls); + Assert.Equal(3, call.SpeakerId); + } + + /// + /// Proves that a segment's own Natural Language Audio Tag speed/volume overrides are + /// unaffected by, and coexist correctly with, a supplied parameter value bag driving + /// speaker-id resolution - the two mechanisms remain independent. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SynthesizeAsync_ParameterValuesSuppliedAlongsideSpeedTag_BothMechanismsApplyIndependently() + { + // Arrange + var backend = new FakeSynthesisEngine(); + var device = CreateAvailablePlaybackDevice(); + var model = new FakeSynthesisModel(resolveSpeakerId: _ => 5); + IReadOnlyDictionary parameterValues = new Dictionary { ["voice"] = 5 }; + await using var session = CreateSession(backend, device, model, parameterValues); + + // Act: "fast" is a natural language audio tag mapped to the model's own "tempo" parameter + await session.SynthesizeAsync("[fast] Hello world.", TestContext.Current.CancellationToken); + + // Assert: the speaker id still resolves from the bag, and the segment's speed differs from 1.0 + var call = Assert.Single(backend.GenerateCalls); + Assert.Equal(5, call.SpeakerId); + Assert.NotEqual(1.0f, call.Speed); + } + + /// + /// Proves that a long, multi-sentence input still produces every segment, in the exact + /// order the sentences appear in the source text, and that + /// is never called concurrently with itself. + /// + [Fact] + public async Task SherpaOnnxSynthesisSession_SynthesizeAsync_LongMultiSentenceInput_ProducesOrderedSegmentsSequentially() + { + // Arrange: 12 short, uniquely numbered sentences, so ordering across many chunks can be + // cross-checked against the source text. + const int sentenceCount = 12; + var sentences = Enumerable.Range(1, sentenceCount).Select(i => $"Sentence number {i}."); + var text = string.Join(" ", sentences); + var backend = new FakeSynthesisEngine(simulatedGenerateDelay: TimeSpan.FromMilliseconds(5)); + var device = CreateAvailablePlaybackDevice(); + await using var session = CreateSession(backend, device); + + // Act + var segments = await session.SynthesizeAsync(text, TestContext.Current.CancellationToken); + + // Assert: every sentence was synthesized exactly once, in source order + Assert.Equal(sentenceCount, backend.GenerateCalls.Count); + for (var i = 0; i < sentenceCount; i++) + { + Assert.Contains($"number {i + 1}", backend.GenerateCalls[i].Text, StringComparison.Ordinal); + } + + Assert.Equal(sentenceCount, segments.Count); + for (var i = 0; i < sentenceCount; i++) + { + Assert.Equal(backend.GenerateCalls[i].Text.Length * 4, segments[i].Samples.Count); + } + + // Assert: Generate was never entered concurrently with itself + Assert.Equal(1, backend.MaxConcurrentGenerateCalls); + } + + /// + /// Builds a substitute playback device reporting itself available with the given format. + /// + private static IAudioPlaybackDevice CreateAvailablePlaybackDevice(int sampleRate = 16000, int channelCount = 1) + { + var device = Substitute.For(); + device.IsAvailable.Returns(true); + device.SampleRate.Returns(sampleRate); + device.ChannelCount.Returns(channelCount); + return device; + } + + /// + /// Builds a directly against a backend and + /// device, with a no-op lease-release callback, for tests that exercise the session in + /// isolation from . + /// + private static SherpaOnnxSynthesisSession CreateSession( + ISynthesisBackend backend, + IAudioPlaybackDevice device, + ISynthesisModel? model = null, + IReadOnlyDictionary? parameterValues = null, + ISpeechDiagnostics? diagnostics = null) => + new(backend, device, model ?? new FakeSynthesisModel(), parameterValues, diagnostics ?? NullSpeechDiagnostics.Instance, () => { }); + + /// + /// Test-only whose signals a + /// semaphore once called, then blocks until the test explicitly releases a second + /// semaphore - simulating a native call that keeps running for a while after the + /// session's cancellation token is cancelled, exactly like the real sherpa-onnx binding, + /// which has no in-flight cancellation primitive of its own and so cannot be interrupted + /// by or an externally cancelled token. + /// + /// + /// Deliberately does not unblock from : + /// the whole point of the regression this fake supports is that + /// must never call (or let + /// a caller do so) while this call is still in flight, so a test relying on + /// to unblock it would either mask that exact bug or deadlock + /// against the fix. Callers must release explicitly to + /// let the in-flight call complete. However, still waits against + /// , so a failed or timed-out test unblocks the + /// background producer thread instead of leaving it blocked for the rest of the test run. + /// + /// Released once is called. + /// Awaited by before it returns. + /// + /// The owning test's own cancellation/timeout token (for example, + /// ), observed only as a failure-mode safety + /// net so the wait never outlives the test, not as part of the behavior under test. + /// + private sealed class BlockingSynthesisEngine( + SemaphoreSlim generateStarted, + SemaphoreSlim generateRelease, + CancellationToken testCancellationToken) : ISynthesisBackend + { + public int SampleRate => 16000; + + /// Gets a value indicating whether the in-flight call has genuinely returned. + public bool GenerateReturned { get; private set; } + + public EngineAudio Generate(string text, float speed, int speakerId) + { + generateStarted.Release(); + + // Blocks until the test explicitly allows this in-flight call to complete, rather + // than observing any cancellation token that the session under test might use - + // matching the real native call this fake stands in for. The test's own + // cancellation/timeout token is still observed here purely as a safety net so a + // failed or timed-out test cannot leave this background thread blocked forever. + generateRelease.Wait(testCancellationToken); + + GenerateReturned = true; + return new EngineAudio([], SampleRate); + } + + public void Dispose() + { + // Intentionally does not unblock Generate(): see remarks above. + } + } +} diff --git a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SpeechSynthesizerFactoryTests.cs b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SpeechSynthesizerFactoryTests.cs index 42adb88..81d2caf 100644 --- a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SpeechSynthesizerFactoryTests.cs +++ b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/SpeechSynthesizerFactoryTests.cs @@ -11,7 +11,7 @@ namespace DemaConsulting.Speech.Tests.SynthesisSubsystem; /// /// Unit tests for , proving that composition never /// throws for an ordinary machine state and honestly degrades to -/// instead. +/// instead. /// public sealed class SpeechSynthesizerFactoryTests : IDisposable { @@ -81,89 +81,63 @@ public void Dispose() /// /// Proves that a model whose files are not installed composes to the honest unavailable - /// synthesizer, and that the engine is never loaded. + /// engine, and that the backend is never loaded. /// [Fact] - public void SpeechSynthesizerFactory_Create_ModelNotInstalled_ReturnsUnavailableSynthesizer() + public async Task SpeechSynthesizerFactory_LoadAsync_ModelNotInstalled_ReturnsUnavailableEngine() { - // Arrange: an available playback device but a directory that does not exist - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + // Arrange: a directory that does not exist + var backendFactory = new FakeSynthesisEngineFactory(); var missingDirectory = Path.Join(_installedModelDirectory, "not-installed"); // Act: compose against the missing model directory - var synthesizer = SpeechSynthesizerFactory.Create( - new FakeSynthesisModel(), missingDirectory, playbackDevice, null, engineFactory); + var engine = await SpeechSynthesizerFactory.LoadAsync( + new FakeSynthesisModel(), missingDirectory, null, backendFactory, null, TestContext.Current.CancellationToken); - // Assert: the honest fallback is returned and no engine was loaded - Assert.Same(UnavailableSpeechSynthesizer.Instance, synthesizer); - Assert.Equal(0, engineFactory.CreateCallCount); - } - - /// - /// Proves that a machine with no usable playback device composes to the honest - /// unavailable synthesizer rather than loading a model that could never be heard. - /// - [Fact] - public void SpeechSynthesizerFactory_Create_PlaybackDeviceUnavailable_ReturnsUnavailableSynthesizer() - { - // Arrange: an installed model but the shared unavailable playback device - var engineFactory = new FakeSynthesisEngineFactory(); - - // Act: compose against the unavailable device - var synthesizer = SpeechSynthesizerFactory.Create( - new FakeSynthesisModel(), - _installedModelDirectory, - UnavailableAudioPlaybackDevice.Instance, - null, - engineFactory); - - // Assert: the honest fallback is returned and no engine was loaded - Assert.Same(UnavailableSpeechSynthesizer.Instance, synthesizer); - Assert.Equal(0, engineFactory.CreateCallCount); + // Assert: the honest fallback is returned and no backend was loaded + Assert.Same(UnavailableSpeechSynthesizerEngine.Instance, engine); + Assert.Equal(0, backendFactory.CreateCallCount); } /// /// Proves that a model declaring a non-synthesis role composes to the honest unavailable - /// synthesizer. + /// engine. /// [Fact] - public void SpeechSynthesizerFactory_Create_ModelRoleIsNotSynthesis_ReturnsUnavailableSynthesizer() + public async Task SpeechSynthesizerFactory_LoadAsync_ModelRoleIsNotSynthesis_ReturnsUnavailableEngine() { - // Arrange: an installed model that declares the recognition role, and an available device - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + // Arrange: an installed model that declares the recognition role + var backendFactory = new FakeSynthesisEngineFactory(); // Act: compose against the wrong-role model - var synthesizer = SpeechSynthesizerFactory.Create( - new WrongRoleSynthesisModel(), _installedModelDirectory, playbackDevice, null, engineFactory); + var engine = await SpeechSynthesizerFactory.LoadAsync( + new WrongRoleSynthesisModel(), _installedModelDirectory, null, backendFactory, null, TestContext.Current.CancellationToken); - // Assert: the honest fallback is returned and no engine was loaded - Assert.Same(UnavailableSpeechSynthesizer.Instance, synthesizer); - Assert.Equal(0, engineFactory.CreateCallCount); + // Assert: the honest fallback is returned and no backend was loaded + Assert.Same(UnavailableSpeechSynthesizerEngine.Instance, engine); + Assert.Equal(0, backendFactory.CreateCallCount); } /// /// Proves that a native-runtime or model-file load failure degrades to the honest - /// unavailable synthesizer instead of propagating out of composition. + /// unavailable engine instead of propagating out of composition. /// [Fact] - public void SpeechSynthesizerFactory_Create_EngineLoadFails_ReturnsUnavailableSynthesizerAndDoesNotThrow() + public async Task SpeechSynthesizerFactory_LoadAsync_EngineLoadFails_ReturnsUnavailableEngineAndDoesNotThrow() { - // Arrange: an installed model, an available device, and an engine factory that faults - var playbackDevice = CreateAvailablePlaybackDevice(); + // Arrange: an installed model and a backend factory that faults var diagnostics = Substitute.For(); - var engineFactory = new FakeSynthesisEngineFactory( + var backendFactory = new FakeSynthesisEngineFactory( createException: new DllNotFoundException("sherpa-onnx-c-api")); // Act: compose, capturing any exception that escapes - ISpeechSynthesizer? synthesizer = null; - var exception = Record.Exception(() => synthesizer = SpeechSynthesizerFactory.Create( - new FakeSynthesisModel(), _installedModelDirectory, playbackDevice, diagnostics, engineFactory)); + ISpeechSynthesizerEngine? engine = null; + var exception = await Record.ExceptionAsync(async () => engine = await SpeechSynthesizerFactory.LoadAsync( + new FakeSynthesisModel(), _installedModelDirectory, diagnostics, backendFactory, null, TestContext.Current.CancellationToken)); // Assert: composition succeeded honestly and reported the fault as a structural fact Assert.Null(exception); - Assert.Same(UnavailableSpeechSynthesizer.Instance, synthesizer); + Assert.Same(UnavailableSpeechSynthesizerEngine.Instance, engine); diagnostics.Received().Report( SpeechDiagnosticLevel.Error, "SynthesisSubsystem", @@ -171,28 +145,25 @@ public void SpeechSynthesizerFactory_Create_EngineLoadFails_ReturnsUnavailableSy } /// - /// Proves that an installed synthesis model plus an available playback device composes a - /// real synthesizer wired to the injected engine factory, with the installed-model - /// directory passed through unchanged. + /// Proves that an installed synthesis model composes a real engine wired to the injected + /// backend factory, with the installed-model directory passed through unchanged. /// [Fact] - public void SpeechSynthesizerFactory_Create_ModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer() + public async Task SpeechSynthesizerFactory_LoadAsync_ModelInstalled_ReturnsRealEngine() { - // Arrange: an installed model, an available device, and a fake engine factory - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + // Arrange: an installed model and a fake backend factory + var backendFactory = new FakeSynthesisEngineFactory(); var model = new FakeSynthesisModel(); - // Act: compose a synthesizer - using var synthesizer = SpeechSynthesizerFactory.Create( - model, _installedModelDirectory, playbackDevice, null, engineFactory); + // Act: compose an engine + await using var engine = await SpeechSynthesizerFactory.LoadAsync( + model, _installedModelDirectory, null, backendFactory, null, TestContext.Current.CancellationToken); - // Assert: a real synthesizer was built from the injected engine, for the right model - Assert.IsType(synthesizer); - Assert.True(synthesizer.IsAvailable); - Assert.Equal(1, engineFactory.CreateCallCount); - Assert.Same(model, engineFactory.RequestedModel); - Assert.Equal(_installedModelDirectory, engineFactory.RequestedInstalledModelDirectory); + // Assert: a real engine was built from the injected backend, for the right model + Assert.True(engine.IsAvailable); + Assert.Equal(1, backendFactory.CreateCallCount); + Assert.Same(model, backendFactory.RequestedModel); + Assert.Equal(_installedModelDirectory, backendFactory.RequestedInstalledModelDirectory); } /// @@ -200,52 +171,52 @@ public void SpeechSynthesizerFactory_Create_ModelInstalledAndDeviceAvailable_Ret /// argument is a programming error rather than an ordinary machine state. /// [Fact] - public void SpeechSynthesizerFactory_Create_NullModel_ThrowsArgumentNullException() + public async Task SpeechSynthesizerFactory_LoadAsync_NullModel_ThrowsArgumentNullException() { - // Arrange: an available playback device - var playbackDevice = CreateAvailablePlaybackDevice(); - // Act & Assert: a null model is rejected - Assert.Throws( - () => SpeechSynthesizerFactory.Create(null!, _installedModelDirectory, playbackDevice)); + await Assert.ThrowsAsync( + () => SpeechSynthesizerFactory.LoadAsync(null!, _installedModelDirectory, cancellationToken: TestContext.Current.CancellationToken)); } /// - /// Proves that the public composition overload rejects a null playback device. + /// Proves that an already-cancelled token throws + /// synchronously from composition, rather than composing anyway. /// [Fact] - public void SpeechSynthesizerFactory_Create_NullPlaybackDevice_ThrowsArgumentNullException() + public async Task SpeechSynthesizerFactory_LoadAsync_CancelledToken_ThrowsOperationCanceledException() { - // Act & Assert: a null playback device is rejected - Assert.Throws( - () => SpeechSynthesizerFactory.Create(new FakeSynthesisModel(), _installedModelDirectory, null!)); + // Arrange + var backendFactory = new FakeSynthesisEngineFactory(); + using var cts = new CancellationTokenSource(); + await cts.CancelAsync(); + + // Act & Assert + await Assert.ThrowsAnyAsync(() => SpeechSynthesizerFactory.LoadAsync( + new FakeSynthesisModel(), _installedModelDirectory, null, backendFactory, null, cts.Token)); } /// - /// Proves that parameterValues passed to - /// reaches the constructed synthesizer's synthesis calls, by round-tripping it through a - /// fake model's ResolveSpeakerId hook into the fake engine's captured speaker id. + /// Proves that parameterValues passed to + /// + /// reaches the constructed engine's synthesis calls, by round-tripping it through a fake + /// model's ResolveSpeakerId hook into the fake engine's captured speaker id. /// [Fact] - public async Task SpeechSynthesizerFactory_Create_ParameterValuesSupplied_ForwardedToSynthesizer() + public async Task SpeechSynthesizerFactory_LoadAsync_ParameterValuesSupplied_ForwardedToEngine() { // Arrange - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + var backendFactory = new FakeSynthesisEngineFactory(); var model = new FakeSynthesisModel( resolveSpeakerId: values => values is not null && values.TryGetValue("voice", out var value) && value is int id ? id : 0); IReadOnlyDictionary parameterValues = new Dictionary { ["voice"] = 4 }; // Act - using var synthesizer = SpeechSynthesizerFactory.Create( - model, _installedModelDirectory, playbackDevice, null, engineFactory, parameterValues); - await foreach (var _ in synthesizer.SynthesizeStreamAsync("Hello.", TestContext.Current.CancellationToken)) - { - // Draining is enough to trigger the engine call. - } + await using var engine = await SpeechSynthesizerFactory.LoadAsync( + model, _installedModelDirectory, null, backendFactory, parameterValues, TestContext.Current.CancellationToken); + await engine.SynthesizeAsync("Hello.", TestContext.Current.CancellationToken); // Assert - var call = Assert.Single(engineFactory.Engine.GenerateCalls); + var call = Assert.Single(backendFactory.Engine.GenerateCalls); Assert.Equal(4, call.SpeakerId); } @@ -255,21 +226,20 @@ public async Task SpeechSynthesizerFactory_Create_ParameterValuesSupplied_Forwar /// supplied. /// [Fact] - public void SpeechSynthesizerFactory_Create_UnrecognizedParameterId_ComposesAndReportsInfo() + public async Task SpeechSynthesizerFactory_LoadAsync_UnrecognizedParameterId_ComposesAndReportsInfo() { // Arrange - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + var backendFactory = new FakeSynthesisEngineFactory(); var model = new FakeSynthesisModel(); var diagnostics = Substitute.For(); IReadOnlyDictionary parameterValues = new Dictionary { ["typo-id"] = 1 }; // Act - using var synthesizer = SpeechSynthesizerFactory.Create( - model, _installedModelDirectory, playbackDevice, diagnostics, engineFactory, parameterValues); + await using var engine = await SpeechSynthesizerFactory.LoadAsync( + model, _installedModelDirectory, diagnostics, backendFactory, parameterValues, TestContext.Current.CancellationToken); - // Assert: still a real, working synthesizer, plus the observability diagnostic - Assert.True(synthesizer.IsAvailable); + // Assert: still a real, working engine, plus the observability diagnostic + Assert.True(engine.IsAvailable); diagnostics.Received(1).Report( SpeechDiagnosticLevel.Info, "SynthesisSubsystem", @@ -278,58 +248,55 @@ public void SpeechSynthesizerFactory_Create_UnrecognizedParameterId_ComposesAndR /// /// Proves that an out-of-range value for a recognized numeric parameter throws - /// synchronously from Create, rather than - /// silently defaulting, and that the engine is never loaded. + /// synchronously from LoadAsync, rather than + /// silently defaulting, and that the backend is never loaded. /// [Fact] - public void SpeechSynthesizerFactory_Create_RecognizedNumericParameterOutOfRange_Throws() + public async Task SpeechSynthesizerFactory_LoadAsync_RecognizedNumericParameterOutOfRange_Throws() { // Arrange: FakeSynthesisModel declares "tempo" bounded to [0.5, 2.0] - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + var backendFactory = new FakeSynthesisEngineFactory(); var model = new FakeSynthesisModel(); IReadOnlyDictionary parameterValues = new Dictionary { ["tempo"] = 5.0 }; // Act / Assert - var exception = Assert.Throws(() => SpeechSynthesizerFactory.Create( - model, _installedModelDirectory, playbackDevice, null, engineFactory, parameterValues)); + var exception = await Assert.ThrowsAsync(() => SpeechSynthesizerFactory.LoadAsync( + model, _installedModelDirectory, null, backendFactory, parameterValues, TestContext.Current.CancellationToken)); Assert.Contains("tempo", exception.Message, StringComparison.Ordinal); - Assert.Equal(0, engineFactory.CreateCallCount); + Assert.Equal(0, backendFactory.CreateCallCount); } /// /// Proves that a wrong CLR type for a recognized numeric parameter throws - /// synchronously from Create. + /// synchronously from LoadAsync. /// [Fact] - public void SpeechSynthesizerFactory_Create_RecognizedNumericParameterWrongType_Throws() + public async Task SpeechSynthesizerFactory_LoadAsync_RecognizedNumericParameterWrongType_Throws() { // Arrange - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + var backendFactory = new FakeSynthesisEngineFactory(); var model = new FakeSynthesisModel(); IReadOnlyDictionary parameterValues = new Dictionary { ["tempo"] = "fast" }; // Act / Assert - var exception = Assert.Throws(() => SpeechSynthesizerFactory.Create( - model, _installedModelDirectory, playbackDevice, null, engineFactory, parameterValues)); + var exception = await Assert.ThrowsAsync(() => SpeechSynthesizerFactory.LoadAsync( + model, _installedModelDirectory, null, backendFactory, parameterValues, TestContext.Current.CancellationToken)); Assert.Contains("tempo", exception.Message, StringComparison.Ordinal); - Assert.Equal(0, engineFactory.CreateCallCount); + Assert.Equal(0, backendFactory.CreateCallCount); } /// /// Proves that a non-integral value for the real, shipped /// 's integer-only /// speaker parameter throws synchronously from - /// Create, rather than silently rounding it (this library's previous, now + /// LoadAsync, rather than silently rounding it (this library's previous, /// deliberately superseded, behavior for this exact case). /// [Fact] - public void SpeechSynthesizerFactory_Create_RecognizedIntegerParameterFractionalValue_Throws() + public async Task SpeechSynthesizerFactory_LoadAsync_RecognizedIntegerParameterFractionalValue_Throws() { // Arrange - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + var backendFactory = new FakeSynthesisEngineFactory(); var model = new SherpaOnnxVitsLibriTtsEnglishSynthesisModel(); IReadOnlyDictionary parameterValues = new Dictionary { @@ -337,24 +304,23 @@ public void SpeechSynthesizerFactory_Create_RecognizedIntegerParameterFractional }; // Act / Assert - var exception = Assert.Throws(() => SpeechSynthesizerFactory.Create( - model, _installedModelDirectory, playbackDevice, null, engineFactory, parameterValues)); + var exception = await Assert.ThrowsAsync(() => SpeechSynthesizerFactory.LoadAsync( + model, _installedModelDirectory, null, backendFactory, parameterValues, TestContext.Current.CancellationToken)); Assert.Contains(SherpaOnnxVitsLibriTtsEnglishSynthesisModel.SpeakerParameterId, exception.Message, StringComparison.Ordinal); Assert.Contains("whole number", exception.Message, StringComparison.Ordinal); - Assert.Equal(0, engineFactory.CreateCallCount); + Assert.Equal(0, backendFactory.CreateCallCount); } /// /// Proves that a value not matching any declared voice for the real, shipped /// 's ChoiceParameter throws - /// synchronously from Create. + /// synchronously from LoadAsync. /// [Fact] - public void SpeechSynthesizerFactory_Create_RecognizedChoiceParameterInvalidOption_Throws() + public async Task SpeechSynthesizerFactory_LoadAsync_RecognizedChoiceParameterInvalidOption_Throws() { // Arrange - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + var backendFactory = new FakeSynthesisEngineFactory(); var model = new SherpaOnnxKokoroEnglishSynthesisModel(); IReadOnlyDictionary parameterValues = new Dictionary { @@ -362,190 +328,196 @@ public void SpeechSynthesizerFactory_Create_RecognizedChoiceParameterInvalidOpti }; // Act / Assert - var exception = Assert.Throws(() => SpeechSynthesizerFactory.Create( - model, _installedModelDirectory, playbackDevice, null, engineFactory, parameterValues)); + var exception = await Assert.ThrowsAsync(() => SpeechSynthesizerFactory.LoadAsync( + model, _installedModelDirectory, null, backendFactory, parameterValues, TestContext.Current.CancellationToken)); Assert.Contains("not-a-declared-voice", exception.Message, StringComparison.Ordinal); - Assert.Equal(0, engineFactory.CreateCallCount); + Assert.Equal(0, backendFactory.CreateCallCount); } /// /// Proves that a model not yet installed in the store composes to the honest unavailable - /// synthesizer, and that the engine is never loaded. + /// engine, and that the backend is never loaded. /// [Fact] - public void SpeechSynthesizerFactory_Create_WithStoreModelNotInstalled_ReturnsUnavailableSynthesizer() + public async Task SpeechSynthesizerFactory_LoadAsync_WithStoreModelNotInstalled_ReturnsUnavailableEngine() { - // Arrange: an available playback device and a store with no installed model directory - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + // Arrange: a store with no installed model directory + var backendFactory = new FakeSynthesisEngineFactory(); var model = new FakeSynthesisModel(); // Act: compose against the store, which resolves to a directory that does not exist - var synthesizer = SpeechSynthesizerFactory.Create(model, _store, playbackDevice, null, engineFactory); + var engine = await SpeechSynthesizerFactory.LoadAsync( + model, _store, null, backendFactory, null, TestContext.Current.CancellationToken); - // Assert: the honest fallback is returned and no engine was loaded - Assert.Same(UnavailableSpeechSynthesizer.Instance, synthesizer); - Assert.Equal(0, engineFactory.CreateCallCount); + // Assert: the honest fallback is returned and no backend was loaded + Assert.Same(UnavailableSpeechSynthesizerEngine.Instance, engine); + Assert.Equal(0, backendFactory.CreateCallCount); } /// - /// Proves that an installed synthesis model plus an available playback device composes a - /// real synthesizer wired to the injected engine factory, with the directory resolved - /// through the store rather than hard-coded. + /// Proves that an installed synthesis model composes a real engine wired to the injected + /// backend factory, with the directory resolved through the store rather than hard-coded. /// [Fact] - public void SpeechSynthesizerFactory_Create_WithStoreModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer() + public async Task SpeechSynthesizerFactory_LoadAsync_WithStoreModelInstalled_ReturnsRealEngine() { - // Arrange: a model installed via the store, an available device, and a fake engine factory - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + // Arrange: a model installed via the store, and a fake backend factory + var backendFactory = new FakeSynthesisEngineFactory(); var model = new FakeSynthesisModel(); Directory.CreateDirectory(_store.GetCurrentDirectory(model.Id)); - // Act: compose a synthesizer through the store overload - using var synthesizer = SpeechSynthesizerFactory.Create(model, _store, playbackDevice, null, engineFactory); + // Act: compose an engine through the store overload + await using var engine = await SpeechSynthesizerFactory.LoadAsync( + model, _store, null, backendFactory, null, TestContext.Current.CancellationToken); - // Assert: a real synthesizer was built, and the directory was resolved through the store - Assert.IsType(synthesizer); - Assert.Equal(1, engineFactory.CreateCallCount); - Assert.Equal(_store.GetCurrentDirectory(model.Id), engineFactory.RequestedInstalledModelDirectory); + // Assert: a real engine was built, and the directory was resolved through the store + Assert.True(engine.IsAvailable); + Assert.Equal(1, backendFactory.CreateCallCount); + Assert.Equal(_store.GetCurrentDirectory(model.Id), backendFactory.RequestedInstalledModelDirectory); } /// /// Proves that the public store-based composition overload rejects a null model. /// [Fact] - public void SpeechSynthesizerFactory_Create_WithStoreNullModel_ThrowsArgumentNullException() + public async Task SpeechSynthesizerFactory_LoadAsync_WithStoreNullModel_ThrowsArgumentNullException() { - // Arrange: an available playback device - var playbackDevice = CreateAvailablePlaybackDevice(); - // Act & Assert: a null model is rejected - Assert.Throws( - () => SpeechSynthesizerFactory.Create(null!, _store, playbackDevice)); + await Assert.ThrowsAsync( + () => SpeechSynthesizerFactory.LoadAsync(null!, _store, cancellationToken: TestContext.Current.CancellationToken)); } /// /// Proves that the public store-based composition overload rejects a null store. /// [Fact] - public void SpeechSynthesizerFactory_Create_WithStoreNullStore_ThrowsArgumentNullException() + public async Task SpeechSynthesizerFactory_LoadAsync_WithStoreNullStore_ThrowsArgumentNullException() { - // Arrange: an available playback device - var playbackDevice = CreateAvailablePlaybackDevice(); - // Act & Assert: a null store is rejected - Assert.Throws( - () => SpeechSynthesizerFactory.Create(new FakeSynthesisModel(), (SpeechModelStore)null!, playbackDevice)); - } - - /// - /// Proves that the public store-based composition overload rejects a null playback - /// device, confirming the delegation still reaches the string-overload's own null check. - /// - [Fact] - public void SpeechSynthesizerFactory_Create_WithStoreNullPlaybackDevice_ThrowsArgumentNullException() - { - // Act & Assert: a null playback device is rejected - Assert.Throws( - () => SpeechSynthesizerFactory.Create(new FakeSynthesisModel(), _store, null!)); + await Assert.ThrowsAsync( + () => SpeechSynthesizerFactory.LoadAsync(new FakeSynthesisModel(), (SpeechModelStore)null!, cancellationToken: TestContext.Current.CancellationToken)); } /// /// Proves that a model not yet installed in the catalog's store composes to the honest - /// unavailable synthesizer, and that the engine is never loaded. + /// unavailable engine, and that the backend is never loaded. /// [Fact] - public void SpeechSynthesizerFactory_Create_WithCatalogModelNotInstalled_ReturnsUnavailableSynthesizer() + public async Task SpeechSynthesizerFactory_LoadAsync_WithCatalogModelNotInstalled_ReturnsUnavailableEngine() { - // Arrange: an available playback device and a catalog whose store has no installed model directory - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + // Arrange: a catalog whose store has no installed model directory + var backendFactory = new FakeSynthesisEngineFactory(); var model = new FakeSynthesisModel(); // Act: compose against the catalog, which resolves to a directory that does not exist - var synthesizer = SpeechSynthesizerFactory.Create(model, _catalog, playbackDevice, null, engineFactory); + var engine = await SpeechSynthesizerFactory.LoadAsync( + model, _catalog, null, backendFactory, null, TestContext.Current.CancellationToken); - // Assert: the honest fallback is returned and no engine was loaded - Assert.Same(UnavailableSpeechSynthesizer.Instance, synthesizer); - Assert.Equal(0, engineFactory.CreateCallCount); + // Assert: the honest fallback is returned and no backend was loaded + Assert.Same(UnavailableSpeechSynthesizerEngine.Instance, engine); + Assert.Equal(0, backendFactory.CreateCallCount); } /// - /// Proves that an installed synthesis model plus an available playback device composes a - /// real synthesizer wired to the injected engine factory, with the directory resolved - /// through the catalog's own store rather than a second, disconnected store. + /// Proves that an installed synthesis model composes a real engine wired to the injected + /// backend factory, with the directory resolved through the catalog's own store rather + /// than a second, disconnected store. /// [Fact] - public void SpeechSynthesizerFactory_Create_WithCatalogModelInstalledAndDeviceAvailable_ReturnsRealSynthesizer() + public async Task SpeechSynthesizerFactory_LoadAsync_WithCatalogModelInstalled_ReturnsRealEngine() { - // Arrange: a model installed via the catalog's store, an available device, and a fake engine factory - var playbackDevice = CreateAvailablePlaybackDevice(); - var engineFactory = new FakeSynthesisEngineFactory(); + // Arrange: a model installed via the catalog's store, and a fake backend factory + var backendFactory = new FakeSynthesisEngineFactory(); var model = new FakeSynthesisModel(); Directory.CreateDirectory(_store.GetCurrentDirectory(model.Id)); - // Act: compose a synthesizer through the catalog overload - using var synthesizer = SpeechSynthesizerFactory.Create(model, _catalog, playbackDevice, null, engineFactory); + // Act: compose an engine through the catalog overload + await using var engine = await SpeechSynthesizerFactory.LoadAsync( + model, _catalog, null, backendFactory, null, TestContext.Current.CancellationToken); - // Assert: a real synthesizer was built, and the directory was resolved through the catalog's store - Assert.IsType(synthesizer); - Assert.Equal(1, engineFactory.CreateCallCount); - Assert.Equal(_catalog.Store.GetCurrentDirectory(model.Id), engineFactory.RequestedInstalledModelDirectory); + // Assert: a real engine was built, and the directory was resolved through the catalog's store + Assert.True(engine.IsAvailable); + Assert.Equal(1, backendFactory.CreateCallCount); + Assert.Equal(_catalog.Store.GetCurrentDirectory(model.Id), backendFactory.RequestedInstalledModelDirectory); } /// /// Proves that the public catalog-based composition overload rejects a null model. /// [Fact] - public void SpeechSynthesizerFactory_Create_WithCatalogNullModel_ThrowsArgumentNullException() + public async Task SpeechSynthesizerFactory_LoadAsync_WithCatalogNullModel_ThrowsArgumentNullException() { - // Arrange: an available playback device - var playbackDevice = CreateAvailablePlaybackDevice(); - // Act & Assert: a null model is rejected - Assert.Throws( - () => SpeechSynthesizerFactory.Create(null!, _catalog, playbackDevice)); + await Assert.ThrowsAsync( + () => SpeechSynthesizerFactory.LoadAsync(null!, _catalog, cancellationToken: TestContext.Current.CancellationToken)); } /// /// Proves that the public catalog-based composition overload rejects a null catalog. /// [Fact] - public void SpeechSynthesizerFactory_Create_WithCatalogNullCatalog_ThrowsArgumentNullException() + public async Task SpeechSynthesizerFactory_LoadAsync_WithCatalogNullCatalog_ThrowsArgumentNullException() { - // Arrange: an available playback device - var playbackDevice = CreateAvailablePlaybackDevice(); - // Act & Assert: a null catalog is rejected - Assert.Throws( - () => SpeechSynthesizerFactory.Create(new FakeSynthesisModel(), (SpeechModelCatalog)null!, playbackDevice)); + await Assert.ThrowsAsync( + () => SpeechSynthesizerFactory.LoadAsync(new FakeSynthesisModel(), (SpeechModelCatalog)null!, cancellationToken: TestContext.Current.CancellationToken)); } /// - /// Proves that the public catalog-based composition overload rejects a null playback - /// device, confirming the delegation still reaches the store-overload's own null check. + /// Proves that when ignores cancellation but + /// still finishes within the dedicated worker's abandon grace period - so the worker + /// genuinely completes rather than being abandoned - LoadAsync still honors the + /// cancellation request (finding 31) rather than reporting a loaded engine, and disposes + /// the backend that was created so it does not leak. /// - [Fact] - public void SpeechSynthesizerFactory_Create_WithCatalogNullPlaybackDevice_ThrowsArgumentNullException() + [Fact(Timeout = 10000)] + public async Task SpeechSynthesizerFactory_LoadAsync_CancelledDuringGraceWindowCreateSucceeds_ThrowsAndDisposesBackend() { - // Act & Assert: a null playback device is rejected - Assert.Throws( - () => SpeechSynthesizerFactory.Create(new FakeSynthesisModel(), _catalog, null!)); + // Arrange: a backend factory whose Create blocks until released, then succeeds, ignoring + // cancellation entirely - exactly like the real native binding, which has no in-flight + // cancellation primitive of its own. + using var createStarted = new SemaphoreSlim(0, 1); + using var createRelease = new SemaphoreSlim(0, 1); + var engine = new FakeSynthesisEngine(); + var backendFactory = new BlockingSynthesisEngineFactory(createStarted, createRelease, engine); + using var cts = new CancellationTokenSource(); + + // Act: start loading, wait until Create has begun, cancel, then let Create finish quickly + // (well within the worker's default abandon timeout) so the worker is not abandoned + var loadTask = SpeechSynthesizerFactory.LoadAsync( + new FakeSynthesisModel(), _installedModelDirectory, null, backendFactory, null, cts.Token); + await createStarted.WaitAsync(TestContext.Current.CancellationToken); + await cts.CancelAsync(); + createRelease.Release(); + + // Assert: the cancellation is still honored even though backend creation genuinely + // succeeded, and the backend that was created is disposed rather than leaked + await Assert.ThrowsAsync(() => loadTask); + Assert.Equal(1, engine.DisposeCallCount); } /// - /// Builds a substitute playback device that reports itself available with a realistic - /// stereo 48 kHz playback format. + /// Test-only whose signals a + /// semaphore once called, then blocks until the test explicitly releases a second + /// semaphore before returning the pre-configured engine - simulating a native load call + /// that ignores cancellation but still finishes within the dedicated worker's abandon + /// grace period. /// - /// The configured substitute playback device. - private static IAudioPlaybackDevice CreateAvailablePlaybackDevice() + private sealed class BlockingSynthesisEngineFactory( + SemaphoreSlim createStarted, + SemaphoreSlim createRelease, + FakeSynthesisEngine engine) : ISynthesisBackendFactory { - var playbackDevice = Substitute.For(); - playbackDevice.IsAvailable.Returns(true); - playbackDevice.SampleRate.Returns(48000); - playbackDevice.ChannelCount.Returns(2); - return playbackDevice; + public ISynthesisBackend Create(ISynthesisModel model, string installedModelDirectory) + { + createStarted.Release(); + + // Bounded wait purely as a safety net so a failed/timed-out test cannot leave this + // dedicated worker thread blocked forever; the behavior under test only relies on the + // release happening promptly. + createRelease.Wait(TimeSpan.FromSeconds(30)); + return engine; + } } /// diff --git a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSpeechSynthesizerTests.cs b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSpeechSynthesizerEngineTests.cs similarity index 52% rename from test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSpeechSynthesizerTests.cs rename to test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSpeechSynthesizerEngineTests.cs index c4d7ab5..62dab06 100644 --- a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSpeechSynthesizerTests.cs +++ b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSpeechSynthesizerEngineTests.cs @@ -1,101 +1,108 @@ +using DemaConsulting.Speech.AudioSubsystem; using DemaConsulting.Speech.SynthesisSubsystem; +using NSubstitute; namespace DemaConsulting.Speech.Tests.SynthesisSubsystem; /// -/// Unit tests for and +/// Unit tests for and /// . /// -public class UnavailableSpeechSynthesizerTests +public class UnavailableSpeechSynthesizerEngineTests { /// - /// Proves that reports itself as - /// unavailable. + /// Proves that reports itself + /// as unavailable. /// [Fact] - public void UnavailableSpeechSynthesizer_IsAvailable_Read_ReturnsFalse() + public void UnavailableSpeechSynthesizerEngine_IsAvailable_Read_ReturnsFalse() { // Act - var isAvailable = UnavailableSpeechSynthesizer.Instance.IsAvailable; + var isAvailable = UnavailableSpeechSynthesizerEngine.Instance.IsAvailable; // Assert Assert.False(isAvailable); } /// - /// Proves that calling throws - /// . + /// Proves that always succeeds, + /// returning rather than throwing. /// [Fact] - public void UnavailableSpeechSynthesizer_SynthesizeStreamAsync_Always_ThrowsSpeechSynthesizerUnavailableException() + public async Task UnavailableSpeechSynthesizerEngine_CreateSessionAsync_Always_ReturnsUnavailableSession() { // Arrange - var synthesizer = UnavailableSpeechSynthesizer.Instance; + var engine = UnavailableSpeechSynthesizerEngine.Instance; + var device = Substitute.For(); - // Act & Assert - Assert.Throws( - () => synthesizer.SynthesizeStreamAsync("hello", TestContext.Current.CancellationToken)); + // Act + var session = await engine.CreateSessionAsync(device, TestContext.Current.CancellationToken); + + // Assert + Assert.Same(UnavailableSynthesisSession.Instance, session); } /// - /// Proves that calling throws - /// . + /// Proves that rejects a null + /// device. /// [Fact] - public async Task UnavailableSpeechSynthesizer_PlayStreamAsync_Always_ThrowsSpeechSynthesizerUnavailableException() + public async Task UnavailableSpeechSynthesizerEngine_CreateSessionAsync_NullDevice_ThrowsArgumentNullException() { // Arrange - var synthesizer = UnavailableSpeechSynthesizer.Instance; + var engine = UnavailableSpeechSynthesizerEngine.Instance; // Act & Assert - await Assert.ThrowsAsync( - () => synthesizer.PlayStreamAsync(EmptyStream(), TestContext.Current.CancellationToken)); + await Assert.ThrowsAsync( + () => engine.CreateSessionAsync(null!, TestContext.Current.CancellationToken)); } /// - /// Proves that calling throws + /// Proves that calling throws /// . /// [Fact] - public async Task UnavailableSpeechSynthesizer_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException() + public async Task UnavailableSpeechSynthesizerEngine_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException() { // Arrange - var synthesizer = UnavailableSpeechSynthesizer.Instance; + var engine = UnavailableSpeechSynthesizerEngine.Instance; + var device = Substitute.For(); // Act & Assert await Assert.ThrowsAsync( - () => synthesizer.SpeakAsync("hello", TestContext.Current.CancellationToken)); + () => engine.SpeakAsync(device, "hello", TestContext.Current.CancellationToken)); } /// - /// Proves that calling throws + /// Proves that calling throws /// . /// [Fact] - public void UnavailableSpeechSynthesizer_Stop_Always_ThrowsSpeechSynthesizerUnavailableException() + public async Task UnavailableSpeechSynthesizerEngine_SynthesizeAsync_Always_ThrowsSpeechSynthesizerUnavailableException() { // Arrange - var synthesizer = UnavailableSpeechSynthesizer.Instance; + var engine = UnavailableSpeechSynthesizerEngine.Instance; // Act & Assert - Assert.Throws(synthesizer.Stop); + await Assert.ThrowsAsync( + () => engine.SynthesizeAsync("hello", TestContext.Current.CancellationToken)); } /// /// Proves that disposing the shared instance is a safe no-op, even when called more than - /// once, so a host wrapping it in a using block never fails. + /// once, so a host wrapping it in an await using block never fails. /// [Fact] - public void UnavailableSpeechSynthesizer_Dispose_CalledTwice_DoesNotThrow() + public async Task UnavailableSpeechSynthesizerEngine_DisposeAsync_CalledTwice_DoesNotThrow() { // Arrange - var synthesizer = UnavailableSpeechSynthesizer.Instance; + var engine = UnavailableSpeechSynthesizerEngine.Instance; // Act - var exception = Record.Exception(() => + var exception = await Record.ExceptionAsync(async () => { - synthesizer.Dispose(); - synthesizer.Dispose(); + await engine.DisposeAsync(); + await engine.DisposeAsync(); }); // Assert @@ -152,11 +159,4 @@ public void SpeechSynthesizerUnavailableException_Constructor_Default_HasNonEmpt // Assert Assert.False(string.IsNullOrEmpty(exception.Message)); } - - /// An empty asynchronous stream, standing in for a real synthesized-speech sequence. - private static async IAsyncEnumerable EmptyStream() - { - await Task.CompletedTask; - yield break; - } } diff --git a/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSynthesisSessionTests.cs b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSynthesisSessionTests.cs new file mode 100644 index 0000000..66c758f --- /dev/null +++ b/test/DemaConsulting.Speech.Tests/SynthesisSubsystem/UnavailableSynthesisSessionTests.cs @@ -0,0 +1,159 @@ +using DemaConsulting.Speech.SynthesisSubsystem; + +namespace DemaConsulting.Speech.Tests.SynthesisSubsystem; + +/// +/// Unit tests for . +/// +public class UnavailableSynthesisSessionTests +{ + /// + /// Proves that reports itself as + /// unavailable. + /// + [Fact] + public void UnavailableSynthesisSession_IsAvailable_Read_ReturnsFalse() + { + // Act + var isAvailable = UnavailableSynthesisSession.Instance.IsAvailable; + + // Assert + Assert.False(isAvailable); + } + + /// + /// Proves that always reports + /// , since it never transitions. + /// + [Fact] + public void UnavailableSynthesisSession_State_Read_ReturnsCreated() + { + // Act + var state = UnavailableSynthesisSession.Instance.State; + + // Assert + Assert.Equal(SynthesisSessionState.Created, state); + } + + /// + /// Proves that subscribing to and unsubscribing from + /// is a safe no-op, since this session never raises it. + /// + [Fact] + public void UnavailableSynthesisSession_StateChanged_SubscribeAndUnsubscribe_DoesNotThrow() + { + // Arrange + var session = UnavailableSynthesisSession.Instance; + EventHandler handler = (_, _) => { }; + + // Act + var exception = Record.Exception(() => + { + session.StateChanged += handler; + session.StateChanged -= handler; + }); + + // Assert + Assert.Null(exception); + } + + /// + /// Proves that calling throws + /// . + /// + [Fact] + public async Task UnavailableSynthesisSession_SpeakAsync_Always_ThrowsSpeechSynthesizerUnavailableException() + { + // Arrange + var session = UnavailableSynthesisSession.Instance; + + // Act & Assert + await Assert.ThrowsAsync( + () => session.SpeakAsync("hello", TestContext.Current.CancellationToken)); + } + + /// + /// Proves that calling with a null text still + /// validates the argument, throwing rather than the + /// unavailable fault. + /// + [Fact] + public async Task UnavailableSynthesisSession_SpeakAsync_NullText_ThrowsArgumentNullException() + { + // Arrange + var session = UnavailableSynthesisSession.Instance; + + // Act & Assert + await Assert.ThrowsAsync( + () => session.SpeakAsync(null!, TestContext.Current.CancellationToken)); + } + + /// + /// Proves that calling throws + /// . + /// + [Fact] + public async Task UnavailableSynthesisSession_SynthesizeAsync_Always_ThrowsSpeechSynthesizerUnavailableException() + { + // Arrange + var session = UnavailableSynthesisSession.Instance; + + // Act & Assert + await Assert.ThrowsAsync( + () => session.SynthesizeAsync("hello", TestContext.Current.CancellationToken)); + } + + /// + /// Proves that calling with a null text + /// still validates the argument, throwing rather than + /// the unavailable fault. + /// + [Fact] + public async Task UnavailableSynthesisSession_SynthesizeAsync_NullText_ThrowsArgumentNullException() + { + // Arrange + var session = UnavailableSynthesisSession.Instance; + + // Act & Assert + await Assert.ThrowsAsync( + () => session.SynthesizeAsync(null!, TestContext.Current.CancellationToken)); + } + + /// + /// Proves that calling with no operation in + /// flight is a safe no-op. + /// + [Fact] + public async Task UnavailableSynthesisSession_StopAsync_NoSessionInFlight_IsNoOp() + { + // Arrange + var session = UnavailableSynthesisSession.Instance; + + // Act + var exception = await Record.ExceptionAsync(() => session.StopAsync(TestContext.Current.CancellationToken)); + + // Assert + Assert.Null(exception); + } + + /// + /// Proves that disposing the shared instance is a safe no-op, even when called more than + /// once, so a host wrapping it in an await using block never fails. + /// + [Fact] + public async Task UnavailableSynthesisSession_DisposeAsync_CalledTwice_DoesNotThrow() + { + // Arrange + var session = UnavailableSynthesisSession.Instance; + + // Act + var exception = await Record.ExceptionAsync(async () => + { + await session.DisposeAsync(); + await session.DisposeAsync(); + }); + + // Assert + Assert.Null(exception); + } +}