diff --git a/.github/workflows/publish-to-pypi.yml b/.github/workflows/publish-to-pypi.yml new file mode 100644 index 00000000..0c87d5a3 --- /dev/null +++ b/.github/workflows/publish-to-pypi.yml @@ -0,0 +1,85 @@ +name: Publish Python 🐍 distribution 📦 to PyPI and TestPyPI + +on: + push: + workflow_dispatch: + +jobs: + build: + name: Build distribution 📦 + runs-on: ubuntu-latest + + steps: + - uses: actions/checkout@v6 + with: + persist-credentials: false + - name: Set up Python + uses: actions/setup-python@v6 + with: + python-version: "3.x" + - name: Install pypa/build + run: >- + cd src/modules/python && + python3 -m + pip install + build + --user + - name: Build a binary wheel and a source tarball + run: >- + cd src/modules/python && + python3 -m build + - name: Store the distribution packages + uses: actions/upload-artifact@v5 + with: + name: python-package-distributions + path: src/modules/python/dist/ + + publish-to-pypi: + name: >- + Publish Python 🐍 distribution 📦 to PyPI + #if: startsWith(github.ref, 'refs/tags/') # only publish to PyPI on tag pushes + needs: + - build + runs-on: ubuntu-latest + environment: + name: pypi + url: https://pypi.org/p/pySpeechModule + permissions: + id-token: write # IMPORTANT: mandatory for trusted publishing + + steps: + - name: Download all the dists + uses: actions/download-artifact@v6 + with: + name: python-package-distributions + path: src/modules/python/dist/ + - name: Publish distribution 📦 to PyPI + uses: pypa/gh-action-pypi-publish@release/v1 + with: + packages-dir: src/modules/python/dist/ + + publish-to-testpypi: + name: Publish Python 🐍 distribution 📦 to TestPyPI + #if: startsWith(github.ref, 'refs/tags/') # only publish to PyPI on tag pushes + needs: + - build + runs-on: ubuntu-latest + + environment: + name: testpypi + url: https://test.pypi.org/p/pySpeechModule + + permissions: + id-token: write # IMPORTANT: mandatory for trusted publishing + + steps: + - name: Download all the dists + uses: actions/download-artifact@v6 + with: + name: python-package-distributions + path: src/modules/python/dist/ + - name: Publish distribution 📦 to TestPyPI + uses: pypa/gh-action-pypi-publish@release/v1 + with: + repository-url: https://test.pypi.org/legacy/ + packages-dir: src/modules/python/dist/ diff --git a/.readthedocs.yaml b/.readthedocs.yaml new file mode 100644 index 00000000..7e433bbd --- /dev/null +++ b/.readthedocs.yaml @@ -0,0 +1,13 @@ +version: "2" + +build: + os: "ubuntu-22.04" + tools: + python: "3.10" + +python: + install: + - requirements: doc/pySpeechModule/requirements.txt + +sphinx: + configuration: doc/pySpeechModule/source/conf.py \ No newline at end of file diff --git a/README.md b/README.md index af7fc099..6dee2e5e 100644 --- a/README.md +++ b/README.md @@ -40,6 +40,7 @@ These speech syntheses are supported: - Pico - Piper - Swift +- Kitten Documentation ------------- @@ -60,6 +61,8 @@ The python binding documentation is available on the shell with `pydoc3 speechd` (or `pydoc speechd`) and online: the [speechd.client module documentation](http://htmlpreview.github.io/?https://github.com/brailcom/speechd/blob/master/doc/speechd.client.html) +Also python documentation for writing server modules can be found [here](https://pyspeechmodule.readthedocs.io/en/latest/index.html) + The key features and the supported TTS engines, output subsystems, client interfaces and client applications known to work with Speech Dispatcher are listed in [overview of speech-dispatcher](README.overview.md) as well as voices @@ -170,7 +173,7 @@ Development team: * Andrei Kholodnyi Contributors: Trevor Saunders, Lukas Loehrer,Gary Cramblitt, Olivier Bert, Jacob -Schmude, Steve Holmes, Gilles Casse, Rui Batista, Marco Skambraks ...and many +Schmude, Steve Holmes, Gilles Casse, Rui Batista, Marco Skambraks, John Settlemyer ...and many others. Licensing diff --git a/doc/pySpeechModule/Makefile b/doc/pySpeechModule/Makefile new file mode 100644 index 00000000..d0c3cbf1 --- /dev/null +++ b/doc/pySpeechModule/Makefile @@ -0,0 +1,20 @@ +# Minimal makefile for Sphinx documentation +# + +# You can set these variables from the command line, and also +# from the environment for the first two. +SPHINXOPTS ?= +SPHINXBUILD ?= sphinx-build +SOURCEDIR = source +BUILDDIR = build + +# Put it first so that "make" without argument is like "make help". +help: + @$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) + +.PHONY: help Makefile + +# Catch-all target: route all unknown targets to Sphinx using the new +# "make mode" option. $(O) is meant as a shortcut for $(SPHINXOPTS). +%: Makefile + @$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O) diff --git a/doc/pySpeechModule/make.bat b/doc/pySpeechModule/make.bat new file mode 100644 index 00000000..747ffb7b --- /dev/null +++ b/doc/pySpeechModule/make.bat @@ -0,0 +1,35 @@ +@ECHO OFF + +pushd %~dp0 + +REM Command file for Sphinx documentation + +if "%SPHINXBUILD%" == "" ( + set SPHINXBUILD=sphinx-build +) +set SOURCEDIR=source +set BUILDDIR=build + +%SPHINXBUILD% >NUL 2>NUL +if errorlevel 9009 ( + echo. + echo.The 'sphinx-build' command was not found. Make sure you have Sphinx + echo.installed, then set the SPHINXBUILD environment variable to point + echo.to the full path of the 'sphinx-build' executable. Alternatively you + echo.may add the Sphinx directory to PATH. + echo. + echo.If you don't have Sphinx installed, grab it from + echo.https://www.sphinx-doc.org/ + exit /b 1 +) + +if "%1" == "" goto help + +%SPHINXBUILD% -M %1 %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O% +goto end + +:help +%SPHINXBUILD% -M help %SOURCEDIR% %BUILDDIR% %SPHINXOPTS% %O% + +:end +popd diff --git a/doc/pySpeechModule/readme.md b/doc/pySpeechModule/readme.md new file mode 100644 index 00000000..889ae432 --- /dev/null +++ b/doc/pySpeechModule/readme.md @@ -0,0 +1,17 @@ +# building the docs locally. + +``` +python -m venv ./venv +source venv/bin/activate + +pip install -U sphinx +pip install sphinx-autobuild +pip install sphinx-rtd-theme myst-parser + +# using sphinx +sphinx-build -M html source build + +# using autobuild +sphinx-autobuild --host 0.0.0.0 --port 8080 source build +# go to localhost:8080/ to see a live preview of the doc's +``` \ No newline at end of file diff --git a/doc/pySpeechModule/requirements.txt b/doc/pySpeechModule/requirements.txt new file mode 100644 index 00000000..9a27eda0 --- /dev/null +++ b/doc/pySpeechModule/requirements.txt @@ -0,0 +1,4 @@ +sphinx==7.1.2 +sphinx-rtd-theme==1.3.0rc1 +myst-parser +soundfile \ No newline at end of file diff --git a/doc/pySpeechModule/source/Advanced_Tutorial.rst b/doc/pySpeechModule/source/Advanced_Tutorial.rst new file mode 100644 index 00000000..d207ace2 --- /dev/null +++ b/doc/pySpeechModule/source/Advanced_Tutorial.rst @@ -0,0 +1,286 @@ +Advanced Tutorial. +================== + +In this tutorial we will create a speechd module that can generate tts output, is able to change speed and voices. We will be using `Kokoro `_ a open-weight TTS model with 82 million parameters to generated audio for us. + +Setup the directories and venv. +------------------------------- + +.. code-block:: bash + + mkdir kokoro_speechd_module + cd kokoro_speechd_module + + python3.12 -m venv venv + source venv/bin/activate + +.. note:: In this case we are using python 3.12. As of writing this, 3.14 is somewhat painful to use due to having to build many packages, I would recommend 3.12 for now. + +Install dependencies. +--------------------- + +.. code-block:: bash + + pip install --upgrade pip + pip install pySpeechModule "kokoro>=0.9.4" soundfile + +Verify that kokoro is working. +------------------------------ + +Start python and run the following. You should get a .wav file back that says "hello world". + +.. code-block:: python3 + + from kokoro import KPipeline + import soundfile as sf + import torch + pipeline = KPipeline(lang_code='a') + text = "Hello world." + generator = pipeline(text, voice='af_heart') + for i, (gs, ps, audio) in enumerate(generator): + print(i, gs, ps) + sf.write(f'{i}.wav', audio, 24000) + + +Create your module. +------------------- + +.. code-block:: python3 + :caption: kokoro_module.py + :name: kokoro-py + + #!/home/user/src/kokoro_speechd_module/venv/bin/python + + import soundfile as sf + from kokoro import KPipeline + import logging + import torch + from pySpeechModule import SpeechServer, SpeechDispatch, parse_ssml + import io + + class KokoroDispatch(SpeechDispatch): + def __init__(self): + self.pipeline = KPipeline(lang_code='a') + self.speed = 1.0 + self.voice = 'af_heart' + self.voices = [{"name": "af_heart", "language": "en", "variant": "none", "kokoro_lang": 'a'}, + {"name": "af_alloy", "language": "en", "variant": "none", "kokoro_lang": 'a'}] + self.max_ahead = 60 + self.min_ahead = 30 + + def change_speed(self, speed): + # callback for changing the speed. + # speechd gives speed between -100 and 100 + # we have to convert that into a multiplier + speed = int(speed) + if speed < 0: + self.speed = 1.0 + (float(speed) / 100.0) * 0.5 + else: + self.speed = 1.0 + (float(speed) / 100.0) * 2.0 + + def change_voice(self, voice_name): + # call back for changing the voice. + logging.critical(f"Changed voice_name: {voice_name}") + try: + filtered = [voice for voice in self.voices if voice["name"] == voice_name][0] + except: + filtered = self.voices[0] + self.voice = filtered["name"] + self.pipeline = KPipeline(lang_code=filtered["kokoro_lang"]) + + def speak(self, ssml): + try: + # parse the ssml into chunk of text each with a mark. + for item in parse_ssml(ssml): + text = item['text'] + mark = item['mark'] + # setup Kokoro's generator + generator = self.pipeline(text, voice=self.voice, speed=self.speed) + # iterate over Kokoro's generator. + for i, (gs, ps, audio) in enumerate(generator): + # use sound file to convert the output audio to a format we can use + buff = io.BytesIO() # create a in memory buffer + buff.name = "tmp.wav" + samplerate = 24000 + sf.write(buff, audio, samplerate, subtype="PCM_16") # write the current audio to the buffer + # you can use sf.available_subtypes() to get a list available subtypes, and match your models input audio format. + buff.seek(0) # set the buffer back to start + audio_data, output_sample_rate = sf.read(buff, dtype="int16") # read the buffer in the correct format. + yield {"audio": audio_data, "rate": output_sample_rate, "mark": mark} # you must yield data even if you have a single output. + except Exception as e: + print("An error occurred:", e) + + SpeechServer().start(KokoroDispatch()) + +This should like somewhat similar to the previous example. But lets go through the methods and explain each of them. + +__init__() +---------- + +.. code-block:: python3 + + def __init__(self): + self.pipeline = KPipeline(lang_code='a') + self.speed = 1.0 + self.voice = 'af_heart' + self.voices = [{"name": "af_heart", "language": "en", "variant": "none", "kokoro_lang": 'a'}, + {"name": "af_alloy", "language": "en", "variant": "none", "kokoro_lang": 'a'}] + self.max_ahead = 60 + self.min_ahead = 30 + +Here we are creating ``pipeline`` which is the core class for our Kokoro pipeline. In addition to that we create a few other attributes. You should pay close attention to ``self.voices``, ``self.max_ahead`` and ``self.min_ahead``. These are special attributes used by the module. ``self.voices`` is used to generated the list of voices you would get when a client asks for a list of voices(for example ``spd-say -L``). ``max_ahead`` and ``min_ahead`` define the max and min number of seconds the generated audio can get ahead of your played audio. This saves your cpu from running too hard and is automatically handled in the background if you provide these values. A good default is to use 60 and 30. + +Change speed. +------------- + +.. code-block:: python3 + + def change_speed(self, speed): + # callback for changing the speed. + # speechd gives speed between -100 and 100 + # we have to convert that into a multiplier + speed = int(speed) + if speed < 0: + self.speed = 1.0 + (float(speed) / 100.0) * 0.5 + else: + self.speed = 1.0 + (float(speed) / 100.0) * 2.0 + +This is a callback that gets called when the client asks to change the speed. There is not much to this since we are simply setting a variable. The only important thing to note is that speed will be between -100 and 100. + +Change voice. +------------- + +.. code-block:: python3 + + def change_voice(self, voice_name): + # call back for changing the voice. + try: + filtered = [voice for voice in sel.voices if voice["name"] == voice_name][0] + except: + filtered = self.voices[0] + self.voice = filtered["name"] + self.pipeline = KPipeline(lang_code=filtered["kokoro_lang"]) + +This is a callback that will get called when we are asked to change the voice being used. Here we also are updating the pipeline since the ``lang_code`` is being set in the pipeline and the voice and language are linked together. In this example we add a extra key to voices `kokoro_lang`. This is necessary because the language code for kokoros is different from the speechd language code. `kokoro_lang` is not a necessary key for the speechd module. + +Speak. +------ + +.. code-block:: python3 + + def speak(self, ssml): + try: + # parse the ssml into chunk of text each with a mark. + for item in parse_ssml(ssml): + text = item['text'] + mark = item['mark'] + # setup Kokoro's generator + generator = self.pipeline(text, voice=self.voice, speed=self.speed) + # iterate over Kokoro's generator. + for i, (gs, ps, audio) in enumerate(generator): + # use sound file to convert the output audio to a format we can use + buff = io.BytesIO() # create a in memory buffer + buff.name = "tmp.wav" + samplerate = 24000 + sf.write(buff, audio, samplerate, subtype="PCM_16") # write the current audio to the buffer + # you can use sf.available_subtypes() to get a list available subtypes, and match your models input audio format. + buff.seek(0) # set the buffer back to start + audio_data, output_sample_rate = sf.read(buff, dtype="int16") # read the buffer in the correct format. + yield {"audio": audio_data, "rate": output_sample_rate, "mark": mark} # you must yield data even if you have a single output. + except Exception as e: + print("An error occurred:", e) + +We showed this callback in the previous tutorial. It is a callback which is called when we are asked to speak something. + +Lets now break it all this down. + +.. code-block:: python3 + + for item in parse_ssml(ssml): + +``parse_ssml`` is a convenience function which you can use to turn the ssml(which is really just like xml) into a python list of dict's. Each dict has the text and the mark. The mark is a way of telling speechd where you are at in the text, you can think of it as a index. if you have a few sentence's of text and you speak up to the second sentence you would send the mark linked to the second sentence to tell speechd what has been spoken. + +.. code-block:: python3 + + generator = self.pipeline(text, voice=self.voice, speed=self.speed) + # iterate over Kokoro's generator. + for i, (gs, ps, audio) in enumerate(generator): + +This is how kokoro generates audio. It also does it own chunking. Which is awesome since we have no problem sending chunked audio out. We only have on issue here kokoro generates audio in a format we can not use directly( but we can convert it). + +.. code-block:: python3 + + buff = io.BytesIO() + buff.name = "tmp.wav" + samplerate = 24000 + sf.write(buff, audio, samplerate, subtype="PCM_16") + # you can use sf.available_subtypes() to get a list available subtypes, and match your models input audio format. + buff.seek(0) # set the buffer back to start + audio_data, output_sample_rate = sf.read(buff, dtype="int16") # read the buffer in the correct format. + +This code converts the audio to a format we can use. The first part of this we are creating a in memory file, using pythons built-in ``BytesIO``. We then can use sound file to write our audio to that BytesIO object. We use ``subtype`` to specify what the input audio format was. Next we need to reset the position on our in memory file, we do this with a seek. Then finally we are able to read the file back out in a format we can use. + +.. tip:: you can use ``sf.available_subtypes()`` to get a list available subtypes, and match your models input audio format. + +.. code-block:: python3 + + yield {"audio": audio_data, "rate": output_sample_rate, "mark": mark} + +This last line is used to "return" the output of a chunk of audio along with the sample rate and mark. You can use a yield inside a loop to return multiple chunks of audio. + +.. code-block:: python3 + + SpeechServer().start(KokoroDispatch()) + +Finally we start the server and pass a instance of the class. + +Setup your config. +------------------ + +When speechd run's you script it needs to file to be executable. You can use ``chmod`` to do this. + +.. code-block:: python3 + + chmod +x kokoro_module.py + +Finally add a configure line to speechd's config file. speechd's configure file is usually found in ``~/.config/speech-dispatcher/speechd.conf``, but may be in a different location depending on your distro. + +.. code-block:: bash + :caption: speechd.conf + :name: speechd-conf-adv + + Timeout 0 + AddModule "kokoropy" "/home/user/src/kokoro_speechd_module/kokoro_module.py" "kokoropy.conf" + +.. warning:: You should change ``/home/user/src/kokoro_speechd_module/kokoro_module.py`` to match your path. + +Testing the module. +------------------- + +First stop any running speech-dispatch processes. + +.. code-block:: bash + + pkill -9 speech-dispatch + +Then send a speak command to speechd using ``spd-say``. + +.. code-block:: bash + + spd-say -o "kokoropy" -y "af_heart" "Hello world." + +You should hear it speaking. + +Lets try a diffrent voice. + +.. code-block:: bash + + spd-say -o "kokoropy" -y "af_alloy" "Hello world." + +Lets make it speak faster. + +.. code-block:: bash + + spd-say -o "kokoropy" -r 25 -y "af_heart" "Hello world." + +Congratulations, you have now create a speechd module that can generate tts output, is able to change speed and voices diff --git a/doc/pySpeechModule/source/Basic_Tutorial.rst b/doc/pySpeechModule/source/Basic_Tutorial.rst new file mode 100644 index 00000000..223e9f4c --- /dev/null +++ b/doc/pySpeechModule/source/Basic_Tutorial.rst @@ -0,0 +1,176 @@ +Basic Tutorial +============== + +This is the same as the quick start but with more explanation on everything. In this Tutorial we will create a speechd module which plays a audio file when ever it is asked to speak something. + + +Setting up a directory. +----------------------- + +Created a directory and cd into it. + +.. code-block:: bash + + mkdir my_first_speechd_module + cd my_first_speechd_module + +.. note:: You can place your directory any where you like. + +Create a venv. +-------------- + +.. code-block:: bash + + python3.12 -m venv venv + source venv/bin/activate + +.. note:: In this case we are using python 3.12, but other versions will also likely work. But as of writing this, 3.14 is somewhat painful to use due to having to build many packages. + +Install dependencies. +--------------------- + +.. code-block:: bash + + pip install --upgrade pip + pip install pySpeechModule soundfile + +Create your module. +------------------- + +Now lets create your module. We are going to walk through each step here, but if you just want the full code the jump to the :ref:`full-code` + +Add a shebang at the top of your python file. + +.. code-block:: python3 + :caption: dummy.py + :name: dummy-py0 + + #!/home/user/src/my_first_speechd_module/venv/bin/python + +.. attention:: + + In our case we are using ``/home/user/src/my_first_speechd_module`` this should be changed to match your directory structure. + + The last part ``venv/bin/python`` points to your venv's python binary. When python starts it automatically sets up your venv if you use this binary. This is exactly what we want to do. + +Now add your imports. + +.. code-block:: python3 + :caption: dummy.py + :name: dummy-py1 + + import logging + import soundfile as sf + from pySpeechModule import SpeechServer, SpeechDispatch + +Here we are using pySpeechModule, soundfile and standard python logging. + +Now lets create a class. + +.. code-block:: python3 + :caption: dummy.py + :name: dummy-py2 + + class Dummy(SpeechDispatch): + +.. warning:: You can name your class anything, but make sure your class inherits from ``SpeechDispatch``. + +Now `download `_ and save the file as ``deep_learning.wav``. Then add a ``__init__`` method to your class. + +.. code-block:: python3 + :caption: dummy.py + :name: dummy-py3 + + def __init__(self): + logging.info("Dummy Started!") + self.file_path = "deep_learning.wav" + +.. note:: ``__init__`` is the were you will want to place anything that should be long lived. Once your class is initialized and passed to the server it will live for the life of the module. In this simple example the only thing we will place here is ``self.file_path`` + +Now lets create the ``speak`` method. This method will get called by the server when ever we get a request to speak text. It will then yield back data with the audio. The yield will return a dict with keys for ``audio`` for audio data, ``rate`` for the audio's sample rate, and ``mark`` for the text index mark. We will explain the mark in a later example, for now it can be set to a empty string. + +.. code-block:: python3 + :caption: dummy.py + :name: dummy-py4 + + def speak(self,text): + try: + data, samplerate = sf.read(self.file_path, dtype='int16') + yield {"audio": data, "rate": samplerate, "mark": ""} + except Exception as e: + logging.error("An error occurred:", e) + +.. warning:: when creating audio you must output it in ``dtype='int16'`` in this simple example we can just read into that format. In the next example we will show you how to convert audio to get that format. + +Last we need to create the server and create a instance of our class. + +.. code-block:: python3 + :caption: dummy.py + :name: dummy-py5 + + SpeechServer().start(Dummy()) + +.. _full-code: + +Full Code. +---------- + +Here is the complete code. You will also need to `download `_ if you have not already and make sure to name it ``deep_learning.wav``. + +.. code-block:: python3 + :caption: dummy.py + :name: dummy-py + + #!/home/user/src/my_first_speechd_module/venv/bin/python + + import logging + import soundfile as sf + from speechd_module import SpeechServer, SpeechDispatch + + class Dummy(SpeechDispatch): + def __init__(self): + logging.info("Dummy Started!") + self.file_path = "deep_learning.wav" + def speak(self,text): + try: + data, samplerate = sf.read(self.file_path, dtype='int16') + yield {"audio": data, "rate": samplerate, "mark": ""} + except Exception as e: + logging.error("An error occurred:", e) + + SpeechServer().start(Dummy()) + +Now you need to make the file executable + +.. code-block:: bash + + chmod +x dummy.py + +Finally add a configure line to speechd's configure file. speechd's configure file is usually found in ``~/.config/speech-dispatcher/speechd.conf``, but may be in a different location depending on your distro. + +.. code-block:: bash + :caption: speechd.conf + :name: speechd-conf-basic + + Timeout 0 + AddModule "dummypy" "/home/user/src/my_first_speechd_module/dummy.py" "dummypy.conf" + +.. warning:: You should change ``/home/user/src/my_first_speechd_module/dummy.py`` to match your path. + +Testing the module. +------------------- + +First stop any running speech-dispatch processes. + +.. code-block:: bash + + pkill -9 speech-dispatch + +Then send a speak command to speechd using ``spd-say``. + +.. code-block:: bash + + spd-say -o "dummypy" -y "en" "Hello world." + +While this is pretty cool its not very useful to just play the same audio file for all speak commands. In the next tutorial we will create a module that actually speaks using Kokoro as our tts model. + diff --git a/doc/pySpeechModule/source/Quick_Start.rst b/doc/pySpeechModule/source/Quick_Start.rst new file mode 100644 index 00000000..f19c0adc --- /dev/null +++ b/doc/pySpeechModule/source/Quick_Start.rst @@ -0,0 +1,65 @@ +Quick Start. +============ + +If you have not already see the Basic_Tutorial for a more detailed explaintion of whats going on. + +.. code-block:: bash + + mkdir my_first_speechd_module + cd my_first_speechd_module + python3.12 -m venv venv + source venv/bin/activate + pip install --upgrade pip + pip install pySpeechModule soundfile + +You need to `download `_ and make sure to name it ``deep_learning.wav``. + +.. code-block:: python3 + :caption: dummy.py + :name: dummy-py-quick + + #!/home/user/src/my_first_speechd_module/venv/bin/python + + import logging + import soundfile as sf + from pySpeechModule import SpeechServer, SpeechDispatch + + class Dummy(SpeechDispatch): + def __init__(self): + logging.info("Dummy Started!") + self.file_path = "deep_learning.wav" + def speak(self,text): + try: + data, samplerate = sf.read(self.file_path, dtype='int16') + yield {"audio": data, "rate": samplerate, "mark": ""} + except Exception as e: + logging.error("An error occurred:", e) + + SpeechServer().start(Dummy()) + +Now you need to make the file executable + +.. code-block:: bash + + chmod +x dummy.py + +Finally add a configure line to speechd's configure file. speechd's configure file is usually found in ``~/.config/speech-dispatcher/speechd.conf``, but may be in a different location depending on your distro. + +.. code-block:: bash + :caption: speechd.conf + :name: speechd-conf-quick + + Timeout 0 + AddModule "dummypy" "/home/user/src/my_first_speechd_module/dummy.py" "dummypy.conf" + +To test the module. First stop any running speech-dispatch processes. + +.. code-block:: bash + + pkill -9 speech-dispatch + +Then send a speak command to speechd using ``spd-say``. + +.. code-block:: bash + + spd-say -o "dummypy" -y "en" "Hello world." \ No newline at end of file diff --git a/doc/pySpeechModule/source/api.rst b/doc/pySpeechModule/source/api.rst new file mode 100644 index 00000000..b7fa641d --- /dev/null +++ b/doc/pySpeechModule/source/api.rst @@ -0,0 +1,7 @@ +API Reference +============= + +.. automodule:: pySpeechModule + :members: + :undoc-members: + :show-inheritance: \ No newline at end of file diff --git a/doc/pySpeechModule/source/conf.py b/doc/pySpeechModule/source/conf.py new file mode 100644 index 00000000..c20c79fb --- /dev/null +++ b/doc/pySpeechModule/source/conf.py @@ -0,0 +1,38 @@ +import sys, os +from pathlib import Path + +sys.path.insert(0, os.path.abspath('../../../src/modules/python/src/')) +print(sys.path) + +# Configuration file for the Sphinx documentation builder. +# +# For the full list of built-in configuration values, see the documentation: +# https://www.sphinx-doc.org/en/master/usage/configuration.html + +# -- Project information ----------------------------------------------------- +# https://www.sphinx-doc.org/en/master/usage/configuration.html#project-information + +project = 'pySpeechModule' +copyright = '2026, John Settlemyer' +author = 'John Settlemyer' + +# -- General configuration --------------------------------------------------- +# https://www.sphinx-doc.org/en/master/usage/configuration.html#general-configuration + +extensions = [ + 'sphinx.ext.duration', + 'sphinx.ext.doctest', + 'sphinx.ext.autodoc', + 'sphinx.ext.autosummary', + 'sphinx.ext.intersphinx', + 'myst_parser', +] + +templates_path = ['_templates'] +exclude_patterns = [] + +# -- Options for HTML output ------------------------------------------------- +# https://www.sphinx-doc.org/en/master/usage/configuration.html#options-for-html-output + +html_theme = 'sphinx_rtd_theme' +html_static_path = ['_static'] \ No newline at end of file diff --git a/doc/pySpeechModule/source/deep_learning.wav b/doc/pySpeechModule/source/deep_learning.wav new file mode 100644 index 00000000..694fa559 Binary files /dev/null and b/doc/pySpeechModule/source/deep_learning.wav differ diff --git a/doc/pySpeechModule/source/index.rst b/doc/pySpeechModule/source/index.rst new file mode 100644 index 00000000..66fedd33 --- /dev/null +++ b/doc/pySpeechModule/source/index.rst @@ -0,0 +1,15 @@ +.. py-speech-module documentation master file, created by + sphinx-quickstart on Tue Sep 15 15:07:09 2026. + You can adapt this file completely to your liking, but it should at least + contain the root `toctree` directive. + +py-speech-module documentation +============================== + +pySpeechModule, is a framework for creating speechd modules in python without having to implement the speechd protocol your self. + +.. toctree:: + Quick_Start + Basic_Tutorial + Advanced_Tutorial + api \ No newline at end of file diff --git a/src/modules/python/LICENSE b/src/modules/python/LICENSE new file mode 100644 index 00000000..3c2c4149 --- /dev/null +++ b/src/modules/python/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 John Settlemyer + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. diff --git a/src/modules/python/README.md b/src/modules/python/README.md new file mode 100644 index 00000000..a79f4f24 --- /dev/null +++ b/src/modules/python/README.md @@ -0,0 +1,4 @@ +# pySpeechModule +A framework for writing python speechd modules. + +documentation: https://pyspeechmodule.readthedocs.io/en/latest/index.html diff --git a/src/modules/python/pyproject.toml b/src/modules/python/pyproject.toml new file mode 100644 index 00000000..8d12c9eb --- /dev/null +++ b/src/modules/python/pyproject.toml @@ -0,0 +1,19 @@ +[project] +name = "pySpeechModule" +version = "1.0.2" +authors = [ + { name="John Settlemyer", email="author@example.com" }, +] +dependencies = [] +description = "A framework for creating speechd modules" +readme = "README.md" +requires-python = ">=3.12" +classifiers = [ + "Programming Language :: Python :: 3", +] +license = "MIT" +license-files = ["LICEN[CS]E*"] + +[project.urls] +Homepage = "https://github.com/brailcom/speechd" +Issues = "https://github.com/brailcom/speechd/issues" diff --git a/src/modules/python/src/pySpeechModule/.gitignore b/src/modules/python/src/pySpeechModule/.gitignore new file mode 100644 index 00000000..ed8ebf58 --- /dev/null +++ b/src/modules/python/src/pySpeechModule/.gitignore @@ -0,0 +1 @@ +__pycache__ \ No newline at end of file diff --git a/src/modules/python/src/pySpeechModule/__init__.py b/src/modules/python/src/pySpeechModule/__init__.py new file mode 100644 index 00000000..72662e66 --- /dev/null +++ b/src/modules/python/src/pySpeechModule/__init__.py @@ -0,0 +1,37 @@ +# MIT License + +# Copyright (c) 2026 John Settlemyer + +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: + +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. + +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + +from .speechd_server import SpeechServer +from .speechd import SpeechDispatch +from .speechd_action_worker import ActionWorker +from .speechd_execution_worker import ExecutionWorker +from .speechd_utilities import hdlc_escape, parse_ssml, strip_ssml + +__all__ = [ + "SpeechServer", + "SpeechDispatch", + "ActionWorker", + "ExecutionWorker", + "hdlc_escape", + "parse_ssml", + "strip_ssml", + ] \ No newline at end of file diff --git a/src/modules/python/src/pySpeechModule/speechd.py b/src/modules/python/src/pySpeechModule/speechd.py new file mode 100644 index 00000000..9ce6645c --- /dev/null +++ b/src/modules/python/src/pySpeechModule/speechd.py @@ -0,0 +1,177 @@ +# MIT License + +# Copyright (c) 2026 John Settlemyer + +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: + +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. + +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + +import logging +import time + +class SpeechDispatch(): + def __init__(self): + """ + you should place the voices list here in __init__ + + .. code-block:: python + + self.voices = [{"name": "Testing", "language": "en", "variant": "none"}, + {"name": "Testing2", "language": "en", "variant": "none"}] + + You should also set your max_ahead and min_ahead here. + + .. code-block:: python + + self.max_ahead = 60 # this sets the max number of second your generation will get ahead of played audio. + self.min_ahead = 30 # this sets how small the ahead number maybe before starting generation again. + + You can also place model init or what ever else you would like to init here. + """ + pass + def _set_init(self,queue, server): + """ + Set the action_queue and server. + + :meta private: + """ + + self.action_queue = queue + self._server = server + def configure(self, configure_file: str): + """ + This method get passed in the file path to the modules dot conf file + If you need to use it you should open the file and parse here. + """ + pass + def change_voice(self, voice: str): + """ + This method gets passed the name of a voice when the user has requested to change the voice. + + :param voice: String containing the voice name. + :type String: String + """ + pass + def change_language(self, language: str): + """ + This method gets passed a language when the user has requested to change the language. + + :param voice: String containing the language. + :type String: String + """ + pass + def change_speed(self, speed: str): + """ + This method gets passed a speed when the user requests to change the speed. + The speed will be between -100 and 100 + + :param voice: String containing the new speed. + :type String: String + """ + pass + def change_pitch(self, pitch: str): + """ + This method gets passed a pitch when the user requests to change the pitch + + :param voice: String containing the new pitch. + :type String: String + """ + pass + def _parse_settings(self,options: dict): + """ + When passed the settings it calls the convenience methods associated with that setting. + + :meta private: + """ + if 'synthesis_voice' in options: + self.change_voice(options['synthesis_voice']) + if 'language' in options: + self.change_language(options['language']) + if 'rate' in options: + self.change_speed(options['rate']) + if 'pitch' in options: + self.change_pitch(options['pitch']) + def _settings(self,options): + """ + Passed the setting, sends it to the parsers. + + :meta private: + """ + self._parse_settings(options) + self.settings(options) + + def settings(self,options: dict): + """ + This gets passed the full settings options + + :param options: Dict containing all the settings options that were sent. + :type Dictionary: Dictionary. + """ + pass + def _speak(self,text: str): + """ + This is called to generate speech + It then iterate's through speech chunks returned by the user. + + :meta private: + """ + self._server.stop_set(False) + self._ahead = 0 + self.action_queue.put({"command": "begin"}) + it = self.speak(text) + while (True): + try: + if (self._server.stop_get()): + break + start_time = time.time() + item = next(it) # get the next audio chunk + end_time = time.time() + total_time = end_time-start_time # total time it took to generate that audio chunk + produced_time = (len(item['audio']) / item['rate']) # amount of audio generate + logging.debug(f"Speak chunk took {total_time} to produce {produced_time} for mark {item['mark']}") + self._ahead = self._ahead + produced_time + self._ahead = self._ahead - total_time + logging.debug(f"Ahead: {self._ahead}") + self.action_queue.put({"command": "speak", "val": item}) + # sleep if we have too much audio generated, only do this is max_ahead and min_ahead are set. + if (hasattr(self, "max_ahead") and hasattr(self, "min_ahead")): + if (self._ahead >= self.max_ahead): + logging.debug(f"Ahead by too much sleeping") + while (self._ahead >= self.min_ahead): + time.sleep(1) + self._ahead = self._ahead - 1 + if (self._server.stop_get()): + break; + except StopIteration: + logging.debug("finished speak chunking") + break + self.action_queue.put({"command": "end"}) + def speak(self,text: str): + """ + This method get called to get the user to generate audio. + Data must be yield'ed in the format + + .. code-block:: python + + yield {"audio": , "rate": , "mark": } + + :param text: String containing ssml text. + :type String: String + :yields: Dict containing keys for "audio", "rate", "mark" + :ytype: Dict + """ + pass \ No newline at end of file diff --git a/src/modules/python/src/pySpeechModule/speechd_action_worker.py b/src/modules/python/src/pySpeechModule/speechd_action_worker.py new file mode 100644 index 00000000..1cd9b4a1 --- /dev/null +++ b/src/modules/python/src/pySpeechModule/speechd_action_worker.py @@ -0,0 +1,63 @@ +# MIT License + +# Copyright (c) 2026 John Settlemyer + +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: + +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. + +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + +import logging +import threading +import queue + +class ActionWorker(threading.Thread): + """ + Worker for running speak commands on the server + + :meta private: + """ + def __init__(self, server): + super().__init__() + self.queue = queue.Queue() + self.server = server + self.daemon = True + + def begin(self): + self.server._begin() + + def end(self): + self.server._end() + + def speak(self,val): + self.server._send_audio(val) + + def stop(self): + """Sends the sentinel value to trigger a graceful shutdown.""" + self.queue.put(None) + + def run(self): + while True: + item = self.queue.get() + logging.debug(f"action worker: got item {item}") + if item is None: + break + if (item['command'] == 'begin'): + self.begin() + if (item['command'] == 'end'): + self.end() + if (item['command'] == 'speak'): + self.speak(item['val']) \ No newline at end of file diff --git a/src/modules/python/src/pySpeechModule/speechd_execution_worker.py b/src/modules/python/src/pySpeechModule/speechd_execution_worker.py new file mode 100644 index 00000000..738b7156 --- /dev/null +++ b/src/modules/python/src/pySpeechModule/speechd_execution_worker.py @@ -0,0 +1,56 @@ +# MIT License + +# Copyright (c) 2026 John Settlemyer + +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: + +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. + +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + +import logging +import threading +import queue + +class ExecutionWorker(threading.Thread): + """ + Worker for running the execution callbacks. + + :meta private: + """ + def __init__(self): + super().__init__() + self.queue = queue.Queue() + self.daemon = True + + def set_callback(self,callback): + self._callback = callback + + def stop(self): + """Sends the sentinel value to trigger a graceful shutdown.""" + self.queue.put(None) + + def run(self): + while True: + item = self.queue.get() + logging.debug(f"execution worker: got item {item}") + if item is None: + break + elif (item['command'] == 'speak'): + self._callback._speak(item['args']) + elif (item['command'] == 'settings'): + self._callback._settings(item['args']) + elif (item['command'] == 'configure'): + self._callback.configure(item['args']) \ No newline at end of file diff --git a/src/modules/python/src/pySpeechModule/speechd_server.py b/src/modules/python/src/pySpeechModule/speechd_server.py new file mode 100644 index 00000000..1664429a --- /dev/null +++ b/src/modules/python/src/pySpeechModule/speechd_server.py @@ -0,0 +1,426 @@ +# MIT License + +# Copyright (c) 2026 John Settlemyer + +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: + +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. + +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + +import sys +import logging +import threading +import select +import queue +import time +import struct + +from .speechd_action_worker import ActionWorker +from .speechd_execution_worker import ExecutionWorker +from .speechd_utilities import hdlc_escape + +stdout = sys.stdout +stdin = sys.stdin +sys.stdout = sys.stderr + +logging.basicConfig( + stream=sys.stderr, + level=logging.DEBUG, + format="[%(asctime)s] %(levelname)s: %(message)s", + datefmt="%Y-%m-%d %H:%M:%S" +) + +class SpeechServer(): + """ + This is the server class, it handles running the protocal. + + .. note:: we set ``sys.stdout = sys.stderr``, this is to ensure that users don't accedentally break the protocal. Anything printed to stdout may break the server. + """ + def __init__(self): + """ + :meta private: + """ + + self.line_count = 0 + + self.execution_worker = ExecutionWorker() + self.action_worker = ActionWorker(server=self) + + self.voices = [] + self.server_lock = threading.RLock() + self.stop_lock = threading.RLock() + self.stop = False # make sure you use stop_lock when using this. + + def _readline(self, block=True): + """ + This is basically pythons readline but with the added ability to do it none blocking. + + :meta private: + """ + + try: + if (not block): + timeout = 0.0 + ready, _, _ = select.select([stdin], [], [], timeout) + if ready: + output = stdin.readline() + logging.debug(f"line {self.line_count}: {output[:-1]}") + self.line_count = self.line_count + 1 + return output + return None + else: + output = stdin.readline() + logging.debug(f"line {self.line_count}: {output[:-1]}") + self.line_count = self.line_count + 1 + return output + except Exception as e: + logging.error(f"_readline: {e}") + + + def _begin(self): + """ + Send command telling speechd we are starting to send audio data. + + :meta private: + """ + + with self.server_lock: + stdout.write("701 BEGIN\n") + stdout.flush() + def _end(self): + """ + Tells speechd we are done sending audio data. + + :meta private: + """ + + with self.server_lock: + stdout.write("702 END\n") + stdout.flush() + def _mark(self, mark): + """ + Tells speechd which mark we are on. + + :meta private: + """ + + if (mark==""): + return + + with self.server_lock: + stdout.write(f"700-{mark}\n700 INDEX MARK\n") + stdout.flush() + + def _send_audio(self, val): + """ + Send audio data to speechd. + + :meta private: + """ + + with self.server_lock: + logging.debug("server: _send_audio") + data = val['audio'] + sample_rate = val['rate'] + mark = val['mark'] + + chunk_size = 5000 + for i in range(0, len(data), chunk_size): + start_time = time.time() + tmp_data = data[i : i + chunk_size] + num_samples = len(tmp_data) + if (type(tmp_data).__name__ == "ndarray"): + # if our data comes from a numpy array use tobytes to turn it into a byte string. + tmp_data = tmp_data.tobytes() + elif (type(tmp_data).__name__ == "list"): + # for a python list we will pack it. + tmp_data = struct.pack('%sh' % len(tmp_data), *tmp_data) + else: + # anything else is a fatal error. + raise Exception("Audio must be list or ndarray.") + + stdout.write(f"705-bits=16\n") + stdout.flush() + stdout.write(f"705-num_channels=1\n") + stdout.flush() + stdout.write(f"705-sample_rate={sample_rate}\n") + stdout.flush() + stdout.write(f"705-num_samples={num_samples}\n") + stdout.flush() + stdout.write(f"705-big_endian=0\n") + stdout.flush() + stdout.write(f"705-AUDIO") + stdout.flush() + stdout.buffer.write(b'\x00') + + tmp_data2 = hdlc_escape(tmp_data) + stdout.buffer.write(tmp_data2) + stdout.write("\n") + stdout.write("705 AUDIO\n") + stdout.flush() + end_time = time.time() + total_time = end_time-start_time + self._process(block=False) + self._mark(mark) + + def _configure(self): + """ + Get the config file path and send it to the client. + + :meta private: + """ + + if len(sys.argv) > 1: + configfile = sys.argv[1] + self.execution_worker.queue.put({"command": "configure", "args": configfile}) + + def parse_set_params(self,ack: int, param_type: str): + """ + Get a settings string. + + :meta private: + """ + + with self.server_lock: + stdout.write(f"{ack} OK RECEIVING {param_type} SETTINGS\n") + stdout.flush() + + output = {} + + while (True): + line = self._readline() + if (line == ".\n"): + break + line_clean = line.rstrip("\n") + var, val = line_clean.split("=", 1) + output[var] = val + + return output + + def _audio(self): + """ + :meta private: + """ + + tmp = self.parse_set_params(207, "AUDIO") + with self.server_lock: + stdout.write("203 OK AUDIO INITIALIZED\n") + stdout.flush() + def _debug(self, line): + """ + We already set up logging so we will not be using this for now. + + :meta private: + """ + pass + + def _loglevel(self): + """ + Sets the logging level from a settings string. + + The logging levels don't quit match up with speechd but we try and get them as close as possible. + see this link for the speechd logging levels. + https://htmlpreview.github.io/?https://github.com/brailcom/speechd/blob/master/doc/speech-dispatcher.html#Log-Levels + + :meta private: + """ + + levels = {'5': logging.DEBUG, '4': logging.INFO, '3': logging.WARNING, '2': logging.ERROR, '1': logging.CRITICAL} + + tmp = self.parse_set_params(207, "LOGLEVEL") + logging.critical(f"Changing log level to {tmp['log_level']}") + if (tmp['log_level'] == '0'): + logging.disable(logging.CRITICAL) # Disable all logging globally. + else: + logging.disable(logging.NOTSET) # Enable all logging globally. + logging.getLogger().setLevel(levels[tmp['log_level']]) # set the logging level. + + with self.server_lock: + stdout.write("203 OK LOGLEVEL SET\n") + stdout.flush() + + def _list_voices(self): + """ + List the voices and send them back to speechd. + + :meta private: + """ + voices = self.voices + with self.server_lock: + for voice in voices: + name = voice['name'] if 'name' in voice else "none" + language = voice['language'] if 'language' in voice else "none" + variant = voice['variant'] if 'variant' in voice else "none" + stdout.write(f"200-{name}\t{language}\t{variant}\n") + + stdout.write("200 OK VOICE LIST SENT\n") + stdout.flush() + + def _setting(self): + """ + Receive setting from speechd. + + :meta private: + """ + tmp = self.parse_set_params(203, "") + self.execution_worker.queue.put({"command": "settings", "args": tmp}) + with self.server_lock: + stdout.write("203 OK SETTINGS RECEIVED\n") + stdout.flush() + + def _speak(self): + """ + Get a speak command from speechd. + + :meta private: + """ + with self.server_lock: + stdout.write("202 OK RECEIVING MESSAGE\n") + stdout.flush() + + full_text = "" + + while (True): + line = self._readline() + if (line == ".\n"): + break + else: + full_text = full_text + line[:-1] + " " + + self.execution_worker.queue.put({"command": "speak", "args": full_text}) + + with self.server_lock: + stdout.write("200 OK SPEAKING\n") + stdout.flush() + + def stop_get(self): + """ + Get the value of the stop var. + + :meta private: + """ + with self.stop_lock: + return self.stop + def stop_set(self, val): + """ + Set the value of the stop var. + + :meta private: + """ + with self.stop_lock: + self.stop = val + def _stop(self): + """ + Stop the client. + + :meta private: + """ + self.stop_set(True) + with self.server_lock: + stdout.write("703 STOP\n") + stdout.flush() + def _pause(self): + """ + Stop the client. + + :meta private: + """ + + self.stop_set(True) + with self.server_lock: + stdout.write("704 PAUSE\n") + stdout.flush() + + def _process(self, block=True): + """ + Process commands from speechd. + + :meta private: + """ + while (True): + line = self._readline(block) + if (line == None): + break + if (line == ""): + break + if (line == "SPEAK\n"): + self._speak() + elif (line == "SOUND_ICON\n"): + self._speak() + elif (line == "CHAR\n"): + self._speak() + elif (line == "KEY\n"): + self._speak() + elif (line[:11] == "LIST VOICES"): + self._list_voices() + elif (line == "SET\n"): + self._setting() + elif (line == "AUDIO\n"): + self._audio() + elif (line == "LOGLEVEL\n"): + self._loglevel() + elif (line == "STOP\n"): + self._stop() + elif (line == "PAUSE\n"): + self._pause() + elif (line[:5] == "DEBUG"): + self._debug(line) + elif (line == "QUIT\n"): + break + else: + logging.error("_process: unknown command") + logging.error(line) + + def start(self, callback): + """ + Starts the mainloop and sets the client callback. + """ + + self._callback = callback + # keep a local copy of the voices + # it should be set at the start and never changed. + if hasattr(callback, "voices"): + self.voices = callback.voices + else: + self.voices = [{"name": "GenericPythonModule", "language": "en", "variant": "none"}] + + # Start the worker threads. + self.execution_worker.set_callback(callback) + self.execution_worker.start() + self.action_worker.start() + self._callback._set_init(self.action_worker.queue, self) + + # run configure. + self._configure() + + line = self._readline() + if (line != "INIT\n"): + logging.error("ERROR: Server did not start with INIT\n") + + msg = "GOOD" + stdout.write(f"299-{msg}\n") + stdout.write("299 OK LOADED SUCCESSFULLY\n") + stdout.flush() + + logging.info("server: Starting _process()") + self._process() + + # let the workers stop before quitting + self.execution_worker.stop() + self.action_worker.stop() + self.execution_worker.join() + self.action_worker.join() + logging.debug("server: Quiting") \ No newline at end of file diff --git a/src/modules/python/src/pySpeechModule/speechd_utilities.py b/src/modules/python/src/pySpeechModule/speechd_utilities.py new file mode 100644 index 00000000..2882212a --- /dev/null +++ b/src/modules/python/src/pySpeechModule/speechd_utilities.py @@ -0,0 +1,115 @@ +# MIT License + +# Copyright (c) 2026 John Settlemyer + +# Permission is hereby granted, free of charge, to any person obtaining a copy +# of this software and associated documentation files (the "Software"), to deal +# in the Software without restriction, including without limitation the rights +# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +# copies of the Software, and to permit persons to whom the Software is +# furnished to do so, subject to the following conditions: + +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. + +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +# SOFTWARE. + +import xml.etree.ElementTree as ET +import logging + +def hdlc_escape(data: bytes) -> bytes: + """ + hdlc_escape is for excaping newline chars using hdlc encoding. + This function likely has little value to the users but is need internally. + + :meta private: + """ + ESCAPE = 0x7D + INVERT = 1 << 5 + out = bytearray() + + for byte in data: + if byte == ESCAPE or byte == ord("\n"): + out.append(ESCAPE) + out.append(byte ^ INVERT) + else: + out.append(byte) + + return bytes(out) + +def parse_ssml(data: bytes) -> list[dict]: + """ + This is a convenience function for parsing SSML into a list of dicts in the format + + .. code-block:: python + + [{"mark": "", "text": ""}] + + :param data: The bytes string for your SSML text + :type bytes: bytes + :return: List of dicts each with mark and text + :rtype: list + """ + + output = [] + buffer = [] + + # Parse the SSML XML data + try: + root = ET.fromstring(data) + except ET.ParseError: + logging.error("parse_ssml: Failed to parse XML") + return output + + # Handle text directly inside before any child elements + if root.text: + buffer.append(root.text) + + # Iterate through direct child elements of + for child in root: + # Check for element + if child.tag.endswith("mark"): + mark_name = child.attrib.get("name", "") + + output.append( + {"mark": mark_name, "text": "".join(buffer)} + ) + + # Reset buffer after encountering a mark + buffer.clear() + + # Append trailing text after a child node (XML tail text) + if child.tail: + buffer.append(child.tail) + + # Add any remaining trailing text after the final mark + if buffer: + output.append({"mark": "", "text": "".join(buffer)}) + + return output + +def strip_ssml(ssml_string: bytes): + """ + This is a convenience function for stripping out all SSML tags leaving just the text. + + :param data: The bytes string for your SSML text + :type bytes: bytes + :return: extracted text + :rtype: string + """ + + # Wrap in a root element if the string is an XML fragment + try: + root = ET.fromstring(ssml_string) + except ET.ParseError: + # Wrap fragments in a temporary root node to ensure valid parsing + root = ET.fromstring(f"{ssml_string}") + + # itertext() yields all text within the element and its children + return "".join(root.itertext()) \ No newline at end of file