UnleashX commited on
Commit
9ec6562
·
verified ·
1 Parent(s): bbb89bf

Upload 217 files

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. audio/Received-from-BOT.txt +0 -0
  2. audio/sent-to-BOT.txt +18 -0
  3. deepgram/__init__.py +383 -0
  4. deepgram/__pycache__/__init__.cpython-313.pyc +0 -0
  5. deepgram/__pycache__/client.cpython-313.pyc +0 -0
  6. deepgram/__pycache__/errors.cpython-313.pyc +0 -0
  7. deepgram/__pycache__/options.cpython-313.pyc +0 -0
  8. deepgram/audio/__init__.py +22 -0
  9. deepgram/audio/__pycache__/__init__.cpython-313.pyc +0 -0
  10. deepgram/audio/microphone/__init__.py +7 -0
  11. deepgram/audio/microphone/__pycache__/__init__.cpython-313.pyc +0 -0
  12. deepgram/audio/microphone/__pycache__/constants.cpython-313.pyc +0 -0
  13. deepgram/audio/microphone/__pycache__/errors.cpython-313.pyc +0 -0
  14. deepgram/audio/microphone/__pycache__/microphone.cpython-313.pyc +0 -0
  15. deepgram/audio/microphone/constants.py +11 -0
  16. deepgram/audio/microphone/errors.py +21 -0
  17. deepgram/audio/microphone/microphone.py +302 -0
  18. deepgram/audio/speaker/__init__.py +7 -0
  19. deepgram/audio/speaker/__pycache__/__init__.cpython-313.pyc +0 -0
  20. deepgram/audio/speaker/__pycache__/constants.cpython-313.pyc +0 -0
  21. deepgram/audio/speaker/__pycache__/errors.cpython-313.pyc +0 -0
  22. deepgram/audio/speaker/__pycache__/speaker.cpython-313.pyc +0 -0
  23. deepgram/audio/speaker/constants.py +15 -0
  24. deepgram/audio/speaker/errors.py +21 -0
  25. deepgram/audio/speaker/speaker.py +380 -0
  26. deepgram/client.py +669 -0
  27. deepgram/clients/__init__.py +381 -0
  28. deepgram/clients/__pycache__/__init__.cpython-313.pyc +0 -0
  29. deepgram/clients/__pycache__/agent_router.cpython-313.pyc +0 -0
  30. deepgram/clients/__pycache__/errors.cpython-313.pyc +0 -0
  31. deepgram/clients/__pycache__/listen_router.cpython-313.pyc +0 -0
  32. deepgram/clients/__pycache__/read_router.cpython-313.pyc +0 -0
  33. deepgram/clients/__pycache__/speak_router.cpython-313.pyc +0 -0
  34. deepgram/clients/agent/__init__.py +56 -0
  35. deepgram/clients/agent/__pycache__/__init__.cpython-313.pyc +0 -0
  36. deepgram/clients/agent/__pycache__/client.cpython-313.pyc +0 -0
  37. deepgram/clients/agent/__pycache__/enums.cpython-313.pyc +0 -0
  38. deepgram/clients/agent/client.py +102 -0
  39. deepgram/clients/agent/enums.py +36 -0
  40. deepgram/clients/agent/v1/__init__.py +60 -0
  41. deepgram/clients/agent/v1/__pycache__/__init__.cpython-313.pyc +0 -0
  42. deepgram/clients/agent/v1/websocket/__init__.py +51 -0
  43. deepgram/clients/agent/v1/websocket/__pycache__/__init__.cpython-313.pyc +0 -0
  44. deepgram/clients/agent/v1/websocket/__pycache__/async_client.cpython-313.pyc +0 -0
  45. deepgram/clients/agent/v1/websocket/__pycache__/client.cpython-313.pyc +0 -0
  46. deepgram/clients/agent/v1/websocket/__pycache__/options.cpython-313.pyc +0 -0
  47. deepgram/clients/agent/v1/websocket/__pycache__/response.cpython-313.pyc +0 -0
  48. deepgram/clients/agent/v1/websocket/async_client.py +688 -0
  49. deepgram/clients/agent/v1/websocket/client.py +677 -0
  50. deepgram/clients/agent/v1/websocket/options.py +453 -0
audio/Received-from-BOT.txt ADDED
The diff for this file is too large to render. See raw diff
 
audio/sent-to-BOT.txt ADDED
@@ -0,0 +1,18 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"timestamp":"2025-01-17 11:07:06","streamId":"C:\\record\\0.1.0.18.0_CallID_9911011200_051025154759.wav","callerId":"9911011200","channelId":"0.1.0.18.0","event":"answer","callDirection":"incoming","did":"0120111222","callId":"0.1.0.18.0_CallID","cid":"CallID","extraParams":""} (15:47:59.580)
2
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"sAIwAzACcAPQAmAFIAVwAjgAcPzQ/BACwAuAEEAJwPWA4ADbgOWQ/YAQgBaAEEAKwArADEAJ0PyA7IDlgOpA9+AFQAzAC+AF6AFYAJD9IPvA90D1wPag+OD7kP3o/tADwAqAEMAN4AVw/cD3YPnw/BADIAZgBWAGoAbACsAKoAUQ/EDzQPTg+vACoAZgBagBkAPADoAXgBNw/QDfAMsAz4DmYAWAGoAdgBqAGIAXgBPgBIDvAN0A2YDkYPhACYAQgBHADEAKIAWw/EDzgOqA6YDvwPdYAWAFQAlADIAQwA/gByj/QPbA9GD4qP5gBSAGwAnAC8AKQAnQA7D9QPbA88D2oPvYAfAD4AXgBJADQAjADUAOsAOA7ADVANOA43j/gBgAIYAbgBWAE8AOIARA84DhANuA5MD3wAmAEoASQA4=","callId":"0.1.0.18.0","extraParams":""} (15:48:09.927)
3
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"gBGAEoASgBeAGIAUwAwYAGD4wPWA74DrgO3A9KD7iACwA2AHQAvACfACUPxg+qD6YPog+7D9SAFgBkALwAzACsAIYAXYAdD94Plg+CD5YPmg+KD5UP0oAbACWAHQAmAFoARgBNACiACYANgAMALQA+D6wPBA9uD5wPKA78DxYPlgBcAKwAqAEYAZgBiAEcAIEAIY/sD3gO6A6IDswPPA92D7UALACcANwAmwAgj/4PtA98D0QPZg+pgA4AVACkAPgBDADUAK4AX4AJD8YPnA9sD0QPRA9iD6WP5IAPACwAhACSAH4AUQA4gAKP9w/VgACP/A8cDwYPvA98DxwPNA9Qj+oAdgBkAKgBaAGIAVgBJADGAH8AJA94DtgOuA7IDuwPHA9sj/QAhACkAIoAegBWgAoPvg+aD6cPzI/pgBYAY=","callId":"0.1.0.18.0","extraParams":""} (15:48:10.020)
4
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"QPWg+KD7iADACcANwA1ADkANQAygB2gAYPtg+cD1wPFA8cDzwPWg+TD8eP/QA6AGoAfgB8AIwAggByAF4AQgBFACGAHIAPj/eP4o/jD9EP3Y/qj/GP/o/ggBWACI/xgAWAC4AKj/SP/o/9gAiACIAXADEP2w/XgAoPog+ED2wPeg+SD7CP7oAOAHwAjACcAKwAnACWAF2ACQ/WD7YPjA9ED1QPbA9+D5oPvQ/QgB8ANgBSAGYAfgB+AG4AUgBZAD8AKoAQgACAAI//D8kP1o/kj++P7Y/zgA6ADoAEgAeAD4AOj/qP74/ij/2P5o/kj/IPuw/SgAIPtg+WD5UPwQ/DD88P0Y/+AF4AQgBCAHoAegB5ADyACY/7j+oPvA92D5YPog+iD7cPz4/vgA0ALQA+AEoAagBiAEoAUgBjAD8AI=","callId":"0.1.0.18.0","extraParams":""} (15:48:10.061)
5
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"SP6Y/5gBUANgBGAFIAYgBmAFIARwAmgA8P3g+6D6oPkg+eD5oPtw/Wj/iAHQA+AFYAagBiAGoAWgBDAC2ACoANj/aP6I/tj/qP+Y/zgAmACIAEgA2P9I/xj/mP6Q/Tj+eP6I/tj+KP8o/wj/SADI/xD9cP3Y/lD94PtQ/JD9GP7o/vj+yP/4AbACEANgBGAFYAUgBWAE8AJoASj/0Pyg++D5oPlg+lD80P1YALADoAWgBiAH4AZgBSAEOAHo/vD9kPyg++D7sP1I/gj/aABIAZgBeAFIAagAKACI/3j+iP7Y/pj+2P5I/4j+SP5Y/nj+MP1Q/Vj+KP9YAMgB8AKQAyAEkAOQAugByACo/wj/8P2Q/fD90P0I/qj+OP+I/gj/WP8I/1j/2P9IAKgAiAHIAfgBEAOQAxAD0AKQAogBiAA=","callId":"0.1.0.18.0","extraParams":""} (15:48:10.103)
6
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"6P6I/0gAKAGYAdgBWAE4AGj/6P5o/ij+qP4IADgBUAJwApgBGADw/fD8UP0I/7gAqAG4AXgAmP9Y/4gAiAFoASgAiP7Q/Tj+mP/4AIgBSAFoAPj/eP8Y/zj/2P6I/hj/qABQAjADEAMIARj/OP6w/XD90P1I/xgB0ALQAxADWAFI/uD7oPtQ/RgAyAFwAlgB6P+I/5gA4ATAC6AGwPWA5YDlwPZADYAZgBSgBwgAqABgBdACQPeA7IDrwPfgBsAOwAygBnACmAFYAYj+YPlA9cD14Pu4AeAF4AbgBWAEMAL4//D8oPrg+SD7KP7oAeAE4AVgBHACaAAY/rD90P0I/jj+6P5wAmAFIAUoAVD8oPow/Dj/2ABoADj/CP8oAdADQArADyAFQPCA4IDkYPnADIAVgBBACKAFYAfgB8j+wPA=","callId":"0.1.0.18.0","extraParams":""} (15:48:11.922)
7
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"QPeoAcAKgBKAGYAdgB+AHIAWgBDACbACcPxg+WD4IPgg+KD44PhA98D0QPPA8kDzwPTA9iD4oPmg+3D9KP9oAWAEYAfAC8APgBKAFIAUgBOAEEAM4AcQA8j/WP4w/eD7oPrA9kDwgOuA6oDqgOyA7YDvwPKg+DgAoAfADYASgBWAF4AVgBLADKAHyAEg+8D2wPTA9ED2wPeg+aD6oPrg+qD7cPwQ/RD9MPzg+9D88P2o/7gB8ANgBkAJQAzADUAPwA/ADsAMQAogB+AE8ANQA6gB6P8g+8DygOuA6IDogOqA7YDtgO/A82D6yAFACUAPgBKAFIAVgBOAEMAMQAmgBHj+oPhA9EDyQPLA80D1QPeg+KD6MP24/2gBuABY/nD84Psw/LD9uP+YAfAD4AZACUALQAxADcAMwAtACkAJwAg=","callId":"0.1.0.18.0","extraParams":""} (15:48:12.020)
8
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"wA2gB+AGwAhACkALwA1AD0AOwA1ADcAOgBWAHIAYoAWA6gDZANMA38DykAMgB9gBkP0wA4ARgB0AIYAYMANA8YDogOpA9TgAIARoAWD7IPlw/cj+wPSA5gDZANUA2YDjYPhACYASgBWAEkAOQAvAC0AMQAngBUAIgBCAGIAcgBqAFYAXgBuAF3ADgOUAzQDFAM+A6Hj/wApACWAHQAqAFIAagBfgB8DzgOiA6ED0cANACsAIiAF4/iAEQAn4/4DqANcAzQDTgOTg+cALgBKAEkAPQA5ADoAQgBDAC/ADKAGgB4ASgBiAGIAQwAzAD4AQoASA7QDXAM0A0YDjQPcgBcAIwAhADYAXgB+AHEAOoPiA6oDowPIQA0ANwAxgBbD88P1ACEALIPkA2wDHAMcA2cDxQAqAFoAUwA/ADEAPgBI=","callId":"0.1.0.18.0","extraParams":""} (15:48:12.100)
9
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"gBuAG4AZgBeAFYAUgBWAFEANqP6A74DmgOeA7UDxQPGA7sDyUPwgB0ALoAcgBKAGwA6AFMAP4AUo/rD8qAHgBaAG8AKg+oDvgOaA5YDrQPCA7oDpgOiA7uD4KAFQAzAC8ALACIAQgBSAFYAUgBSAFoAWgBaAFIARwAlY/8D3wPRA9sD2QPTA8cDxQPYw/WgBOACw/BD9MAJACkANwAnQAgj+WP/QAvADmAGQ/aD4QPXA88DzwPNA8YDtgOyA7sDyQPdg+SD6EP0QA0AKQA7AD4ARgBSAGIAZgBeAFYARwAtgBBD9IPhA9sD1QPRA8kDxQPOg+ND80P0w/bj+UAPACUAMwAqgBtAD0AMgBGgBUP3g+ED0QPCA7oDtgOyA64DpgOuA7sDxQPZg+eD7uABgB0AOgBGAE4AXgBqAHYAegB0=","callId":"0.1.0.18.0","extraParams":""} (15:48:12.140)
10
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"gBmAE0ALoARI/+D7IPnA9UDzwPFA80D3IPoQ/Lj+OACwAuAGQAnACMAIwAjACKAHYAQoAGD7QPWA74DqgOeA5oDkgOKA5IDngOxA8sD2IPpI/yAFQAvAD4ASgBSAGIAcgB8AIYAfgBqAFEANYAeQAsj+oPrA9kD0wPPA9SD4YPkg+9D8CP4oASAEYATwA9ADMAOQAxADCAFw/SD4QPKA7YDqgOmA54DlgOaA6IDrQPDA9WD6KADgBUALwA+AEoAUgBaAGYAdgB8AIYAdgBdAD0AIeAHg+8D2QPKA7oDtgO9A8kD2YPoI/tgA4ARACMAKwAvACsAIoAbQAzgBCP7g+kD3wPKA7oDrgOiA54DngOeA6IDqgO7A8sD3EP0wA0AJwA6AEoAVgBaAGIAagBuAG4AagBaAEUAL4AX4/yD7QPc=","callId":"0.1.0.18.0","extraParams":""} (15:48:12.164)
11
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"cAKgBsAKwA6AEoAWgBmAGYAYgBaAFIAQQAxACHADmP/Q/KD64PjA90D3wPfg+CD6YPsQ/dD9OP74/8gAOAD4/3j/kP0w/OD6oPjA9sD1wPTA80DzQPNA88DzQPTA9cD3IPmg+6j+eAEgBcAIwAvADoARgBSAFIATgBKAEEANQApgB5ADKAFI/5D9UPyg+yD74Ptw/Qj+mP5o/zj/aP84AOgAWABo/6j+MP2g+2D6IPnA9kD1wPRA88DywPLA8kDzwPPA9MD2oPig+pD9KAAQA6AGwAlADEAPgBGAE4ATgBOAE4AQwA1AC+AH4ATQAqgAeP7w/KD7oPtQ/HD80PyQ/bD9OP64/wgBKAF4ABgA6P5w/RD8YPrA98D1wPTA8sDxwPBA8MDwQPFA8kD0wPag+RD9KACwA+AHQAtADoARgBQ=","callId":"0.1.0.18.0","extraParams":""} (15:48:12.239)
12
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"gBWAFoAWgBSAEoAQwAxACaAF6AGo/uD7IPlA98D1wPVA9sD2IPig+aD6cPzY/rgAqAG4AcgB+ACo//D94Psg+UD2QPVA80DxQPBA8MDwwPFA80D2oPgQ/Kj/8ALgBkALwA6AEYAUgBeAGIAYgBeAFYATQA9ACyAHsAKo/iD7IPhA9UD0QPNA9ED1wPag+GD6kPyo/sgAUALwAvACsAKYARgA0P2g++D4wPbA9MDyQPFA8MDwQPFA80D1wPeg+lj+mAFgBcAJQA2AEIAUgBaAGIAYgBeAFoATwA9ACyAHcAKQ/aD5wPXA8sDwgO9A8EDxQPNA9qD4YPtY/ggBcANgBSAGIAagBSAEEAJI/3D8IPnA9kD0wPGA74DvgO+A78DxwPTA9mD6iP4wAiAGwApADoARgBWAF4AZgBmAGYAXgBQ=","callId":"0.1.0.18.0","extraParams":""} (15:48:12.260)
13
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"gBHADKAHMANI/iD6wPbA88DxwPDA8MDxwPNA9mD4IPvQ/ZgAsAJgBOAE4ARgBPACCAGY/hD8YPlA90D1wPLA8UDwQPHA8cDzQPXA9yD7uP4wAiAFwAhAC0AOgBCAEoAUgBWAFYAVgBOAEMANwAngBYgBUP2g+UD2wPPA8UDxwPFA8kD0QPZg+OD6cP2Y/3gBkAIwA1AD0AK4ASgAeP5w/KD64PhA90D1wPRA9ED0QPXA9mD4oPqw/agAcAOgBkAJwAvADoAQgBKAE4ATgBOAEsAPwAxACSAF2ACw/CD5wPXA88DxwPFA8kDzQPXA9yD6kPzo/mgBUAPgBGAFoAVgBaAEEAMIAej+sPxg+mD4QPbA9MDzQPPA88D0wPXA9yD6sP3YAPADoAdAC8ANgBCAEoAUgBSAFIAUgBJAD8ALoAc=","callId":"0.1.0.18.0","extraParams":""} (15:48:12.280)
14
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"4AbADoAQgBOAGIAbgBmAGYAYgBWAEoAQwAogBsAIsAJA8sD28PyA5QDfwPOA7wDfgO3IAED3YPlADoARwA6AFoAagBSAE8AOoAawA3D9wPXA8sDxgOyA5oDogOqA54DmgO5A8kDxIPhoALgBIARAC8AOgBCAE4AWgBWAFYAVgBKAEYAQQAngBEALUAOA78D2mP+A5QDdwPTA8wDfgO2gBlgAcPyAEIAZgBOAFoAagBKAEEAOsALg+3D8wPWA7IDtQPCA6YDkgOqA7IDpgO3A80D1oPlYABACoAVADEAOwA6AE4AWgBSAFIAVgBLAD8APQAmgBsAKeP+A7yD4oPmA4wDfwPFA8oDmQPIwAxACoASAEIATgBKAFoAUwAvAC8AKOP5A9mD5QPaA7IDsQPGA7IDogO5A8IDvwPJA9qD4GP4=","callId":"0.1.0.18.0","extraParams":""} (15:48:12.401)
15
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"MANACEAMwA1ADoAQgBKAEYAQQA/ADsANQAxAC0AIIAbgBmAHQAloASD4IPrg+EDxgOyA6oDsgOzA8MD0wPYo/iAGwAlADIAQgBOAEoAQQA7ACBACOP7A98DwgOuA6YDngOWA5oDqgO7A9OD6GADgBcALwA6AEIASgBOAE4ARwA/ADkANQAxACaAGYATwAsgBaAFoAaD7IPig+cD2wPPA8sDzwPHA8cD0wPUg+fD8iADwAyAHwArADMAOQA7ADEALwAjQA/j/MPzA98DzQPCA7oDtgO2A70DywPWg+fD96AGgBUAIwApADEANwA1ADkANQA1ADEALwAmgB+AFsAPYAUgAGP/Q/KD64Png+MD3QPfA9kD2QPZA9yD4oPng+5D9+P/4AfADoAXgBqAHYAcgByAGIARQAogASP4w/CD64Pg=","callId":"0.1.0.18.0","extraParams":""} (15:48:12.440)
16
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"+AH4ABgAKP8o/nD9EP3w/BD9kP0I/pj+SP/4/8gAKAG4ARACcAJwAlACUALoAYgBKAEYASgBKAFIAagB6AHYAcgBaAG4ANj/6P4o/pD9sPxQ/HD80PxQ/Tj+KAB4AXACcAOwA5AD0AOwAzACeP/Q/ZD8IPvg+aD5IPrg+hD8UP2I/uj/KAFwAlAD4ATgBKAEUAJ4AJj/GP4w/dD8sP3Q/Xj++P/YAMgBsAJwA7ADoAdACfADeADY/8j+4Pvg+qD7wPbA9RD8IARgB8AIQAtAC0AJQAvAC6AH6P4g+UD2QPJA8EDwgO2A7cDxQPag+Tj+IATgB8AIQArADMAMQAggBCgAkP3Q/OD7IPug+hD8UP3IAWAFIAagBqAHoAegB0AMQAlQ/MD3wPXA90D34Pig+EDzwPUoAcAOgBHAD4AQwA0=","callId":"0.1.0.18.0","extraParams":""} (15:48:12.498)
17
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"gOwQ/YAWACOAFeAHgBGAGYAQSAFA8IDggOFA8MD2QPJA9FgBQArADEAKKABA9sD3eP4Y/6D72P9gB8ANgBBADGAFMAPgBJAD6P8w/DD8CADgBeAHEANo/3j+4AWAGoAbQPYA1YDiYPqg+4DrgOWA7Ej/gBoAJYAUoAaAE4AegBPo/4DsgOCA5EDyQPSA7IDsYPkgB0AMwAi4/qD48PygBGAEkPzg+tACwAvADUAJCAFI/iAEQAgwAmD6IPvQAsAJwAygB2j/UP2YACAHgBKAEMD0AN+A6jD9IPqA64DmgO94AIAUgBtADqAGgBSAHYASSABA8YDogOvA80D0gO/A8ND9QAlACyAHaAAw/bj/CABg+8D2IPoQA2AHYAbwA7ACYATACEAJ6AEw/NgAQAhACWAFqAAY//gAEAIgBcAMwAs=","callId":"0.1.0.18.0","extraParams":""} (15:48:12.559)
18
+ {"sampleRate":16000,"streamId":"0.1.0.18.0","chunk":1,"chunk_durn_ms":20,"channelId":"","event":"media","payload":"wAjQA0gBOAAoAKAFQApgBuD6wPVg+WD7oPkg+MD3oPoQ/bADgBKAGUAIgOYA34DvYPjA9sDyQPQwA4AdACWAFqAE8ANACOAEoPnA8UD0WABADcALWP5g+SAEQAxQAoDtgOfA8CD64Pog+ED34PrgBkANoAf4AaAFwAvACRACeAFgBGAFIAagB6AFCP+g+zD8IPrA96D4oPrg++D6kPxgBYASgBlIAADbAN3A88D3wPHA8uD7gBEAJwAnwA84/yAFwAo4/0DywPJg+iAFQA7ACaD4QPfgB0AJwPOA4oDpwPcY/hj/kPzg+egAQA7ADLgAqP/ACMAL4AawA6AEoAfADYAQQAloAHD9CP6g+sDzQPHA86D44Ptg+mD48P3ADYARwPMA2YDmYPvg+UD04PmgB4AaACeAHaAHeP9AC0AM4Pk=","callId":"0.1.0.18.0","extraParams":""} (15:48:12.700)
deepgram/__init__.py ADDED
@@ -0,0 +1,383 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2023-2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ # version
6
+ __version__ = "0.0.0"
7
+
8
+ # entry point for the deepgram python sdk
9
+ import logging
10
+ from .utils import VerboseLogger
11
+ from .utils import (
12
+ NOTICE,
13
+ SPAM,
14
+ SUCCESS,
15
+ VERBOSE,
16
+ WARNING,
17
+ ERROR,
18
+ FATAL,
19
+ CRITICAL,
20
+ INFO,
21
+ DEBUG,
22
+ NOTSET,
23
+ )
24
+
25
+ from .client import Deepgram, DeepgramClient
26
+ from .client import DeepgramClientOptions, ClientOptionsFromEnv
27
+ from .client import (
28
+ DeepgramError,
29
+ DeepgramTypeError,
30
+ DeepgramModuleError,
31
+ DeepgramApiError,
32
+ DeepgramUnknownApiError,
33
+ )
34
+ from .errors import DeepgramApiKeyError
35
+
36
+ # listen/read client
37
+ from .client import ListenRouter, ReadRouter, SpeakRouter, AgentRouter
38
+
39
+ # common
40
+ from .client import (
41
+ TextSource,
42
+ BufferSource,
43
+ StreamSource,
44
+ FileSource,
45
+ UrlSource,
46
+ )
47
+ from .client import BaseResponse
48
+ from .client import (
49
+ Average,
50
+ Intent,
51
+ Intents,
52
+ IntentsInfo,
53
+ Segment,
54
+ SentimentInfo,
55
+ Sentiment,
56
+ Sentiments,
57
+ SummaryInfo,
58
+ Topic,
59
+ Topics,
60
+ TopicsInfo,
61
+ )
62
+ from .client import (
63
+ ModelInfo,
64
+ Hit,
65
+ Search,
66
+ )
67
+ from .client import (
68
+ OpenResponse,
69
+ CloseResponse,
70
+ UnhandledResponse,
71
+ ErrorResponse,
72
+ )
73
+
74
+ # speect-to-text WS
75
+ from .client import LiveClient, AsyncLiveClient # backward compat
76
+ from .client import ListenWebSocketClient, AsyncListenWebSocketClient
77
+ from .client import LiveTranscriptionEvents
78
+ from .client import LiveOptions, ListenWebSocketOptions
79
+ from .client import (
80
+ #### top level
81
+ LiveResultResponse,
82
+ ListenWSMetadataResponse,
83
+ SpeechStartedResponse,
84
+ UtteranceEndResponse,
85
+ #### common websocket response
86
+ # OpenResponse,
87
+ # CloseResponse,
88
+ # UnhandledResponse,
89
+ # ErrorResponse,
90
+ #### unique
91
+ ListenWSMetadata,
92
+ ListenWSAlternative,
93
+ ListenWSChannel,
94
+ ListenWSWord,
95
+ )
96
+
97
+ # prerecorded
98
+ from .client import PreRecordedClient, AsyncPreRecordedClient # backward compat
99
+ from .client import ListenRESTClient, AsyncListenRESTClient
100
+ from .client import (
101
+ # common
102
+ # UrlSource,
103
+ # BufferSource,
104
+ # StreamSource,
105
+ # TextSource,
106
+ # FileSource,
107
+ # unique
108
+ PreRecordedStreamSource,
109
+ PrerecordedSource,
110
+ ListenRestSource,
111
+ SpeakRESTSource,
112
+ )
113
+ from .client import (
114
+ ListenRESTOptions,
115
+ PrerecordedOptions,
116
+ )
117
+ from .client import (
118
+ #### top level
119
+ AsyncPrerecordedResponse,
120
+ PrerecordedResponse,
121
+ SyncPrerecordedResponse,
122
+ #### shared
123
+ # Average,
124
+ # Alternative,
125
+ # Channel,
126
+ # Intent,
127
+ # Intents,
128
+ # IntentsInfo,
129
+ # Segment,
130
+ # SentimentInfo,
131
+ # Sentiment,
132
+ # Sentiments,
133
+ # SummaryInfo,
134
+ # Topic,
135
+ # Topics,
136
+ # TopicsInfo,
137
+ # Word,
138
+ #### unique
139
+ Entity,
140
+ Hit,
141
+ ListenRESTMetadata,
142
+ ModelInfo,
143
+ Paragraph,
144
+ Paragraphs,
145
+ ListenRESTResults,
146
+ Search,
147
+ Sentence,
148
+ Summaries,
149
+ SummaryV1,
150
+ SummaryV2,
151
+ Translation,
152
+ Utterance,
153
+ Warning,
154
+ ListenRESTAlternative,
155
+ ListenRESTChannel,
156
+ ListenRESTWord,
157
+ )
158
+
159
+ # read
160
+ from .client import ReadClient, AsyncReadClient
161
+ from .client import AnalyzeClient, AsyncAnalyzeClient
162
+ from .client import (
163
+ AnalyzeOptions,
164
+ AnalyzeStreamSource,
165
+ AnalyzeSource,
166
+ )
167
+ from .client import (
168
+ #### top level
169
+ AsyncAnalyzeResponse,
170
+ SyncAnalyzeResponse,
171
+ AnalyzeResponse,
172
+ #### shared
173
+ # Average,
174
+ # Intent,
175
+ # Intents,
176
+ # IntentsInfo,
177
+ # Segment,
178
+ # SentimentInfo,
179
+ # Sentiment,
180
+ # Sentiments,
181
+ # SummaryInfo,
182
+ # Topic,
183
+ # Topics,
184
+ # TopicsInfo,
185
+ #### unique
186
+ AnalyzeMetadata,
187
+ AnalyzeResults,
188
+ AnalyzeSummary,
189
+ )
190
+
191
+ # speak
192
+ ## speak REST
193
+ from .client import (
194
+ #### top level
195
+ SpeakRESTOptions,
196
+ SpeakOptions, # backward compat
197
+ #### common
198
+ # TextSource,
199
+ # BufferSource,
200
+ # StreamSource,
201
+ # FileSource,
202
+ #### unique
203
+ SpeakSource,
204
+ SpeakRestSource,
205
+ )
206
+
207
+ from .client import (
208
+ SpeakClient, # backward compat
209
+ SpeakRESTClient,
210
+ AsyncSpeakRESTClient,
211
+ )
212
+
213
+ from .client import (
214
+ SpeakResponse, # backward compat
215
+ SpeakRESTResponse,
216
+ )
217
+
218
+ ## speak WebSocket
219
+ from .client import SpeakWebSocketEvents, SpeakWebSocketMessage
220
+
221
+ from .client import (
222
+ SpeakWSOptions,
223
+ )
224
+
225
+ from .client import (
226
+ SpeakWebSocketClient,
227
+ AsyncSpeakWebSocketClient,
228
+ SpeakWSClient,
229
+ AsyncSpeakWSClient,
230
+ )
231
+
232
+ from .client import (
233
+ #### top level
234
+ SpeakWSMetadataResponse,
235
+ FlushedResponse,
236
+ ClearedResponse,
237
+ WarningResponse,
238
+ #### common websocket response
239
+ # OpenResponse,
240
+ # CloseResponse,
241
+ # UnhandledResponse,
242
+ # ErrorResponse,
243
+ )
244
+
245
+ # manage
246
+ from .client import ManageClient, AsyncManageClient
247
+ from .client import (
248
+ ProjectOptions,
249
+ KeyOptions,
250
+ ScopeOptions,
251
+ InviteOptions,
252
+ UsageRequestOptions,
253
+ UsageSummaryOptions,
254
+ UsageFieldsOptions,
255
+ )
256
+
257
+ # manage client responses
258
+ from .client import (
259
+ #### top level
260
+ Message,
261
+ ProjectsResponse,
262
+ ModelResponse,
263
+ ModelsResponse,
264
+ MembersResponse,
265
+ KeyResponse,
266
+ KeysResponse,
267
+ ScopesResponse,
268
+ InvitesResponse,
269
+ UsageRequest,
270
+ UsageResponse,
271
+ UsageRequestsResponse,
272
+ UsageSummaryResponse,
273
+ UsageFieldsResponse,
274
+ BalancesResponse,
275
+ #### shared
276
+ Project,
277
+ STTDetails,
278
+ TTSMetadata,
279
+ TTSDetails,
280
+ Member,
281
+ Key,
282
+ Invite,
283
+ Config,
284
+ STTUsageDetails,
285
+ Callback,
286
+ TokenDetail,
287
+ SpeechSegment,
288
+ TTSUsageDetails,
289
+ STTTokens,
290
+ TTSTokens,
291
+ UsageSummaryResults,
292
+ Resolution,
293
+ UsageModel,
294
+ Balance,
295
+ )
296
+
297
+ # selfhosted
298
+ from .client import (
299
+ OnPremClient,
300
+ AsyncOnPremClient,
301
+ SelfHostedClient,
302
+ AsyncSelfHostedClient,
303
+ )
304
+
305
+
306
+ # agent
307
+ from .client import AgentWebSocketEvents
308
+
309
+ # websocket
310
+ from .client import (
311
+ AgentWebSocketClient,
312
+ AsyncAgentWebSocketClient,
313
+ )
314
+
315
+ from .client import (
316
+ #### common websocket response
317
+ # OpenResponse,
318
+ # CloseResponse,
319
+ # ErrorResponse,
320
+ # UnhandledResponse,
321
+ #### unique
322
+ WelcomeResponse,
323
+ SettingsAppliedResponse,
324
+ ConversationTextResponse,
325
+ UserStartedSpeakingResponse,
326
+ AgentThinkingResponse,
327
+ FunctionCallRequest,
328
+ AgentStartedSpeakingResponse,
329
+ AgentAudioDoneResponse,
330
+ InjectionRefusedResponse,
331
+ )
332
+
333
+ from .client import (
334
+ # top level
335
+ SettingsOptions,
336
+ UpdatePromptOptions,
337
+ UpdateSpeakOptions,
338
+ InjectAgentMessageOptions,
339
+ FunctionCallResponse,
340
+ AgentKeepAlive,
341
+ # sub level
342
+ Listen,
343
+ ListenProvider,
344
+ Speak,
345
+ SpeakProvider,
346
+ Header,
347
+ Item,
348
+ Properties,
349
+ Parameters,
350
+ Function,
351
+ Think,
352
+ ThinkProvider,
353
+ Agent,
354
+ Input,
355
+ Output,
356
+ Audio,
357
+ Endpoint,
358
+ )
359
+
360
+ # utilities
361
+ # pylint: disable=wrong-import-position
362
+ from .audio import Microphone, DeepgramMicrophoneError
363
+ from .audio import (
364
+ INPUT_LOGGING,
365
+ INPUT_CHANNELS,
366
+ INPUT_RATE,
367
+ INPUT_CHUNK,
368
+ )
369
+
370
+ LOGGING = INPUT_LOGGING
371
+ CHANNELS = INPUT_CHANNELS
372
+ RATE = INPUT_RATE
373
+ CHUNK = INPUT_CHUNK
374
+
375
+ from .audio import Speaker
376
+ from .audio import (
377
+ OUTPUT_LOGGING,
378
+ OUTPUT_CHANNELS,
379
+ OUTPUT_RATE,
380
+ OUTPUT_CHUNK,
381
+ )
382
+
383
+ # pylint: enable=wrong-import-position
deepgram/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (6.65 kB). View file
 
deepgram/__pycache__/client.cpython-313.pyc ADDED
Binary file (18.2 kB). View file
 
deepgram/__pycache__/errors.cpython-313.pyc ADDED
Binary file (885 Bytes). View file
 
deepgram/__pycache__/options.cpython-313.pyc ADDED
Binary file (11.6 kB). View file
 
deepgram/audio/__init__.py ADDED
@@ -0,0 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2023-2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ from .microphone import Microphone
6
+ from .microphone import DeepgramMicrophoneError
7
+ from .microphone import (
8
+ LOGGING as INPUT_LOGGING,
9
+ CHANNELS as INPUT_CHANNELS,
10
+ RATE as INPUT_RATE,
11
+ CHUNK as INPUT_CHUNK,
12
+ )
13
+
14
+ from .speaker import Speaker
15
+ from .speaker import DeepgramSpeakerError
16
+ from .speaker import (
17
+ LOGGING as OUTPUT_LOGGING,
18
+ CHANNELS as OUTPUT_CHANNELS,
19
+ RATE as OUTPUT_RATE,
20
+ CHUNK as OUTPUT_CHUNK,
21
+ PLAYBACK_DELTA as OUTPUT_PLAYBACK_DELTA,
22
+ )
deepgram/audio/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (669 Bytes). View file
 
deepgram/audio/microphone/__init__.py ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ # Copyright 2023-2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ from .microphone import Microphone
6
+ from .constants import LOGGING, CHANNELS, RATE, CHUNK
7
+ from .errors import DeepgramMicrophoneError
deepgram/audio/microphone/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (389 Bytes). View file
 
deepgram/audio/microphone/__pycache__/constants.cpython-313.pyc ADDED
Binary file (356 Bytes). View file
 
deepgram/audio/microphone/__pycache__/errors.cpython-313.pyc ADDED
Binary file (1.16 kB). View file
 
deepgram/audio/microphone/__pycache__/microphone.cpython-313.pyc ADDED
Binary file (12.9 kB). View file
 
deepgram/audio/microphone/constants.py ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2023-2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ from ...utils import verboselogs
6
+
7
+ # Constants for microphone
8
+ LOGGING = verboselogs.WARNING
9
+ CHANNELS = 1
10
+ RATE = 16000
11
+ CHUNK = 8194
deepgram/audio/microphone/errors.py ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2023-2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+
6
+ # exceptions for microphone
7
+ class DeepgramMicrophoneError(Exception):
8
+ """
9
+ Exception raised for known errors related to Microphone library.
10
+
11
+ Attributes:
12
+ message (str): The error message describing the exception.
13
+ """
14
+
15
+ def __init__(self, message: str):
16
+ super().__init__(message)
17
+ self.name = "DeepgramMicrophoneError"
18
+ self.message = message
19
+
20
+ def __str__(self):
21
+ return f"{self.name}: {self.message}"
deepgram/audio/microphone/microphone.py ADDED
@@ -0,0 +1,302 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2023-2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ import inspect
6
+ import asyncio
7
+ import threading
8
+ from typing import Optional, Callable, Union, TYPE_CHECKING
9
+ import logging
10
+
11
+ from ...utils import verboselogs
12
+
13
+ from .constants import LOGGING, CHANNELS, RATE, CHUNK
14
+
15
+ if TYPE_CHECKING:
16
+ import pyaudio
17
+
18
+
19
+ class Microphone: # pylint: disable=too-many-instance-attributes
20
+ """
21
+ This implements a microphone for local audio input. This uses PyAudio under the hood.
22
+ """
23
+
24
+ _logger: verboselogs.VerboseLogger
25
+
26
+ _audio: Optional["pyaudio.PyAudio"] = None
27
+ _stream: Optional["pyaudio.Stream"] = None
28
+
29
+ _chunk: int
30
+ _rate: int
31
+ _format: int
32
+ _channels: int
33
+ _input_device_index: Optional[int]
34
+ _is_muted: bool
35
+
36
+ _asyncio_loop: asyncio.AbstractEventLoop
37
+ _asyncio_thread: Optional[threading.Thread] = None
38
+ _exit: threading.Event
39
+
40
+ _push_callback_org: Optional[Callable] = None
41
+ _push_callback: Optional[Callable] = None
42
+
43
+ def __init__(
44
+ self,
45
+ push_callback: Optional[Callable] = None,
46
+ verbose: int = LOGGING,
47
+ rate: int = RATE,
48
+ chunk: int = CHUNK,
49
+ channels: int = CHANNELS,
50
+ input_device_index: Optional[int] = None,
51
+ ): # pylint: disable=too-many-positional-arguments
52
+ # dynamic import of pyaudio as not to force the requirements on the SDK (and users)
53
+ import pyaudio # pylint: disable=import-outside-toplevel
54
+
55
+ self._logger = verboselogs.VerboseLogger(__name__)
56
+ self._logger.addHandler(logging.StreamHandler())
57
+ self._logger.setLevel(verbose)
58
+
59
+ self._exit = threading.Event()
60
+
61
+ self._audio = pyaudio.PyAudio()
62
+ self._chunk = chunk
63
+ self._rate = rate
64
+ self._format = pyaudio.paInt16
65
+ self._channels = channels
66
+ self._is_muted = False
67
+
68
+ self._input_device_index = input_device_index
69
+ self._push_callback_org = push_callback
70
+
71
+ def _start_asyncio_loop(self) -> None:
72
+ self._asyncio_loop = asyncio.new_event_loop()
73
+ self._asyncio_loop.run_forever()
74
+
75
+ def is_active(self) -> bool:
76
+ """
77
+ is_active - returns the state of the stream
78
+
79
+ Args:
80
+ None
81
+
82
+ Returns:
83
+ True if the stream is active, False otherwise
84
+ """
85
+ self._logger.debug("Microphone.is_active ENTER")
86
+
87
+ if self._stream is None:
88
+ self._logger.error("stream is None")
89
+ self._logger.debug("Microphone.is_active LEAVE")
90
+ return False
91
+
92
+ val = self._stream.is_active()
93
+ self._logger.info("is_active: %s", val)
94
+ self._logger.info("is_exiting: %s", self._exit.is_set())
95
+ self._logger.debug("Microphone.is_active LEAVE")
96
+ return val
97
+
98
+ def set_callback(self, push_callback: Callable) -> None:
99
+ """
100
+ set_callback - sets the callback function to be called when data is received.
101
+
102
+ Args:
103
+ push_callback (Callable): The callback function to be called when data is received.
104
+ This should be the websocket send function.
105
+
106
+ Returns:
107
+ None
108
+ """
109
+ self._push_callback_org = push_callback
110
+
111
+ def start(self) -> bool:
112
+ """
113
+ starts - starts the microphone stream
114
+
115
+ Returns:
116
+ bool: True if the stream was started, False otherwise
117
+ """
118
+ self._logger.debug("Microphone.start ENTER")
119
+
120
+ self._logger.info("format: %s", self._format)
121
+ self._logger.info("channels: %d", self._channels)
122
+ self._logger.info("rate: %d", self._rate)
123
+ self._logger.info("chunk: %d", self._chunk)
124
+ # self._logger.info("input_device_id: %d", self._input_device_index)
125
+
126
+ if self._push_callback_org is None:
127
+ self._logger.error("start failed. No callback set.")
128
+ self._logger.debug("Microphone.start LEAVE")
129
+ return False
130
+
131
+ if inspect.iscoroutinefunction(self._push_callback_org):
132
+ self._logger.verbose("async/await callback - wrapping")
133
+ # Run our own asyncio loop.
134
+ self._asyncio_thread = threading.Thread(target=self._start_asyncio_loop)
135
+ self._asyncio_thread.start()
136
+
137
+ self._push_callback = lambda data: (
138
+ asyncio.run_coroutine_threadsafe(
139
+ self._push_callback_org(data), self._asyncio_loop
140
+ ).result()
141
+ if self._push_callback_org
142
+ else None
143
+ )
144
+ else:
145
+ self._logger.verbose("regular threaded callback")
146
+ self._asyncio_thread = None
147
+ self._push_callback = self._push_callback_org
148
+
149
+ if self._audio is not None:
150
+ self._stream = self._audio.open(
151
+ format=self._format,
152
+ channels=self._channels,
153
+ rate=self._rate,
154
+ input=True,
155
+ output=False,
156
+ frames_per_buffer=self._chunk,
157
+ input_device_index=self._input_device_index,
158
+ stream_callback=self._callback,
159
+ )
160
+
161
+ if self._stream is None:
162
+ self._logger.error("start failed. No stream created.")
163
+ self._logger.debug("Microphone.start LEAVE")
164
+ return False
165
+
166
+ self._exit.clear()
167
+ if self._stream is not None:
168
+ self._stream.start_stream()
169
+
170
+ self._logger.notice("start succeeded")
171
+ self._logger.debug("Microphone.start LEAVE")
172
+ return True
173
+
174
+ def mute(self) -> bool:
175
+ """
176
+ mute - mutes the microphone stream
177
+
178
+ Returns:
179
+ bool: True if the stream was muted, False otherwise
180
+ """
181
+ self._logger.verbose("Microphone.mute ENTER")
182
+
183
+ if self._stream is None:
184
+ self._logger.error("mute failed. Library not initialized.")
185
+ self._logger.verbose("Microphone.mute LEAVE")
186
+ return False
187
+
188
+ self._is_muted = True
189
+
190
+ self._logger.notice("mute succeeded")
191
+ self._logger.verbose("Microphone.mute LEAVE")
192
+ return True
193
+
194
+ def unmute(self) -> bool:
195
+ """
196
+ unmute - unmutes the microphone stream
197
+
198
+ Returns:
199
+ bool: True if the stream was unmuted, False otherwise
200
+ """
201
+ self._logger.verbose("Microphone.unmute ENTER")
202
+
203
+ if self._stream is None:
204
+ self._logger.error("unmute failed. Library not initialized.")
205
+ self._logger.verbose("Microphone.unmute LEAVE")
206
+ return False
207
+
208
+ self._is_muted = False
209
+
210
+ self._logger.notice("unmute succeeded")
211
+ self._logger.verbose("Microphone.unmute LEAVE")
212
+ return True
213
+
214
+ def is_muted(self) -> bool:
215
+ """
216
+ is_muted - returns the state of the stream
217
+
218
+ Args:
219
+ None
220
+
221
+ Returns:
222
+ True if the stream is muted, False otherwise
223
+ """
224
+ self._logger.spam("Microphone.is_muted ENTER")
225
+
226
+ if self._stream is None:
227
+ self._logger.spam("is_muted: stream is None")
228
+ self._logger.spam("Microphone.is_muted LEAVE")
229
+ return False
230
+
231
+ val = self._is_muted
232
+
233
+ self._logger.spam("is_muted: %s", val)
234
+ self._logger.spam("Microphone.is_muted LEAVE")
235
+ return val
236
+
237
+ def finish(self) -> bool:
238
+ """
239
+ finish - stops the microphone stream
240
+
241
+ Returns:
242
+ bool: True if the stream was stopped, False otherwise
243
+ """
244
+ self._logger.debug("Microphone.finish ENTER")
245
+
246
+ self._logger.notice("signal exit")
247
+ self._exit.set()
248
+
249
+ # Stop the stream.
250
+ if self._stream is not None:
251
+ self._logger.notice("stopping stream...")
252
+ self._stream.stop_stream()
253
+ self._stream.close()
254
+ self._logger.notice("stream stopped")
255
+
256
+ # clean up the thread
257
+ if (
258
+ # inspect.iscoroutinefunction(self._push_callback_org)
259
+ # and
260
+ self._asyncio_thread
261
+ is not None
262
+ ):
263
+ self._logger.notice("stopping _asyncio_loop...")
264
+ self._asyncio_loop.call_soon_threadsafe(self._asyncio_loop.stop)
265
+ self._asyncio_thread.join()
266
+ self._logger.notice("_asyncio_thread joined")
267
+ self._stream = None
268
+ self._asyncio_thread = None
269
+
270
+ self._logger.notice("finish succeeded")
271
+ self._logger.debug("Microphone.finish LEAVE")
272
+
273
+ return True
274
+
275
+ def _callback(
276
+ self, input_data, frame_count, time_info, status_flags
277
+ ): # pylint: disable=unused-argument
278
+ """
279
+ The callback used to process data in callback mode.
280
+ """
281
+ # dynamic import of pyaudio as not to force the requirements on the SDK (and users)
282
+ import pyaudio # pylint: disable=import-outside-toplevel
283
+
284
+ if self._exit.is_set():
285
+ self._logger.notice("_callback exit is Set. stopping...")
286
+ return None, pyaudio.paAbort
287
+
288
+ if input_data is None:
289
+ self._logger.warning("input_data is None")
290
+ return None, pyaudio.paContinue
291
+
292
+ try:
293
+ if self._is_muted:
294
+ size = len(input_data)
295
+ input_data = b"\x00" * size
296
+
297
+ self._push_callback(input_data)
298
+ except Exception as e:
299
+ self._logger.error("Error while sending: %s", str(e))
300
+ raise
301
+
302
+ return input_data, pyaudio.paContinue
deepgram/audio/speaker/__init__.py ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ # Copyright 2023-2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ from .speaker import Speaker
6
+ from .errors import DeepgramSpeakerError
7
+ from .constants import LOGGING, CHANNELS, RATE, CHUNK, PLAYBACK_DELTA
deepgram/audio/speaker/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (402 Bytes). View file
 
deepgram/audio/speaker/__pycache__/constants.cpython-313.pyc ADDED
Binary file (412 Bytes). View file
 
deepgram/audio/speaker/__pycache__/errors.cpython-313.pyc ADDED
Binary file (1.14 kB). View file
 
deepgram/audio/speaker/__pycache__/speaker.cpython-313.pyc ADDED
Binary file (19.2 kB). View file
 
deepgram/audio/speaker/constants.py ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ from ...utils import verboselogs
6
+
7
+ # Constants for speaker
8
+ LOGGING = verboselogs.WARNING
9
+ TIMEOUT = 0.050
10
+ CHANNELS = 1
11
+ RATE = 16000
12
+ CHUNK = 8194
13
+
14
+ # Constants for speaker
15
+ PLAYBACK_DELTA = 2000
deepgram/audio/speaker/errors.py ADDED
@@ -0,0 +1,21 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+
6
+ # exceptions for speaker
7
+ class DeepgramSpeakerError(Exception):
8
+ """
9
+ Exception raised for known errors related to Speaker library.
10
+
11
+ Attributes:
12
+ message (str): The error message describing the exception.
13
+ """
14
+
15
+ def __init__(self, message: str):
16
+ super().__init__(message)
17
+ self.name = "DeepgramSpeakerError"
18
+ self.message = message
19
+
20
+ def __str__(self):
21
+ return f"{self.name}: {self.message}"
deepgram/audio/speaker/speaker.py ADDED
@@ -0,0 +1,380 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ import asyncio
6
+ import inspect
7
+ import queue
8
+ import threading
9
+ from typing import Optional, Callable, Union, TYPE_CHECKING
10
+ import logging
11
+ from datetime import datetime
12
+
13
+ import websockets
14
+
15
+ from ...utils import verboselogs
16
+ from .constants import LOGGING, CHANNELS, RATE, CHUNK, TIMEOUT, PLAYBACK_DELTA
17
+
18
+ from ..microphone import Microphone
19
+
20
+ if TYPE_CHECKING:
21
+ import pyaudio
22
+
23
+ HALF_SECOND = 0.5
24
+
25
+
26
+ class Speaker: # pylint: disable=too-many-instance-attributes
27
+ """
28
+ This implements a speaker for local audio output. This uses PyAudio under the hood.
29
+ """
30
+
31
+ _logger: verboselogs.VerboseLogger
32
+
33
+ _audio: Optional["pyaudio.PyAudio"] = None
34
+ _stream: Optional["pyaudio.Stream"] = None
35
+
36
+ _chunk: int
37
+ _rate: int
38
+ _channels: int
39
+ _output_device_index: Optional[int] = None
40
+
41
+ # last time we received audio
42
+ _last_datagram: datetime = datetime.now()
43
+ _last_play_delta_in_ms: int
44
+ _lock_wait: threading.Lock
45
+
46
+ _queue: queue.Queue
47
+ _exit: threading.Event
48
+
49
+ _thread: Optional[threading.Thread] = None
50
+ # _asyncio_loop: asyncio.AbstractEventLoop
51
+ # _asyncio_thread: threading.Thread
52
+ _receiver_thread: Optional[threading.Thread] = None
53
+ _loop: Optional[asyncio.AbstractEventLoop] = None
54
+
55
+ _push_callback_org: Optional[Callable] = None
56
+ _push_callback: Optional[Callable] = None
57
+ _pull_callback_org: Optional[Callable] = None
58
+ _pull_callback: Optional[Callable] = None
59
+
60
+ _microphone: Optional[Microphone] = None
61
+
62
+ def __init__(
63
+ self,
64
+ pull_callback: Optional[Callable] = None,
65
+ push_callback: Optional[Callable] = None,
66
+ verbose: int = LOGGING,
67
+ rate: int = RATE,
68
+ chunk: int = CHUNK,
69
+ channels: int = CHANNELS,
70
+ last_play_delta_in_ms: int = PLAYBACK_DELTA,
71
+ output_device_index: Optional[int] = None,
72
+ microphone: Optional[Microphone] = None,
73
+ ): # pylint: disable=too-many-positional-arguments
74
+ # dynamic import of pyaudio as not to force the requirements on the SDK (and users)
75
+ import pyaudio # pylint: disable=import-outside-toplevel
76
+
77
+ self._logger = verboselogs.VerboseLogger(__name__)
78
+ self._logger.addHandler(logging.StreamHandler())
79
+ self._logger.setLevel(verbose)
80
+
81
+ self._exit = threading.Event()
82
+ self._queue = queue.Queue()
83
+
84
+ self._last_datagram = datetime.now()
85
+ self._lock_wait = threading.Lock()
86
+
87
+ self._microphone = microphone
88
+
89
+ self._audio = pyaudio.PyAudio()
90
+ self._chunk = chunk
91
+ self._rate = rate
92
+ self._format = pyaudio.paInt16
93
+ self._channels = channels
94
+ self._last_play_delta_in_ms = last_play_delta_in_ms
95
+ self._output_device_index = output_device_index
96
+
97
+ self._push_callback_org = push_callback
98
+ self._pull_callback_org = pull_callback
99
+
100
+ def set_push_callback(self, push_callback: Callable) -> None:
101
+ """
102
+ set_push_callback - sets the callback function to be called when data is sent.
103
+
104
+ Args:
105
+ push_callback (Callable): The callback function to be called when data is send.
106
+ This should be the websocket handle message function.
107
+
108
+ Returns:
109
+ None
110
+ """
111
+ self._push_callback_org = push_callback
112
+
113
+ def set_pull_callback(self, pull_callback: Callable) -> None:
114
+ """
115
+ set_pull_callback - sets the callback function to be called when data is received.
116
+
117
+ Args:
118
+ pull_callback (Callable): The callback function to be called when data is received.
119
+ This should be the websocket recv function.
120
+
121
+ Returns:
122
+ None
123
+ """
124
+ self._pull_callback_org = pull_callback
125
+
126
+ def start(self, active_loop: Optional[asyncio.AbstractEventLoop] = None) -> bool:
127
+ """
128
+ starts - starts the Speaker stream
129
+
130
+ Args:
131
+ socket (Union[SyncClientConnection, AsyncClientConnection]): The socket to receive audio data from.
132
+
133
+ Returns:
134
+ bool: True if the stream was started, False otherwise
135
+ """
136
+ self._logger.debug("Speaker.start ENTER")
137
+
138
+ self._logger.info("format: %s", self._format)
139
+ self._logger.info("channels: %d", self._channels)
140
+ self._logger.info("rate: %d", self._rate)
141
+ self._logger.info("chunk: %d", self._chunk)
142
+ # self._logger.info("output_device_id: %d", self._output_device_index)
143
+
144
+ # Automatically get the current running event loop
145
+ if inspect.iscoroutinefunction(self._push_callback_org) and active_loop is None:
146
+ self._logger.verbose("get default running asyncio loop")
147
+ self._loop = asyncio.get_running_loop()
148
+
149
+ self._exit.clear()
150
+ self._queue = queue.Queue()
151
+
152
+ if self._audio is not None:
153
+ self._stream = self._audio.open(
154
+ format=self._format,
155
+ channels=self._channels,
156
+ rate=self._rate,
157
+ input=False,
158
+ output=True,
159
+ frames_per_buffer=self._chunk,
160
+ output_device_index=self._output_device_index,
161
+ )
162
+
163
+ if self._stream is None:
164
+ self._logger.error("start failed. No stream created.")
165
+ self._logger.debug("Speaker.start LEAVE")
166
+ return False
167
+
168
+ self._push_callback = self._push_callback_org
169
+ self._pull_callback = self._pull_callback_org
170
+
171
+ # start the play thread
172
+ self._thread = threading.Thread(
173
+ target=self._play, args=(self._queue, self._stream, self._exit), daemon=True
174
+ )
175
+ self._thread.start()
176
+
177
+ # Start the stream
178
+ if self._stream is not None:
179
+ self._stream.start_stream()
180
+
181
+ # Start the receiver thread within the start function
182
+ self._logger.verbose("Starting receiver thread...")
183
+ self._receiver_thread = threading.Thread(target=self._start_receiver)
184
+ self._receiver_thread.start()
185
+
186
+ self._logger.notice("start succeeded")
187
+ self._logger.debug("Speaker.start LEAVE")
188
+
189
+ return True
190
+
191
+ def wait_for_complete_with_mute(self, mic: Microphone):
192
+ """
193
+ This method will mute/unmute a Microphone and block until the speak is done playing sound.
194
+ """
195
+ self._logger.debug("Speaker.wait_for_complete ENTER")
196
+
197
+ if self._microphone is not None:
198
+ mic.mute()
199
+ self.wait_for_complete()
200
+ if self._microphone is not None:
201
+ mic.unmute()
202
+
203
+ self._logger.debug("Speaker.wait_for_complete LEAVE")
204
+
205
+ def wait_for_complete(self):
206
+ """
207
+ This method will block until the speak is done playing sound.
208
+ """
209
+ self._logger.debug("Speaker.wait_for_complete ENTER")
210
+
211
+ delta_in_ms = float(self._last_play_delta_in_ms)
212
+ self._logger.debug("Last Play delta: %f", delta_in_ms)
213
+
214
+ # set to now
215
+ with self._lock_wait:
216
+ self._last_datagram = datetime.now()
217
+
218
+ while True:
219
+ # sleep for a bit
220
+ self._exit.wait(HALF_SECOND)
221
+
222
+ # check if we should exit
223
+ if self._exit.is_set():
224
+ self._logger.debug("Exiting wait_for_complete _exit is set")
225
+ break
226
+
227
+ # check the time
228
+ with self._lock_wait:
229
+ delta = datetime.now() - self._last_datagram
230
+ diff_in_ms = delta.total_seconds() * 1000
231
+ if diff_in_ms < delta_in_ms:
232
+ self._logger.debug("LastPlay delta is less than threshold")
233
+ continue
234
+
235
+ # if we get here, we are done playing audio
236
+ self._logger.debug("LastPlay delta is greater than threshold. Exit wait!")
237
+ break
238
+
239
+ self._logger.debug("Speaker.wait_for_complete LEAVE")
240
+
241
+ def _start_receiver(self):
242
+ # Check if the socket is an asyncio WebSocket
243
+ if inspect.iscoroutinefunction(self._pull_callback_org):
244
+ self._logger.verbose("Starting asyncio receiver...")
245
+ asyncio.run_coroutine_threadsafe(self._start_asyncio_receiver(), self._loop)
246
+ else:
247
+ self._logger.verbose("Starting threaded receiver...")
248
+ self._start_threaded_receiver()
249
+
250
+ async def _start_asyncio_receiver(self):
251
+ try:
252
+ while True:
253
+ if self._exit.is_set():
254
+ self._logger.verbose("Exiting receiver thread...")
255
+ break
256
+
257
+ message = await self._pull_callback()
258
+ if message is None:
259
+ self._logger.verbose("No message received...")
260
+ continue
261
+
262
+ if isinstance(message, str):
263
+ self._logger.verbose("Received control message...")
264
+ await self._push_callback(message)
265
+ elif isinstance(message, bytes):
266
+ self._logger.verbose("Received audio data...")
267
+ await self._push_callback(message)
268
+ self.add_audio_to_queue(message)
269
+ except websockets.exceptions.ConnectionClosedOK as e:
270
+ self._logger.debug("send() exiting gracefully: %d", e.code)
271
+ except websockets.exceptions.ConnectionClosed as e:
272
+ if e.code in [1000, 1001]:
273
+ self._logger.debug("send() exiting gracefully: %d", e.code)
274
+ return
275
+ self._logger.error("_start_asyncio_receiver - ConnectionClosed: %s", str(e))
276
+ except websockets.exceptions.WebSocketException as e:
277
+ self._logger.error(
278
+ "_start_asyncio_receiver- WebSocketException: %s", str(e)
279
+ )
280
+ except Exception as e: # pylint: disable=broad-except
281
+ self._logger.error("_start_asyncio_receiver exception: %s", str(e))
282
+
283
+ def _start_threaded_receiver(self):
284
+ try:
285
+ while True:
286
+ if self._exit.is_set():
287
+ self._logger.verbose("Exiting receiver thread...")
288
+ break
289
+
290
+ message = self._pull_callback()
291
+ if message is None:
292
+ self._logger.verbose("No message received...")
293
+ continue
294
+
295
+ if isinstance(message, str):
296
+ self._logger.verbose("Received control message...")
297
+ self._push_callback(message)
298
+ elif isinstance(message, bytes):
299
+ self._logger.verbose("Received audio data...")
300
+ self._push_callback(message)
301
+ self.add_audio_to_queue(message)
302
+ except Exception as e: # pylint: disable=broad-except
303
+ self._logger.notice("_start_threaded_receiver exception: %s", str(e))
304
+
305
+ def add_audio_to_queue(self, data: bytes) -> None:
306
+ """
307
+ add_audio_to_queue - adds audio data to the Speaker queue
308
+
309
+ Args:
310
+ data (bytes): The audio data to add to the queue
311
+ """
312
+ self._queue.put(data)
313
+
314
+ def finish(self) -> bool:
315
+ """
316
+ finish - stops the Speaker stream
317
+
318
+ Returns:
319
+ bool: True if the stream was stopped, False otherwise
320
+ """
321
+ self._logger.debug("Speaker.finish ENTER")
322
+
323
+ self._logger.notice("signal exit")
324
+ self._exit.set()
325
+
326
+ if self._stream is not None:
327
+ self._logger.notice("stopping stream...")
328
+ self._stream.stop_stream()
329
+ self._stream.close()
330
+ self._logger.notice("stream stopped")
331
+
332
+ if self._thread is not None:
333
+ self._logger.notice("joining _thread...")
334
+ self._thread.join()
335
+ self._logger.notice("thread stopped")
336
+
337
+ if self._receiver_thread is not None:
338
+ self._logger.notice("stopping _receiver_thread...")
339
+ self._receiver_thread.join()
340
+ self._logger.notice("_receiver_thread joined")
341
+
342
+ with self._queue.mutex:
343
+ self._queue.queue.clear()
344
+
345
+ self._stream = None
346
+ self._thread = None
347
+ self._receiver_thread = None
348
+
349
+ self._logger.notice("finish succeeded")
350
+ self._logger.debug("Speaker.finish LEAVE")
351
+
352
+ return True
353
+
354
+ def _play(self, audio_out, stream, stop):
355
+ """
356
+ _play - plays audio data from the Speaker queue callback for portaudio
357
+ """
358
+ while not stop.is_set():
359
+ try:
360
+ if self._microphone is not None and self._microphone.is_muted():
361
+ with self._lock_wait:
362
+ delta = datetime.now() - self._last_datagram
363
+ diff_in_ms = delta.total_seconds() * 1000
364
+ if diff_in_ms > float(self._last_play_delta_in_ms):
365
+ self._logger.debug(
366
+ "LastPlay delta is greater than threshold. Unmute!"
367
+ )
368
+ self._microphone.unmute()
369
+
370
+ data = audio_out.get(True, TIMEOUT)
371
+ with self._lock_wait:
372
+ self._last_datagram = datetime.now()
373
+ if self._microphone is not None and not self._microphone.is_muted():
374
+ self._logger.debug("New speaker sound detected. Mute!")
375
+ self._microphone.mute()
376
+ stream.write(data)
377
+ except queue.Empty:
378
+ pass
379
+ except Exception as e: # pylint: disable=broad-except
380
+ self._logger.error("_play exception: %s", str(e))
deepgram/client.py ADDED
@@ -0,0 +1,669 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2023-2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ from typing import Optional
6
+ from importlib import import_module
7
+ import os
8
+ import logging
9
+ import deprecation # type: ignore
10
+
11
+ from . import __version__
12
+ from .utils import verboselogs
13
+
14
+ # common
15
+ # pylint: disable=unused-import
16
+ from .clients import (
17
+ TextSource,
18
+ BufferSource,
19
+ StreamSource,
20
+ FileSource,
21
+ UrlSource,
22
+ )
23
+ from .clients import BaseResponse
24
+ from .clients import (
25
+ Average,
26
+ Intent,
27
+ Intents,
28
+ IntentsInfo,
29
+ Segment,
30
+ SentimentInfo,
31
+ Sentiment,
32
+ Sentiments,
33
+ SummaryInfo,
34
+ Topic,
35
+ Topics,
36
+ TopicsInfo,
37
+ )
38
+ from .clients import (
39
+ ModelInfo,
40
+ Hit,
41
+ Search,
42
+ )
43
+ from .clients import (
44
+ OpenResponse,
45
+ CloseResponse,
46
+ UnhandledResponse,
47
+ ErrorResponse,
48
+ )
49
+ from .clients import (
50
+ DeepgramError,
51
+ DeepgramTypeError,
52
+ DeepgramModuleError,
53
+ DeepgramApiError,
54
+ DeepgramUnknownApiError,
55
+ )
56
+
57
+ # listen client
58
+ from .clients import ListenRouter, ReadRouter, SpeakRouter, AgentRouter
59
+
60
+ # speech-to-text
61
+ from .clients import LiveClient, AsyncLiveClient # backward compat
62
+ from .clients import (
63
+ ListenWebSocketClient,
64
+ AsyncListenWebSocketClient,
65
+ )
66
+ from .clients import (
67
+ ListenWebSocketOptions,
68
+ LiveOptions,
69
+ LiveTranscriptionEvents,
70
+ )
71
+
72
+ # live client responses
73
+ from .clients import (
74
+ #### top level
75
+ LiveResultResponse,
76
+ ListenWSMetadataResponse,
77
+ SpeechStartedResponse,
78
+ UtteranceEndResponse,
79
+ #### common websocket response
80
+ # OpenResponse,
81
+ # CloseResponse,
82
+ # ErrorResponse,
83
+ # UnhandledResponse,
84
+ #### unique
85
+ ListenWSMetadata,
86
+ ListenWSAlternative,
87
+ ListenWSChannel,
88
+ ListenWSWord,
89
+ )
90
+
91
+ # prerecorded
92
+ from .clients import (
93
+ # common
94
+ # UrlSource,
95
+ # BufferSource,
96
+ # StreamSource,
97
+ # TextSource,
98
+ # FileSource,
99
+ # unique
100
+ PreRecordedStreamSource,
101
+ PrerecordedSource,
102
+ ListenRestSource,
103
+ )
104
+
105
+ from .clients import (
106
+ PreRecordedClient,
107
+ AsyncPreRecordedClient,
108
+ ) # backward compat
109
+ from .clients import (
110
+ ListenRESTClient,
111
+ AsyncListenRESTClient,
112
+ )
113
+ from .clients import (
114
+ ListenRESTOptions,
115
+ PrerecordedOptions,
116
+ )
117
+
118
+ # rest client responses
119
+ from .clients import (
120
+ #### top level
121
+ AsyncPrerecordedResponse,
122
+ PrerecordedResponse,
123
+ SyncPrerecordedResponse,
124
+ #### shared
125
+ # Average,
126
+ # Intent,
127
+ # Intents,
128
+ # IntentsInfo,
129
+ # Segment,
130
+ # SentimentInfo,
131
+ # Sentiment,
132
+ # Sentiments,
133
+ # SummaryInfo,
134
+ # Topic,
135
+ # Topics,
136
+ # TopicsInfo,
137
+ #### between rest and websocket
138
+ # ModelInfo,
139
+ # Alternative,
140
+ # Hit,
141
+ # Search,
142
+ # Channel,
143
+ # Word,
144
+ # unique
145
+ Entity,
146
+ ListenRESTMetadata,
147
+ Paragraph,
148
+ Paragraphs,
149
+ ListenRESTResults,
150
+ Sentence,
151
+ Summaries,
152
+ SummaryV1,
153
+ SummaryV2,
154
+ Translation,
155
+ Utterance,
156
+ Warning,
157
+ ListenRESTAlternative,
158
+ ListenRESTChannel,
159
+ ListenRESTWord,
160
+ )
161
+
162
+ # read
163
+ from .clients import ReadClient, AsyncReadClient
164
+ from .clients import AnalyzeClient, AsyncAnalyzeClient
165
+ from .clients import (
166
+ AnalyzeOptions,
167
+ AnalyzeStreamSource,
168
+ AnalyzeSource,
169
+ )
170
+
171
+ # read client responses
172
+ from .clients import (
173
+ #### top level
174
+ AsyncAnalyzeResponse,
175
+ SyncAnalyzeResponse,
176
+ AnalyzeResponse,
177
+ #### shared
178
+ # Average,
179
+ # Intent,
180
+ # Intents,
181
+ # IntentsInfo,
182
+ # Segment,
183
+ # SentimentInfo,
184
+ # Sentiment,
185
+ # Sentiments,
186
+ # SummaryInfo,
187
+ # Topic,
188
+ # Topics,
189
+ # TopicsInfo,
190
+ #### unique
191
+ AnalyzeMetadata,
192
+ AnalyzeResults,
193
+ AnalyzeSummary,
194
+ )
195
+
196
+ # speak
197
+ ## speak REST
198
+ from .clients import (
199
+ #### top level
200
+ SpeakRESTOptions,
201
+ SpeakOptions, # backward compat
202
+ #### common
203
+ # TextSource,
204
+ # BufferSource,
205
+ # StreamSource,
206
+ # FileSource,
207
+ #### unique
208
+ SpeakSource,
209
+ SpeakRestSource,
210
+ SpeakRESTSource,
211
+ )
212
+
213
+ from .clients import (
214
+ SpeakClient, # backward compat
215
+ SpeakRESTClient,
216
+ AsyncSpeakRESTClient,
217
+ )
218
+
219
+ from .clients import (
220
+ SpeakResponse, # backward compat
221
+ SpeakRESTResponse,
222
+ )
223
+
224
+ ## speak WebSocket
225
+ from .clients import SpeakWebSocketEvents, SpeakWebSocketMessage
226
+
227
+ from .clients import (
228
+ SpeakWSOptions,
229
+ )
230
+
231
+ from .clients import (
232
+ SpeakWebSocketClient,
233
+ AsyncSpeakWebSocketClient,
234
+ SpeakWSClient,
235
+ AsyncSpeakWSClient,
236
+ )
237
+
238
+ from .clients import (
239
+ #### top level
240
+ SpeakWSMetadataResponse,
241
+ FlushedResponse,
242
+ ClearedResponse,
243
+ WarningResponse,
244
+ #### common websocket response
245
+ # OpenResponse,
246
+ # CloseResponse,
247
+ # UnhandledResponse,
248
+ # ErrorResponse,
249
+ )
250
+
251
+ # auth client classes
252
+ from .clients import AuthRESTClient, AsyncAuthRESTClient
253
+
254
+ # auth client responses
255
+ from .clients import (
256
+ GrantTokenResponse,
257
+ )
258
+
259
+ # manage client classes/input
260
+ from .clients import ManageClient, AsyncManageClient
261
+ from .clients import (
262
+ ProjectOptions,
263
+ KeyOptions,
264
+ ScopeOptions,
265
+ InviteOptions,
266
+ UsageRequestOptions,
267
+ UsageSummaryOptions,
268
+ UsageFieldsOptions,
269
+ )
270
+
271
+ # manage client responses
272
+ from .clients import (
273
+ #### top level
274
+ Message,
275
+ ProjectsResponse,
276
+ ModelResponse,
277
+ ModelsResponse,
278
+ MembersResponse,
279
+ KeyResponse,
280
+ KeysResponse,
281
+ ScopesResponse,
282
+ InvitesResponse,
283
+ UsageRequest,
284
+ UsageResponse,
285
+ UsageRequestsResponse,
286
+ UsageSummaryResponse,
287
+ UsageFieldsResponse,
288
+ BalancesResponse,
289
+ #### shared
290
+ Project,
291
+ STTDetails,
292
+ TTSMetadata,
293
+ TTSDetails,
294
+ Member,
295
+ Key,
296
+ Invite,
297
+ Config,
298
+ STTUsageDetails,
299
+ Callback,
300
+ TokenDetail,
301
+ SpeechSegment,
302
+ TTSUsageDetails,
303
+ STTTokens,
304
+ TTSTokens,
305
+ UsageSummaryResults,
306
+ Resolution,
307
+ UsageModel,
308
+ Balance,
309
+ )
310
+
311
+ # on-prem
312
+ from .clients import (
313
+ OnPremClient,
314
+ AsyncOnPremClient,
315
+ SelfHostedClient,
316
+ AsyncSelfHostedClient,
317
+ )
318
+
319
+
320
+ # agent
321
+ from .clients import AgentWebSocketEvents
322
+
323
+ # websocket
324
+ from .clients import (
325
+ AgentWebSocketClient,
326
+ AsyncAgentWebSocketClient,
327
+ )
328
+
329
+ from .clients import (
330
+ #### common websocket response
331
+ # OpenResponse,
332
+ # CloseResponse,
333
+ # ErrorResponse,
334
+ # UnhandledResponse,
335
+ #### unique
336
+ WelcomeResponse,
337
+ SettingsAppliedResponse,
338
+ ConversationTextResponse,
339
+ UserStartedSpeakingResponse,
340
+ AgentThinkingResponse,
341
+ FunctionCallRequest,
342
+ AgentStartedSpeakingResponse,
343
+ AgentAudioDoneResponse,
344
+ InjectionRefusedResponse,
345
+ )
346
+
347
+ from .clients import (
348
+ # top level
349
+ SettingsOptions,
350
+ UpdatePromptOptions,
351
+ UpdateSpeakOptions,
352
+ InjectAgentMessageOptions,
353
+ FunctionCallResponse,
354
+ AgentKeepAlive,
355
+ # sub level
356
+ Listen,
357
+ ListenProvider,
358
+ Speak,
359
+ SpeakProvider,
360
+ Header,
361
+ Item,
362
+ Properties,
363
+ Parameters,
364
+ Function,
365
+ Think,
366
+ ThinkProvider,
367
+ Agent,
368
+ Input,
369
+ Output,
370
+ Audio,
371
+ Endpoint,
372
+ )
373
+
374
+
375
+ # client errors and options
376
+ from .options import DeepgramClientOptions, ClientOptionsFromEnv
377
+ from .errors import DeepgramApiKeyError
378
+
379
+ # pylint: enable=unused-import
380
+
381
+
382
+ class Deepgram: # pylint: disable=broad-exception-raised
383
+ """
384
+ The Deepgram class is no longer a class in version 3 of this SDK.
385
+ """
386
+
387
+ def __init__(self, *anything):
388
+ raise Exception(
389
+ """
390
+ FATAL ERROR:
391
+ You are attempting to instantiate a Deepgram object, which is no longer a class in version 3 of this SDK.
392
+
393
+ To fix this issue:
394
+ 1. You need to revert to the previous version 2 of the SDK: pip install deepgram-sdk==2.12.0
395
+ 2. or, update your application's code to use version 3 of this SDK. See the README for more information.
396
+
397
+ Things to consider:
398
+
399
+ - This Version 3 of the SDK requires Python 3.10 or higher.
400
+ Older versions (3.9 and lower) of Python are nearing end-of-life: https://devguide.python.org/versions/
401
+ Understand the risks of using a version of Python nearing EOL.
402
+
403
+ - Version 2 of the SDK will receive maintenance updates in the form of security fixes only.
404
+ No new features will be added to version 2 of the SDK.
405
+ """
406
+ )
407
+
408
+
409
+ class DeepgramClient:
410
+ """
411
+ Represents a client for interacting with the Deepgram API.
412
+
413
+ This class provides a client for making requests to the Deepgram API with various configuration options.
414
+
415
+ Attributes:
416
+ api_key (str): The Deepgram API key used for authentication.
417
+ config_options (DeepgramClientOptions): An optional configuration object specifying client options.
418
+
419
+ Raises:
420
+ DeepgramApiKeyError: If the API key is missing or invalid.
421
+
422
+ Methods:
423
+ listen: Returns a ListenClient instance for interacting with Deepgram's transcription services.
424
+
425
+ manage: (Preferred) Returns a Threaded ManageClient instance for managing Deepgram resources.
426
+ selfhosted: (Preferred) Returns an Threaded SelfHostedClient instance for interacting with Deepgram's on-premises API.
427
+
428
+ asyncmanage: Returns an (Async) ManageClient instance for managing Deepgram resources.
429
+ asyncselfhosted: Returns an (Async) SelfHostedClient instance for interacting with Deepgram's on-premises API.
430
+ """
431
+
432
+ _config: DeepgramClientOptions
433
+ _logger: verboselogs.VerboseLogger
434
+
435
+ def __init__(
436
+ self,
437
+ api_key: str = "",
438
+ config: Optional[DeepgramClientOptions] = None,
439
+ ):
440
+ self._logger = verboselogs.VerboseLogger(__name__)
441
+ self._logger.addHandler(logging.StreamHandler())
442
+
443
+ if api_key == "" and config is not None:
444
+ self._logger.info("Attempting to set API key from config object")
445
+ api_key = config.api_key
446
+ if api_key == "":
447
+ self._logger.info("Attempting to set API key from environment variable")
448
+ api_key = os.getenv("DEEPGRAM_API_KEY", "")
449
+ if api_key == "":
450
+ self._logger.warning("WARNING: API key is missing")
451
+
452
+ self.api_key = api_key
453
+ if config is None: # Use default configuration
454
+ self._config = DeepgramClientOptions(self.api_key)
455
+ else:
456
+ config.set_apikey(self.api_key)
457
+ self._config = config
458
+
459
+ @property
460
+ def listen(self):
461
+ """
462
+ Returns a Listen dot-notation router for interacting with Deepgram's transcription services.
463
+ """
464
+ return ListenRouter(self._config)
465
+
466
+ @property
467
+ def read(self):
468
+ """
469
+ Returns a Read dot-notation router for interacting with Deepgram's read services.
470
+ """
471
+ return ReadRouter(self._config)
472
+
473
+ @property
474
+ def speak(self):
475
+ """
476
+ Returns a Speak dot-notation router for interacting with Deepgram's speak services.
477
+ """
478
+ return SpeakRouter(self._config)
479
+
480
+ @property
481
+ @deprecation.deprecated(
482
+ deprecated_in="3.4.0",
483
+ removed_in="4.0.0",
484
+ current_version=__version__,
485
+ details="deepgram.asyncspeak is deprecated. Use deepgram.speak.asyncrest instead.",
486
+ )
487
+ def asyncspeak(self):
488
+ """
489
+ DEPRECATED: deepgram.asyncspeak is deprecated. Use deepgram.speak.asyncrest instead.
490
+ """
491
+ return self.Version(self._config, "asyncspeak")
492
+
493
+ @property
494
+ def manage(self):
495
+ """
496
+ Returns a ManageClient instance for managing Deepgram resources.
497
+ """
498
+ return self.Version(self._config, "manage")
499
+
500
+ @property
501
+ def asyncmanage(self):
502
+ """
503
+ Returns an AsyncManageClient instance for managing Deepgram resources.
504
+ """
505
+ return self.Version(self._config, "asyncmanage")
506
+
507
+ @property
508
+ def auth(self):
509
+ """
510
+ Returns an AuthRESTClient instance for managing short-lived tokens.
511
+ """
512
+ return self.Version(self._config, "auth")
513
+
514
+ @property
515
+ def asyncauth(self):
516
+ """
517
+ Returns an AsyncAuthRESTClient instance for managing short-lived tokens.
518
+ """
519
+ return self.Version(self._config, "asyncauth")
520
+
521
+ @property
522
+ @deprecation.deprecated(
523
+ deprecated_in="3.4.0",
524
+ removed_in="4.0.0",
525
+ current_version=__version__,
526
+ details="deepgram.onprem is deprecated. Use deepgram.speak.selfhosted instead.",
527
+ )
528
+ def onprem(self):
529
+ """
530
+ DEPRECATED: deepgram.onprem is deprecated. Use deepgram.speak.selfhosted instead.
531
+ """
532
+ return self.Version(self._config, "selfhosted")
533
+
534
+ @property
535
+ def selfhosted(self):
536
+ """
537
+ Returns an SelfHostedClient instance for interacting with Deepgram's on-premises API.
538
+ """
539
+ return self.Version(self._config, "selfhosted")
540
+
541
+ @property
542
+ @deprecation.deprecated(
543
+ deprecated_in="3.4.0",
544
+ removed_in="4.0.0",
545
+ current_version=__version__,
546
+ details="deepgram.asynconprem is deprecated. Use deepgram.speak.asyncselfhosted instead.",
547
+ )
548
+ def asynconprem(self):
549
+ """
550
+ DEPRECATED: deepgram.asynconprem is deprecated. Use deepgram.speak.asyncselfhosted instead.
551
+ """
552
+ return self.Version(self._config, "asyncselfhosted")
553
+
554
+ @property
555
+ def asyncselfhosted(self):
556
+ """
557
+ Returns an AsyncSelfHostedClient instance for interacting with Deepgram's on-premises API.
558
+ """
559
+ return self.Version(self._config, "asyncselfhosted")
560
+
561
+ @property
562
+ def agent(self):
563
+ """
564
+ Returns a Agent dot-notation router for interacting with Deepgram's speak services.
565
+ """
566
+ return AgentRouter(self._config)
567
+
568
+ # INTERNAL CLASSES
569
+ class Version:
570
+ """
571
+ Represents a version of the Deepgram API.
572
+ """
573
+
574
+ _logger: verboselogs.VerboseLogger
575
+ _config: DeepgramClientOptions
576
+ _parent: str
577
+
578
+ def __init__(self, config, parent: str):
579
+ self._logger = verboselogs.VerboseLogger(__name__)
580
+ self._logger.addHandler(logging.StreamHandler())
581
+ self._logger.setLevel(config.verbose)
582
+
583
+ self._config = config
584
+ self._parent = parent
585
+
586
+ # FUTURE VERSIONING:
587
+ # When v2 or v1.1beta1 or etc. This allows easy access to the latest version of the API.
588
+ # @property
589
+ # def latest(self):
590
+ # match self._parent:
591
+ # case "manage":
592
+ # return ManageClient(self._config)
593
+ # case "selfhosted":
594
+ # return SelfHostedClient(self._config)
595
+ # case _:
596
+ # raise DeepgramModuleError("Invalid parent")
597
+
598
+ def v(self, version: str = ""):
599
+ # pylint: disable-msg=too-many-statements
600
+ """
601
+ Returns a client for the specified version of the API.
602
+ """
603
+ self._logger.debug("Version.v ENTER")
604
+ self._logger.info("version: %s", version)
605
+ if len(version) == 0:
606
+ self._logger.error("version is empty")
607
+ self._logger.debug("Version.v LEAVE")
608
+ raise DeepgramModuleError("Invalid module version")
609
+
610
+ parent = ""
611
+ filename = ""
612
+ classname = ""
613
+ match self._parent:
614
+ case "manage":
615
+ parent = "manage"
616
+ filename = "client"
617
+ classname = "ManageClient"
618
+ case "asyncmanage":
619
+ parent = "manage"
620
+ filename = "async_client"
621
+ classname = "AsyncManageClient"
622
+ case "asyncspeak":
623
+ return AsyncSpeakRESTClient(self._config)
624
+ case "selfhosted":
625
+ parent = "selfhosted"
626
+ filename = "client"
627
+ classname = "SelfHostedClient"
628
+ case "asyncselfhosted":
629
+ parent = "selfhosted"
630
+ filename = "async_client"
631
+ classname = "AsyncSelfHostedClient"
632
+ case "auth":
633
+ parent = "auth"
634
+ filename = "client"
635
+ classname = "AuthRESTClient"
636
+ case "asyncauth":
637
+ parent = "auth"
638
+ filename = "async_client"
639
+ classname = "AsyncAuthRESTClient"
640
+ case _:
641
+ self._logger.error("parent unknown: %s", self._parent)
642
+ self._logger.debug("Version.v LEAVE")
643
+ raise DeepgramModuleError("Invalid parent type")
644
+
645
+ # create class path
646
+ path = f"deepgram.clients.{parent}.v{version}.{filename}"
647
+ self._logger.info("path: %s", path)
648
+ self._logger.info("classname: %s", classname)
649
+
650
+ # import class
651
+ mod = import_module(path)
652
+ if mod is None:
653
+ self._logger.error("module path is None")
654
+ self._logger.debug("Version.v LEAVE")
655
+ raise DeepgramModuleError("Unable to find package")
656
+
657
+ my_class = getattr(mod, classname)
658
+ if my_class is None:
659
+ self._logger.error("my_class is None")
660
+ self._logger.debug("Version.v LEAVE")
661
+ raise DeepgramModuleError("Unable to find class")
662
+
663
+ # instantiate class
664
+ my_class_instance = my_class(self._config)
665
+ self._logger.notice("Version.v succeeded")
666
+ self._logger.debug("Version.v LEAVE")
667
+ return my_class_instance
668
+
669
+ # pylint: enable-msg=too-many-statements
deepgram/clients/__init__.py ADDED
@@ -0,0 +1,381 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2023-2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ # common
6
+ from .common import (
7
+ TextSource,
8
+ BufferSource,
9
+ StreamSource,
10
+ FileSource,
11
+ UrlSource,
12
+ )
13
+ from .common import BaseResponse
14
+
15
+ # common (shared between analze and prerecorded)
16
+ from .common import (
17
+ Average,
18
+ Intent,
19
+ Intents,
20
+ IntentsInfo,
21
+ Segment,
22
+ SentimentInfo,
23
+ Sentiment,
24
+ Sentiments,
25
+ SummaryInfo,
26
+ Topic,
27
+ Topics,
28
+ TopicsInfo,
29
+ )
30
+
31
+ # common (shared between listen rest and websocket)
32
+ from .common import (
33
+ ModelInfo,
34
+ Hit,
35
+ Search,
36
+ )
37
+ from .common import (
38
+ OpenResponse,
39
+ CloseResponse,
40
+ UnhandledResponse,
41
+ ErrorResponse,
42
+ )
43
+ from .common import (
44
+ DeepgramError,
45
+ DeepgramTypeError,
46
+ DeepgramApiError,
47
+ DeepgramUnknownApiError,
48
+ )
49
+ from .errors import DeepgramModuleError
50
+
51
+ from .listen_router import ListenRouter
52
+ from .read_router import ReadRouter
53
+ from .speak_router import SpeakRouter
54
+ from .agent_router import AgentRouter
55
+
56
+ # listen
57
+ from .listen import LiveTranscriptionEvents
58
+
59
+ ## backward compat
60
+ from .prerecorded import (
61
+ PreRecordedClient,
62
+ AsyncPreRecordedClient,
63
+ )
64
+ from .live import (
65
+ LiveClient,
66
+ AsyncLiveClient,
67
+ )
68
+
69
+ # speech-to-text rest
70
+ from .listen import ListenRESTClient, AsyncListenRESTClient
71
+
72
+ ## input
73
+ from .listen import (
74
+ # common
75
+ # UrlSource,
76
+ # BufferSource,
77
+ # StreamSource,
78
+ # TextSource,
79
+ # FileSource,
80
+ # unique
81
+ PreRecordedStreamSource,
82
+ PrerecordedSource,
83
+ ListenRestSource,
84
+ )
85
+
86
+ from .listen import (
87
+ ListenRESTOptions,
88
+ PrerecordedOptions,
89
+ )
90
+
91
+ ## output
92
+ from .listen import (
93
+ #### top level
94
+ AsyncPrerecordedResponse,
95
+ PrerecordedResponse,
96
+ SyncPrerecordedResponse,
97
+ #### shared
98
+ # Average,
99
+ # Intent,
100
+ # Intents,
101
+ # IntentsInfo,
102
+ # Segment,
103
+ # SentimentInfo,
104
+ # Sentiment,
105
+ # Sentiments,
106
+ # SummaryInfo,
107
+ # Topic,
108
+ # Topics,
109
+ # TopicsInfo,
110
+ #### between rest and websocket
111
+ # ModelInfo,
112
+ # Alternative,
113
+ # Hit,
114
+ # Search,
115
+ # Channel,
116
+ # Word,
117
+ # unique
118
+ Entity,
119
+ ListenRESTMetadata,
120
+ Paragraph,
121
+ Paragraphs,
122
+ ListenRESTResults,
123
+ Sentence,
124
+ Summaries,
125
+ SummaryV1,
126
+ SummaryV2,
127
+ Translation,
128
+ Utterance,
129
+ Warning,
130
+ ListenRESTAlternative,
131
+ ListenRESTChannel,
132
+ ListenRESTWord,
133
+ )
134
+
135
+
136
+ # speech-to-text websocket
137
+ from .listen import ListenWebSocketClient, AsyncListenWebSocketClient
138
+
139
+ ## input
140
+ from .listen import (
141
+ ListenWebSocketOptions,
142
+ LiveOptions,
143
+ )
144
+
145
+ ## output
146
+ from .listen import (
147
+ #### top level
148
+ LiveResultResponse,
149
+ ListenWSMetadataResponse,
150
+ SpeechStartedResponse,
151
+ UtteranceEndResponse,
152
+ #### common websocket response
153
+ # OpenResponse,
154
+ # CloseResponse,
155
+ # ErrorResponse,
156
+ # UnhandledResponse,
157
+ #### uniqye
158
+ ListenWSMetadata,
159
+ ListenWSWord,
160
+ ListenWSAlternative,
161
+ ListenWSChannel,
162
+ )
163
+
164
+ ## clients
165
+ from .listen import (
166
+ ListenWebSocketClient,
167
+ AsyncListenWebSocketClient,
168
+ )
169
+
170
+
171
+ # read/analyze
172
+ from .analyze import ReadClient, AsyncReadClient
173
+ from .analyze import AnalyzeClient, AsyncAnalyzeClient
174
+ from .analyze import AnalyzeOptions
175
+ from .analyze import (
176
+ # common
177
+ # UrlSource,
178
+ # TextSource,
179
+ # BufferSource,
180
+ # StreamSource,
181
+ # FileSource
182
+ # unique
183
+ AnalyzeStreamSource,
184
+ AnalyzeSource,
185
+ )
186
+ from .analyze import (
187
+ #### top level
188
+ AsyncAnalyzeResponse,
189
+ SyncAnalyzeResponse,
190
+ AnalyzeResponse,
191
+ #### shared between analyze and pre-recorded
192
+ # Average,
193
+ # Intent,
194
+ # Intents,
195
+ # IntentsInfo,
196
+ # Segment,
197
+ # SentimentInfo,
198
+ # Sentiment,
199
+ # Sentiments,
200
+ # SummaryInfo,
201
+ # Topic,
202
+ # Topics,
203
+ # TopicsInfo,
204
+ #### unique
205
+ AnalyzeMetadata,
206
+ AnalyzeResults,
207
+ AnalyzeSummary,
208
+ )
209
+
210
+ # text-to-speech
211
+ ## text-to-speech REST
212
+ from .speak import (
213
+ #### top level
214
+ SpeakRESTOptions,
215
+ SpeakOptions,
216
+ # common
217
+ # TextSource,
218
+ # BufferSource,
219
+ # StreamSource,
220
+ # FileSource,
221
+ # unique
222
+ SpeakSource,
223
+ SpeakRestSource,
224
+ SpeakRESTSource,
225
+ )
226
+
227
+ from .speak import (
228
+ SpeakClient, # backward compat
229
+ SpeakRESTClient,
230
+ AsyncSpeakRESTClient,
231
+ )
232
+
233
+ from .speak import (
234
+ SpeakResponse, # backward compat
235
+ SpeakRESTResponse,
236
+ )
237
+
238
+ ## text-to-speech WebSocket
239
+ from .speak import SpeakWebSocketEvents, SpeakWebSocketMessage
240
+
241
+ from .speak import (
242
+ SpeakWSOptions,
243
+ )
244
+
245
+ from .speak import (
246
+ SpeakWebSocketClient,
247
+ AsyncSpeakWebSocketClient,
248
+ SpeakWSClient,
249
+ AsyncSpeakWSClient,
250
+ )
251
+
252
+ from .speak import (
253
+ #### top level
254
+ SpeakWSMetadataResponse,
255
+ FlushedResponse,
256
+ ClearedResponse,
257
+ WarningResponse,
258
+ #### common websocket response
259
+ # OpenResponse,
260
+ # CloseResponse,
261
+ # UnhandledResponse,
262
+ # ErrorResponse,
263
+ )
264
+
265
+ # manage
266
+ from .manage import ManageClient, AsyncManageClient
267
+ from .manage import (
268
+ ProjectOptions,
269
+ KeyOptions,
270
+ ScopeOptions,
271
+ InviteOptions,
272
+ UsageRequestOptions,
273
+ UsageSummaryOptions,
274
+ UsageFieldsOptions,
275
+ )
276
+ from .manage import (
277
+ #### top level
278
+ Message,
279
+ ProjectsResponse,
280
+ ModelResponse,
281
+ ModelsResponse,
282
+ MembersResponse,
283
+ KeyResponse,
284
+ KeysResponse,
285
+ ScopesResponse,
286
+ InvitesResponse,
287
+ UsageRequest,
288
+ UsageResponse,
289
+ UsageRequestsResponse,
290
+ UsageSummaryResponse,
291
+ UsageFieldsResponse,
292
+ BalancesResponse,
293
+ #### shared
294
+ Project,
295
+ STTDetails,
296
+ TTSMetadata,
297
+ TTSDetails,
298
+ Member,
299
+ Key,
300
+ Invite,
301
+ Config,
302
+ STTUsageDetails,
303
+ Callback,
304
+ TokenDetail,
305
+ SpeechSegment,
306
+ TTSUsageDetails,
307
+ STTTokens,
308
+ TTSTokens,
309
+ UsageSummaryResults,
310
+ Resolution,
311
+ UsageModel,
312
+ Balance,
313
+ )
314
+
315
+ # auth
316
+ from .auth import AuthRESTClient, AsyncAuthRESTClient
317
+ from .auth import (
318
+ GrantTokenResponse,
319
+ )
320
+
321
+ # selfhosted
322
+ from .selfhosted import (
323
+ OnPremClient,
324
+ AsyncOnPremClient,
325
+ SelfHostedClient,
326
+ AsyncSelfHostedClient,
327
+ )
328
+
329
+ # agent
330
+ from .agent import AgentWebSocketEvents
331
+
332
+ # websocket
333
+ from .agent import (
334
+ AgentWebSocketClient,
335
+ AsyncAgentWebSocketClient,
336
+ )
337
+
338
+ from .agent import (
339
+ #### common websocket response
340
+ # OpenResponse,
341
+ # CloseResponse,
342
+ # ErrorResponse,
343
+ # UnhandledResponse,
344
+ #### unique
345
+ WelcomeResponse,
346
+ SettingsAppliedResponse,
347
+ ConversationTextResponse,
348
+ UserStartedSpeakingResponse,
349
+ AgentThinkingResponse,
350
+ FunctionCallRequest,
351
+ AgentStartedSpeakingResponse,
352
+ AgentAudioDoneResponse,
353
+ InjectionRefusedResponse,
354
+ )
355
+
356
+ from .agent import (
357
+ # top level
358
+ SettingsOptions,
359
+ UpdatePromptOptions,
360
+ UpdateSpeakOptions,
361
+ InjectAgentMessageOptions,
362
+ FunctionCallResponse,
363
+ AgentKeepAlive,
364
+ # sub level
365
+ Listen,
366
+ ListenProvider,
367
+ Speak,
368
+ SpeakProvider,
369
+ Header,
370
+ Item,
371
+ Properties,
372
+ Parameters,
373
+ Function,
374
+ Think,
375
+ ThinkProvider,
376
+ Agent,
377
+ Input,
378
+ Output,
379
+ Audio,
380
+ Endpoint,
381
+ )
deepgram/clients/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (6.06 kB). View file
 
deepgram/clients/__pycache__/agent_router.cpython-313.pyc ADDED
Binary file (5.82 kB). View file
 
deepgram/clients/__pycache__/errors.cpython-313.pyc ADDED
Binary file (892 Bytes). View file
 
deepgram/clients/__pycache__/listen_router.cpython-313.pyc ADDED
Binary file (9.48 kB). View file
 
deepgram/clients/__pycache__/read_router.cpython-313.pyc ADDED
Binary file (5.78 kB). View file
 
deepgram/clients/__pycache__/speak_router.cpython-313.pyc ADDED
Binary file (7.47 kB). View file
 
deepgram/clients/agent/__init__.py ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2023-2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ from .enums import AgentWebSocketEvents
6
+
7
+ # websocket
8
+ from .client import (
9
+ AgentWebSocketClient,
10
+ AsyncAgentWebSocketClient,
11
+ )
12
+
13
+ from .client import (
14
+ #### common websocket response
15
+ OpenResponse,
16
+ CloseResponse,
17
+ ErrorResponse,
18
+ UnhandledResponse,
19
+ #### unique
20
+ WelcomeResponse,
21
+ SettingsAppliedResponse,
22
+ ConversationTextResponse,
23
+ UserStartedSpeakingResponse,
24
+ AgentThinkingResponse,
25
+ FunctionCallRequest,
26
+ AgentStartedSpeakingResponse,
27
+ AgentAudioDoneResponse,
28
+ InjectionRefusedResponse,
29
+ )
30
+
31
+ from .client import (
32
+ # top level
33
+ SettingsOptions,
34
+ UpdatePromptOptions,
35
+ UpdateSpeakOptions,
36
+ InjectAgentMessageOptions,
37
+ FunctionCallResponse,
38
+ AgentKeepAlive,
39
+ # sub level
40
+ Listen,
41
+ ListenProvider,
42
+ Speak,
43
+ SpeakProvider,
44
+ Header,
45
+ Item,
46
+ Properties,
47
+ Parameters,
48
+ Function,
49
+ Think,
50
+ ThinkProvider,
51
+ Agent,
52
+ Input,
53
+ Output,
54
+ Audio,
55
+ Endpoint,
56
+ )
deepgram/clients/agent/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (1.28 kB). View file
 
deepgram/clients/agent/__pycache__/client.cpython-313.pyc ADDED
Binary file (2.45 kB). View file
 
deepgram/clients/agent/__pycache__/enums.cpython-313.pyc ADDED
Binary file (1.43 kB). View file
 
deepgram/clients/agent/client.py ADDED
@@ -0,0 +1,102 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ # websocket
6
+ from .v1 import (
7
+ AgentWebSocketClient as LatestAgentWebSocketClient,
8
+ AsyncAgentWebSocketClient as LatestAsyncAgentWebSocketClient,
9
+ )
10
+
11
+ from .v1 import (
12
+ #### common websocket response
13
+ BaseResponse as LatestBaseResponse,
14
+ OpenResponse as LatestOpenResponse,
15
+ CloseResponse as LatestCloseResponse,
16
+ ErrorResponse as LatestErrorResponse,
17
+ UnhandledResponse as LatestUnhandledResponse,
18
+ #### unique
19
+ WelcomeResponse as LatestWelcomeResponse,
20
+ SettingsAppliedResponse as LatestSettingsAppliedResponse,
21
+ ConversationTextResponse as LatestConversationTextResponse,
22
+ UserStartedSpeakingResponse as LatestUserStartedSpeakingResponse,
23
+ AgentThinkingResponse as LatestAgentThinkingResponse,
24
+ FunctionCallRequest as LatestFunctionCallRequest,
25
+ AgentStartedSpeakingResponse as LatestAgentStartedSpeakingResponse,
26
+ AgentAudioDoneResponse as LatestAgentAudioDoneResponse,
27
+ InjectionRefusedResponse as LatestInjectionRefusedResponse,
28
+ )
29
+
30
+ from .v1 import (
31
+ # top level
32
+ SettingsOptions as LatestSettingsOptions,
33
+ UpdatePromptOptions as LatestUpdatePromptOptions,
34
+ UpdateSpeakOptions as LatestUpdateSpeakOptions,
35
+ InjectAgentMessageOptions as LatestInjectAgentMessageOptions,
36
+ FunctionCallResponse as LatestFunctionCallResponse,
37
+ AgentKeepAlive as LatestAgentKeepAlive,
38
+ # sub level
39
+ Listen as LatestListen,
40
+ ListenProvider as LatestListenProvider,
41
+ Speak as LatestSpeak,
42
+ SpeakProvider as LatestSpeakProvider,
43
+ Header as LatestHeader,
44
+ Item as LatestItem,
45
+ Properties as LatestProperties,
46
+ Parameters as LatestParameters,
47
+ Function as LatestFunction,
48
+ Think as LatestThink,
49
+ ThinkProvider as LatestThinkProvider,
50
+ Agent as LatestAgent,
51
+ Input as LatestInput,
52
+ Output as LatestOutput,
53
+ Audio as LatestAudio,
54
+ Endpoint as LatestEndpoint,
55
+ )
56
+
57
+
58
+ # The vX/client.py points to the current supported version in the SDK.
59
+ # Older versions are supported in the SDK for backwards compatibility.
60
+
61
+ AgentWebSocketClient = LatestAgentWebSocketClient
62
+ AsyncAgentWebSocketClient = LatestAsyncAgentWebSocketClient
63
+
64
+ OpenResponse = LatestOpenResponse
65
+ CloseResponse = LatestCloseResponse
66
+ ErrorResponse = LatestErrorResponse
67
+ UnhandledResponse = LatestUnhandledResponse
68
+
69
+ WelcomeResponse = LatestWelcomeResponse
70
+ SettingsAppliedResponse = LatestSettingsAppliedResponse
71
+ ConversationTextResponse = LatestConversationTextResponse
72
+ UserStartedSpeakingResponse = LatestUserStartedSpeakingResponse
73
+ AgentThinkingResponse = LatestAgentThinkingResponse
74
+ FunctionCallRequest = LatestFunctionCallRequest
75
+ AgentStartedSpeakingResponse = LatestAgentStartedSpeakingResponse
76
+ AgentAudioDoneResponse = LatestAgentAudioDoneResponse
77
+ InjectionRefusedResponse = LatestInjectionRefusedResponse
78
+
79
+
80
+ SettingsOptions = LatestSettingsOptions
81
+ UpdatePromptOptions = LatestUpdatePromptOptions
82
+ UpdateSpeakOptions = LatestUpdateSpeakOptions
83
+ InjectAgentMessageOptions = LatestInjectAgentMessageOptions
84
+ FunctionCallResponse = LatestFunctionCallResponse
85
+ AgentKeepAlive = LatestAgentKeepAlive
86
+
87
+ Listen = LatestListen
88
+ ListenProvider = LatestListenProvider
89
+ Speak = LatestSpeak
90
+ SpeakProvider = LatestSpeakProvider
91
+ Header = LatestHeader
92
+ Item = LatestItem
93
+ Properties = LatestProperties
94
+ Parameters = LatestParameters
95
+ Function = LatestFunction
96
+ Think = LatestThink
97
+ ThinkProvider = LatestThinkProvider
98
+ Agent = LatestAgent
99
+ Input = LatestInput
100
+ Output = LatestOutput
101
+ Audio = LatestAudio
102
+ Endpoint = LatestEndpoint
deepgram/clients/agent/enums.py ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2023-2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ from aenum import StrEnum
6
+
7
+ # Constants mapping to events from the Deepgram API
8
+
9
+
10
+ class AgentWebSocketEvents(StrEnum):
11
+ """
12
+ Enumerates the possible Agent API events that can be received from the Deepgram API
13
+ """
14
+
15
+ # server
16
+ Open: str = "Open"
17
+ Close: str = "Close"
18
+ AudioData: str = "AudioData"
19
+ Welcome: str = "Welcome"
20
+ SettingsApplied: str = "SettingsApplied"
21
+ ConversationText: str = "ConversationText"
22
+ UserStartedSpeaking: str = "UserStartedSpeaking"
23
+ AgentThinking: str = "AgentThinking"
24
+ FunctionCallRequest: str = "FunctionCallRequest"
25
+ AgentStartedSpeaking: str = "AgentStartedSpeaking"
26
+ AgentAudioDone: str = "AgentAudioDone"
27
+ Error: str = "Error"
28
+ Unhandled: str = "Unhandled"
29
+
30
+ # client
31
+ Settings: str = "Settings"
32
+ UpdatePrompt: str = "UpdatePrompt"
33
+ UpdateSpeak: str = "UpdateSpeak"
34
+ InjectAgentMessage: str = "InjectAgentMessage"
35
+ InjectionRefused: str = "InjectionRefused"
36
+ AgentKeepAlive: str = "AgentKeepAlive"
deepgram/clients/agent/v1/__init__.py ADDED
@@ -0,0 +1,60 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ # common websocket
6
+ from ...common import (
7
+ OpenResponse,
8
+ CloseResponse,
9
+ UnhandledResponse,
10
+ ErrorResponse,
11
+ )
12
+
13
+ # websocket
14
+ from .websocket import AgentWebSocketClient, AsyncAgentWebSocketClient
15
+
16
+ from .websocket import (
17
+ #### common websocket response
18
+ BaseResponse,
19
+ OpenResponse,
20
+ CloseResponse,
21
+ ErrorResponse,
22
+ UnhandledResponse,
23
+ #### unique
24
+ WelcomeResponse,
25
+ SettingsAppliedResponse,
26
+ ConversationTextResponse,
27
+ UserStartedSpeakingResponse,
28
+ AgentThinkingResponse,
29
+ FunctionCallRequest,
30
+ AgentStartedSpeakingResponse,
31
+ AgentAudioDoneResponse,
32
+ InjectionRefusedResponse,
33
+ )
34
+
35
+ from .websocket import (
36
+ # top level
37
+ SettingsOptions,
38
+ UpdatePromptOptions,
39
+ UpdateSpeakOptions,
40
+ InjectAgentMessageOptions,
41
+ FunctionCallResponse,
42
+ AgentKeepAlive,
43
+ # sub level
44
+ Listen,
45
+ ListenProvider,
46
+ Speak,
47
+ SpeakProvider,
48
+ Header,
49
+ Item,
50
+ Properties,
51
+ Parameters,
52
+ Function,
53
+ Think,
54
+ ThinkProvider,
55
+ Agent,
56
+ Input,
57
+ Output,
58
+ Audio,
59
+ Endpoint,
60
+ )
deepgram/clients/agent/v1/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (1.33 kB). View file
 
deepgram/clients/agent/v1/websocket/__init__.py ADDED
@@ -0,0 +1,51 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ from .client import AgentWebSocketClient
6
+ from .async_client import AsyncAgentWebSocketClient
7
+
8
+ from .response import (
9
+ #### common websocket response
10
+ BaseResponse,
11
+ OpenResponse,
12
+ CloseResponse,
13
+ ErrorResponse,
14
+ UnhandledResponse,
15
+ #### unique
16
+ WelcomeResponse,
17
+ SettingsAppliedResponse,
18
+ ConversationTextResponse,
19
+ UserStartedSpeakingResponse,
20
+ AgentThinkingResponse,
21
+ FunctionCallRequest,
22
+ AgentStartedSpeakingResponse,
23
+ AgentAudioDoneResponse,
24
+ InjectionRefusedResponse,
25
+ )
26
+ from .options import (
27
+ # top level
28
+ SettingsOptions,
29
+ UpdatePromptOptions,
30
+ UpdateSpeakOptions,
31
+ InjectAgentMessageOptions,
32
+ FunctionCallResponse,
33
+ AgentKeepAlive,
34
+ # sub level
35
+ Listen,
36
+ ListenProvider,
37
+ Speak,
38
+ SpeakProvider,
39
+ Header,
40
+ Item,
41
+ Properties,
42
+ Parameters,
43
+ Function,
44
+ Think,
45
+ ThinkProvider,
46
+ Agent,
47
+ Input,
48
+ Output,
49
+ Audio,
50
+ Endpoint,
51
+ )
deepgram/clients/agent/v1/websocket/__pycache__/__init__.cpython-313.pyc ADDED
Binary file (1.31 kB). View file
 
deepgram/clients/agent/v1/websocket/__pycache__/async_client.cpython-313.pyc ADDED
Binary file (34.2 kB). View file
 
deepgram/clients/agent/v1/websocket/__pycache__/client.cpython-313.pyc ADDED
Binary file (31.6 kB). View file
 
deepgram/clients/agent/v1/websocket/__pycache__/options.cpython-313.pyc ADDED
Binary file (21.6 kB). View file
 
deepgram/clients/agent/v1/websocket/__pycache__/response.cpython-313.pyc ADDED
Binary file (4.08 kB). View file
 
deepgram/clients/agent/v1/websocket/async_client.py ADDED
@@ -0,0 +1,688 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ import asyncio
6
+ import json
7
+ import logging
8
+ from typing import Dict, Union, Optional, cast, Any, Callable
9
+ import threading
10
+
11
+ from .....utils import verboselogs
12
+ from .....options import DeepgramClientOptions
13
+ from ...enums import AgentWebSocketEvents
14
+ from ....common import AbstractAsyncWebSocketClient
15
+ from ....common import DeepgramError
16
+
17
+ from .response import (
18
+ OpenResponse,
19
+ WelcomeResponse,
20
+ SettingsAppliedResponse,
21
+ ConversationTextResponse,
22
+ UserStartedSpeakingResponse,
23
+ AgentThinkingResponse,
24
+ FunctionCallRequest,
25
+ AgentStartedSpeakingResponse,
26
+ AgentAudioDoneResponse,
27
+ InjectionRefusedResponse,
28
+ CloseResponse,
29
+ ErrorResponse,
30
+ UnhandledResponse,
31
+ )
32
+ from .options import (
33
+ SettingsOptions,
34
+ UpdatePromptOptions,
35
+ UpdateSpeakOptions,
36
+ InjectAgentMessageOptions,
37
+ FunctionCallResponse,
38
+ AgentKeepAlive,
39
+ )
40
+
41
+ from .....audio.speaker import (
42
+ Speaker,
43
+ RATE as SPEAKER_RATE,
44
+ CHANNELS as SPEAKER_CHANNELS,
45
+ PLAYBACK_DELTA as SPEAKER_PLAYBACK_DELTA,
46
+ )
47
+ from .....audio.microphone import (
48
+ Microphone,
49
+ RATE as MICROPHONE_RATE,
50
+ CHANNELS as MICROPHONE_CHANNELS,
51
+ )
52
+
53
+ ONE_SECOND = 1
54
+ HALF_SECOND = 0.5
55
+ DEEPGRAM_INTERVAL = 5
56
+
57
+
58
+ class AsyncAgentWebSocketClient(
59
+ AbstractAsyncWebSocketClient
60
+ ): # pylint: disable=too-many-instance-attributes
61
+ """
62
+ Client for interacting with Deepgram's live transcription services over WebSockets.
63
+
64
+ This class provides methods to establish a WebSocket connection for live transcription and handle real-time transcription events.
65
+
66
+ Args:
67
+ config (DeepgramClientOptions): all the options for the client.
68
+ """
69
+
70
+ _logger: verboselogs.VerboseLogger
71
+ _config: DeepgramClientOptions
72
+ _endpoint: str
73
+
74
+ _event_handlers: Dict[AgentWebSocketEvents, list]
75
+
76
+ _keep_alive_thread: Union[asyncio.Task, None]
77
+
78
+ _kwargs: Optional[Dict] = None
79
+ _addons: Optional[Dict] = None
80
+ # note the distinction here. We can't use _config because it's already used in the parent
81
+ _settings: Optional[SettingsOptions] = None
82
+ _headers: Optional[Dict] = None
83
+
84
+ _speaker_created: bool = False
85
+ _speaker: Optional[Speaker] = None
86
+ _microphone_created: bool = False
87
+ _microphone: Optional[Microphone] = None
88
+
89
+ def __init__(self, config: DeepgramClientOptions):
90
+ if config is None:
91
+ raise DeepgramError("Config is required")
92
+
93
+ self._logger = verboselogs.VerboseLogger(__name__)
94
+ self._logger.addHandler(logging.StreamHandler())
95
+ self._logger.setLevel(config.verbose)
96
+
97
+ self._config = config
98
+
99
+ # needs to be "wss://agent.deepgram.com/agent"
100
+ self._endpoint = "v1/agent/converse"
101
+
102
+ # override the endpoint since it needs to be "wss://agent.deepgram.com/agent"
103
+ self._config.url = "agent.deepgram.com"
104
+ self._keep_alive_thread = None
105
+
106
+ # init handlers
107
+ self._event_handlers = {
108
+ event: [] for event in AgentWebSocketEvents.__members__.values()
109
+ }
110
+
111
+ if self._config.options.get("microphone_record") == "true":
112
+ self._logger.info("microphone_record is enabled")
113
+ rate = self._config.options.get("microphone_record_rate", MICROPHONE_RATE)
114
+ channels = self._config.options.get(
115
+ "microphone_record_channels", MICROPHONE_CHANNELS
116
+ )
117
+ device_index = self._config.options.get("microphone_record_device_index")
118
+
119
+ self._logger.debug("rate: %s", rate)
120
+ self._logger.debug("channels: %s", channels)
121
+ if device_index is not None:
122
+ self._logger.debug("device_index: %s", device_index)
123
+
124
+ self._microphone_created = True
125
+
126
+ if device_index is not None:
127
+ self._microphone = Microphone(
128
+ rate=rate,
129
+ channels=channels,
130
+ verbose=self._config.verbose,
131
+ input_device_index=device_index,
132
+ )
133
+ else:
134
+ self._microphone = Microphone(
135
+ rate=rate,
136
+ channels=channels,
137
+ verbose=self._config.verbose,
138
+ )
139
+
140
+ if self._config.options.get("speaker_playback") == "true":
141
+ self._logger.info("speaker_playback is enabled")
142
+ rate = self._config.options.get("speaker_playback_rate", SPEAKER_RATE)
143
+ channels = self._config.options.get(
144
+ "speaker_playback_channels", SPEAKER_CHANNELS
145
+ )
146
+ playback_delta_in_ms = self._config.options.get(
147
+ "speaker_playback_delta_in_ms", SPEAKER_PLAYBACK_DELTA
148
+ )
149
+ device_index = self._config.options.get("speaker_playback_device_index")
150
+
151
+ self._logger.debug("rate: %s", rate)
152
+ self._logger.debug("channels: %s", channels)
153
+
154
+ self._speaker_created = True
155
+
156
+ if device_index is not None:
157
+ self._logger.debug("device_index: %s", device_index)
158
+
159
+ self._speaker = Speaker(
160
+ rate=rate,
161
+ channels=channels,
162
+ last_play_delta_in_ms=playback_delta_in_ms,
163
+ verbose=self._config.verbose,
164
+ output_device_index=device_index,
165
+ microphone=self._microphone,
166
+ )
167
+ else:
168
+ self._speaker = Speaker(
169
+ rate=rate,
170
+ channels=channels,
171
+ last_play_delta_in_ms=playback_delta_in_ms,
172
+ verbose=self._config.verbose,
173
+ microphone=self._microphone,
174
+ )
175
+ # call the parent constructor
176
+ super().__init__(self._config, self._endpoint)
177
+
178
+ # pylint: disable=too-many-branches,too-many-statements
179
+ async def start(
180
+ self,
181
+ options: Optional[SettingsOptions] = None,
182
+ addons: Optional[Dict] = None,
183
+ headers: Optional[Dict] = None,
184
+ members: Optional[Dict] = None,
185
+ **kwargs,
186
+ ) -> bool:
187
+ """
188
+ Starts the WebSocket connection for agent API.
189
+ """
190
+ self._logger.debug("AsyncAgentWebSocketClient.start ENTER")
191
+ self._logger.info("settings: %s", options)
192
+ self._logger.info("addons: %s", addons)
193
+ self._logger.info("headers: %s", headers)
194
+ self._logger.info("members: %s", members)
195
+ self._logger.info("kwargs: %s", kwargs)
196
+
197
+ if isinstance(options, SettingsOptions) and not options.check():
198
+ self._logger.error("settings.check failed")
199
+ self._logger.debug("AsyncAgentWebSocketClient.start LEAVE")
200
+ raise DeepgramError("Fatal agent settings error")
201
+
202
+ self._addons = addons
203
+ self._headers = headers
204
+
205
+ # add "members" as members of the class
206
+ if members is not None:
207
+ self.__dict__.update(members)
208
+
209
+ # set kwargs as members of the class
210
+ if kwargs is not None:
211
+ self._kwargs = kwargs
212
+ else:
213
+ self._kwargs = {}
214
+
215
+ if isinstance(options, SettingsOptions):
216
+ self._logger.info("options is class")
217
+ self._settings = options
218
+ elif isinstance(options, dict):
219
+ self._logger.info("options is dict")
220
+ self._settings = SettingsOptions.from_dict(options)
221
+ elif isinstance(options, str):
222
+ self._logger.info("options is json")
223
+ self._settings = SettingsOptions.from_json(options)
224
+ else:
225
+ raise DeepgramError("Invalid options type")
226
+
227
+ if self._settings.agent.listen.provider.keyterms is not None and self._settings.agent.listen.provider.model is not None and not self._settings.agent.listen.provider.model.startswith("nova-3"):
228
+ raise DeepgramError("Keyterms are only supported for nova-3 models")
229
+
230
+ try:
231
+ # speaker substitutes the listening thread
232
+ if self._speaker is not None:
233
+ self._logger.notice("passing speaker to delegate_listening")
234
+ super().delegate_listening(self._speaker)
235
+
236
+ # call parent start
237
+ if (
238
+ await super().start(
239
+ {},
240
+ self._addons,
241
+ self._headers,
242
+ **dict(cast(Dict[Any, Any], self._kwargs)),
243
+ )
244
+ is False
245
+ ):
246
+ self._logger.error("AsyncAgentWebSocketClient.start failed")
247
+ self._logger.debug("AsyncAgentWebSocketClient.start LEAVE")
248
+ return False
249
+
250
+ if self._speaker is not None:
251
+ self._logger.notice("speaker is delegate_listening. Starting speaker")
252
+ self._speaker.start()
253
+
254
+ if self._speaker is not None and self._microphone is not None:
255
+ self._logger.notice(
256
+ "speaker is delegate_listening. Starting microphone"
257
+ )
258
+ self._microphone.set_callback(self.send)
259
+ self._microphone.start()
260
+
261
+ # debug the threads
262
+ for thread in threading.enumerate():
263
+ self._logger.debug("after running thread: %s", thread.name)
264
+ self._logger.debug("number of active threads: %s", threading.active_count())
265
+
266
+ # keepalive thread
267
+ if self._config.is_keep_alive_enabled():
268
+ self._logger.notice("keepalive is enabled")
269
+ self._keep_alive_thread = asyncio.create_task(self._keep_alive())
270
+ else:
271
+ self._logger.notice("keepalive is disabled")
272
+
273
+ # debug the threads
274
+ for thread in threading.enumerate():
275
+ self._logger.debug("after running thread: %s", thread.name)
276
+ self._logger.debug("number of active threads: %s", threading.active_count())
277
+
278
+ # send the configurationsetting message
279
+ self._logger.notice("Sending Settings...")
280
+ ret_send_cs = await self.send(str(self._settings))
281
+ if not ret_send_cs:
282
+ self._logger.error("Settings failed")
283
+
284
+ err_error: ErrorResponse = ErrorResponse(
285
+ "Exception in AsyncAgentWebSocketClient.start",
286
+ "Settings failed to send",
287
+ "Exception",
288
+ )
289
+ await self._emit(
290
+ AgentWebSocketEvents(AgentWebSocketEvents.Error),
291
+ error=err_error,
292
+ **dict(cast(Dict[Any, Any], self._kwargs)),
293
+ )
294
+
295
+ self._logger.debug("AgentWebSocketClient.start LEAVE")
296
+ return False
297
+
298
+ self._logger.notice("start succeeded")
299
+ self._logger.debug("AsyncAgentWebSocketClient.start LEAVE")
300
+ return True
301
+
302
+ except Exception as e: # pylint: disable=broad-except
303
+ self._logger.error(
304
+ "WebSocketException in AsyncAgentWebSocketClient.start: %s", e
305
+ )
306
+ self._logger.debug("AsyncAgentWebSocketClient.start LEAVE")
307
+ if self._config.options.get("termination_exception_connect") is True:
308
+ raise e
309
+ return False
310
+
311
+ # pylint: enable=too-many-branches,too-many-statements
312
+
313
+ def on(self, event: AgentWebSocketEvents, handler: Callable) -> None:
314
+ """
315
+ Registers event handlers for specific events.
316
+ """
317
+ self._logger.info("event subscribed: %s", event)
318
+ if event in AgentWebSocketEvents.__members__.values() and callable(handler):
319
+ self._event_handlers[event].append(handler)
320
+
321
+ async def _emit(self, event: AgentWebSocketEvents, *args, **kwargs) -> None:
322
+ """
323
+ Emits events to the registered event handlers.
324
+ """
325
+ self._logger.debug("AsyncAgentWebSocketClient._emit ENTER")
326
+ self._logger.debug("callback handlers for: %s", event)
327
+
328
+ # debug the threads
329
+ for thread in threading.enumerate():
330
+ self._logger.debug("after running thread: %s", thread.name)
331
+ self._logger.debug("number of active threads: %s", threading.active_count())
332
+
333
+ self._logger.debug("callback handlers for: %s", event)
334
+ tasks = []
335
+ for handler in self._event_handlers[event]:
336
+ task = asyncio.create_task(handler(self, *args, **kwargs))
337
+ tasks.append(task)
338
+
339
+ if tasks:
340
+ self._logger.debug("waiting for tasks to finish...")
341
+ await asyncio.gather(*tasks, return_exceptions=True)
342
+ tasks.clear()
343
+
344
+ # debug the threads
345
+ for thread in threading.enumerate():
346
+ self._logger.debug("after running thread: %s", thread.name)
347
+ self._logger.debug("number of active threads: %s", threading.active_count())
348
+
349
+ self._logger.debug("AsyncAgentWebSocketClient._emit LEAVE")
350
+
351
+ # pylint: disable=too-many-locals,too-many-statements
352
+ async def _process_text(self, message: str) -> None:
353
+ """
354
+ Processes messages received over the WebSocket connection.
355
+ """
356
+ self._logger.debug("AsyncAgentWebSocketClient._process_text ENTER")
357
+
358
+ try:
359
+ self._logger.debug("Text data received")
360
+ if len(message) == 0:
361
+ self._logger.debug("message is empty")
362
+ self._logger.debug("AsyncAgentWebSocketClient._process_text LEAVE")
363
+ return
364
+
365
+ data = json.loads(message)
366
+ response_type = data.get("type")
367
+ self._logger.debug("response_type: %s, data: %s", response_type, data)
368
+
369
+ match response_type:
370
+ case AgentWebSocketEvents.Open:
371
+ open_result: OpenResponse = OpenResponse.from_json(message)
372
+ self._logger.verbose("OpenResponse: %s", open_result)
373
+ await self._emit(
374
+ AgentWebSocketEvents(AgentWebSocketEvents.Open),
375
+ open=open_result,
376
+ **dict(cast(Dict[Any, Any], self._kwargs)),
377
+ )
378
+ case AgentWebSocketEvents.Welcome:
379
+ welcome_result: WelcomeResponse = WelcomeResponse.from_json(message)
380
+ self._logger.verbose("WelcomeResponse: %s", welcome_result)
381
+ await self._emit(
382
+ AgentWebSocketEvents(AgentWebSocketEvents.Welcome),
383
+ welcome=welcome_result,
384
+ **dict(cast(Dict[Any, Any], self._kwargs)),
385
+ )
386
+ case AgentWebSocketEvents.SettingsApplied:
387
+ settings_applied_result: SettingsAppliedResponse = (
388
+ SettingsAppliedResponse.from_json(message)
389
+ )
390
+ self._logger.verbose(
391
+ "SettingsAppliedResponse: %s", settings_applied_result
392
+ )
393
+ await self._emit(
394
+ AgentWebSocketEvents(AgentWebSocketEvents.SettingsApplied),
395
+ settings_applied=settings_applied_result,
396
+ **dict(cast(Dict[Any, Any], self._kwargs)),
397
+ )
398
+ case AgentWebSocketEvents.ConversationText:
399
+ conversation_text_result: ConversationTextResponse = (
400
+ ConversationTextResponse.from_json(message)
401
+ )
402
+ self._logger.verbose(
403
+ "ConversationTextResponse: %s", conversation_text_result
404
+ )
405
+ await self._emit(
406
+ AgentWebSocketEvents(AgentWebSocketEvents.ConversationText),
407
+ conversation_text=conversation_text_result,
408
+ **dict(cast(Dict[Any, Any], self._kwargs)),
409
+ )
410
+ case AgentWebSocketEvents.UserStartedSpeaking:
411
+ user_started_speaking_result: UserStartedSpeakingResponse = (
412
+ UserStartedSpeakingResponse.from_json(message)
413
+ )
414
+ self._logger.verbose(
415
+ "UserStartedSpeakingResponse: %s", user_started_speaking_result
416
+ )
417
+ await self._emit(
418
+ AgentWebSocketEvents(AgentWebSocketEvents.UserStartedSpeaking),
419
+ user_started_speaking=user_started_speaking_result,
420
+ **dict(cast(Dict[Any, Any], self._kwargs)),
421
+ )
422
+ case AgentWebSocketEvents.AgentThinking:
423
+ agent_thinking_result: AgentThinkingResponse = (
424
+ AgentThinkingResponse.from_json(message)
425
+ )
426
+ self._logger.verbose(
427
+ "AgentThinkingResponse: %s", agent_thinking_result
428
+ )
429
+ await self._emit(
430
+ AgentWebSocketEvents(AgentWebSocketEvents.AgentThinking),
431
+ agent_thinking=agent_thinking_result,
432
+ **dict(cast(Dict[Any, Any], self._kwargs)),
433
+ )
434
+ case AgentWebSocketEvents.FunctionCallRequest:
435
+ function_call_request_result: FunctionCallRequest = (
436
+ FunctionCallRequest.from_json(message)
437
+ )
438
+ self._logger.verbose(
439
+ "FunctionCallRequest: %s", function_call_request_result
440
+ )
441
+ await self._emit(
442
+ AgentWebSocketEvents(AgentWebSocketEvents.FunctionCallRequest),
443
+ function_call_request=function_call_request_result,
444
+ **dict(cast(Dict[Any, Any], self._kwargs)),
445
+ )
446
+ case AgentWebSocketEvents.AgentStartedSpeaking:
447
+ agent_started_speaking_result: AgentStartedSpeakingResponse = (
448
+ AgentStartedSpeakingResponse.from_json(message)
449
+ )
450
+ self._logger.verbose(
451
+ "AgentStartedSpeakingResponse: %s",
452
+ agent_started_speaking_result,
453
+ )
454
+ await self._emit(
455
+ AgentWebSocketEvents(AgentWebSocketEvents.AgentStartedSpeaking),
456
+ agent_started_speaking=agent_started_speaking_result,
457
+ **dict(cast(Dict[Any, Any], self._kwargs)),
458
+ )
459
+ case AgentWebSocketEvents.AgentAudioDone:
460
+ agent_audio_done_result: AgentAudioDoneResponse = (
461
+ AgentAudioDoneResponse.from_json(message)
462
+ )
463
+ self._logger.verbose(
464
+ "AgentAudioDoneResponse: %s", agent_audio_done_result
465
+ )
466
+ await self._emit(
467
+ AgentWebSocketEvents(AgentWebSocketEvents.AgentAudioDone),
468
+ agent_audio_done=agent_audio_done_result,
469
+ **dict(cast(Dict[Any, Any], self._kwargs)),
470
+ )
471
+ case AgentWebSocketEvents.InjectionRefused:
472
+ injection_refused_result: InjectionRefusedResponse = (
473
+ InjectionRefusedResponse.from_json(message)
474
+ )
475
+ self._logger.verbose(
476
+ "InjectionRefused: %s", injection_refused_result
477
+ )
478
+ await self._emit(
479
+ AgentWebSocketEvents(AgentWebSocketEvents.InjectionRefused),
480
+ injection_refused=injection_refused_result,
481
+ **dict(cast(Dict[Any, Any], self._kwargs)),
482
+ )
483
+ case AgentWebSocketEvents.Close:
484
+ close_result: CloseResponse = CloseResponse.from_json(message)
485
+ self._logger.verbose("CloseResponse: %s", close_result)
486
+ await self._emit(
487
+ AgentWebSocketEvents(AgentWebSocketEvents.Close),
488
+ close=close_result,
489
+ **dict(cast(Dict[Any, Any], self._kwargs)),
490
+ )
491
+ case AgentWebSocketEvents.Error:
492
+ err_error: ErrorResponse = ErrorResponse.from_json(message)
493
+ self._logger.verbose("ErrorResponse: %s", err_error)
494
+ await self._emit(
495
+ AgentWebSocketEvents(AgentWebSocketEvents.Error),
496
+ error=err_error,
497
+ **dict(cast(Dict[Any, Any], self._kwargs)),
498
+ )
499
+ case _:
500
+ self._logger.warning(
501
+ "Unknown Message: response_type: %s, data: %s",
502
+ response_type,
503
+ data,
504
+ )
505
+ unhandled_error: UnhandledResponse = UnhandledResponse(
506
+ type=AgentWebSocketEvents(AgentWebSocketEvents.Unhandled),
507
+ raw=message,
508
+ )
509
+ await self._emit(
510
+ AgentWebSocketEvents(AgentWebSocketEvents.Unhandled),
511
+ unhandled=unhandled_error,
512
+ **dict(cast(Dict[Any, Any], self._kwargs)),
513
+ )
514
+
515
+ self._logger.notice("_process_text Succeeded")
516
+ self._logger.debug("AsyncAgentWebSocketClient._process_text LEAVE")
517
+
518
+ except Exception as e: # pylint: disable=broad-except
519
+ self._logger.error(
520
+ "Exception in AsyncAgentWebSocketClient._process_text: %s", e
521
+ )
522
+ e_error: ErrorResponse = ErrorResponse(
523
+ "Exception in AsyncAgentWebSocketClient._process_text",
524
+ f"{e}",
525
+ "Exception",
526
+ )
527
+ await self._emit(
528
+ AgentWebSocketEvents(AgentWebSocketEvents.Error),
529
+ error=e_error,
530
+ **dict(cast(Dict[Any, Any], self._kwargs)),
531
+ )
532
+
533
+ # signal exit and close
534
+ await super()._signal_exit()
535
+
536
+ self._logger.debug("AsyncAgentWebSocketClient._process_text LEAVE")
537
+
538
+ if self._config.options.get("termination_exception") is True:
539
+ raise
540
+ return
541
+
542
+ # pylint: enable=too-many-locals,too-many-statements
543
+
544
+ async def _process_binary(self, message: bytes) -> None:
545
+ self._logger.debug("AsyncAgentWebSocketClient._process_binary ENTER")
546
+ self._logger.debug("Binary data received")
547
+
548
+ await self._emit(
549
+ AgentWebSocketEvents(AgentWebSocketEvents.AudioData),
550
+ data=message,
551
+ **dict(cast(Dict[Any, Any], self._kwargs)),
552
+ )
553
+
554
+ self._logger.notice("_process_binary Succeeded")
555
+ self._logger.debug("AsyncAgentWebSocketClient._process_binary LEAVE")
556
+
557
+ # pylint: disable=too-many-return-statements
558
+ async def _keep_alive(self) -> None:
559
+ """
560
+ Sends keepalive messages to the WebSocket connection.
561
+ """
562
+ self._logger.debug("AsyncAgentWebSocketClient._keep_alive ENTER")
563
+
564
+ counter = 0
565
+ while True:
566
+ try:
567
+ counter += 1
568
+ await asyncio.sleep(ONE_SECOND)
569
+
570
+ if self._exit_event.is_set():
571
+ self._logger.notice("_keep_alive exiting gracefully")
572
+ self._logger.debug("AsyncAgentWebSocketClient._keep_alive LEAVE")
573
+ return
574
+
575
+ # deepgram keepalive
576
+ if counter % DEEPGRAM_INTERVAL == 0:
577
+ await self.keep_alive()
578
+
579
+ except Exception as e: # pylint: disable=broad-except
580
+ self._logger.error(
581
+ "Exception in AsyncAgentWebSocketClient._keep_alive: %s", e
582
+ )
583
+ e_error: ErrorResponse = ErrorResponse(
584
+ "Exception in AsyncAgentWebSocketClient._keep_alive",
585
+ f"{e}",
586
+ "Exception",
587
+ )
588
+ self._logger.error(
589
+ "Exception in AsyncAgentWebSocketClient._keep_alive: %s", str(e)
590
+ )
591
+ await self._emit(
592
+ AgentWebSocketEvents(AgentWebSocketEvents.Error),
593
+ error=e_error,
594
+ **dict(cast(Dict[Any, Any], self._kwargs)),
595
+ )
596
+
597
+ # signal exit and close
598
+ await super()._signal_exit()
599
+
600
+ self._logger.debug("AsyncAgentWebSocketClient._keep_alive LEAVE")
601
+
602
+ if self._config.options.get("termination_exception") is True:
603
+ raise
604
+ return
605
+
606
+ async def keep_alive(self) -> bool:
607
+ """
608
+ Sends a KeepAlive message
609
+ """
610
+ self._logger.spam("AsyncAgentWebSocketClient.keep_alive ENTER")
611
+
612
+ self._logger.notice("Sending KeepAlive...")
613
+ ret = await self.send(json.dumps({"type": "KeepAlive"}))
614
+
615
+ if not ret:
616
+ self._logger.error("keep_alive failed")
617
+ self._logger.spam("AsyncAgentWebSocketClient.keep_alive LEAVE")
618
+ return False
619
+
620
+ self._logger.notice("keep_alive succeeded")
621
+ self._logger.spam("AsyncAgentWebSocketClient.keep_alive LEAVE")
622
+
623
+ return True
624
+
625
+ async def _close_message(self) -> bool:
626
+ # TODO: No known API close message # pylint: disable=fixme
627
+ # return await self.send(json.dumps({"type": "Close"}))
628
+ return True
629
+
630
+ async def finish(self) -> bool:
631
+ """
632
+ Closes the WebSocket connection gracefully.
633
+ """
634
+ self._logger.debug("AsyncAgentWebSocketClient.finish ENTER")
635
+
636
+ # stop the threads
637
+ self._logger.verbose("cancelling tasks...")
638
+ try:
639
+ # call parent finish
640
+ if await super().finish() is False:
641
+ self._logger.error("AsyncAgentWebSocketClient.finish failed")
642
+
643
+ if self._microphone is not None and self._microphone_created:
644
+ self._microphone.finish()
645
+ self._microphone_created = False
646
+
647
+ if self._speaker is not None and self._speaker_created:
648
+ self._speaker.finish()
649
+ self._speaker_created = False
650
+
651
+ # Before cancelling, check if the tasks were created
652
+ # debug the threads
653
+ for thread in threading.enumerate():
654
+ self._logger.debug("before running thread: %s", thread.name)
655
+ self._logger.debug("number of active threads: %s", threading.active_count())
656
+
657
+ tasks = []
658
+ if self._keep_alive_thread is not None:
659
+ self._keep_alive_thread.cancel()
660
+ tasks.append(self._keep_alive_thread)
661
+ self._logger.notice("processing _keep_alive_thread cancel...")
662
+
663
+ # Use asyncio.gather to wait for tasks to be cancelled
664
+ # Prevent indefinite waiting by setting a timeout
665
+ await asyncio.wait_for(asyncio.gather(*tasks), timeout=10)
666
+ self._logger.notice("threads joined")
667
+
668
+ self._speaker = None
669
+ self._microphone = None
670
+
671
+ # debug the threads
672
+ for thread in threading.enumerate():
673
+ self._logger.debug("after running thread: %s", thread.name)
674
+ self._logger.debug("number of active threads: %s", threading.active_count())
675
+
676
+ self._logger.notice("finish succeeded")
677
+ self._logger.spam("AsyncAgentWebSocketClient.finish LEAVE")
678
+ return True
679
+
680
+ except asyncio.CancelledError as e:
681
+ self._logger.error("tasks cancelled error: %s", e)
682
+ self._logger.debug("AsyncAgentWebSocketClient.finish LEAVE")
683
+ return False
684
+
685
+ except asyncio.TimeoutError as e:
686
+ self._logger.error("tasks cancellation timed out: %s", e)
687
+ self._logger.debug("AsyncAgentWebSocketClient.finish LEAVE")
688
+ return False
deepgram/clients/agent/v1/websocket/client.py ADDED
@@ -0,0 +1,677 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ import json
6
+ import logging
7
+ from typing import Dict, Union, Optional, cast, Any, Callable
8
+ import threading
9
+ import time
10
+
11
+ from .....utils import verboselogs
12
+ from .....options import DeepgramClientOptions
13
+ from ...enums import AgentWebSocketEvents
14
+ from ....common import AbstractSyncWebSocketClient
15
+ from ....common import DeepgramError
16
+
17
+ from .response import (
18
+ OpenResponse,
19
+ WelcomeResponse,
20
+ SettingsAppliedResponse,
21
+ ConversationTextResponse,
22
+ UserStartedSpeakingResponse,
23
+ AgentThinkingResponse,
24
+ FunctionCallRequest,
25
+ AgentStartedSpeakingResponse,
26
+ AgentAudioDoneResponse,
27
+ InjectionRefusedResponse,
28
+ CloseResponse,
29
+ ErrorResponse,
30
+ UnhandledResponse,
31
+ )
32
+ from .options import (
33
+ SettingsOptions,
34
+ UpdatePromptOptions,
35
+ UpdateSpeakOptions,
36
+ InjectAgentMessageOptions,
37
+ FunctionCallResponse,
38
+ AgentKeepAlive,
39
+ )
40
+
41
+ from .....audio.speaker import (
42
+ Speaker,
43
+ RATE as SPEAKER_RATE,
44
+ CHANNELS as SPEAKER_CHANNELS,
45
+ PLAYBACK_DELTA as SPEAKER_PLAYBACK_DELTA,
46
+ )
47
+ from .....audio.microphone import (
48
+ Microphone,
49
+ RATE as MICROPHONE_RATE,
50
+ CHANNELS as MICROPHONE_CHANNELS,
51
+ )
52
+
53
+ ONE_SECOND = 1
54
+ HALF_SECOND = 0.5
55
+ DEEPGRAM_INTERVAL = 5
56
+
57
+
58
+ class AgentWebSocketClient(
59
+ AbstractSyncWebSocketClient
60
+ ): # pylint: disable=too-many-instance-attributes
61
+ """
62
+ Client for interacting with Deepgram's live transcription services over WebSockets.
63
+
64
+ This class provides methods to establish a WebSocket connection for live transcription and handle real-time transcription events.
65
+
66
+ Args:
67
+ config (DeepgramClientOptions): all the options for the client.
68
+ """
69
+
70
+ _logger: verboselogs.VerboseLogger
71
+ _config: DeepgramClientOptions
72
+ _endpoint: str
73
+
74
+ _event_handlers: Dict[AgentWebSocketEvents, list]
75
+
76
+ _keep_alive_thread: Union[threading.Thread, None]
77
+
78
+ _kwargs: Optional[Dict] = None
79
+ _addons: Optional[Dict] = None
80
+ # note the distinction here. We can't use _config because it's already used in the parent
81
+ _settings: Optional[SettingsOptions] = None
82
+ _headers: Optional[Dict] = None
83
+
84
+ _speaker_created: bool = False
85
+ _speaker: Optional[Speaker] = None
86
+ _microphone_created: bool = False
87
+ _microphone: Optional[Microphone] = None
88
+
89
+ def __init__(self, config: DeepgramClientOptions):
90
+ if config is None:
91
+ raise DeepgramError("Config is required")
92
+
93
+ self._logger = verboselogs.VerboseLogger(__name__)
94
+ self._logger.addHandler(logging.StreamHandler())
95
+ self._logger.setLevel(config.verbose)
96
+
97
+ self._config = config
98
+
99
+ # needs to be "wss://agent.deepgram.com/agent"
100
+ self._endpoint = "v1/agent/converse"
101
+
102
+ # override the endpoint since it needs to be "wss://agent.deepgram.com/agent"
103
+ self._config.url = "agent.deepgram.com"
104
+
105
+ self._keep_alive_thread = None
106
+
107
+ # init handlers
108
+ self._event_handlers = {
109
+ event: [] for event in AgentWebSocketEvents.__members__.values()
110
+ }
111
+
112
+ if self._config.options.get("microphone_record") == "true":
113
+ self._logger.info("microphone_record is enabled")
114
+ rate = self._config.options.get("microphone_record_rate", MICROPHONE_RATE)
115
+ channels = self._config.options.get(
116
+ "microphone_record_channels", MICROPHONE_CHANNELS
117
+ )
118
+ device_index = self._config.options.get("microphone_record_device_index")
119
+
120
+ self._logger.debug("rate: %s", rate)
121
+ self._logger.debug("channels: %s", channels)
122
+
123
+ self._microphone_created = True
124
+
125
+ if device_index is not None:
126
+ self._logger.debug("device_index: %s", device_index)
127
+ self._microphone = Microphone(
128
+ rate=rate,
129
+ channels=channels,
130
+ verbose=self._config.verbose,
131
+ input_device_index=device_index,
132
+ )
133
+ else:
134
+ self._microphone = Microphone(
135
+ rate=rate,
136
+ channels=channels,
137
+ verbose=self._config.verbose,
138
+ )
139
+
140
+ if self._config.options.get("speaker_playback") == "true":
141
+ self._logger.info("speaker_playback is enabled")
142
+ rate = self._config.options.get("speaker_playback_rate", SPEAKER_RATE)
143
+ channels = self._config.options.get(
144
+ "speaker_playback_channels", SPEAKER_CHANNELS
145
+ )
146
+ playback_delta_in_ms = self._config.options.get(
147
+ "speaker_playback_delta_in_ms", SPEAKER_PLAYBACK_DELTA
148
+ )
149
+ device_index = self._config.options.get("speaker_playback_device_index")
150
+
151
+ self._logger.debug("rate: %s", rate)
152
+ self._logger.debug("channels: %s", channels)
153
+
154
+ self._speaker_created = True
155
+
156
+ if device_index is not None:
157
+ self._logger.debug("device_index: %s", device_index)
158
+
159
+ self._speaker = Speaker(
160
+ rate=rate,
161
+ channels=channels,
162
+ last_play_delta_in_ms=playback_delta_in_ms,
163
+ verbose=self._config.verbose,
164
+ output_device_index=device_index,
165
+ microphone=self._microphone,
166
+ )
167
+ else:
168
+ self._speaker = Speaker(
169
+ rate=rate,
170
+ channels=channels,
171
+ last_play_delta_in_ms=playback_delta_in_ms,
172
+ verbose=self._config.verbose,
173
+ microphone=self._microphone,
174
+ )
175
+
176
+ # call the parent constructor
177
+ super().__init__(self._config, self._endpoint)
178
+
179
+ # pylint: disable=too-many-statements,too-many-branches
180
+ def start(
181
+ self,
182
+ options: Optional[SettingsOptions] = None,
183
+ addons: Optional[Dict] = None,
184
+ headers: Optional[Dict] = None,
185
+ members: Optional[Dict] = None,
186
+ **kwargs,
187
+ ) -> bool:
188
+ """
189
+ Starts the WebSocket connection for agent API.
190
+ """
191
+ self._logger.debug("AgentWebSocketClient.start ENTER")
192
+ self._logger.info("settings: %s", options)
193
+ self._logger.info("addons: %s", addons)
194
+ self._logger.info("headers: %s", headers)
195
+ self._logger.info("members: %s", members)
196
+ self._logger.info("kwargs: %s", kwargs)
197
+
198
+ if isinstance(options, SettingsOptions) and not options.check():
199
+ self._logger.error("settings.check failed")
200
+ self._logger.debug("AgentWebSocketClient.start LEAVE")
201
+ raise DeepgramError("Fatal agent settings error")
202
+
203
+ self._addons = addons
204
+ self._headers = headers
205
+
206
+ # add "members" as members of the class
207
+ if members is not None:
208
+ self.__dict__.update(members)
209
+
210
+ # set kwargs as members of the class
211
+ if kwargs is not None:
212
+ self._kwargs = kwargs
213
+ else:
214
+ self._kwargs = {}
215
+
216
+ if isinstance(options, SettingsOptions):
217
+ self._logger.info("options is class")
218
+ self._settings = options
219
+ elif isinstance(options, dict):
220
+ self._logger.info("options is dict")
221
+ self._settings = SettingsOptions.from_dict(options)
222
+ elif isinstance(options, str):
223
+ self._logger.info("options is json")
224
+ self._settings = SettingsOptions.from_json(options)
225
+ else:
226
+ raise DeepgramError("Invalid options type")
227
+
228
+ if (
229
+ self._settings.agent.listen.provider
230
+ and self._settings.agent.listen.provider.keyterms is not None
231
+ and self._settings.agent.listen.provider.model is not None
232
+ and not self._settings.agent.listen.provider.model.startswith("nova-3")
233
+ ):
234
+ raise DeepgramError("Keyterms are only supported for nova-3 models")
235
+
236
+ try:
237
+ # speaker substitutes the listening thread
238
+ if self._speaker is not None:
239
+ self._logger.notice("passing speaker to delegate_listening")
240
+ super().delegate_listening(self._speaker)
241
+
242
+ # call parent start
243
+ if (
244
+ super().start(
245
+ {},
246
+ self._addons,
247
+ self._headers,
248
+ **dict(cast(Dict[Any, Any], self._kwargs)),
249
+ )
250
+ is False
251
+ ):
252
+ self._logger.error("AgentWebSocketClient.start failed")
253
+ self._logger.debug("AgentWebSocketClient.start LEAVE")
254
+ return False
255
+
256
+ if self._speaker is not None:
257
+ self._logger.notice("speaker is delegate_listening. Starting speaker")
258
+ self._speaker.start()
259
+
260
+ if self._speaker is not None and self._microphone is not None:
261
+ self._logger.notice(
262
+ "speaker is delegate_listening. Starting microphone"
263
+ )
264
+ self._microphone.set_callback(self.send)
265
+ self._microphone.start()
266
+
267
+ # debug the threads
268
+ for thread in threading.enumerate():
269
+ self._logger.debug("after running thread: %s", thread.name)
270
+ self._logger.debug("number of active threads: %s", threading.active_count())
271
+
272
+ # keepalive thread
273
+ if self._config.is_keep_alive_enabled():
274
+ self._logger.notice("keepalive is enabled")
275
+ self._keep_alive_thread = threading.Thread(target=self._keep_alive)
276
+ self._keep_alive_thread.start()
277
+ else:
278
+ self._logger.notice("keepalive is disabled")
279
+
280
+ # debug the threads
281
+ for thread in threading.enumerate():
282
+ self._logger.debug("after running thread: %s", thread.name)
283
+ self._logger.debug("number of active threads: %s", threading.active_count())
284
+
285
+ # send the Settings message
286
+ self._logger.notice("Sending Settings...")
287
+ ret_send_cs = self.send(str(self._settings))
288
+ if not ret_send_cs:
289
+ self._logger.error("Settings failed")
290
+
291
+ err_error: ErrorResponse = ErrorResponse(
292
+ "Exception in AgentWebSocketClient.start",
293
+ "Settings failed to send",
294
+ "Exception",
295
+ )
296
+ self._emit(
297
+ AgentWebSocketEvents(AgentWebSocketEvents.Error),
298
+ error=err_error,
299
+ **dict(cast(Dict[Any, Any], self._kwargs)),
300
+ )
301
+
302
+ self._logger.debug("AgentWebSocketClient.start LEAVE")
303
+ return False
304
+
305
+ self._logger.notice("start succeeded")
306
+ self._logger.debug("AgentWebSocketClient.start LEAVE")
307
+ return True
308
+
309
+ except Exception as e: # pylint: disable=broad-except
310
+ self._logger.error(
311
+ "WebSocketException in AgentWebSocketClient.start: %s", e
312
+ )
313
+ self._logger.debug("AgentWebSocketClient.start LEAVE")
314
+ if self._config.options.get("termination_exception_connect") is True:
315
+ raise e
316
+ return False
317
+
318
+ # pylint: enable=too-many-statements,too-many-branches
319
+
320
+ def on(self, event: AgentWebSocketEvents, handler: Callable) -> None:
321
+ """
322
+ Registers event handlers for specific events.
323
+ """
324
+ self._logger.info("event subscribed: %s", event)
325
+ if event in AgentWebSocketEvents.__members__.values() and callable(handler):
326
+ self._event_handlers[event].append(handler)
327
+
328
+ def _emit(self, event: AgentWebSocketEvents, *args, **kwargs) -> None:
329
+ """
330
+ Emits events to the registered event handlers.
331
+ """
332
+ self._logger.debug("AgentWebSocketClient._emit ENTER")
333
+ self._logger.debug("callback handlers for: %s", event)
334
+
335
+ # debug the threads
336
+ for thread in threading.enumerate():
337
+ self._logger.debug("after running thread: %s", thread.name)
338
+ self._logger.debug("number of active threads: %s", threading.active_count())
339
+
340
+ self._logger.debug("callback handlers for: %s", event)
341
+ for handler in self._event_handlers[event]:
342
+ handler(self, *args, **kwargs)
343
+
344
+ # debug the threads
345
+ for thread in threading.enumerate():
346
+ self._logger.debug("after running thread: %s", thread.name)
347
+ self._logger.debug("number of active threads: %s", threading.active_count())
348
+
349
+ self._logger.debug("AgentWebSocketClient._emit LEAVE")
350
+
351
+ # pylint: disable=too-many-return-statements,too-many-statements,too-many-locals,too-many-branches
352
+ def _process_text(self, message: str) -> None:
353
+ """
354
+ Processes messages received over the WebSocket connection.
355
+ """
356
+ self._logger.debug("AgentWebSocketClient._process_text ENTER")
357
+
358
+ try:
359
+ self._logger.debug("Text data received")
360
+ if len(message) == 0:
361
+ self._logger.debug("message is empty")
362
+ self._logger.debug("AgentWebSocketClient._process_text LEAVE")
363
+ return
364
+
365
+ data = json.loads(message)
366
+ response_type = data.get("type")
367
+ self._logger.debug("response_type: %s, data: %s", response_type, data)
368
+
369
+ match response_type:
370
+ case AgentWebSocketEvents.Open:
371
+ open_result: OpenResponse = OpenResponse.from_json(message)
372
+ self._logger.verbose("OpenResponse: %s", open_result)
373
+ self._emit(
374
+ AgentWebSocketEvents(AgentWebSocketEvents.Open),
375
+ open=open_result,
376
+ **dict(cast(Dict[Any, Any], self._kwargs)),
377
+ )
378
+ case AgentWebSocketEvents.Welcome:
379
+ welcome_result: WelcomeResponse = WelcomeResponse.from_json(message)
380
+ self._logger.verbose("WelcomeResponse: %s", welcome_result)
381
+ self._emit(
382
+ AgentWebSocketEvents(AgentWebSocketEvents.Welcome),
383
+ welcome=welcome_result,
384
+ **dict(cast(Dict[Any, Any], self._kwargs)),
385
+ )
386
+ case AgentWebSocketEvents.SettingsApplied:
387
+ settings_applied_result: SettingsAppliedResponse = (
388
+ SettingsAppliedResponse.from_json(message)
389
+ )
390
+ self._logger.verbose(
391
+ "SettingsAppliedResponse: %s", settings_applied_result
392
+ )
393
+ self._emit(
394
+ AgentWebSocketEvents(AgentWebSocketEvents.SettingsApplied),
395
+ settings_applied=settings_applied_result,
396
+ **dict(cast(Dict[Any, Any], self._kwargs)),
397
+ )
398
+ case AgentWebSocketEvents.ConversationText:
399
+ conversation_text_result: ConversationTextResponse = (
400
+ ConversationTextResponse.from_json(message)
401
+ )
402
+ self._logger.verbose(
403
+ "ConversationTextResponse: %s", conversation_text_result
404
+ )
405
+ self._emit(
406
+ AgentWebSocketEvents(AgentWebSocketEvents.ConversationText),
407
+ conversation_text=conversation_text_result,
408
+ **dict(cast(Dict[Any, Any], self._kwargs)),
409
+ )
410
+ case AgentWebSocketEvents.UserStartedSpeaking:
411
+ user_started_speaking_result: UserStartedSpeakingResponse = (
412
+ UserStartedSpeakingResponse.from_json(message)
413
+ )
414
+ self._logger.verbose(
415
+ "UserStartedSpeakingResponse: %s", user_started_speaking_result
416
+ )
417
+ self._emit(
418
+ AgentWebSocketEvents(AgentWebSocketEvents.UserStartedSpeaking),
419
+ user_started_speaking=user_started_speaking_result,
420
+ **dict(cast(Dict[Any, Any], self._kwargs)),
421
+ )
422
+ case AgentWebSocketEvents.AgentThinking:
423
+ agent_thinking_result: AgentThinkingResponse = (
424
+ AgentThinkingResponse.from_json(message)
425
+ )
426
+ self._logger.verbose(
427
+ "AgentThinkingResponse: %s", agent_thinking_result
428
+ )
429
+ self._emit(
430
+ AgentWebSocketEvents(AgentWebSocketEvents.AgentThinking),
431
+ agent_thinking=agent_thinking_result,
432
+ **dict(cast(Dict[Any, Any], self._kwargs)),
433
+ )
434
+ case AgentWebSocketEvents.FunctionCallRequest:
435
+ function_call_request_result: FunctionCallRequest = (
436
+ FunctionCallRequest.from_json(message)
437
+ )
438
+ self._logger.verbose(
439
+ "FunctionCallRequest: %s", function_call_request_result
440
+ )
441
+ self._emit(
442
+ AgentWebSocketEvents(AgentWebSocketEvents.FunctionCallRequest),
443
+ function_call_request=function_call_request_result,
444
+ **dict(cast(Dict[Any, Any], self._kwargs)),
445
+ )
446
+ case AgentWebSocketEvents.AgentStartedSpeaking:
447
+ agent_started_speaking_result: AgentStartedSpeakingResponse = (
448
+ AgentStartedSpeakingResponse.from_json(message)
449
+ )
450
+ self._logger.verbose(
451
+ "AgentStartedSpeakingResponse: %s",
452
+ agent_started_speaking_result,
453
+ )
454
+ self._emit(
455
+ AgentWebSocketEvents(AgentWebSocketEvents.AgentStartedSpeaking),
456
+ agent_started_speaking=agent_started_speaking_result,
457
+ **dict(cast(Dict[Any, Any], self._kwargs)),
458
+ )
459
+ case AgentWebSocketEvents.AgentAudioDone:
460
+ agent_audio_done_result: AgentAudioDoneResponse = (
461
+ AgentAudioDoneResponse.from_json(message)
462
+ )
463
+ self._logger.verbose(
464
+ "AgentAudioDoneResponse: %s", agent_audio_done_result
465
+ )
466
+ self._emit(
467
+ AgentWebSocketEvents(AgentWebSocketEvents.AgentAudioDone),
468
+ agent_audio_done=agent_audio_done_result,
469
+ **dict(cast(Dict[Any, Any], self._kwargs)),
470
+ )
471
+ case AgentWebSocketEvents.InjectionRefused:
472
+ injection_refused_result: InjectionRefusedResponse = (
473
+ InjectionRefusedResponse.from_json(message)
474
+ )
475
+ self._logger.verbose(
476
+ "InjectionRefused: %s", injection_refused_result
477
+ )
478
+ self._emit(
479
+ AgentWebSocketEvents(AgentWebSocketEvents.InjectionRefused),
480
+ injection_refused=injection_refused_result,
481
+ **dict(cast(Dict[Any, Any], self._kwargs)),
482
+ )
483
+ case AgentWebSocketEvents.Close:
484
+ close_result: CloseResponse = CloseResponse.from_json(message)
485
+ self._logger.verbose("CloseResponse: %s", close_result)
486
+ self._emit(
487
+ AgentWebSocketEvents(AgentWebSocketEvents.Close),
488
+ close=close_result,
489
+ **dict(cast(Dict[Any, Any], self._kwargs)),
490
+ )
491
+ case AgentWebSocketEvents.Error:
492
+ err_error: ErrorResponse = ErrorResponse.from_json(message)
493
+ self._logger.verbose("ErrorResponse: %s", err_error)
494
+ self._emit(
495
+ AgentWebSocketEvents(AgentWebSocketEvents.Error),
496
+ error=err_error,
497
+ **dict(cast(Dict[Any, Any], self._kwargs)),
498
+ )
499
+ case _:
500
+ self._logger.warning(
501
+ "Unknown Message: response_type: %s, data: %s",
502
+ response_type,
503
+ data,
504
+ )
505
+ unhandled_error: UnhandledResponse = UnhandledResponse(
506
+ type=AgentWebSocketEvents(AgentWebSocketEvents.Unhandled),
507
+ raw=message,
508
+ )
509
+ self._emit(
510
+ AgentWebSocketEvents(AgentWebSocketEvents.Unhandled),
511
+ unhandled=unhandled_error,
512
+ **dict(cast(Dict[Any, Any], self._kwargs)),
513
+ )
514
+
515
+ self._logger.notice("_process_text Succeeded")
516
+ self._logger.debug("SpeakStreamClient._process_text LEAVE")
517
+
518
+ except Exception as e: # pylint: disable=broad-except
519
+ self._logger.error("Exception in AgentWebSocketClient._process_text: %s", e)
520
+ e_error: ErrorResponse = ErrorResponse(
521
+ "Exception in AgentWebSocketClient._process_text",
522
+ f"{e}",
523
+ "Exception",
524
+ )
525
+ self._logger.error(
526
+ "Exception in AgentWebSocketClient._process_text: %s", str(e)
527
+ )
528
+ self._emit(
529
+ AgentWebSocketEvents(AgentWebSocketEvents.Error),
530
+ error=e_error,
531
+ **dict(cast(Dict[Any, Any], self._kwargs)),
532
+ )
533
+
534
+ # signal exit and close
535
+ super()._signal_exit()
536
+
537
+ self._logger.debug("AgentWebSocketClient._process_text LEAVE")
538
+
539
+ if self._config.options.get("termination_exception") is True:
540
+ raise
541
+ return
542
+
543
+ # pylint: enable=too-many-return-statements,too-many-statements
544
+
545
+ def _process_binary(self, message: bytes) -> None:
546
+ self._logger.debug("AgentWebSocketClient._process_binary ENTER")
547
+ self._logger.debug("Binary data received")
548
+
549
+ self._emit(
550
+ AgentWebSocketEvents(AgentWebSocketEvents.AudioData),
551
+ data=message,
552
+ **dict(cast(Dict[Any, Any], self._kwargs)),
553
+ )
554
+
555
+ self._logger.notice("_process_binary Succeeded")
556
+ self._logger.debug("AgentWebSocketClient._process_binary LEAVE")
557
+
558
+ # pylint: disable=too-many-return-statements
559
+ def _keep_alive(self) -> None:
560
+ """
561
+ Sends keepalive messages to the WebSocket connection.
562
+ """
563
+ self._logger.debug("AgentWebSocketClient._keep_alive ENTER")
564
+
565
+ counter = 0
566
+ while True:
567
+ try:
568
+ counter += 1
569
+ self._exit_event.wait(timeout=ONE_SECOND)
570
+
571
+ if self._exit_event.is_set():
572
+ self._logger.notice("_keep_alive exiting gracefully")
573
+ self._logger.debug("AgentWebSocketClient._keep_alive LEAVE")
574
+ return
575
+
576
+ # deepgram keepalive
577
+ if counter % DEEPGRAM_INTERVAL == 0:
578
+ self.keep_alive()
579
+
580
+ except Exception as e: # pylint: disable=broad-except
581
+ self._logger.error(
582
+ "Exception in AgentWebSocketClient._keep_alive: %s", e
583
+ )
584
+ e_error: ErrorResponse = ErrorResponse(
585
+ "Exception in AgentWebSocketClient._keep_alive",
586
+ f"{e}",
587
+ "Exception",
588
+ )
589
+ self._logger.error(
590
+ "Exception in AgentWebSocketClient._keep_alive: %s", str(e)
591
+ )
592
+ self._emit(
593
+ AgentWebSocketEvents(AgentWebSocketEvents.Error),
594
+ error=e_error,
595
+ **dict(cast(Dict[Any, Any], self._kwargs)),
596
+ )
597
+
598
+ # signal exit and close
599
+ super()._signal_exit()
600
+
601
+ self._logger.debug("AgentWebSocketClient._keep_alive LEAVE")
602
+
603
+ if self._config.options.get("termination_exception") is True:
604
+ raise
605
+ return
606
+
607
+ def keep_alive(self) -> bool:
608
+ """
609
+ Sends a KeepAlive message
610
+ """
611
+ self._logger.spam("AgentWebSocketClient.keep_alive ENTER")
612
+
613
+ self._logger.notice("Sending KeepAlive...")
614
+ ret = self.send(json.dumps({"type": "KeepAlive"}))
615
+
616
+ if not ret:
617
+ self._logger.error("keep_alive failed")
618
+ self._logger.spam("AgentWebSocketClient.keep_alive LEAVE")
619
+ return False
620
+
621
+ self._logger.notice("keep_alive succeeded")
622
+ self._logger.spam("AgentWebSocketClient.keep_alive LEAVE")
623
+
624
+ return True
625
+
626
+ def _close_message(self) -> bool:
627
+ # TODO: No known API close message # pylint: disable=fixme
628
+ # return self.send(json.dumps({"type": "Close"}))
629
+ return True
630
+
631
+ # closes the WebSocket connection gracefully
632
+ def finish(self) -> bool:
633
+ """
634
+ Closes the WebSocket connection gracefully.
635
+ """
636
+ self._logger.spam("AgentWebSocketClient.finish ENTER")
637
+
638
+ # call parent finish
639
+ if super().finish() is False:
640
+ self._logger.error("AgentWebSocketClient.finish failed")
641
+
642
+ if self._microphone is not None and self._microphone_created:
643
+ self._microphone.finish()
644
+ self._microphone_created = False
645
+
646
+ if self._speaker is not None and self._speaker_created:
647
+ self._speaker.finish()
648
+ self._speaker_created = False
649
+
650
+ # debug the threads
651
+ for thread in threading.enumerate():
652
+ self._logger.debug("before running thread: %s", thread.name)
653
+ self._logger.debug("number of active threads: %s", threading.active_count())
654
+
655
+ # stop the threads
656
+ self._logger.verbose("cancelling tasks...")
657
+ if self._keep_alive_thread is not None:
658
+ self._keep_alive_thread.join()
659
+ self._keep_alive_thread = None
660
+ self._logger.notice("processing _keep_alive_thread thread joined")
661
+
662
+ if self._listen_thread is not None:
663
+ self._listen_thread.join()
664
+ self._listen_thread = None
665
+ self._logger.notice("listening thread joined")
666
+
667
+ self._speaker = None
668
+ self._microphone = None
669
+
670
+ # debug the threads
671
+ for thread in threading.enumerate():
672
+ self._logger.debug("before running thread: %s", thread.name)
673
+ self._logger.debug("number of active threads: %s", threading.active_count())
674
+
675
+ self._logger.notice("finish succeeded")
676
+ self._logger.spam("AgentWebSocketClient.finish LEAVE")
677
+ return True
deepgram/clients/agent/v1/websocket/options.py ADDED
@@ -0,0 +1,453 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Copyright 2024 Deepgram SDK contributors. All Rights Reserved.
2
+ # Use of this source code is governed by a MIT license that can be found in the LICENSE file.
3
+ # SPDX-License-Identifier: MIT
4
+
5
+ from typing import List, Optional, Union, Any, Tuple
6
+ import logging
7
+
8
+ from dataclasses import dataclass, field
9
+ from dataclasses_json import config as dataclass_config
10
+
11
+ from deepgram.utils import verboselogs
12
+
13
+ from ...enums import AgentWebSocketEvents
14
+ from ....common import BaseResponse
15
+
16
+
17
+ # ConfigurationSettings
18
+
19
+
20
+ @dataclass
21
+ class Header(BaseResponse):
22
+ """
23
+ This class defines a single key/value pair for a header.
24
+ """
25
+
26
+ key: str
27
+ value: str
28
+
29
+
30
+ @dataclass
31
+ class Item(BaseResponse):
32
+ """
33
+ This class defines a single item in a list of items.
34
+ """
35
+
36
+ type: str
37
+ description: str
38
+
39
+
40
+ @dataclass
41
+ class Properties(BaseResponse):
42
+ """
43
+ This class defines the properties which is just a list of items.
44
+ """
45
+
46
+ item: Item
47
+
48
+ def __getitem__(self, key):
49
+ _dict = self.to_dict()
50
+ if "item" in _dict:
51
+ _dict["item"] = [Item.from_dict(item) for item in _dict["item"]]
52
+ return _dict[key]
53
+
54
+
55
+ @dataclass
56
+ class Parameters(BaseResponse):
57
+ """
58
+ This class defines the parameters for a function.
59
+ """
60
+
61
+ type: str
62
+ properties: Properties
63
+ required: List[str]
64
+
65
+ def __getitem__(self, key):
66
+ _dict = self.to_dict()
67
+ if "properties" in _dict:
68
+ _dict["properties"] = _dict["properties"].copy()
69
+ return _dict[key]
70
+
71
+
72
+ @dataclass
73
+ class Endpoint(BaseResponse):
74
+ """
75
+ Define a custom endpoint for the agent.
76
+ """
77
+
78
+ method: Optional[str] = field(default="POST")
79
+ url: str = field(default="")
80
+ headers: Optional[List[Header]] = field(
81
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
82
+ )
83
+
84
+ def __getitem__(self, key):
85
+ _dict = self.to_dict()
86
+ if "headers" in _dict:
87
+ _dict["headers"] = [
88
+ Header.from_dict(headers) for headers in _dict["headers"]
89
+ ]
90
+ return _dict[key]
91
+
92
+
93
+ @dataclass
94
+ class Function(BaseResponse):
95
+ """
96
+ This class defines a function for the Think model.
97
+ """
98
+
99
+ name: str
100
+ description: str
101
+ url: str
102
+ method: str
103
+ headers: Optional[List[Header]] = field(
104
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
105
+ )
106
+ parameters: Optional[Parameters] = field(
107
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
108
+ )
109
+ endpoint: Optional[Endpoint] = field(
110
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
111
+ )
112
+
113
+ def __getitem__(self, key):
114
+ _dict = self.to_dict()
115
+ if "parameters" in _dict and isinstance(_dict["parameters"], dict):
116
+ _dict["parameters"] = Parameters.from_dict(_dict["parameters"])
117
+ if "headers" in _dict and isinstance(_dict["headers"], list):
118
+ _dict["headers"] = [Header.from_dict(header) for header in _dict["headers"]]
119
+ if "endpoint" in _dict and isinstance(_dict["endpoint"], dict):
120
+ _dict["endpoint"] = Endpoint.from_dict(_dict["endpoint"])
121
+ return _dict[key]
122
+
123
+
124
+ @dataclass
125
+ class CartesiaVoice(BaseResponse):
126
+ """
127
+ This class defines the voice for the Cartesia model.
128
+ """
129
+
130
+ mode: str = field(
131
+ default="", metadata=dataclass_config(exclude=lambda f: f is None or f == "")
132
+ )
133
+ id: str = field(
134
+ default="", metadata=dataclass_config(exclude=lambda f: f is None or f == "")
135
+ )
136
+
137
+
138
+ @dataclass
139
+ class ListenProvider(BaseResponse):
140
+ """
141
+ This class defines the provider for the Listen model.
142
+ """
143
+
144
+ type: str = field(default="")
145
+ model: str = field(default="")
146
+ keyterms: Optional[List[str]] = field(
147
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
148
+ )
149
+
150
+ def __getitem__(self, key):
151
+ _dict = self.to_dict()
152
+ if "keyterms" in _dict and isinstance(_dict["keyterms"], list):
153
+ _dict["keyterms"] = [str(keyterm) for keyterm in _dict["keyterms"]]
154
+ return _dict[key]
155
+
156
+
157
+ @dataclass
158
+ class ThinkProvider(BaseResponse):
159
+ """
160
+ This class defines the provider for the Think model.
161
+ """
162
+
163
+ type: Optional[str] = field(default=None)
164
+ model: Optional[str] = field(default=None)
165
+ temperature: Optional[float] = field(
166
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
167
+ )
168
+
169
+
170
+ @dataclass
171
+ class SpeakProvider(BaseResponse):
172
+ """
173
+ This class defines the provider for the Speak model.
174
+ """
175
+
176
+ type: Optional[str] = field(default="deepgram")
177
+ """
178
+ Deepgram OR OpenAI model to use.
179
+ """
180
+ model: Optional[str] = field(
181
+ default="aura-2-thalia-en",
182
+ metadata=dataclass_config(exclude=lambda f: f is None),
183
+ )
184
+ """
185
+ ElevenLabs or Cartesia model to use.
186
+ """
187
+ model_id: Optional[str] = field(
188
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
189
+ )
190
+ """
191
+ Cartesia voice configuration.
192
+ """
193
+ voice: Optional[CartesiaVoice] = field(
194
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
195
+ )
196
+ """
197
+ Cartesia language.
198
+ """
199
+ language: Optional[str] = field(
200
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
201
+ )
202
+ """
203
+ ElevenLabs language.
204
+ """
205
+ language_code: Optional[str] = field(
206
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
207
+ )
208
+
209
+ def __getitem__(self, key):
210
+ _dict = self.to_dict()
211
+ if "voice" in _dict and isinstance(_dict["voice"], dict):
212
+ _dict["voice"] = CartesiaVoice.from_dict(_dict["voice"])
213
+ return _dict[key]
214
+
215
+
216
+ @dataclass
217
+ class Think(BaseResponse):
218
+ """
219
+ This class defines any configuration settings for the Think model.
220
+ """
221
+
222
+ provider: ThinkProvider = field(default_factory=ThinkProvider)
223
+ functions: Optional[List[Function]] = field(
224
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
225
+ )
226
+ endpoint: Optional[Endpoint] = field(
227
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
228
+ )
229
+ prompt: Optional[str] = field(
230
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
231
+ )
232
+
233
+ def __getitem__(self, key):
234
+ _dict = self.to_dict()
235
+ if "provider" in _dict and isinstance(_dict["provider"], dict):
236
+ _dict["provider"] = ThinkProvider.from_dict(_dict["provider"])
237
+ if "functions" in _dict and isinstance(_dict["functions"], list):
238
+ _dict["functions"] = [
239
+ Function.from_dict(function) for function in _dict["functions"]
240
+ ]
241
+ if "endpoint" in _dict and isinstance(_dict["endpoint"], dict):
242
+ _dict["endpoint"] = Endpoint.from_dict(_dict["endpoint"])
243
+ return _dict[key]
244
+
245
+
246
+ @dataclass
247
+ class Listen(BaseResponse):
248
+ """
249
+ This class defines any configuration settings for the Listen model.
250
+ """
251
+
252
+ provider: ListenProvider = field(default_factory=ListenProvider)
253
+
254
+ def __getitem__(self, key):
255
+ _dict = self.to_dict()
256
+ if "provider" in _dict and isinstance(_dict["provider"], dict):
257
+ _dict["provider"] = ListenProvider.from_dict(_dict["provider"])
258
+ return _dict[key]
259
+
260
+
261
+ @dataclass
262
+ class Speak(BaseResponse):
263
+ """
264
+ This class defines any configuration settings for the Speak model.
265
+ """
266
+
267
+ provider: SpeakProvider = field(default_factory=SpeakProvider)
268
+ endpoint: Optional[Endpoint] = field(
269
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
270
+ )
271
+
272
+ def __getitem__(self, key):
273
+ _dict = self.to_dict()
274
+ if "provider" in _dict and isinstance(_dict["provider"], dict):
275
+ _dict["provider"] = SpeakProvider.from_dict(_dict["provider"])
276
+ if "endpoint" in _dict and isinstance(_dict["endpoint"], dict):
277
+ _dict["endpoint"] = Endpoint.from_dict(_dict["endpoint"])
278
+ return _dict[key]
279
+
280
+
281
+ @dataclass
282
+ class Agent(BaseResponse):
283
+ """
284
+ This class defines any configuration settings for the Agent model.
285
+ """
286
+
287
+ listen: Listen = field(default_factory=Listen)
288
+ think: Think = field(default_factory=Think)
289
+ speak: Speak = field(default_factory=Speak)
290
+ greeting: Optional[str] = field(
291
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
292
+ )
293
+
294
+ def __getitem__(self, key):
295
+ _dict = self.to_dict()
296
+ if "listen" in _dict and isinstance(_dict["listen"], dict):
297
+ _dict["listen"] = Listen.from_dict(_dict["listen"])
298
+ if "think" in _dict and isinstance(_dict["think"], dict):
299
+ _dict["think"] = Think.from_dict(_dict["think"])
300
+ if "speak" in _dict and isinstance(_dict["speak"], dict):
301
+ _dict["speak"] = Speak.from_dict(_dict["speak"])
302
+ return _dict[key]
303
+
304
+
305
+ @dataclass
306
+ class Input(BaseResponse):
307
+ """
308
+ This class defines any configuration settings for the input audio.
309
+ """
310
+
311
+ encoding: Optional[str] = field(default="linear16")
312
+ sample_rate: int = field(default=16000)
313
+
314
+
315
+ @dataclass
316
+ class Output(BaseResponse):
317
+ """
318
+ This class defines any configuration settings for the output audio.
319
+ """
320
+
321
+ encoding: Optional[str] = field(default="linear16")
322
+ sample_rate: Optional[int] = field(default=16000)
323
+ bitrate: Optional[int] = field(
324
+ default=None, metadata=dataclass_config(exclude=lambda f: f is None)
325
+ )
326
+ container: Optional[str] = field(default="none")
327
+
328
+
329
+ @dataclass
330
+ class Audio(BaseResponse):
331
+ """
332
+ This class defines any configuration settings for the audio.
333
+ """
334
+
335
+ input: Optional[Input] = field(default_factory=Input)
336
+ output: Optional[Output] = field(default_factory=Output)
337
+
338
+ def __getitem__(self, key):
339
+ _dict = self.to_dict()
340
+ if "input" in _dict and isinstance(_dict["input"], dict):
341
+ _dict["input"] = Input.from_dict(_dict["input"])
342
+ if "output" in _dict and isinstance(_dict["output"], dict):
343
+ _dict["output"] = Output.from_dict(_dict["output"])
344
+ return _dict[key]
345
+
346
+
347
+ @dataclass
348
+ class Language(BaseResponse):
349
+ """
350
+ Define the language for the agent.
351
+ """
352
+
353
+ type: str = field(default="en")
354
+
355
+
356
+ @dataclass
357
+ class SettingsOptions(BaseResponse):
358
+ """
359
+ The client should send a Settings message immediately after opening the websocket and before sending any audio.
360
+ """
361
+
362
+ experimental: Optional[bool] = field(default=False)
363
+ type: str = str(AgentWebSocketEvents.Settings)
364
+ audio: Audio = field(default_factory=Audio)
365
+ agent: Agent = field(default_factory=Agent)
366
+
367
+ def __getitem__(self, key):
368
+ _dict = self.to_dict()
369
+ if "audio" in _dict and isinstance(_dict["audio"], dict):
370
+ _dict["audio"] = Audio.from_dict(_dict["audio"])
371
+ if "agent" in _dict and isinstance(_dict["agent"], dict):
372
+ _dict["agent"] = Agent.from_dict(_dict["agent"])
373
+ return _dict[key]
374
+
375
+ def check(self):
376
+ """
377
+ Check the options for any deprecated or soon-to-be-deprecated options.
378
+ """
379
+ logger = verboselogs.VerboseLogger(__name__)
380
+ logger.addHandler(logging.StreamHandler())
381
+ prev = logger.level
382
+ logger.setLevel(verboselogs.ERROR)
383
+
384
+ # do we need to check anything here?
385
+
386
+ logger.setLevel(prev)
387
+
388
+ return True
389
+
390
+
391
+ # UpdatePrompt
392
+
393
+
394
+ @dataclass
395
+ class UpdatePromptOptions(BaseResponse):
396
+ """
397
+ The client can send an UpdatePrompt message to provide a new prompt to the Think model in the middle of a conversation.
398
+ """
399
+
400
+ type: str = str(AgentWebSocketEvents.UpdatePrompt)
401
+ prompt: str = field(default="")
402
+
403
+
404
+ # UpdateSpeak
405
+
406
+
407
+ @dataclass
408
+ class UpdateSpeakOptions(BaseResponse):
409
+ """
410
+ The client can send an UpdateSpeak message to change the Speak model in the middle of a conversation.
411
+ """
412
+
413
+ type: str = str(AgentWebSocketEvents.UpdateSpeak)
414
+ speak: Speak = field(default_factory=Speak)
415
+
416
+
417
+ # InjectAgentMessage
418
+
419
+
420
+ @dataclass
421
+ class InjectAgentMessageOptions(BaseResponse):
422
+ """
423
+ The client can send an InjectAgentMessage to immediately trigger an agent statement. If the injection request arrives while the user is speaking, or while the server is in the middle of sending audio for an agent response, then the request will be ignored and the server will reply with an InjectionRefused.
424
+ """
425
+
426
+ type: str = str(AgentWebSocketEvents.InjectAgentMessage)
427
+ message: str = field(default="")
428
+
429
+
430
+ # Function Call Response
431
+
432
+
433
+ @dataclass
434
+ class FunctionCallResponse(BaseResponse):
435
+ """
436
+ TheFunctionCallResponse message is a JSON command that the client should reply with every time there is a FunctionCallRequest received.
437
+ """
438
+
439
+ type: str = "FunctionCallResponse"
440
+ function_call_id: str = field(default="")
441
+ output: str = field(default="")
442
+
443
+
444
+ # Agent Keep Alive
445
+
446
+
447
+ @dataclass
448
+ class AgentKeepAlive(BaseResponse):
449
+ """
450
+ The KeepAlive message is a JSON command that you can use to ensure that the server does not close the connection.
451
+ """
452
+
453
+ type: str = "KeepAlive"