|
|
@@ -0,0 +1,228 @@
|
|
|
+#!/usr/bin/env python
|
|
|
+# -*- coding: utf-8 -*-
|
|
|
+
|
|
|
+import win32com.client
|
|
|
+import platform
|
|
|
+import wave
|
|
|
+import numpy as np
|
|
|
+from pydub import AudioSegment, silence
|
|
|
+from pydub.playback import play
|
|
|
+from playsound import playsound
|
|
|
+import re
|
|
|
+import librosa
|
|
|
+from pydub import effects
|
|
|
+import os
|
|
|
+
|
|
|
+class Voice:
|
|
|
+
|
|
|
+
|
|
|
+ instance = None
|
|
|
+ @staticmethod
|
|
|
+ def getInstance():
|
|
|
+ if Voice.instance==None:
|
|
|
+ Voice.instance = Voice()
|
|
|
+ return Voice.instance
|
|
|
+
|
|
|
+ def __init__(self):
|
|
|
+ self.sys = platform.system()
|
|
|
+ if self.sys=="Windows":
|
|
|
+ self.speaker = win32com.client.Dispatch("SAPI.SpVoice")
|
|
|
+ else:
|
|
|
+ self.speaker = None
|
|
|
+
|
|
|
+ def Say(self, text, save:bool = True):
|
|
|
+ if self.speaker != None:
|
|
|
+ self.speaker.Rate = 0.0
|
|
|
+ if save:
|
|
|
+ stream = win32com.client.Dispatch("SAPI.SpFileStream")
|
|
|
+ stream.Open("cache/spoken.wav", 3, False)
|
|
|
+ defAudioOutput = self.speaker.AudioOutputStream
|
|
|
+ self.speaker.AudioOutputStream = stream
|
|
|
+ self.speaker.Speak(text)
|
|
|
+ if save:
|
|
|
+ stream.Close()
|
|
|
+ self.Edit("cache/spoken.wav","cache/pitched.wav")
|
|
|
+ playsound("cache/pitched.wav")
|
|
|
+ self.speaker.AudioOutputStream = defAudioOutput
|
|
|
+ else:
|
|
|
+ self.__svox(text)
|
|
|
+
|
|
|
+ def EmotionalSay(self, text):
|
|
|
+ files = list()
|
|
|
+ defAudioOutput = self.speaker.AudioOutputStream
|
|
|
+ collection = re.finditer('(?:\[(?P<args>.*?)\])?(?P<text>[^\[]*)', text, re.UNICODE)
|
|
|
+ count=0
|
|
|
+ combined_wav = AudioSegment.empty()
|
|
|
+ self.speaker.Rate = -1
|
|
|
+ for x in collection:
|
|
|
+ args = self.__getArgs(x.group("args"))
|
|
|
+ itm = self.__sayToFile(x.group("text"),"cache/tmp_"+str(count)+".wav")
|
|
|
+
|
|
|
+ # Remove silence at the beginning/end
|
|
|
+ start_trim = silence.detect_leading_silence(itm)
|
|
|
+ end_trim = silence.detect_leading_silence(itm.reverse())
|
|
|
+ duration = len(itm)
|
|
|
+ itm = itm[start_trim:-end_trim]
|
|
|
+
|
|
|
+ if args != "":
|
|
|
+ itm_e = self.EditSegment(itm, args)
|
|
|
+ combined_wav += itm_e
|
|
|
+ else:
|
|
|
+ combined_wav += itm
|
|
|
+ #os.remove("cache/tmp_"+str(count)+".wav")
|
|
|
+ self.speaker.AudioOutputStream = defAudioOutput
|
|
|
+ play(combined_wav)
|
|
|
+
|
|
|
+ def __sayToFile(self, text:str, file:str):
|
|
|
+ if text == "":
|
|
|
+ return AudioSegment.empty()
|
|
|
+
|
|
|
+ stream = win32com.client.Dispatch("SAPI.SpFileStream")
|
|
|
+ stream.Open(file, 3, False)
|
|
|
+ self.speaker.AudioOutputStream = stream
|
|
|
+ self.speaker.Speak(text)
|
|
|
+ stream.Close()
|
|
|
+ return AudioSegment.from_file(file)
|
|
|
+
|
|
|
+ def PlaySound(file:str):
|
|
|
+ if self.sys=="Windows":
|
|
|
+ import winsound
|
|
|
+ winsound.PlaySound("SystemExit", winsound.SND_ALIAS)
|
|
|
+
|
|
|
+ def EditSegment(self, segment, args):
|
|
|
+ print(args)
|
|
|
+ if "vol" in args:
|
|
|
+ segment += args["vol"]
|
|
|
+ if "pitch" in args:
|
|
|
+ segment = self.pitch_shift(segment, args["pitch"])
|
|
|
+ if "speedup" in args:
|
|
|
+ segment = segment.speedup(args["speedup"])
|
|
|
+ if "speed" in args and args["speed"]>0:
|
|
|
+ segment = self.speed_change(segment, 1.15)
|
|
|
+ #segment = self.speed_change(segment, args["speed"])
|
|
|
+ if args["speed"]>1:
|
|
|
+ segment = self.pitch_shift(segment, -(1-args["speed"]))
|
|
|
+ else:
|
|
|
+ #pitch = (1/((args["speed"]/12)))*1
|
|
|
+ pitch = 6.55
|
|
|
+ #print("Speed Pitch: "+str(pitch))
|
|
|
+ #segment = self.pitch_shift(segment, 1)
|
|
|
+ #segment = segment.speedup(2)
|
|
|
+ if "fadein" in args:
|
|
|
+ segment = segment.fade_in(args["fadein"])
|
|
|
+ if "fadeout" in args:
|
|
|
+ segment = segment.fade_out(args["fadeout"])
|
|
|
+ if "break" in args and args["break"]>=1:
|
|
|
+ pauseSegment=AudioSegment.silent(args["break"])
|
|
|
+ segment = pauseSegment + segment
|
|
|
+ if "effect" in args:
|
|
|
+ if os.path.exists("sounds/"+args["effect"]):
|
|
|
+ fxSound=AudioSegment.from_file(args["effect"])
|
|
|
+ segment = fxSound + segment
|
|
|
+ else:
|
|
|
+ print(" Effect not found")
|
|
|
+ return segment
|
|
|
+
|
|
|
+ def removeSilence(self, sound):
|
|
|
+ non_sil_times = detect_nonsilent(sound, min_silence_len=50, silence_thresh=sound.dBFS * 1.5)
|
|
|
+ if len(non_sil_times) > 0:
|
|
|
+ non_sil_times_concat = [non_sil_times[0]]
|
|
|
+ if len(non_sil_times) > 1:
|
|
|
+ for t in non_sil_times[1:]:
|
|
|
+ if t[0] - non_sil_times_concat[-1][-1] < 200:
|
|
|
+ non_sil_times_concat[-1][-1] = t[1]
|
|
|
+ else:
|
|
|
+ non_sil_times_concat.append(t)
|
|
|
+ non_sil_times = [t for t in non_sil_times_concat if t[1] - t[0] > 350]
|
|
|
+ return sound[non_sil_times[0][0]: non_sil_times[-1][1]]
|
|
|
+
|
|
|
+ def pitch_shift(self, sound, n_steps:float):
|
|
|
+ y = np.frombuffer(sound._data, dtype=np.int16).astype(np.float32)/2**15
|
|
|
+ y = librosa.effects.pitch_shift(y, sound.frame_rate, n_steps=n_steps)
|
|
|
+ a = AudioSegment(np.array(y * (1<<15), dtype=np.int16).tobytes(), frame_rate = sound.frame_rate, sample_width=2, channels = 1)
|
|
|
+ return a
|
|
|
+
|
|
|
+ def speed_change(self, sound, speed=1.0):
|
|
|
+ # Manually override the frame_rate. This tells the computer how many
|
|
|
+ # samples to play per second
|
|
|
+ sound_with_altered_frame_rate = sound._spawn(sound.raw_data, overrides={
|
|
|
+ "frame_rate": int(sound.frame_rate * speed)
|
|
|
+ })
|
|
|
+ # convert the sound with altered frame rate to a standard frame rate
|
|
|
+ # so that regular playback programs will work right. They often only
|
|
|
+ # know how to play audio at standard frame rate (like 44.1k)
|
|
|
+ return sound_with_altered_frame_rate.set_frame_rate(sound.frame_rate)
|
|
|
+
|
|
|
+ def detect_leading_silence(self, sound, silence_threshold=-50.0, chunk_size=10):
|
|
|
+ '''
|
|
|
+ sound is a pydub.AudioSegment
|
|
|
+ silence_threshold in dB
|
|
|
+ chunk_size in ms
|
|
|
+
|
|
|
+ iterate over chunks until you find the first one with sound
|
|
|
+ '''
|
|
|
+ trim_ms = 0 # ms
|
|
|
+ assert chunk_size > 0 # to avoid infinite loop
|
|
|
+ while sound[trim_ms:trim_ms+chunk_size].dBFS < silence_threshold and trim_ms < len(sound):
|
|
|
+ trim_ms += chunk_size
|
|
|
+ return trim_ms
|
|
|
+
|
|
|
+ def __getArgs(self, args:str):
|
|
|
+ res = dict()
|
|
|
+ if args is None or args=="":
|
|
|
+ return res
|
|
|
+ expl = args.split(",")
|
|
|
+ for x in expl:
|
|
|
+ x = x.strip()
|
|
|
+ if ":" in x:
|
|
|
+ k,v = x.split(":",2)
|
|
|
+ print("getArgs -> '"+k+"' : '"+v+"'")
|
|
|
+ if k.lower().strip()=="p" or k.lower().strip()=="pitch":
|
|
|
+ res["pitch"]=float(v)
|
|
|
+ elif k.lower().strip()=="v" or k.lower().strip()=="vol" or k.lower().strip()=="volumen":
|
|
|
+ res["vol"]=int(v)
|
|
|
+ elif k.lower().strip()=="su" or k.lower().strip()=="speedup":
|
|
|
+ res["speedup"]=float(v)
|
|
|
+ elif k.lower().strip()=="s" or k.lower().strip()=="slow" or k.lower().strip()=="speed":
|
|
|
+ res["speed"]=float(v)
|
|
|
+ elif k.lower().strip()=="b" or k.lower().strip()=="break" or k.lower().strip()=="pause" or k.lower().strip()=="p":
|
|
|
+ res["break"]=float(v)
|
|
|
+ elif k.lower().strip()=="fade" or k.lower().strip()=="f":
|
|
|
+ res["fadein"]=int(v)
|
|
|
+ res["fadeout"]=int(v)
|
|
|
+ elif k.lower().strip()=="fadein" or k.lower().strip()=="fi" or k.lower().strip()=="fade in" or k.lower().strip()=="fade-in":
|
|
|
+ res["fadein"]=int(v)
|
|
|
+ elif k.lower().strip()=="fadeout" or k.lower().strip()=="fo" or k.lower().strip()=="fade out" or k.lower().strip()=="fade-out":
|
|
|
+ res["fadeout"]=int(v)
|
|
|
+ elif k.lower().strip()=="fx" or k.lower().strip()=="effect":
|
|
|
+ res["effect"]=v
|
|
|
+ else:
|
|
|
+ if k.lower().strip()=="b" or k.lower().strip()=="break" or k.lower().strip()=="pause" or k.lower().strip()=="p":
|
|
|
+ res["break"]=1500
|
|
|
+ return res
|
|
|
+
|
|
|
+ def RobotEffect(self, input:str, output:str, frequence:int = 2):
|
|
|
+ wr = wave.open(input, 'r')
|
|
|
+ par = list(wr.getparams())
|
|
|
+ par[3] = 0 # The number of samples will be set by writeframes.
|
|
|
+ par = tuple(par)
|
|
|
+ ww = wave.open(output, 'w')
|
|
|
+ ww.setparams(par)
|
|
|
+ sz = wr.getframerate()//frequence # Read and process 1/fr second at a time.
|
|
|
+ # A larger number for fr means less reverb.
|
|
|
+ c = int(wr.getnframes()/sz) # count of the whole file
|
|
|
+ shift = 100//frequence # shifting 100 Hz
|
|
|
+ for num in range(c):
|
|
|
+ da = np.fromstring(wr.readframes(sz), dtype=np.int16)
|
|
|
+ left, right = da[0::2], da[1::2] # left and right channel
|
|
|
+ lf, rf = np.fft.rfft(left), np.fft.rfft(right)
|
|
|
+ lf, rf = np.roll(lf, shift), np.roll(rf, shift)
|
|
|
+ lf[0:shift], rf[0:shift] = 0, 0
|
|
|
+ nl, nr = np.fft.irfft(lf), np.fft.irfft(rf)
|
|
|
+ ns = np.column_stack((nl, nr)).ravel().astype(np.int16)
|
|
|
+ ww.writeframes(ns.tostring())
|
|
|
+ wr.close()
|
|
|
+ ww.close()
|
|
|
+
|
|
|
+ def __svox(self, text):
|
|
|
+ pass
|