diff --git a/style_bert_vits2/text_processing/chinese/__init__.py b/style_bert_vits2/text_processing/chinese/__init__.py index 92c8214..23acbf3 100644 --- a/style_bert_vits2/text_processing/chinese/__init__.py +++ b/style_bert_vits2/text_processing/chinese/__init__.py @@ -2,10 +2,12 @@ import os import re import cn2an +import jieba.posseg as psg from pypinyin import lazy_pinyin, Style +from style_bert_vits2.text_processing.chinese.tone_sandhi import ToneSandhi from style_bert_vits2.text_processing.symbols import PUNCTUATIONS -from text.tone_sandhi import ToneSandhi + current_file_path = os.path.dirname(__file__) pinyin_to_symbol_map = { @@ -13,8 +15,6 @@ pinyin_to_symbol_map = { for line in open(os.path.join(current_file_path, "opencpop-strict.txt")).readlines() } -import jieba.posseg as psg - rep_map = { ":": ",", diff --git a/style_bert_vits2/text_processing/chinese/tone_sandhi.py b/style_bert_vits2/text_processing/chinese/tone_sandhi.py index 38f3137..5832434 100644 --- a/style_bert_vits2/text_processing/chinese/tone_sandhi.py +++ b/style_bert_vits2/text_processing/chinese/tone_sandhi.py @@ -11,8 +11,6 @@ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. -from typing import List -from typing import Tuple import jieba from pypinyin import lazy_pinyin @@ -463,7 +461,7 @@ class ToneSandhi: # word: "家里" # pos: "s" # finals: ['ia1', 'i3'] - def _neural_sandhi(self, word: str, pos: str, finals: List[str]) -> List[str]: + def _neural_sandhi(self, word: str, pos: str, finals: list[str]) -> list[str]: # reduplication words for n. and v. e.g. 奶奶, 试试, 旺旺 for j, item in enumerate(word): if ( @@ -522,7 +520,7 @@ class ToneSandhi: finals = sum(finals_list, []) return finals - def _bu_sandhi(self, word: str, finals: List[str]) -> List[str]: + def _bu_sandhi(self, word: str, finals: list[str]) -> list[str]: # e.g. 看不懂 if len(word) == 3 and word[1] == "不": finals[1] = finals[1][:-1] + "5" @@ -533,7 +531,7 @@ class ToneSandhi: finals[i] = finals[i][:-1] + "2" return finals - def _yi_sandhi(self, word: str, finals: List[str]) -> List[str]: + def _yi_sandhi(self, word: str, finals: list[str]) -> list[str]: # "一" in number sequences, e.g. 一零零, 二一零 if word.find("一") != -1 and all( [item.isnumeric() for item in word if item != "一"] @@ -558,9 +556,9 @@ class ToneSandhi: finals[i] = finals[i][:-1] + "4" return finals - def _split_word(self, word: str) -> List[str]: + def _split_word(self, word: str) -> list[str]: word_list = jieba.cut_for_search(word) - word_list = sorted(word_list, key=lambda i: len(i), reverse=False) + word_list = sorted(word_list, key=lambda i: len(i), reverse=False) # type: ignore first_subword = word_list[0] first_begin_idx = word.find(first_subword) if first_begin_idx == 0: @@ -571,7 +569,7 @@ class ToneSandhi: new_word_list = [second_subword, first_subword] return new_word_list - def _three_sandhi(self, word: str, finals: List[str]) -> List[str]: + def _three_sandhi(self, word: str, finals: list[str]) -> list[str]: if len(word) == 2 and self._all_tone_three(finals): finals[0] = finals[0][:-1] + "2" elif len(word) == 3: @@ -611,12 +609,12 @@ class ToneSandhi: return finals - def _all_tone_three(self, finals: List[str]) -> bool: + def _all_tone_three(self, finals: list[str]) -> bool: return all(x[-1] == "3" for x in finals) # merge "不" and the word behind it # if don't merge, "不" sometimes appears alone according to jieba, which may occur sandhi error - def _merge_bu(self, seg: List[Tuple[str, str]]) -> List[Tuple[str, str]]: + def _merge_bu(self, seg: list[tuple[str, str]]) -> list[tuple[str, str]]: new_seg = [] last_word = "" for word, pos in seg: @@ -636,7 +634,7 @@ class ToneSandhi: # e.g. # input seg: [('听', 'v'), ('一', 'm'), ('听', 'v')] # output seg: [['听一听', 'v']] - def _merge_yi(self, seg: List[Tuple[str, str]]) -> List[Tuple[str, str]]: + def _merge_yi(self, seg: list[tuple[str, str]]) -> list[tuple[str, str]]: new_seg = [] * len(seg) # function 1 i = 0 @@ -674,8 +672,8 @@ class ToneSandhi: # the first and the second words are all_tone_three def _merge_continuous_three_tones( - self, seg: List[Tuple[str, str]] - ) -> List[Tuple[str, str]]: + self, seg: list[tuple[str, str]] + ) -> list[tuple[str, str]]: new_seg = [] sub_finals_list = [ lazy_pinyin(word, neutral_tone_with_five=True, style=Style.FINALS_TONE3) @@ -709,8 +707,8 @@ class ToneSandhi: # the last char of first word and the first char of second word is tone_three def _merge_continuous_three_tones_2( - self, seg: List[Tuple[str, str]] - ) -> List[Tuple[str, str]]: + self, seg: list[tuple[str, str]] + ) -> list[tuple[str, str]]: new_seg = [] sub_finals_list = [ lazy_pinyin(word, neutral_tone_with_five=True, style=Style.FINALS_TONE3) @@ -738,7 +736,7 @@ class ToneSandhi: new_seg.append([word, pos]) return new_seg - def _merge_er(self, seg: List[Tuple[str, str]]) -> List[Tuple[str, str]]: + def _merge_er(self, seg: list[tuple[str, str]]) -> list[tuple[str, str]]: new_seg = [] for i, (word, pos) in enumerate(seg): if i - 1 >= 0 and word == "儿" and seg[i - 1][0] != "#": @@ -747,7 +745,7 @@ class ToneSandhi: new_seg.append([word, pos]) return new_seg - def _merge_reduplication(self, seg: List[Tuple[str, str]]) -> List[Tuple[str, str]]: + def _merge_reduplication(self, seg: list[tuple[str, str]]) -> list[tuple[str, str]]: new_seg = [] for i, (word, pos) in enumerate(seg): if new_seg and word == new_seg[-1][0]: @@ -756,7 +754,7 @@ class ToneSandhi: new_seg.append([word, pos]) return new_seg - def pre_merge_for_modify(self, seg: List[Tuple[str, str]]) -> List[Tuple[str, str]]: + def pre_merge_for_modify(self, seg: list[tuple[str, str]]) -> list[tuple[str, str]]: seg = self._merge_bu(seg) try: seg = self._merge_yi(seg) @@ -768,7 +766,7 @@ class ToneSandhi: seg = self._merge_er(seg) return seg - def modified_tone(self, word: str, pos: str, finals: List[str]) -> List[str]: + def modified_tone(self, word: str, pos: str, finals: list[str]) -> list[str]: finals = self._bu_sandhi(word, finals) finals = self._yi_sandhi(word, finals) finals = self._neural_sandhi(word, pos, finals)