Fix: import error
This commit is contained in:
@@ -2,10 +2,12 @@ import os
|
|||||||
import re
|
import re
|
||||||
|
|
||||||
import cn2an
|
import cn2an
|
||||||
|
import jieba.posseg as psg
|
||||||
from pypinyin import lazy_pinyin, Style
|
from pypinyin import lazy_pinyin, Style
|
||||||
|
|
||||||
|
from style_bert_vits2.text_processing.chinese.tone_sandhi import ToneSandhi
|
||||||
from style_bert_vits2.text_processing.symbols import PUNCTUATIONS
|
from style_bert_vits2.text_processing.symbols import PUNCTUATIONS
|
||||||
from text.tone_sandhi import ToneSandhi
|
|
||||||
|
|
||||||
current_file_path = os.path.dirname(__file__)
|
current_file_path = os.path.dirname(__file__)
|
||||||
pinyin_to_symbol_map = {
|
pinyin_to_symbol_map = {
|
||||||
@@ -13,8 +15,6 @@ pinyin_to_symbol_map = {
|
|||||||
for line in open(os.path.join(current_file_path, "opencpop-strict.txt")).readlines()
|
for line in open(os.path.join(current_file_path, "opencpop-strict.txt")).readlines()
|
||||||
}
|
}
|
||||||
|
|
||||||
import jieba.posseg as psg
|
|
||||||
|
|
||||||
|
|
||||||
rep_map = {
|
rep_map = {
|
||||||
":": ",",
|
":": ",",
|
||||||
|
|||||||
@@ -11,8 +11,6 @@
|
|||||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
# See the License for the specific language governing permissions and
|
# See the License for the specific language governing permissions and
|
||||||
# limitations under the License.
|
# limitations under the License.
|
||||||
from typing import List
|
|
||||||
from typing import Tuple
|
|
||||||
|
|
||||||
import jieba
|
import jieba
|
||||||
from pypinyin import lazy_pinyin
|
from pypinyin import lazy_pinyin
|
||||||
@@ -463,7 +461,7 @@ class ToneSandhi:
|
|||||||
# word: "家里"
|
# word: "家里"
|
||||||
# pos: "s"
|
# pos: "s"
|
||||||
# finals: ['ia1', 'i3']
|
# finals: ['ia1', 'i3']
|
||||||
def _neural_sandhi(self, word: str, pos: str, finals: List[str]) -> List[str]:
|
def _neural_sandhi(self, word: str, pos: str, finals: list[str]) -> list[str]:
|
||||||
# reduplication words for n. and v. e.g. 奶奶, 试试, 旺旺
|
# reduplication words for n. and v. e.g. 奶奶, 试试, 旺旺
|
||||||
for j, item in enumerate(word):
|
for j, item in enumerate(word):
|
||||||
if (
|
if (
|
||||||
@@ -522,7 +520,7 @@ class ToneSandhi:
|
|||||||
finals = sum(finals_list, [])
|
finals = sum(finals_list, [])
|
||||||
return finals
|
return finals
|
||||||
|
|
||||||
def _bu_sandhi(self, word: str, finals: List[str]) -> List[str]:
|
def _bu_sandhi(self, word: str, finals: list[str]) -> list[str]:
|
||||||
# e.g. 看不懂
|
# e.g. 看不懂
|
||||||
if len(word) == 3 and word[1] == "不":
|
if len(word) == 3 and word[1] == "不":
|
||||||
finals[1] = finals[1][:-1] + "5"
|
finals[1] = finals[1][:-1] + "5"
|
||||||
@@ -533,7 +531,7 @@ class ToneSandhi:
|
|||||||
finals[i] = finals[i][:-1] + "2"
|
finals[i] = finals[i][:-1] + "2"
|
||||||
return finals
|
return finals
|
||||||
|
|
||||||
def _yi_sandhi(self, word: str, finals: List[str]) -> List[str]:
|
def _yi_sandhi(self, word: str, finals: list[str]) -> list[str]:
|
||||||
# "一" in number sequences, e.g. 一零零, 二一零
|
# "一" in number sequences, e.g. 一零零, 二一零
|
||||||
if word.find("一") != -1 and all(
|
if word.find("一") != -1 and all(
|
||||||
[item.isnumeric() for item in word if item != "一"]
|
[item.isnumeric() for item in word if item != "一"]
|
||||||
@@ -558,9 +556,9 @@ class ToneSandhi:
|
|||||||
finals[i] = finals[i][:-1] + "4"
|
finals[i] = finals[i][:-1] + "4"
|
||||||
return finals
|
return finals
|
||||||
|
|
||||||
def _split_word(self, word: str) -> List[str]:
|
def _split_word(self, word: str) -> list[str]:
|
||||||
word_list = jieba.cut_for_search(word)
|
word_list = jieba.cut_for_search(word)
|
||||||
word_list = sorted(word_list, key=lambda i: len(i), reverse=False)
|
word_list = sorted(word_list, key=lambda i: len(i), reverse=False) # type: ignore
|
||||||
first_subword = word_list[0]
|
first_subword = word_list[0]
|
||||||
first_begin_idx = word.find(first_subword)
|
first_begin_idx = word.find(first_subword)
|
||||||
if first_begin_idx == 0:
|
if first_begin_idx == 0:
|
||||||
@@ -571,7 +569,7 @@ class ToneSandhi:
|
|||||||
new_word_list = [second_subword, first_subword]
|
new_word_list = [second_subword, first_subword]
|
||||||
return new_word_list
|
return new_word_list
|
||||||
|
|
||||||
def _three_sandhi(self, word: str, finals: List[str]) -> List[str]:
|
def _three_sandhi(self, word: str, finals: list[str]) -> list[str]:
|
||||||
if len(word) == 2 and self._all_tone_three(finals):
|
if len(word) == 2 and self._all_tone_three(finals):
|
||||||
finals[0] = finals[0][:-1] + "2"
|
finals[0] = finals[0][:-1] + "2"
|
||||||
elif len(word) == 3:
|
elif len(word) == 3:
|
||||||
@@ -611,12 +609,12 @@ class ToneSandhi:
|
|||||||
|
|
||||||
return finals
|
return finals
|
||||||
|
|
||||||
def _all_tone_three(self, finals: List[str]) -> bool:
|
def _all_tone_three(self, finals: list[str]) -> bool:
|
||||||
return all(x[-1] == "3" for x in finals)
|
return all(x[-1] == "3" for x in finals)
|
||||||
|
|
||||||
# merge "不" and the word behind it
|
# merge "不" and the word behind it
|
||||||
# if don't merge, "不" sometimes appears alone according to jieba, which may occur sandhi error
|
# if don't merge, "不" sometimes appears alone according to jieba, which may occur sandhi error
|
||||||
def _merge_bu(self, seg: List[Tuple[str, str]]) -> List[Tuple[str, str]]:
|
def _merge_bu(self, seg: list[tuple[str, str]]) -> list[tuple[str, str]]:
|
||||||
new_seg = []
|
new_seg = []
|
||||||
last_word = ""
|
last_word = ""
|
||||||
for word, pos in seg:
|
for word, pos in seg:
|
||||||
@@ -636,7 +634,7 @@ class ToneSandhi:
|
|||||||
# e.g.
|
# e.g.
|
||||||
# input seg: [('听', 'v'), ('一', 'm'), ('听', 'v')]
|
# input seg: [('听', 'v'), ('一', 'm'), ('听', 'v')]
|
||||||
# output seg: [['听一听', 'v']]
|
# output seg: [['听一听', 'v']]
|
||||||
def _merge_yi(self, seg: List[Tuple[str, str]]) -> List[Tuple[str, str]]:
|
def _merge_yi(self, seg: list[tuple[str, str]]) -> list[tuple[str, str]]:
|
||||||
new_seg = [] * len(seg)
|
new_seg = [] * len(seg)
|
||||||
# function 1
|
# function 1
|
||||||
i = 0
|
i = 0
|
||||||
@@ -674,8 +672,8 @@ class ToneSandhi:
|
|||||||
|
|
||||||
# the first and the second words are all_tone_three
|
# the first and the second words are all_tone_three
|
||||||
def _merge_continuous_three_tones(
|
def _merge_continuous_three_tones(
|
||||||
self, seg: List[Tuple[str, str]]
|
self, seg: list[tuple[str, str]]
|
||||||
) -> List[Tuple[str, str]]:
|
) -> list[tuple[str, str]]:
|
||||||
new_seg = []
|
new_seg = []
|
||||||
sub_finals_list = [
|
sub_finals_list = [
|
||||||
lazy_pinyin(word, neutral_tone_with_five=True, style=Style.FINALS_TONE3)
|
lazy_pinyin(word, neutral_tone_with_five=True, style=Style.FINALS_TONE3)
|
||||||
@@ -709,8 +707,8 @@ class ToneSandhi:
|
|||||||
|
|
||||||
# the last char of first word and the first char of second word is tone_three
|
# the last char of first word and the first char of second word is tone_three
|
||||||
def _merge_continuous_three_tones_2(
|
def _merge_continuous_three_tones_2(
|
||||||
self, seg: List[Tuple[str, str]]
|
self, seg: list[tuple[str, str]]
|
||||||
) -> List[Tuple[str, str]]:
|
) -> list[tuple[str, str]]:
|
||||||
new_seg = []
|
new_seg = []
|
||||||
sub_finals_list = [
|
sub_finals_list = [
|
||||||
lazy_pinyin(word, neutral_tone_with_five=True, style=Style.FINALS_TONE3)
|
lazy_pinyin(word, neutral_tone_with_five=True, style=Style.FINALS_TONE3)
|
||||||
@@ -738,7 +736,7 @@ class ToneSandhi:
|
|||||||
new_seg.append([word, pos])
|
new_seg.append([word, pos])
|
||||||
return new_seg
|
return new_seg
|
||||||
|
|
||||||
def _merge_er(self, seg: List[Tuple[str, str]]) -> List[Tuple[str, str]]:
|
def _merge_er(self, seg: list[tuple[str, str]]) -> list[tuple[str, str]]:
|
||||||
new_seg = []
|
new_seg = []
|
||||||
for i, (word, pos) in enumerate(seg):
|
for i, (word, pos) in enumerate(seg):
|
||||||
if i - 1 >= 0 and word == "儿" and seg[i - 1][0] != "#":
|
if i - 1 >= 0 and word == "儿" and seg[i - 1][0] != "#":
|
||||||
@@ -747,7 +745,7 @@ class ToneSandhi:
|
|||||||
new_seg.append([word, pos])
|
new_seg.append([word, pos])
|
||||||
return new_seg
|
return new_seg
|
||||||
|
|
||||||
def _merge_reduplication(self, seg: List[Tuple[str, str]]) -> List[Tuple[str, str]]:
|
def _merge_reduplication(self, seg: list[tuple[str, str]]) -> list[tuple[str, str]]:
|
||||||
new_seg = []
|
new_seg = []
|
||||||
for i, (word, pos) in enumerate(seg):
|
for i, (word, pos) in enumerate(seg):
|
||||||
if new_seg and word == new_seg[-1][0]:
|
if new_seg and word == new_seg[-1][0]:
|
||||||
@@ -756,7 +754,7 @@ class ToneSandhi:
|
|||||||
new_seg.append([word, pos])
|
new_seg.append([word, pos])
|
||||||
return new_seg
|
return new_seg
|
||||||
|
|
||||||
def pre_merge_for_modify(self, seg: List[Tuple[str, str]]) -> List[Tuple[str, str]]:
|
def pre_merge_for_modify(self, seg: list[tuple[str, str]]) -> list[tuple[str, str]]:
|
||||||
seg = self._merge_bu(seg)
|
seg = self._merge_bu(seg)
|
||||||
try:
|
try:
|
||||||
seg = self._merge_yi(seg)
|
seg = self._merge_yi(seg)
|
||||||
@@ -768,7 +766,7 @@ class ToneSandhi:
|
|||||||
seg = self._merge_er(seg)
|
seg = self._merge_er(seg)
|
||||||
return seg
|
return seg
|
||||||
|
|
||||||
def modified_tone(self, word: str, pos: str, finals: List[str]) -> List[str]:
|
def modified_tone(self, word: str, pos: str, finals: list[str]) -> list[str]:
|
||||||
finals = self._bu_sandhi(word, finals)
|
finals = self._bu_sandhi(word, finals)
|
||||||
finals = self._yi_sandhi(word, finals)
|
finals = self._yi_sandhi(word, finals)
|
||||||
finals = self._neural_sandhi(word, pos, finals)
|
finals = self._neural_sandhi(word, pos, finals)
|
||||||
|
|||||||
Reference in New Issue
Block a user