add thirdparty langsegment

This commit is contained in:
NZqian
2025-03-23 20:19:20 +08:00
parent 9be55c908e
commit fdaabdf72a
10 changed files with 1408 additions and 821 deletions
+1
View File
@@ -4,5 +4,6 @@ pretrained/
.vscode
.idea
.DS_Store
*.wav
ckpt
wandb
+1 -1
View File
@@ -6,7 +6,7 @@
from g2p.g2p import cleaners
from tokenizers import Tokenizer
from g2p.g2p.text_tokenizers import TextTokenizer
import LangSegment
from thirdparty.LangSegment import LangSegment
import json
import re
-816
View File
@@ -1,816 +0,0 @@
# Copyright (c) 2024 Amphion.
#
# This source code is licensed under the MIT license found in the
# LICENSE file in the root directory of this source tree.
import io, re, os, sys, time, argparse, pdb, json
from io import StringIO
from typing import Optional
import numpy as np
import traceback
import pyopenjtalk
from pykakasi import kakasi
punctuation = [",", ".", "!", "?", ":", ";", "'", ""]
jp_xphone2ipa = [
" a a",
" i i",
" u ɯ",
" e e",
" o o",
" a: aː",
" i: iː",
" u: ɯː",
" e: eː",
" o: oː",
" k k",
" s s",
" t t",
" n n",
" h ç",
" f ɸ",
" m m",
" y j",
" r ɾ",
" w ɰᵝ",
" N ɴ",
" g g",
" j d ʑ",
" z z",
" d d",
" b b",
" p p",
" q q",
" v v",
" : :",
" by b j",
" ch t ɕ",
" dy d e j",
" ty t e j",
" gy g j",
" gw g ɯ",
" hy ç j",
" ky k j",
" kw k ɯ",
" my m j",
" ny n j",
" py p j",
" ry ɾ j",
" sh ɕ",
" ts t s ɯ",
]
_mora_list_minimum: list[tuple[str, Optional[str], str]] = [
("ヴォ", "v", "o"),
("ヴェ", "v", "e"),
("ヴィ", "v", "i"),
("ヴァ", "v", "a"),
("", "v", "u"),
("", None, "N"),
("", "w", "a"),
("", "r", "o"),
("", "r", "e"),
("", "r", "u"),
("リョ", "ry", "o"),
("リュ", "ry", "u"),
("リャ", "ry", "a"),
("リェ", "ry", "e"),
("", "r", "i"),
("", "r", "a"),
("", "y", "o"),
("", "y", "u"),
("", "y", "a"),
("", "m", "o"),
("", "m", "e"),
("", "m", "u"),
("ミョ", "my", "o"),
("ミュ", "my", "u"),
("ミャ", "my", "a"),
("ミェ", "my", "e"),
("", "m", "i"),
("", "m", "a"),
("", "p", "o"),
("", "b", "o"),
("", "h", "o"),
("", "p", "e"),
("", "b", "e"),
("", "h", "e"),
("", "p", "u"),
("", "b", "u"),
("フォ", "f", "o"),
("フェ", "f", "e"),
("フィ", "f", "i"),
("ファ", "f", "a"),
("", "f", "u"),
("ピョ", "py", "o"),
("ピュ", "py", "u"),
("ピャ", "py", "a"),
("ピェ", "py", "e"),
("", "p", "i"),
("ビョ", "by", "o"),
("ビュ", "by", "u"),
("ビャ", "by", "a"),
("ビェ", "by", "e"),
("", "b", "i"),
("ヒョ", "hy", "o"),
("ヒュ", "hy", "u"),
("ヒャ", "hy", "a"),
("ヒェ", "hy", "e"),
("", "h", "i"),
("", "p", "a"),
("", "b", "a"),
("", "h", "a"),
("", "n", "o"),
("", "n", "e"),
("", "n", "u"),
("ニョ", "ny", "o"),
("ニュ", "ny", "u"),
("ニャ", "ny", "a"),
("ニェ", "ny", "e"),
("", "n", "i"),
("", "n", "a"),
("ドゥ", "d", "u"),
("", "d", "o"),
("トゥ", "t", "u"),
("", "t", "o"),
("デョ", "dy", "o"),
("デュ", "dy", "u"),
("デャ", "dy", "a"),
# ("デェ", "dy", "e"),
("ディ", "d", "i"),
("", "d", "e"),
("テョ", "ty", "o"),
("テュ", "ty", "u"),
("テャ", "ty", "a"),
("ティ", "t", "i"),
("", "t", "e"),
("ツォ", "ts", "o"),
("ツェ", "ts", "e"),
("ツィ", "ts", "i"),
("ツァ", "ts", "a"),
("", "ts", "u"),
("", None, "q"), # 「cl」から「q」に変更
("チョ", "ch", "o"),
("チュ", "ch", "u"),
("チャ", "ch", "a"),
("チェ", "ch", "e"),
("", "ch", "i"),
("", "d", "a"),
("", "t", "a"),
("", "z", "o"),
("", "s", "o"),
("", "z", "e"),
("", "s", "e"),
("ズィ", "z", "i"),
("", "z", "u"),
("スィ", "s", "i"),
("", "s", "u"),
("ジョ", "j", "o"),
("ジュ", "j", "u"),
("ジャ", "j", "a"),
("ジェ", "j", "e"),
("", "j", "i"),
("ショ", "sh", "o"),
("シュ", "sh", "u"),
("シャ", "sh", "a"),
("シェ", "sh", "e"),
("", "sh", "i"),
("", "z", "a"),
("", "s", "a"),
("", "g", "o"),
("", "k", "o"),
("", "g", "e"),
("", "k", "e"),
("グヮ", "gw", "a"),
("", "g", "u"),
("クヮ", "kw", "a"),
("", "k", "u"),
("ギョ", "gy", "o"),
("ギュ", "gy", "u"),
("ギャ", "gy", "a"),
("ギェ", "gy", "e"),
("", "g", "i"),
("キョ", "ky", "o"),
("キュ", "ky", "u"),
("キャ", "ky", "a"),
("キェ", "ky", "e"),
("", "k", "i"),
("", "g", "a"),
("", "k", "a"),
("", None, "o"),
("", None, "e"),
("ウォ", "w", "o"),
("ウェ", "w", "e"),
("ウィ", "w", "i"),
("", None, "u"),
("イェ", "y", "e"),
("", None, "i"),
("", None, "a"),
]
_mora_list_additional: list[tuple[str, Optional[str], str]] = [
("ヴョ", "by", "o"),
("ヴュ", "by", "u"),
("ヴャ", "by", "a"),
("", None, "o"),
("", None, "e"),
("", None, "i"),
("", "w", "a"),
("", "y", "o"),
("", "y", "u"),
("", "z", "u"),
("", "j", "i"),
("", "k", "e"),
("", "y", "a"),
("", None, "o"),
("", None, "e"),
("", None, "u"),
("", None, "i"),
("", None, "a"),
]
# 例: "vo" -> "ヴォ", "a" -> "ア"
mora_phonemes_to_mora_kata: dict[str, str] = {
(consonant or "") + vowel: kana for [kana, consonant, vowel] in _mora_list_minimum
}
# 例: "ヴォ" -> ("v", "o"), "ア" -> (None, "a")
mora_kata_to_mora_phonemes: dict[str, tuple[Optional[str], str]] = {
kana: (consonant, vowel)
for [kana, consonant, vowel] in _mora_list_minimum + _mora_list_additional
}
# 正規化で記号を変換するための辞書
rep_map = {
"": ":",
"": ";",
"": ",",
"": ".",
"": "!",
"": "?",
"\n": ".",
"": ".",
"": "",
"···": "",
"・・・": "",
"·": ",",
"": ",",
"": ",",
"": ",",
"$": ".",
# "“": "'",
# "”": "'",
# '"': "'",
"": "'",
"": "'",
# "": "'",
# "": "'",
# "(": "'",
# ")": "'",
# "《": "'",
# "》": "'",
# "【": "'",
# "】": "'",
# "[": "'",
# "]": "'",
# "——": "-",
# "": "-",
# "-": "-",
# "『": "'",
# "』": "'",
# "〈": "'",
# "〉": "'",
# "«": "'",
# "»": "'",
# # "": "-", # これは長音記号「ー」として扱うよう変更
# # "~": "-", # これは長音記号「ー」として扱うよう変更
# "「": "'",
# "」": "'",
}
def _numeric_feature_by_regex(regex, s):
match = re.search(regex, s)
if match is None:
return -50
return int(match.group(1))
def replace_punctuation(text: str) -> str:
"""句読点等を「.」「,」「!」「?」「'」「-」に正規化し、OpenJTalkで読みが取得できるもののみ残す:
漢字・平仮名・カタカナ、アルファベット、ギリシャ文字
"""
pattern = re.compile("|".join(re.escape(p) for p in rep_map.keys()))
# print("before: ", text)
# 句読点を辞書で置換
replaced_text = pattern.sub(lambda x: rep_map[x.group()], text)
replaced_text = re.sub(
# ↓ ひらがな、カタカナ、漢字
r"[^\u3040-\u309F\u30A0-\u30FF\u4E00-\u9FFF\u3400-\u4DBF\u3005"
# ↓ 半角アルファベット(大文字と小文字)
+ r"\u0041-\u005A\u0061-\u007A"
# ↓ 全角アルファベット(大文字と小文字)
+ r"\uFF21-\uFF3A\uFF41-\uFF5A"
# ↓ ギリシャ文字
+ r"\u0370-\u03FF\u1F00-\u1FFF"
# ↓ "!", "?", "…", ",", ".", "'", "-", 但し`…`はすでに`...`に変換されている
+ "".join(punctuation) + r"]+",
# 上述以外の文字を削除
"",
replaced_text,
)
# print("after: ", replaced_text)
return replaced_text
def fix_phone_tone(phone_tone_list: list[tuple[str, int]]) -> list[tuple[str, int]]:
"""
`phone_tone_list`のtone(アクセントの値)を0か1の範囲に修正する。
例: [(a, 0), (i, -1), (u, -1)] → [(a, 1), (i, 0), (u, 0)]
"""
tone_values = set(tone for _, tone in phone_tone_list)
if len(tone_values) == 1:
assert tone_values == {0}, tone_values
return phone_tone_list
elif len(tone_values) == 2:
if tone_values == {0, 1}:
return phone_tone_list
elif tone_values == {-1, 0}:
return [
(letter, 0 if tone == -1 else 1) for letter, tone in phone_tone_list
]
else:
raise ValueError(f"Unexpected tone values: {tone_values}")
else:
raise ValueError(f"Unexpected tone values: {tone_values}")
def fix_phone_tone_wplen(phone_tone_list, word_phone_length_list):
phones = []
tones = []
w_p_len = []
p_len = len(phone_tone_list)
idx = 0
w_idx = 0
while idx < p_len:
offset = 0
if phone_tone_list[idx] == "":
w_p_len.append(w_idx + 1)
curr_w_p_len = word_phone_length_list[w_idx]
for i in range(curr_w_p_len):
p, t = phone_tone_list[idx]
if p == ":" and len(phones) > 0:
if phones[-1][-1] != ":":
phones[-1] += ":"
offset -= 1
else:
phones.append(p)
tones.append(str(t))
idx += 1
if idx >= p_len:
break
w_p_len.append(curr_w_p_len + offset)
w_idx += 1
# print(w_p_len)
return phones, tones, w_p_len
def g2phone_tone_wo_punct(prosodies) -> list[tuple[str, int]]:
"""
テキストに対して、音素とアクセント(0か1)のペアのリストを返す。
ただし「!」「.」「?」等の非音素記号(punctuation)は全て消える(ポーズ記号も残さない)。
非音素記号を含める処理は`align_tones()`で行われる。
また「っ」は「cl」でなく「q」に変換される(「ん」は「N」のまま)。
例: "こんにちは、世界ー。。元気?!"
[('k', 0), ('o', 0), ('N', 1), ('n', 1), ('i', 1), ('ch', 1), ('i', 1), ('w', 1), ('a', 1), ('s', 1), ('e', 1), ('k', 0), ('a', 0), ('i', 0), ('i', 0), ('g', 1), ('e', 1), ('N', 0), ('k', 0), ('i', 0)]
"""
result: list[tuple[str, int]] = []
current_phrase: list[tuple[str, int]] = []
current_tone = 0
last_accent = ""
for i, letter in enumerate(prosodies):
# 特殊記号の処理
# 文頭記号、無視する
if letter == "^":
assert i == 0, "Unexpected ^"
# アクセント句の終わりに来る記号
elif letter in ("$", "?", "_", "#"):
# 保持しているフレーズを、アクセント数値を0-1に修正し結果に追加
result.extend(fix_phone_tone(current_phrase))
# 末尾に来る終了記号、無視(文中の疑問文は`_`になる)
if letter in ("$", "?"):
assert i == len(prosodies) - 1, f"Unexpected {letter}"
# あとは"_"(ポーズ)と"#"(アクセント句の境界)のみ
# これらは残さず、次のアクセント句に備える。
current_phrase = []
# 0を基準点にしてそこから上昇・下降する(負の場合は上の`fix_phone_tone`で直る)
current_tone = 0
last_accent = ""
# アクセント上昇記号
elif letter == "[":
if last_accent != letter:
current_tone = current_tone + 1
last_accent = letter
# アクセント下降記号
elif letter == "]":
if last_accent != letter:
current_tone = current_tone - 1
last_accent = letter
# それ以外は通常の音素
else:
if letter == "cl": # 「っ」の処理
letter = "q"
current_phrase.append((letter, current_tone))
return result
def handle_long(sep_phonemes: list[list[str]]) -> list[list[str]]:
for i in range(len(sep_phonemes)):
if sep_phonemes[i][0] == "":
# sep_phonemes[i][0] = sep_phonemes[i - 1][-1]
sep_phonemes[i][0] = ":"
if "" in sep_phonemes[i]:
for j in range(len(sep_phonemes[i])):
if sep_phonemes[i][j] == "":
# sep_phonemes[i][j] = sep_phonemes[i][j - 1][-1]
sep_phonemes[i][j] = ":"
return sep_phonemes
def handle_long_word(sep_phonemes: list[list[str]]) -> list[list[str]]:
res = []
for i in range(len(sep_phonemes)):
if sep_phonemes[i][0] == "":
sep_phonemes[i][0] = sep_phonemes[i - 1][-1]
# sep_phonemes[i][0] = ':'
if "" in sep_phonemes[i]:
for j in range(len(sep_phonemes[i])):
if sep_phonemes[i][j] == "":
sep_phonemes[i][j] = sep_phonemes[i][j - 1][-1]
# sep_phonemes[i][j] = ':'
res.append(sep_phonemes[i])
res.append("")
return res
def align_tones(
phones_with_punct: list[str], phone_tone_list: list[tuple[str, int]]
) -> list[tuple[str, int]]:
"""
例:
…私は、、そう思う。
phones_with_punct:
[".", ".", ".", "w", "a", "t", "a", "sh", "i", "w", "a", ",", ",", "s", "o", "o", "o", "m", "o", "u", "."]
phone_tone_list:
[("w", 0), ("a", 0), ("t", 1), ("a", 1), ("sh", 1), ("i", 1), ("w", 1), ("a", 1), ("s", 0), ("o", 0), ("o", 1), ("o", 1), ("m", 1), ("o", 1), ("u", 0))]
Return:
[(".", 0), (".", 0), (".", 0), ("w", 0), ("a", 0), ("t", 1), ("a", 1), ("sh", 1), ("i", 1), ("w", 1), ("a", 1), (",", 0), (",", 0), ("s", 0), ("o", 0), ("o", 1), ("o", 1), ("m", 1), ("o", 1), ("u", 0), (".", 0)]
"""
result: list[tuple[str, int]] = []
tone_index = 0
for phone in phones_with_punct:
if tone_index >= len(phone_tone_list):
# 余ったpunctuationがある場合 → (punctuation, 0)を追加
result.append((phone, 0))
elif phone == phone_tone_list[tone_index][0]:
# phone_tone_listの現在の音素と一致する場合 → toneをそこから取得、(phone, tone)を追加
result.append((phone, phone_tone_list[tone_index][1]))
# 探すindexを1つ進める
tone_index += 1
elif phone in punctuation or phone == "":
# phoneがpunctuationの場合 → (phone, 0)を追加
result.append((phone, 0))
else:
print(f"phones: {phones_with_punct}")
print(f"phone_tone_list: {phone_tone_list}")
print(f"result: {result}")
print(f"tone_index: {tone_index}")
print(f"phone: {phone}")
raise ValueError(f"Unexpected phone: {phone}")
return result
def kata2phoneme_list(text: str) -> list[str]:
"""
原則カタカナの`text`を受け取り、それをそのままいじらずに音素記号のリストに変換。
注意点:
- punctuationが来た場合(punctuationが1文字の場合がありうる)、処理せず1文字のリストを返す
- 冒頭に続く「ー」はそのまま「ー」のままにする(`handle_long()`で処理される)
- 文中の「ー」は前の音素記号の最後の音素記号に変換される。
例:
`ーーソーナノカーー` → ["", "", "s", "o", "o", "n", "a", "n", "o", "k", "a", "a", "a"]
`?` → ["?"]
"""
if text in punctuation:
return [text]
# `text`がカタカナ(`ー`含む)のみからなるかどうかをチェック
if re.fullmatch(r"[\u30A0-\u30FF]+", text) is None:
raise ValueError(f"Input must be katakana only: {text}")
sorted_keys = sorted(mora_kata_to_mora_phonemes.keys(), key=len, reverse=True)
pattern = "|".join(map(re.escape, sorted_keys))
def mora2phonemes(mora: str) -> str:
cosonant, vowel = mora_kata_to_mora_phonemes[mora]
if cosonant is None:
return f" {vowel}"
return f" {cosonant} {vowel}"
spaced_phonemes = re.sub(pattern, lambda m: mora2phonemes(m.group()), text)
# 長音記号「ー」の処理
long_pattern = r"(\w)(ー*)"
long_replacement = lambda m: m.group(1) + (" " + m.group(1)) * len(m.group(2))
spaced_phonemes = re.sub(long_pattern, long_replacement, spaced_phonemes)
# spaced_phonemes += ' ▁'
return spaced_phonemes.strip().split(" ")
def frontend2phoneme(labels, drop_unvoiced_vowels=False):
N = len(labels)
phones = []
for n in range(N):
lab_curr = labels[n]
# print(lab_curr)
# current phoneme
p3 = re.search(r"\-(.*?)\+", lab_curr).group(1)
# deal unvoiced vowels as normal vowels
if drop_unvoiced_vowels and p3 in "AEIOU":
p3 = p3.lower()
# deal with sil at the beginning and the end of text
if p3 == "sil":
# assert n == 0 or n == N - 1
# if n == 0:
# phones.append("^")
# elif n == N - 1:
# # check question form or not
# e3 = _numeric_feature_by_regex(r"!(\d+)_", lab_curr)
# if e3 == 0:
# phones.append("$")
# elif e3 == 1:
# phones.append("?")
continue
elif p3 == "pau":
phones.append("_")
continue
else:
phones.append(p3)
# accent type and position info (forward or backward)
a1 = _numeric_feature_by_regex(r"/A:([0-9\-]+)\+", lab_curr)
a2 = _numeric_feature_by_regex(r"\+(\d+)\+", lab_curr)
a3 = _numeric_feature_by_regex(r"\+(\d+)/", lab_curr)
# number of mora in accent phrase
f1 = _numeric_feature_by_regex(r"/F:(\d+)_", lab_curr)
a2_next = _numeric_feature_by_regex(r"\+(\d+)\+", labels[n + 1])
# accent phrase border
# print(p3, a1, a2, a3, f1, a2_next, lab_curr)
if a3 == 1 and a2_next == 1 and p3 in "aeiouAEIOUNcl":
phones.append("#")
# pitch falling
elif a1 == 0 and a2_next == a2 + 1 and a2 != f1:
phones.append("]")
# pitch rising
elif a2 == 1 and a2_next == 2:
phones.append("[")
# phones = ' '.join(phones)
return phones
class JapanesePhoneConverter(object):
def __init__(self, lexicon_path=None, ipa_dict_path=None):
# lexicon_lines = open(lexicon_path, 'r', encoding='utf-8').readlines()
# self.lexicon = {}
# self.single_dict = {}
# self.double_dict = {}
# for curr_line in lexicon_lines:
# k,v = curr_line.strip().split('+',1)
# self.lexicon[k] = v
# if len(k) == 2:
# self.double_dict[k] = v
# elif len(k) == 1:
# self.single_dict[k] = v
self.ipa_dict = {}
for curr_line in jp_xphone2ipa:
k, v = curr_line.strip().split(" ", 1)
self.ipa_dict[k] = re.sub("\s", "", v)
# kakasi1 = kakasi()
# kakasi1.setMode("H","K")
# kakasi1.setMode("J","K")
# kakasi1.setMode("r","Hepburn")
self.japan_JH2K = kakasi()
self.table = {ord(f): ord(t) for f, t in zip("67", "")}
def text2sep_kata(self, parsed) -> tuple[list[str], list[str]]:
"""
`text_normalize`で正規化済みの`norm_text`を受け取り、それを単語分割し、
分割された単語リストとその読み(カタカナor記号1文字)のリストのタプルを返す。
単語分割結果は、`g2p()`の`word2ph`で1文字あたりに割り振る音素記号の数を決めるために使う。
例:
`私はそう思う!って感じ?` →
["", "", "そう", "思う", "!", "って", "感じ", "?"], ["ワタシ", "", "ソー", "オモウ", "!", "ッテ", "カンジ", "?"]
"""
# parsed: OpenJTalkの解析結果
sep_text: list[str] = []
sep_kata: list[str] = []
fix_parsed = []
i = 0
while i <= len(parsed) - 1:
# word: 実際の単語の文字列
# yomi: その読み、但し無声化サインの`’`は除去
# print(parsed)
yomi = parsed[i]["pron"]
tmp_parsed = parsed[i]
if i != len(parsed) - 1 and parsed[i + 1]["string"] in [
"",
"",
"",
"",
"",
"",
]:
word = parsed[i]["string"] + parsed[i + 1]["string"]
i += 1
else:
word = parsed[i]["string"]
word, yomi = replace_punctuation(word), yomi.replace("", "")
"""
ここで`yomi`の取りうる値は以下の通りのはず。
- `word`が通常単語 → 通常の読み(カタカナ)
(カタカナからなり、長音記号も含みうる、`アー` 等)
- `word`が`ー` から始まる → `ーラー` や `ーーー` など
- `word`が句読点や空白等 → `、`
- `word`が`?` → ``(全角になる)
他にも`word`が読めないキリル文字アラビア文字等が来ると`、`になるが、正規化でこの場合は起きないはず。
また元のコードでは`yomi`が空白の場合の処理があったが、これは起きないはず。
処理すべきは`yomi`が`、`の場合のみのはず。
"""
assert yomi != "", f"Empty yomi: {word}"
if yomi == "":
# wordは正規化されているので、`.`, `,`, `!`, `'`, `-`のいずれか
if word not in (
".",
",",
"!",
"'",
"-",
"?",
":",
";",
"",
"",
):
# ここはpyopenjtalkが読めない文字等のときに起こる
#print(
# "{}Cannot read:{}, yomi:{}, new_word:{};".format(
# parsed, word, yomi, self.japan_JH2K.convert(word)[0]["kana"]
# )
#)
# raise ValueError(word)
word = self.japan_JH2K.convert(word)[0]["kana"]
# print(word, self.japan_JH2K.convert(word)[0]['kana'], kata2phoneme_list(self.japan_JH2K.convert(word)[0]['kana']))
tmp_parsed["pron"] = word
# yomi = "-"
# word = ','
# yomiは元の記号のままに変更
# else:
# parsed[i]['pron'] = parsed[i]["string"]
yomi = word
elif yomi == "":
assert word == "?", f"yomi `` comes from: {word}"
yomi = "?"
if word == "":
i += 1
continue
sep_text.append(word)
sep_kata.append(yomi)
# print(word, yomi, parts)
fix_parsed.append(tmp_parsed)
i += 1
# print(sep_text, sep_kata)
return sep_text, sep_kata, fix_parsed
def getSentencePhone(self, sentence, blank_mode=True, phoneme_mode=False):
# print("origin:", sentence)
words = []
words_phone_len = []
short_char_flag = False
output_duration_flag = []
output_before_sil_flag = []
normed_text = []
sentence = sentence.strip().strip("'")
sentence = re.sub(r"\s+", "", sentence)
output_res = []
failed_words = []
last_long_pause = 4
last_word = None
frontend_text = pyopenjtalk.run_frontend(sentence)
# print("frontend_text: ", frontend_text)
try:
frontend_text = pyopenjtalk.estimate_accent(frontend_text)
except:
pass
# print("estimate_accent: ", frontend_text)
# sep_text: 単語単位の単語のリスト
# sep_kata: 単語単位の単語のカタカナ読みのリスト
sep_text, sep_kata, frontend_text = self.text2sep_kata(frontend_text)
# print("sep_text: ", sep_text)
# print("sep_kata: ", sep_kata)
# print("frontend_text: ", frontend_text)
# sep_phonemes: 各単語ごとの音素のリストのリスト
sep_phonemes = handle_long_word([kata2phoneme_list(i) for i in sep_kata])
# print("sep_phonemes: ", sep_phonemes)
pron_text = [x["pron"].strip().replace("", "") for x in frontend_text]
# pdb.set_trace()
prosodys = pyopenjtalk.make_label(frontend_text)
prosodys = frontend2phoneme(prosodys, drop_unvoiced_vowels=True)
# print("prosodys: ", ' '.join(prosodys))
# print("pron_text: ", pron_text)
normed_text = [x["string"].strip() for x in frontend_text]
# punctuationがすべて消えた、音素とアクセントのタプルのリスト
phone_tone_list_wo_punct = g2phone_tone_wo_punct(prosodys)
# print("phone_tone_list_wo_punct: ", phone_tone_list_wo_punct)
# phone_w_punct: sep_phonemesを結合した、punctuationを元のまま保持した音素列
phone_w_punct: list[str] = []
w_p_len = []
for i in sep_phonemes:
phone_w_punct += i
w_p_len.append(len(i))
phone_w_punct = phone_w_punct[:-1]
# punctuation無しのアクセント情報を使って、punctuationを含めたアクセント情報を作る
# print("phone_w_punct: ", phone_w_punct)
# print("phone_tone_list_wo_punct: ", phone_tone_list_wo_punct)
phone_tone_list = align_tones(phone_w_punct, phone_tone_list_wo_punct)
jp_item = {}
jp_p = ""
jp_t = ""
# mye rye pye bye nye
# je she
# print(phone_tone_list)
for p, t in phone_tone_list:
if p in self.ipa_dict:
curr_p = self.ipa_dict[p]
jp_p += curr_p
jp_t += str(t + 6) * len(curr_p)
elif p in punctuation:
jp_p += p
jp_t += "0"
elif p == "":
jp_p += p
jp_t += " "
else:
print(p, t)
jp_p += "|"
jp_t += "0"
# return phones, tones, w_p_len
jp_p = jp_p.replace("", " ")
jp_t = jp_t.translate(self.table)
jp_l = ""
for t in jp_t:
if t == " ":
jp_l += " "
else:
jp_l += "2"
# print(jp_p)
# print(jp_t)
# print(jp_l)
# print(len(jp_p_len), sum(w_p_len), len(jp_p), sum(jp_p_len))
assert len(jp_p) == len(jp_t) and len(jp_p) == len(jp_l)
jp_item["jp_p"] = jp_p.replace("| |", "|").rstrip("|")
jp_item["jp_t"] = jp_t
jp_item["jp_l"] = jp_l
jp_item["jp_normed_text"] = " ".join(normed_text)
jp_item["jp_pron_text"] = " ".join(pron_text)
# jp_item['jp_ruoma'] = sep_phonemes
# print(len(normed_text), len(sep_phonemes))
# print(normed_text)
return jp_item
jpc = JapanesePhoneConverter()
def japanese_to_ipa(text, text_tokenizer):
# phonemes = text_tokenizer(text)
if type(text) == str:
return jpc.getSentencePhone(text)["jp_p"]
else:
result_ph = []
for t in text:
result_ph.append(jpc.getSentencePhone(t)["jp_p"])
return result_ph
+1 -1
View File
@@ -169,7 +169,7 @@ if __name__ == "__main__":
chunked=args.chunked,
)
e_t = time.time() - s_t
print(f"inference cost {e_t} seconds")
print(f"inference cost {e_t:.2f} seconds")
output_dir = args.output_dir
os.makedirs(output_dir, exist_ok=True)
+1 -3
View File
@@ -12,7 +12,6 @@ prefigure==0.0.10
bitsandbytes
muq==0.1.0
mutagen==1.47.0
pyopenjtalk==0.4.0
pykakasi==2.3.0
jieba==0.42.1
cn2an==0.5.23
@@ -20,5 +19,4 @@ pypinyin==0.53.0
onnxruntime
Unidecode==1.3.8
phonemizer==3.3.0
LangSegment==0.3.5
inflect==7.5.0
inflect==7.5.0
File diff suppressed because it is too large Load Diff
+9
View File
@@ -0,0 +1,9 @@
from .LangSegment import LangSegment,getTexts,classify,getCounts,printList,setfilters,getfilters,setPriorityThreshold,getPriorityThreshold,setEnablePreview,getEnablePreview,setKeepPinyin,getKeepPinyin,setLangMerge,getLangMerge
# release
__version__ = '0.3.5'
# develop
__develop__ = 'dev-0.0.1'
View File
+327
View File
@@ -0,0 +1,327 @@
# Copyright (c) 2021 PaddlePaddle Authors. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# Digital processing from GPT_SoVITS num.py thanks
"""
Rules to verbalize numbers into Chinese characters.
https://zh.wikipedia.org/wiki/中文数字#現代中文
"""
import re
from collections import OrderedDict
from typing import List
DIGITS = {str(i): tran for i, tran in enumerate('零一二三四五六七八九')}
UNITS = OrderedDict({
1: '',
2: '',
3: '',
4: '',
8: '亿',
})
COM_QUANTIFIERS = '(处|台|架|枚|趟|幅|平|方|堵|间|床|株|批|项|例|列|篇|栋|注|亩|封|艘|把|目|套|段|人|所|朵|匹|张|座|回|场|尾|条|个|首|阙|阵|网|炮|顶|丘|棵|只|支|袭|辆|挑|担|颗|壳|窠|曲|墙|群|腔|砣|座|客|贯|扎|捆|刀|令|打|手|罗|坡|山|岭|江|溪|钟|队|单|双|对|出|口|头|脚|板|跳|枝|件|贴|针|线|管|名|位|身|堂|课|本|页|家|户|层|丝|毫|厘|分|钱|两|斤|担|铢|石|钧|锱|忽|(千|毫|微)克|毫|厘|(公)分|分|寸|尺|丈|里|寻|常|铺|程|(千|分|厘|毫|微)米|米|撮|勺|合|升|斗|石|盘|碗|碟|叠|桶|笼|盆|盒|杯|钟|斛|锅|簋|篮|盘|桶|罐|瓶|壶|卮|盏|箩|箱|煲|啖|袋|钵|年|月|日|季|刻|时|周|天|秒|分|小时|旬|纪|岁|世|更|夜|春|夏|秋|冬|代|伏|辈|丸|泡|粒|颗|幢|堆|条|根|支|道|面|片|张|颗|块|元|(亿|千万|百万|万|千|百)|(亿|千万|百万|万|千|百|美|)元|(亿|千万|百万|万|千|百|十|)吨|(亿|千万|百万|万|千|百|)块|角|毛|分)'
# 分数表达式
RE_FRAC = re.compile(r'(-?)(\d+)/(\d+)')
def replace_frac(match) -> str:
"""
Args:
match (re.Match)
Returns:
str
"""
sign = match.group(1)
nominator = match.group(2)
denominator = match.group(3)
sign: str = "" if sign else ""
nominator: str = num2str(nominator)
denominator: str = num2str(denominator)
result = f"{sign}{denominator}分之{nominator}"
return result
# 百分数表达式
RE_PERCENTAGE = re.compile(r'(-?)(\d+(\.\d+)?)%')
def replace_percentage(match) -> str:
"""
Args:
match (re.Match)
Returns:
str
"""
sign = match.group(1)
percent = match.group(2)
sign: str = "" if sign else ""
percent: str = num2str(percent)
result = f"{sign}百分之{percent}"
return result
# 整数表达式
# 带负号的整数 -10
RE_INTEGER = re.compile(r'(-)' r'(\d+)')
def replace_negative_num(match) -> str:
"""
Args:
match (re.Match)
Returns:
str
"""
sign = match.group(1)
number = match.group(2)
sign: str = "" if sign else ""
number: str = num2str(number)
result = f"{sign}{number}"
return result
# 编号-无符号整形
# 00078
RE_DEFAULT_NUM = re.compile(r'\d{3}\d*')
def replace_default_num(match):
"""
Args:
match (re.Match)
Returns:
str
"""
number = match.group(0)
return verbalize_digit(number, alt_one=True)
# 加减乘除
# RE_ASMD = re.compile(
# r'((-?)((\d+)(\.\d+)?)|(\.(\d+)))([\+\-\×÷=])((-?)((\d+)(\.\d+)?)|(\.(\d+)))')
RE_ASMD = re.compile(
r'((-?)((\d+)(\.\d+)?[⁰¹²³⁴⁵⁶⁷⁸⁹ˣʸⁿ]*)|(\.\d+[⁰¹²³⁴⁵⁶⁷⁸⁹ˣʸⁿ]*)|([A-Za-z][⁰¹²³⁴⁵⁶⁷⁸⁹ˣʸⁿ]*))([\+\-\×÷=])((-?)((\d+)(\.\d+)?[⁰¹²³⁴⁵⁶⁷⁸⁹ˣʸⁿ]*)|(\.\d+[⁰¹²³⁴⁵⁶⁷⁸⁹ˣʸⁿ]*)|([A-Za-z][⁰¹²³⁴⁵⁶⁷⁸⁹ˣʸⁿ]*))')
asmd_map = {
'+': '',
'-': '',
'×': '',
'÷': '',
'=': '等于'
}
def replace_asmd(match) -> str:
"""
Args:
match (re.Match)
Returns:
str
"""
result = match.group(1) + asmd_map[match.group(8)] + match.group(9)
return result
# 次方专项
RE_POWER = re.compile(r'[⁰¹²³⁴⁵⁶⁷⁸⁹ˣʸⁿ]+')
power_map = {
'': '0',
'¹': '1',
'²': '2',
'³': '3',
'': '4',
'': '5',
'': '6',
'': '7',
'': '8',
'': '9',
'ˣ': 'x',
'ʸ': 'y',
'': 'n'
}
def replace_power(match) -> str:
"""
Args:
match (re.Match)
Returns:
str
"""
power_num = ""
for m in match.group(0):
power_num += power_map[m]
result = "" + power_num + "次方"
return result
# 数字表达式
# 纯小数
RE_DECIMAL_NUM = re.compile(r'(-?)((\d+)(\.\d+))' r'|(\.(\d+))')
# 正整数 + 量词
RE_POSITIVE_QUANTIFIERS = re.compile(r"(\d+)([多余几\+])?" + COM_QUANTIFIERS)
RE_NUMBER = re.compile(r'(-?)((\d+)(\.\d+)?)' r'|(\.(\d+))')
def replace_positive_quantifier(match) -> str:
"""
Args:
match (re.Match)
Returns:
str
"""
number = match.group(1)
match_2 = match.group(2)
if match_2 == "+":
match_2 = ""
match_2: str = match_2 if match_2 else ""
quantifiers: str = match.group(3)
number: str = num2str(number)
result = f"{number}{match_2}{quantifiers}"
return result
def replace_number(match) -> str:
"""
Args:
match (re.Match)
Returns:
str
"""
sign = match.group(1)
number = match.group(2)
pure_decimal = match.group(5)
if pure_decimal:
result = num2str(pure_decimal)
else:
sign: str = "" if sign else ""
number: str = num2str(number)
result = f"{sign}{number}"
return result
# 范围表达式
# match.group(1) and match.group(8) are copy from RE_NUMBER
RE_RANGE = re.compile(
r"""
(?<![\d\+\-\×÷=]) # 使用反向前瞻以确保数字范围之前没有其他数字和操作符
((-?)((\d+)(\.\d+)?)) # 匹配范围起始的负数或正数(整数或小数)
[-~] # 匹配范围分隔符
((-?)((\d+)(\.\d+)?)) # 匹配范围结束的负数或正数(整数或小数)
(?![\d\+\-\×÷=]) # 使用正向前瞻以确保数字范围之后没有其他数字和操作符
""", re.VERBOSE)
def replace_range(match) -> str:
"""
Args:
match (re.Match)
Returns:
str
"""
first, second = match.group(1), match.group(6)
first = RE_NUMBER.sub(replace_number, first)
second = RE_NUMBER.sub(replace_number, second)
result = f"{first}{second}"
return result
# ~至表达式
RE_TO_RANGE = re.compile(
r'((-?)((\d+)(\.\d+)?)|(\.(\d+)))(%|°C|℃|度|摄氏度|cm2|cm²|cm3|cm³|cm|db|ds|kg|km|m2|m²|m³|m3|ml|m|mm|s)[~]((-?)((\d+)(\.\d+)?)|(\.(\d+)))(%|°C|℃|度|摄氏度|cm2|cm²|cm3|cm³|cm|db|ds|kg|km|m2|m²|m³|m3|ml|m|mm|s)')
def replace_to_range(match) -> str:
"""
Args:
match (re.Match)
Returns:
str
"""
result = match.group(0).replace('~', '')
return result
def _get_value(value_string: str, use_zero: bool=True) -> List[str]:
stripped = value_string.lstrip('0')
if len(stripped) == 0:
return []
elif len(stripped) == 1:
if use_zero and len(stripped) < len(value_string):
return [DIGITS['0'], DIGITS[stripped]]
else:
return [DIGITS[stripped]]
else:
largest_unit = next(
power for power in reversed(UNITS.keys()) if power < len(stripped))
first_part = value_string[:-largest_unit]
second_part = value_string[-largest_unit:]
return _get_value(first_part) + [UNITS[largest_unit]] + _get_value(
second_part)
def verbalize_cardinal(value_string: str) -> str:
if not value_string:
return ''
# 000 -> '零' , 0 -> '零'
value_string = value_string.lstrip('0')
if len(value_string) == 0:
return DIGITS['0']
result_symbols = _get_value(value_string)
# verbalized number starting with '一十*' is abbreviated as `十*`
if len(result_symbols) >= 2 and result_symbols[0] == DIGITS[
'1'] and result_symbols[1] == UNITS[1]:
result_symbols = result_symbols[1:]
return ''.join(result_symbols)
def verbalize_digit(value_string: str, alt_one=False) -> str:
result_symbols = [DIGITS[digit] for digit in value_string]
result = ''.join(result_symbols)
if alt_one:
result = result.replace("", "")
return result
def num2str(value_string: str) -> str:
integer_decimal = value_string.split('.')
if len(integer_decimal) == 1:
integer = integer_decimal[0]
decimal = ''
elif len(integer_decimal) == 2:
integer, decimal = integer_decimal
else:
raise ValueError(
f"The value string: '${value_string}' has more than one point in it."
)
result = verbalize_cardinal(integer)
decimal = decimal.rstrip('0')
if decimal:
# '.22' is verbalized as '零点二二'
# '3.20' is verbalized as '三点二
result = result if result else ""
result += '' + verbalize_digit(decimal)
return result
if __name__ == "__main__":
text = ""
text = num2str(text)
print(text)
pass
Binary file not shown.

Before

Width:  |  Height:  |  Size: 136 KiB