delete 4 files not required (#208)

* Create all_process.py

* Create asr_transcript.py

* Update config.py

* Create extract_list.py

* Create clean_list.py

* Create custom.css

* Create compress_model.py

* Update all_process.py

* Update resample.py

* Update resample.py

* configs/config.json copy utils

* mirror: openi + token
bert models optimize

* text/__init__.py platform compatibility
compress_model.py output

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* default_config.yml fix 'config_path'
all_process.py fix 'config_path'

* default_config.yml fix 'config_path'
all_process.py fix 'config_path'

* [pre-commit.ci] auto fixes from pre-commit.com hooks

for more information, see https://pre-commit.ci

* 'config_path': config.json

* Delete all_process.py

* Delete asr_transcript.py

* Delete clean_list.py

* Delete extract_list.py

---------

Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
This commit is contained in:
spicysama
2023-11-30 20:08:51 +08:00
committed by GitHub
parent 92438187e4
commit d4a2bc225e
4 changed files with 0 additions and 1584 deletions

File diff suppressed because it is too large Load Diff

View File

@@ -1,102 +0,0 @@
import argparse
import concurrent.futures
import os
from loguru import logger
from modelscope.pipelines import pipeline
from modelscope.utils.constant import Tasks
from tqdm import tqdm
os.environ["MODELSCOPE_CACHE"] = "./"
def transcribe_worker(file_path: str, inference_pipeline, language):
"""
Worker function for transcribing a segment of an audio file.
"""
rec_result = inference_pipeline(audio_in=file_path)
text = str(rec_result.get("text", "")).strip()
text_without_spaces = text.replace(" ", "")
logger.info(file_path)
if language != "EN":
logger.info("text: " + text_without_spaces)
return text_without_spaces
else:
logger.info("text: " + text)
return text
def transcribe_folder_parallel(folder_path, language, max_workers=4):
"""
Transcribe all .wav files in the given folder using ThreadPoolExecutor.
"""
logger.critical(f"parallel transcribe: {folder_path}|{language}|{max_workers}")
if language == "JP":
workers = [
pipeline(
task=Tasks.auto_speech_recognition,
model="damo/speech_UniASR_asr_2pass-ja-16k-common-vocab93-tensorflow1-offline",
)
for _ in range(max_workers)
]
elif language == "ZH":
workers = [
pipeline(
task=Tasks.auto_speech_recognition,
model="damo/speech_paraformer-large-vad-punc_asr_nat-zh-cn-16k-common-vocab8404-pytorch",
model_revision="v1.2.4",
)
for _ in range(max_workers)
]
else:
workers = [
pipeline(
task=Tasks.auto_speech_recognition,
model="damo/speech_UniASR_asr_2pass-en-16k-common-vocab1080-tensorflow1-offline",
)
for _ in range(max_workers)
]
file_paths = []
langs = []
for root, _, files in os.walk(folder_path):
for file in files:
if file.lower().endswith(".wav"):
file_path = os.path.join(root, file)
lab_file_path = os.path.splitext(file_path)[0] + ".lab"
file_paths.append(file_path)
langs.append(language)
all_workers = (
workers * (len(file_paths) // max_workers)
+ workers[: len(file_paths) % max_workers]
)
with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
for i in tqdm(range(0, len(file_paths), max_workers), desc="转写进度: "):
l, r = i, min(i + max_workers, len(file_paths))
transcriptions = list(
executor.map(
transcribe_worker, file_paths[l:r], all_workers[l:r], langs[l:r]
)
)
for file_path, transcription in zip(file_paths[l:r], transcriptions):
if transcription:
lab_file_path = os.path.splitext(file_path)[0] + ".lab"
with open(lab_file_path, "w", encoding="utf-8") as lab_file:
lab_file.write(transcription)
logger.critical("已经将wav文件转写为同名的.lab文件")
if __name__ == "__main__":
parser = argparse.ArgumentParser()
parser.add_argument(
"-f", "--filepath", default="./raw/lzy_zh", help="path of your model"
)
parser.add_argument("-l", "--language", default="ZH", help="language")
parser.add_argument("-w", "--workers", default="1", help="trans workers")
args = parser.parse_args()
transcribe_folder_parallel(args.filepath, args.language, int(args.workers))
print("转写结束!")

View File

@@ -1,48 +0,0 @@
import argparse
import shutil
from tempfile import NamedTemporaryFile
from loguru import logger
def remove_chars_from_file(chars_to_remove, input_file, output_file):
rm_cnt = 0
with open(input_file, "r", encoding="utf-8") as f_in, NamedTemporaryFile(
"w", delete=False, encoding="utf-8"
) as f_tmp:
for line in f_in:
if any(char in line for char in chars_to_remove):
logger.info(f"删除了这一行:\n {line.strip()}")
rm_cnt += 1
else:
f_tmp.write(line)
shutil.move(f_tmp.name, output_file)
logger.critical(f"总计移除了: {rm_cnt}")
if __name__ == "__main__":
parser = argparse.ArgumentParser(
description="Remove lines from a file containing specified characters."
)
parser.add_argument(
"-c",
"--chars",
type=str,
required=True,
help="String of characters. If a line contains any of these characters, it will be removed.",
)
parser.add_argument(
"-i", "--input", type=str, required=True, help="Path to the input file."
)
parser.add_argument(
"-o", "--output", type=str, required=True, help="Path to the output file."
)
args = parser.parse_args()
# Setting up basic logging configuration for loguru
logger.add("removed_lines.log", rotation="1 MB")
remove_chars_from_file(args.chars, args.input, args.output)

View File

@@ -1,53 +0,0 @@
import argparse
import os
from loguru import logger
def extract_list(folder_path, language, name, transcript_txt_file):
logger.info(f"extracting list: {folder_path}|{name}|{language}")
current_dir = os.getcwd()
relative_path = os.path.relpath(folder_path, current_dir)
print(relative_path)
os.makedirs(os.path.dirname(transcript_txt_file), exist_ok=True)
with open(transcript_txt_file, "w", encoding="utf-8") as f:
# 遍历 raw 文件夹下的所有子文件夹
for root, _, files in os.walk(relative_path):
for file in files:
if file.endswith(".lab"):
lab_file_path = os.path.join(root, file)
# 读取转写文本
with open(lab_file_path, "r", encoding="utf-8") as lab_file:
transcription = lab_file.read().strip()
if len(transcription) == 0:
continue
# 获取对应的 WAV 文件路径
# ./Data/宵宫/audios/raw
# ./Data/宵宫/audios/wavs
wav_file_path = os.path.splitext(lab_file_path)[0] + ".wav"
if os.path.isfile(wav_file_path):
wav_file_path = wav_file_path.replace("\\", "/").replace(
"/raw", "/wavs"
)
# 写入数据到总的转写文本文件
line = f"{wav_file_path}|{name}|{language}|{transcription}\n"
f.write(line)
else:
print("not exists!")
return f"转写文本 {transcript_txt_file} 生成完成"
if __name__ == "__main__":
parser = argparse.ArgumentParser()
parser.add_argument(
"-f",
"--filepath",
required=True,
help="path of your rawaudios, e.g. ./Data/xxx/audios/raw",
)
parser.add_argument("-l", "--language", default="ZH", help="language")
parser.add_argument("-n", "--name", required=True, help="name of the character")
parser.add_argument("-o", "--outfile", required=True, help="outfile")
args = parser.parse_args()
status_str = extract_list(args.filepath, args.language, args.name, args.outfile)
logger.critical(status_str)