Prevent repeatedly loading the BERT model from the disk. (#37)
* Prevent repeatedly loading the BERT model from the disk. * [pre-commit.ci] auto fixes from pre-commit.com hooks for more information, see https://pre-commit.ci --------- Co-authored-by: pre-commit-ci[bot] <66853113+pre-commit-ci[bot]@users.noreply.github.com>
This commit is contained in:
@@ -4,6 +4,8 @@ from transformers import AutoTokenizer, AutoModelForMaskedLM
|
|||||||
|
|
||||||
tokenizer = AutoTokenizer.from_pretrained("./bert/chinese-roberta-wwm-ext-large")
|
tokenizer = AutoTokenizer.from_pretrained("./bert/chinese-roberta-wwm-ext-large")
|
||||||
|
|
||||||
|
models = dict()
|
||||||
|
|
||||||
|
|
||||||
def get_bert_feature(text, word2ph, device=None):
|
def get_bert_feature(text, word2ph, device=None):
|
||||||
if (
|
if (
|
||||||
@@ -14,14 +16,15 @@ def get_bert_feature(text, word2ph, device=None):
|
|||||||
device = "mps"
|
device = "mps"
|
||||||
if not device:
|
if not device:
|
||||||
device = "cuda"
|
device = "cuda"
|
||||||
model = AutoModelForMaskedLM.from_pretrained(
|
if device not in models.keys():
|
||||||
|
models[device] = AutoModelForMaskedLM.from_pretrained(
|
||||||
"./bert/chinese-roberta-wwm-ext-large"
|
"./bert/chinese-roberta-wwm-ext-large"
|
||||||
).to(device)
|
).to(device)
|
||||||
with torch.no_grad():
|
with torch.no_grad():
|
||||||
inputs = tokenizer(text, return_tensors="pt")
|
inputs = tokenizer(text, return_tensors="pt")
|
||||||
for i in inputs:
|
for i in inputs:
|
||||||
inputs[i] = inputs[i].to(device)
|
inputs[i] = inputs[i].to(device)
|
||||||
res = model(**inputs, output_hidden_states=True)
|
res = models[device](**inputs, output_hidden_states=True)
|
||||||
res = torch.cat(res["hidden_states"][-3:-2], -1)[0].cpu()
|
res = torch.cat(res["hidden_states"][-3:-2], -1)[0].cpu()
|
||||||
|
|
||||||
assert len(word2ph) == len(text) + 2
|
assert len(word2ph) == len(text) + 2
|
||||||
|
|||||||
@@ -4,6 +4,8 @@ import sys
|
|||||||
|
|
||||||
tokenizer = AutoTokenizer.from_pretrained("./bert/bert-base-japanese-v3")
|
tokenizer = AutoTokenizer.from_pretrained("./bert/bert-base-japanese-v3")
|
||||||
|
|
||||||
|
models = dict()
|
||||||
|
|
||||||
|
|
||||||
def get_bert_feature(text, word2ph, device=None):
|
def get_bert_feature(text, word2ph, device=None):
|
||||||
if (
|
if (
|
||||||
@@ -14,14 +16,15 @@ def get_bert_feature(text, word2ph, device=None):
|
|||||||
device = "mps"
|
device = "mps"
|
||||||
if not device:
|
if not device:
|
||||||
device = "cuda"
|
device = "cuda"
|
||||||
model = AutoModelForMaskedLM.from_pretrained("./bert/bert-base-japanese-v3").to(
|
if device not in models.keys():
|
||||||
device
|
models[device] = AutoModelForMaskedLM.from_pretrained(
|
||||||
)
|
"./bert/bert-base-japanese-v3"
|
||||||
|
).to(device)
|
||||||
with torch.no_grad():
|
with torch.no_grad():
|
||||||
inputs = tokenizer(text, return_tensors="pt")
|
inputs = tokenizer(text, return_tensors="pt")
|
||||||
for i in inputs:
|
for i in inputs:
|
||||||
inputs[i] = inputs[i].to(device)
|
inputs[i] = inputs[i].to(device)
|
||||||
res = model(**inputs, output_hidden_states=True)
|
res = models[device](**inputs, output_hidden_states=True)
|
||||||
res = torch.cat(res["hidden_states"][-3:-2], -1)[0].cpu()
|
res = torch.cat(res["hidden_states"][-3:-2], -1)[0].cpu()
|
||||||
assert inputs["input_ids"].shape[-1] == len(word2ph)
|
assert inputs["input_ids"].shape[-1] == len(word2ph)
|
||||||
word2phone = word2ph
|
word2phone = word2ph
|
||||||
|
|||||||
Reference in New Issue
Block a user