add dictionary file process

This commit is contained in:
Wolfang Torres
2026-06-12 20:26:50 +08:00
parent 9b0d23b8ac
commit f9fc887d05
6 changed files with 194 additions and 111 deletions

View File

@@ -1,84 +1,101 @@
"""processor.py"""
# Standard Library
import csv
# Pip
import argostranslate.translate
import torchaudio
# Local
from .constants import LANGUAGES
from .utility import TTS, ProcessFile, TranslationResult # , CCCEDICT
from .utility import CCCEDICT, TTS, DictionaryResult, ProcessFile, TranslationResult
# Constants
FIELDNAMES = ["simplified", "traditional", "pinyin", "meaning"]
# Results Classes
def translator_process(
text_lines: list[str],
process_file: ProcessFile,
language_id: str,
text_lines: list[str], process_file: ProcessFile
) -> list[TranslationResult]:
"""Process for phases or sentence translation"""
results = []
for n, line in enumerate(text_lines):
line = line.strip()
audio_path = process_file.resources / f"N{n:03n}.wav"
audio_path = process_file.resources / f"N{n:03n}.wav"
if not audio_path.exists():
audio = TTS.MODEL.generate(f"{line}", language_id=LANGUAGES.CN)
torchaudio.save(audio_path, audio, TTS.MODEL.sr)
translated = argostranslate.translate.translate(line, LANGUAGES.CN, language_id)
results.append(TranslationResult(language_id, translated, line, audio_path))
translated = argostranslate.translate.translate(
line, LANGUAGES.CN, process_file.language_id
)
results.append(
TranslationResult(process_file.language_id, translated, line, audio_path)
)
return results
# def dictionary_process(dictionary, tts, in_file, resources):
# """Process dictionary files"""
# words_list = in_file.open(encoding="utf8").read().strip().split("\n")
# results = []
# try:
# with in_file.open("w", encoding="utf8") as input_file:
# for words in words_list:
# word = words.split()[0]
# pinyin = " ".join(words.split()[1:]) if len(words.split()) > 1 else None
# if v := dictionary.get(word):
# if len(v) > 1:
# print(f"\nWARNING: {word} has multiple meanings:")
# if pinyin and pinyin != "ERROR":
# ml = list(filter(lambda x: x.pinyin == pinyin, v))
# else:
# ml = v
# if len(ml) > 1:
# for n, w in enumerate(ml):
# print(f"{n+1} - {w}")
# for m in w.meanings:
# print(f"\t{m}")
# s = None
# while (
# not s
# or not s.isnumeric()
# or not (1 <= int(s) <= len(v))
# ):
# s = input(
# f"Please select the correct word [1-{len(v)}]: "
# )
# v = v[int(s) - 1]
# else:
# v = ml[0]
# else:
# v = v[0]
# audio_path = resources / f"{word}.wav"
# if not audio_path.exists():
# audio = tts.generate(f"{word}。", language_id="zh")
# torchaudio.save(audio_path, audio, tts.sr)
# input_file.write(f"{word}\t{v.pinyin}\n")
# results.append((v, audio_path))
# else:
# print("============================================")
# print(f"===================>ERROR: {word} not found")
# print("============================================")
# input_file.write(f"{word}\tERROR\n")
# except Exception:
# with in_file.open("w", encoding="utf8") as input_file:
# input_file.write("\n".join(words_list))
# return results
def dictionary_pre_process(words_list: list[str], process_file: ProcessFile):
"""Pre Process dictionary files into a intermediary resources file"""
dictionary = CCCEDICT.create_cedict(process_file.language_id)
with process_file.dictionary_resource_file.open(
"w", encoding="utf8", newline=""
) as resource_file:
tsv_writer = csv.writer(
resource_file, dialect="excel-tab", fieldnames=FIELDNAMES
)
tsv_writer.writeheader()
for words in words_list:
word = words.split()[0]
pinyin = " ".join(words.split()[1:]) if len(words.split()) > 1 else None
if entries := dictionary.get(word):
if pinyin is not None:
entries = list(filter(lambda x: x.pinyin == pinyin, entries))
if len(entries) > 1:
print(f"\nWARNING: {word} has multiple meanings:")
for entry in entries:
for meaning in entry.meanings:
tsv_writer.writerow(
{
"simplified": entry.simplified,
"traditional": entry.traditional,
"pinyin": entry.pinyin,
"meaning": meaning,
}
)
else:
print("============================================")
print(f"===================>ERROR: {word} not found")
print("============================================")
tsv_writer.writerow(
{
"simplified": word,
"traditional": None,
"pinyin": None,
"meaning": None,
}
)
def dictionary_process(process_file: ProcessFile) -> list[DictionaryResult]:
"""Process a dictionary_resource_file into a final result"""
results = []
with process_file.dictionary_resource_file.open(
"w", encoding="utf8", newline=""
) as resource_file:
reader = csv.DictReader(resource_file)
for line in reader:
audio_path = process_file.resources / f"{line['pinyin']}.wav"
if not audio_path.exists():
audio = TTS.MODEL.generate(f"{line['simplified']}", language_id="zh")
torchaudio.save(audio_path, audio, TTS.MODEL.sr)
result = DictionaryResult(**line, audio_path=audio_path)
results.append(result)
return results
# def output_tsv(out_file, results):
# """writes the output as a tsv file"""