如何基于Transformer实现英西翻译数据处理全流程
Transformer 英语-西班牙语翻译练习代码实现
以下是各练习模块的可直接运行实现,对应要求的7项功能:
1. 实现load_data()数据加载函数
功能:输入语料文件路径,返回英语目标语句、西班牙语输入语句两个列表。anki下载的平行语料每行格式为英语句子\t西班牙语句子\t来源备注,按制表符分割取前两列即可。
from typing import List, Tuple import pathlib def load_data(path: pathlib.Path) -> Tuple[List[str], List[str]]: target = [] # 存储英语目标语句 input_texts = [] # 存储西班牙语输入语句,避开内置函数input命名 with open(path, encoding='utf-8') as f: lines = f.read().split('\n') for line in lines: if not line: continue # 按制表符分割,丢弃第三列备注信息 eng, spa, _ = line.split('\t') target.append(eng) input_texts.append(spa) return target, input_texts
2. 下载数据集并调用加载函数
功能:给keras的get_file传入正确文件名,解压后定位语料路径,调用加载函数读取数据。
import random import tensorflow as tf from tensorflow import keras # 压缩包解压后的语料相对路径 file_name = "spa-eng/spa.txt" text_file = keras.utils.get_file( fname=file_name, origin="http://storage.googleapis.com/download.tensorflow.org/data/spa-eng.zip", extract=True) # 加载全量语料 target, input_texts = load_data(pathlib.Path(text_file))
3. 查看语料样例
直接运行代码随机抽取5对双语句对,验证加载结果是否正确:
for _ in range(5): index = random.choice(range(len(target))) print(f'#{index}\nENG: {target[index]}\nESP: {input_texts[index]}')
4. 实现数据集划分函数
功能:构造英西句对,打乱数据后按70%训练集、15%验证集、15%测试集的比例拆分。
def make_train_validation_test_sets(target: List[str], input_texts: List[str]) -> Tuple[List[Tuple[str,str]],List[Tuple[str,str]],List[Tuple[str,str]],List[Tuple[str,str]]]: # 拼接英西句对 all_pairs = list(zip(target, input_texts)) # 全局打乱 random.shuffle(all_pairs) total = len(all_pairs) # 计算拆分节点 train_end = int(total * 0.7) val_end = train_end + int(total * 0.15) # 切分数据集 train_pairs = all_pairs[:train_end] val_pairs = all_pairs[train_end:val_end] test_pairs = all_pairs[val_end:] return all_pairs, train_pairs, val_pairs, test_pairs all_pairs, train_pairs, val_pairs, test_pairs = make_train_validation_test_sets(target, input_texts) print(f"{len(all_pairs)} total pairs") print(f"{len(train_pairs)} training pairs") print(f"{len(val_pairs)} validation pairs") print(f"{len(test_pairs)} test pairs")
5. 实现西班牙语文本自定义标准化函数
功能:基于tensorflow_text实现Unicode NFKD归一化、转小写、清理特殊标点、添加序列起止标记。
import tensorflow_text as tf_text def custom_standardization(text: tf.Tensor) -> tf.Tensor: # NFKD格式归一化,分解西语重音字符 text = tf_text.normalize_utf8(text, 'NFKD') # 统一转小写 text = tf.strings.lower(text) # 保留西语字母、空格、常用标点,其余特殊字符替换为空格 text = tf.strings.regex_replace(text, r'[^ a-záéíóúüñ¿¡.!?,]', ' ') # 合并连续空格,去除首尾空格 text = tf.strings.regex_replace(text, r'\s+', ' ') text = tf.strings.strip(text) # 句子首尾添加起止标记 text = tf.strings.join(['[start]', text, '[end]'], separator=' ') return text # 功能测试样例 example_text = "¿Hola, cómo estás hoy? " example = tf.constant(example_text) print(example.numpy().decode()) print(custom_standardization(example).numpy().decode())
6. 构建文本向量化层
功能:分别为英语、西班牙语定义TextVectorization实例,西语层传入自定义标准化函数,基于训练集拟合词表。
from tensorflow.keras.layers import TextVectorization vocab_size = 15000 sequence_length = 20 # 英语向量化层,使用默认文本清洗规则 eng_vectorization = TextVectorization( max_tokens=vocab_size, output_mode='int', output_sequence_length=sequence_length ) # 西班牙语向量化层,使用自定义标准化函数 spa_vectorization = TextVectorization( max_tokens=vocab_size, output_mode='int', output_sequence_length=sequence_length, standardize=custom_standardization ) # 从训练集提取对应语种文本拟合词表 train_eng = [pair[0] for pair in train_pairs] train_spa = [pair[1] for pair in train_pairs] eng_vectorization.adapt(train_eng) spa_vectorization.adapt(train_spa) print(f'english vocabulary: {eng_vectorization.get_vocabulary()[:10]}') print(f'foreign vocabulary: {spa_vectorization.get_vocabulary()[:10]}')
7. 实现数据集格式化函数
功能:按照Transformer输入要求拆分序列:完整英语序列作为编码器输入,西班牙语序列去掉最后一位作为解码器输入,西班牙语序列去掉第一位作为预测标签,最终构造批处理、可并行加载的tf.data数据集。
def format_dataset(eng:str, spa:str): eng_vec = eng_vectorization(eng) spa_vec = spa_vectorization(spa) # 编码器输入:完整英语token序列 encoder_inputs = eng_vec # 解码器输入:西语序列去掉最后一位end标记(对应训练时techer forcing的右移输入) decoder_inputs = spa_vec[:, :-1] # 预测标签:西语序列去掉第一位start标记 target_label = spa_vec[:, 1:] return ({"encoder_inputs": encoder_inputs, "decoder_inputs": decoder_inputs,}, target_label) # 构造训练、验证数据流 BUFFER_SIZE = 2048 BATCH_SIZE = 16 def make_dataset(pairs: List[Tuple[str, str]]) -> tf.data.Dataset: eng_texts, spa_texts = zip(*pairs) eng_texts = list(eng_texts) spa_texts = list(spa_texts) dataset = tf.data.Dataset.from_tensor_slices((eng_texts, spa_texts)) dataset = dataset.batch(BATCH_SIZE) dataset = dataset.map(format_dataset, num_parallel_calls=tf.data.AUTOTUNE) return dataset.shuffle(BUFFER_SIZE).prefetch(tf.data.AUTOTUNE).cache() train_ds = make_dataset(train_pairs) val_ds = make_dataset(val_pairs)
运行说明
代码运行前需安装依赖:pip install tensorflow tensorflow-text pathlib,首次运行会自动下载数据集,本地生成缓存后不会重复下载。
内容的提问来源于stack exchange,提问作者Mohammed
相关产品推荐
相关产品推荐

