-
Notifications
You must be signed in to change notification settings - Fork 1
/
extract_yt_testset.py
70 lines (58 loc) · 2.24 KB
/
extract_yt_testset.py
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
# Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved.
# SPDX-License-Identifier: CC-BY-4.0
import os
import argparse
"""Extract from raw file delivered from vendor to get:
test.src, test.tgt, and test.pe.tgt
* File format, sample from the En-Fr pair:
FILE: josh_miles_unboxing_video_0.yaml
S:Okay. [DO NOT EDIT]
T:D'accord.
PE:Bon.
S:This is my new toy. [DO NOT EDIT]
T:C'est mon nouveau jouet.
PE:C'est mon nouveau jouet.
"""
parser = argparse.ArgumentParser()
parser.add_argument('--input', required=True,
help='Input file for data delivered from vendor.')
parser.add_argument('--src-lang', default="en",
help="Source language id.")
parser.add_argument('--tgt-lang', required=True,
help="Target language id.")
parser.add_argument('--path', required=True,
help='Output path.')
if __name__ == '__main__':
args = parser.parse_args()
source, target, postedit = [], [], []
keys = ['S:', 'T:', 'PE:']
path_files = []
files_ = ["test." + args.src_lang, "test." + args.tgt_lang, "test.pe." + args.tgt_lang]
os.makedirs(args.path, exist_ok=True)
for file_ in files_:
path_files.append(os.path.join(args.path, file_))
print(path_files)
with open(args.input, 'r') as input_f:
# get src, tgt, pe samples
for line__ in input_f:
line_ = line__.strip()
if line_.startswith(keys[0]):
line_src = line_.split(keys[0])[1].split("[DO NOT EDIT]")[0].strip()
source.append(line_src)
elif line_.startswith(keys[1]):
line_tgt = line_.split(keys[1])[1]
target.append(line_tgt)
elif line_.startswith(keys[2]):
line_pe = line_.split(keys[2])[1]
postedit.append(line_pe)
else:
pass
# check multi-way parallel
assert len(source) == len(target) == len(postedit)
with open(path_files[0], 'w') as src_f, \
open(path_files[1], 'w') as tgt_f, \
open(path_files[2], 'w') as pe_f:
for src, tgt, pe in zip(source, target, postedit):
src_f.write(src + "\n")
tgt_f.write(tgt + "\n")
pe_f.write(pe + "\n")