-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathHandler.py
More file actions
96 lines (86 loc) · 3.25 KB
/
Copy pathHandler.py
File metadata and controls
96 lines (86 loc) · 3.25 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
import json
import datetime
import codecs
import pandas as pd
from tqdm import tqdm
from collections import defaultdict
tqdm.pandas()
# utility
def process_data_dict(data_dict: dict):
data_df = pd.DataFrame(data_dict)
data_df.drop(index=[0], inplace=True)
return data_df
def get_json(path_to_file):
with open(path_to_file, 'r') as openfile:
articles = json.load(openfile)
return articles
# generate structured information from aws transcribe transcripts
def process_aws_transcribe_output(filename: str):
data_dict = defaultdict(list)
print ("Filename: ", filename)
with codecs.open(filename+'.txt', 'w', 'utf-8') as w:
with codecs.open(filename, 'r', 'utf-8') as f:
data=json.loads(f.read())
labels = data['results']['speaker_labels']['segments']
speaker_start_times={}
for label in labels:
for item in label['items']:
speaker_start_times[item['start_time']] =item['speaker_label']
items = data['results']['items']
lines=[]
line=''
time=0
speaker='null'
i=0
for item in items:
i=i+1
content = item['alternatives'][0]['content']
if item.get('start_time'):
current_speaker=speaker_start_times[item['start_time']]
elif item['type'] == 'punctuation':
line = line+content
if current_speaker != speaker:
if speaker:
lines.append({'speaker':speaker, 'line':line, 'time':time})
line=content
speaker=current_speaker
time=item['start_time']
elif item['type'] != 'punctuation':
line = line + ' ' + content
lines.append({'speaker':speaker, 'line':line,'time':time})
for dicts in lines:
data_dict['speaker'].append(dicts['speaker'])
data_dict['line'].append(dicts['line'])
data_dict['time'].append(dicts['time'])
# save to txt
sorted_lines = sorted(lines,key=lambda k: float(k['time']))
for line_data in sorted_lines:
line='[' + str(datetime.timedelta(seconds=int(round(float(line_data['time']))))) + '] ' + line_data.get('speaker') + ': ' + line_data.get('line')
w.write(line + '\n\n')
return process_data_dict(data_dict)
# generate final normalized multi-speaker transcripts
def generate_normalized_output(data_df: pd.DataFrame):
unique_speakers = list(data_df['speaker'].unique())
person_text = "PERSON"
speaker_mappings = dict()
count = 0
for speaker in unique_speakers:
speaker_mappings[speaker] = f"{person_text}{count}"
count += 1
for key in speaker_mappings.keys():
data_df['speaker'].replace(to_replace=[key], value = speaker_mappings[key], inplace=True)
sentences = data_df.progress_apply(
lambda row: f"({row['speaker']}) {row['line']}",
axis=1
)
return sentences
def process_single(filename: str, code):
data_df = process_aws_transcribe_output(filename)
sentences = generate_normalized_output(data_df)
output_dir = "output//processed-transcripts"
save_location = output_dir + '//' + code + '.txt'
with open(save_location, 'w') as filehandle:
for sentence in sentences:
filehandle.write('%s\n\n' % sentence)
# if __name__ == "__main__":
# process_single(filename="output//raw-transcripts//asrOutput.json", code="asefasawdac")