-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathparseTranscript.js
More file actions
131 lines (110 loc) · 3.58 KB
/
Copy pathparseTranscript.js
File metadata and controls
131 lines (110 loc) · 3.58 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
/* AWS Transcript JSON format
*
*Object
* accountId:
* jobName:
* status:
* results:
* transcripts: []
* transcript:
* speaker_labels:
* speakers:
* segments: []
* start_time:
* speaker_label:
* end_time:
* items: []
* start_time:
* speaker_label:
* end_time:
* items: []
* alternatives:
* confidence:
* content:
* end_time:
* start_time:
* type:
* */
function parseTranscriptJson(json, speller) {
const speakers = json.results.speaker_labels.segments[Symbol.iterator]();
const words = json.results.items[Symbol.iterator]();
let transcript = { phrases: [] };
let word = words.next();
function isSpokenBy(speaker, word) {
const isPunctuation = word.type === 'punctuation';
const inSpeakerPeriod = parseFloat(word.start_time) < parseFloat(speaker.end_time);
return isPunctuation || inSpeakerPeriod;
}
for (let speaker of speakers) {
const phrase = {
speaker_label: speaker.speaker_label,
start_time: speaker.start_time,
end_time: speaker.end_time,
words: []
};
while (!word.done && isSpokenBy(speaker, word.value)) {
//Custom spelling correction on non-punctuation.
let word_content = word.value.alternatives[0].content;
if (word.value.type !== 'punctuation') {
word_content = speller(word_content);
}
phrase.words.push({
content: word.value.alternatives[0].content,
type: word.value.type
});
word = words.next();
}
//Don't keep phrases without words.
if (phrase.words.length > 0) {
last_phrase = transcript.phrases.pop();
//Add word list to previous phrase if the speakers are the same.
if (last_phrase) {
if (last_phrase.speaker_label === phrase.speaker_label) {
phrase.words = [...last_phrase.words, ...phrase.words];
} else {
transcript.phrases.push(last_phrase);
}
}
transcript.phrases.push(phrase);
}
}
return transcript;
}
/* Transcript Object Format
*
*Object:
* phrases: []
* speaker_label:
* start_time:
* end_time:
* words: []
* content:
* type:
*/
//opportunity to double reduce.
function stringifyTranscriptObject(transcript) {
let text = ` `;
for (phrase of transcript.phrases) {
const timestamp = prettifyTime(parseFloat(phrase.start_time));
const start_of_phrase = `[${timestamp}] ${phrase.speaker_label}:\n`
const segment = phrase.words.reduce((spiel, word) => {
const prefix = (word.type === 'punctuation') ? '' : ' ';
return spiel + prefix + word.content;
}, start_of_phrase);
text += `${segment}\n\n`;
}
return text;
}
function prettifyTime(total_seconds) {
function padTime(time) {
return (time < 10) ? `0${time}` : `${time}`;
}
const hours = padTime(Math.floor(total_seconds / 3600));
const minutes = padTime(Math.floor((total_seconds % 3600) / 60));
const seconds = padTime((total_seconds % 60).toFixed(2));
return `${hours}:${minutes}:${seconds}`;
}
module.exports = {
parseTranscriptJson: parseTranscriptJson,
stringifyTranscriptObject: stringifyTranscriptObject
}