Repository navigation
Expand file tree
/
Copy pathfix_mechanical.py
More file actions
210 lines (170 loc) · 6.44 KB
/
Copy pathfix_mechanical.py
File metadata and controls
210 lines (170 loc) · 6.44 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
"""Fix the deterministic defect classes found by audit_library.py.
Only touches things with a single correct answer:
- timestamp label text that disagrees with its own t= value
- href="=" from a bad substitution
- off-by-one <details> tags
Anything requiring judgement (fabricated timestamps, missing content) is left
for a rebuild or the review agents. Every file is backed up before editing.
"""
import io, json, os, re, shutil, sys
BACKUP = r'C:\Users\Bob\AppData\Local\Temp\mechanical_fix_backups'
def label_for(total_seconds):
t = int(total_seconds)
h, rem = divmod(t, 3600)
m, s = divmod(rem, 60)
if h:
return f'[{h}:{m:02d}:{s:02d}]'
return f'[{m}:{s:02d}]'
def fix_labels(s):
"""Rewrite each timestamp label to match its own t= value."""
n = [0]
def repl(m):
prefix, secs, label = m.group(1), int(m.group(2)), m.group(3)
correct = label_for(secs)
if f'[{label}]' != correct:
n[0] += 1
return f'{prefix}{correct}'
return m.group(0)
# ?t=123s ... >[1:23]
s = re.sub(r'((?:[?&])t=(\d+)s"[^>]*>)\[([\d:]+)\]', repl, s)
# #t=123 ... >[1:23]
s = re.sub(r'((?:#)t=(\d+)"[^>]*>)\[([\d:]+)\]', repl, s)
return s, n[0]
def fix_time_string_in_t(s):
"""Convert t=MM:SSs / t=H:MM:SSs into the seconds value YouTube expects.
Some summaries carry `&t=01:21s`, which YouTube cannot parse, so the link
silently fails to seek. The intended time is unambiguous, so it converts
cleanly to `&t=81s`.
"""
n = [0]
def repl(m):
parts = [int(x) for x in m.group(1).split(':')]
secs = parts[0] * 3600 + parts[1] * 60 + parts[2] if len(parts) == 3 else parts[0] * 60 + parts[1]
n[0] += 1
return f't={secs}s'
s = re.sub(r't=(\d+:\d{2}(?::\d{2})?)s', repl, s)
return s, n[0]
def fix_missing_t_param(s):
"""Repair anchors where t= was mangled into the target attribute.
Seen as: href="...watch?v=ID&target="yt-player">[35:54]
The visible label supplies the intended time, so the anchor can be rebuilt.
"""
n = [0]
def repl(m):
vid, label = m.group(1), m.group(2)
parts = [int(x) for x in label.split(':')]
secs = parts[0] * 3600 + parts[1] * 60 + parts[2] if len(parts) == 3 else parts[0] * 60 + parts[1]
n[0] += 1
return (f'<a href="https://www.youtube.com/watch?v={vid}&t={secs}s" '
f'target="yt-player">[{label}]')
s = re.sub(
r'<a href="https://www\.youtube\.com/watch\?v=([\w-]+)&target="yt-player">\[([\d:]+)\]',
repl, s,
)
return s, n[0]
def fix_stray_closing_tags(s):
"""Remove premature </body></html> appearing before the real end of file.
Caused by an edited-transcript paragraph being extracted through end-of-file
and carrying that page's closing tags into a summary dropdown.
"""
n = 0
while True:
i = s.find('</body>')
if i == -1 or i >= s.rfind('</body>'):
break
seg = s[i:i + 40]
cut = len('</body>')
rest = s[i + cut:]
m = re.match(r'\s*</html>', rest)
if m:
cut += m.end()
s = s[:i] + s[i + cut:]
n += 1
return s, n
def fix_mojibake(s):
"""Repair UTF-8 bytes that were decoded as latin-1 somewhere upstream.
Em dash, curly quotes, arrows and accented characters all arrive as
multi-character garbage ("â" for an em dash). The transformation has an
exact inverse, so each candidate run is re-encoded to bytes and decoded as
UTF-8. Anything that fails to round-trip is left untouched, so genuine text
that merely resembles the pattern cannot be corrupted.
"""
n = [0]
def repl(m):
try:
out = m.group(0).encode('latin-1').decode('utf-8')
except (UnicodeEncodeError, UnicodeDecodeError):
return m.group(0)
n[0] += 1
return out
# UTF-8 lead byte followed by its continuation bytes, as seen through latin-1
s = re.sub(r'[\xC2-\xF4][\x80-\xBF]{1,3}', repl, s)
return s, n[0]
def fix_href(s):
before = s
s = s.replace('href="="', 'href="')
s = s.replace('="="', '="')
return s, (0 if s == before else 1)
def fix_details(s):
o, c = s.count('<details>'), s.count('</details>')
if o == c:
return s, 0
if o > c:
add = o - c
marker = '</body>'
if marker in s:
s = s.replace(marker, '</details>' * add + '\n' + marker, 1)
else:
s = s + '</details>' * add
return s, add
# more closers than openers: drop the trailing strays
extra = c - o
for _ in range(extra):
i = s.rfind('</details>')
if i == -1:
break
s = s[:i] + s[i + len('</details>'):]
return s, extra
def main():
dry = '--dry-run' in sys.argv
report = json.load(io.open('audit_report.json', encoding='utf-8'))
os.makedirs(BACKUP, exist_ok=True)
touched = 0
totals = {'labels': 0, 'href': 0, 'details': 0, 'tstr': 0}
for entry in report:
if not entry['issues']:
continue
path = entry['path']
sfs = [f for f in os.listdir(path) if f.startswith('summary - ')]
if not sfs:
continue
sp = os.path.join(path, sfs[0])
orig = io.open(sp, encoding='utf-8').read()
# Convert time-string t= values BEFORE label normalization, so the
# labels are recomputed from corrected seconds.
s, n_tstr = fix_time_string_in_t(orig)
s, n_miss = fix_missing_t_param(s)
n_tstr += n_miss
s, n_lab = fix_labels(s)
s, n_href = fix_href(s)
s, n_det = fix_details(s)
if s == orig:
continue
touched += 1
totals['labels'] += n_lab
totals['tstr'] += n_tstr
totals['href'] += n_href
totals['details'] += n_det
print(f'{entry["title"][:46]:46} tstr={n_tstr} labels={n_lab} href={n_href} details={n_det}')
if not dry:
shutil.copy2(sp, os.path.join(BACKUP, f'{entry["video_id"]}.html'))
io.open(sp, 'w', encoding='utf-8').write(s)
print(f'\n{"DRY RUN - " if dry else ""}{touched} files changed')
print(f' t= time-strings : {totals["tstr"]}')
print(f' labels corrected : {totals["labels"]}')
print(f' hrefs fixed : {totals["href"]}')
print(f' details balanced : {totals["details"]}')
if not dry:
print(f' backups in : {BACKUP}')
if __name__ == '__main__':
main()