-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathaudit_library.py
More file actions
181 lines (152 loc) · 6.14 KB
/
Copy pathaudit_library.py
File metadata and controls
181 lines (152 loc) · 6.14 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
"""Mechanical audit of every summary in the library.
Catches the deterministic defect classes so the LLM review agents can spend
their attention on judgment calls (fidelity, fabrication, quality) instead.
"""
import io, json, os, re, sys
import processor
def load_all():
base = processor.load_config().get('output_dir', './library')
out = []
for cat in sorted(os.listdir(base)):
cp = os.path.join(base, cat)
if not os.path.isdir(cp):
continue
for d in sorted(os.listdir(cp)):
p = os.path.join(cp, d)
mp = os.path.join(p, 'meta.json')
if not os.path.exists(mp):
continue
try:
meta = json.load(io.open(mp, encoding='utf-8'))
except Exception:
continue
out.append((cat, p, meta))
return out
def check(cat, path, meta):
issues = []
dur = meta.get('duration_seconds') or 0
sf = [f for f in os.listdir(path) if f.startswith('summary - ')]
tf = [f for f in os.listdir(path) if f.startswith('transcript - ')]
if not sf:
return ['NO_SUMMARY']
if not tf:
issues.append('NO_TRANSCRIPT')
s = io.open(os.path.join(path, sf[0]), encoding='utf-8').read()
# placeholder labels never replaced with a real time
if re.search(r'>\s*\[?timestamp\]?\s*<', s, re.I):
issues.append('PLACEHOLDER_LABEL')
# markdown fence leaked into the HTML
if '```' in s:
issues.append('MARKDOWN_FENCE')
# malformed href from a bad string substitution
if 'href="="' in s or '="="' in s:
issues.append('MALFORMED_HREF')
# trim-bug placeholder
if 'Trimmed input' in s:
issues.append('TRIM_CORRUPTION')
# closing tag missing its bracket, e.g. "</strong" -- breaks rendering of
# everything after it
malformed = re.findall(r'</(?:strong|em|li|ul|h2|a|p|details|summary)(?![a-z>])(?!>)', s)
if malformed:
issues.append(f'MALFORMED_TAG x{len(malformed)}')
# UTF-8 decoded as latin-1 somewhere upstream
moji = re.findall(r'[\xC2-\xF4][\x80-\xBF]{1,3}', s)
if moji:
issues.append(f'MOJIBAKE x{len(moji)}')
# '&t-123s' instead of '&t=123s' -- the link silently seeks to 0:00
broken_anchor = re.findall(r'&t-\d+s', s)
if broken_anchor:
issues.append(f'BROKEN_ANCHOR_DELIM x{len(broken_anchor)}')
# CJK characters injected mid-token by the model, seen inside CSS
# declarations and substituted for letters in English words
cjk = re.findall(r'[一-鿿-ヿ가-]', s)
if cjk:
issues.append(f'INJECTED_CJK x{len(cjk)}')
# label vs t= disagreement
bad = 0
for secs, label in re.findall(r't=(\d+)s"[^>]*>\[([\d:]+)\]', s):
parts = [int(x) for x in label.split(':')]
val = parts[0] * 3600 + parts[1] * 60 + parts[2] if len(parts) == 3 else parts[0] * 60 + parts[1]
if val != int(secs):
bad += 1
if bad:
issues.append(f'LABEL_MISMATCH x{bad}')
# YouTube videos use ?t=123s; local videos use /player/<id>#t=123
ts = [int(x) for x in re.findall(r'[?&]t=(\d+)s', s)] + \
[int(x) for x in re.findall(r'#t=(\d+)', s)]
if not ts:
issues.append('NO_TIMESTAMPS')
elif dur:
# timestamps past the end of the video are fabricated
over = [t for t in ts if t > dur + max(15, dur * 0.02)]
if over:
issues.append(f'TS_OVERRUN x{len(over)} (max {max(over)}s vs {dur}s)')
chapters = meta.get('chapters') or []
cov = max(ts) / dur
if len(chapters) >= 3:
last_ch = max(c['start'] for c in chapters)
if max(ts) < last_ch * 0.95:
issues.append(f'LOW_COVERAGE {cov*100:.0f}% (chapter-flow)')
elif dur >= 600 and cov < 0.85:
issues.append(f'LOW_COVERAGE {cov*100:.0f}%')
# sections should run forward in time
# Scope to WITHIN each heading. A greedy match reaches past </h2> into the
# #tN anchors inside dropdown blocks and reports nonsense ordering.
h2s = re.findall(r'<h2>.*?</h2>', s, re.S)
sec = []
unlinked = 0
for h in h2s:
m = re.search(r'[?&]t=(\d+)s', h) or re.search(r'#t=(\d+)', h)
if m:
sec.append(int(m.group(1)))
elif re.search(r'\[\d+:\d{2}(?::\d{2})?\]', h):
unlinked += 1
if unlinked:
issues.append(f'HEADING_TS_NOT_LINKED x{unlinked}')
if sec != sorted(sec):
issues.append('SECTIONS_OUT_OF_ORDER')
if s.count('<h2>') == 0:
issues.append('NO_SECTIONS')
# unbalanced tags suggest truncation
for tag in ('h2', 'ul', 'li', 'details'):
if s.count(f'<{tag}>') != s.count(f'</{tag}>'):
issues.append(f'UNBALANCED_{tag.upper()}')
if not s.rstrip().endswith('</html>') and '</body>' not in s:
issues.append('TRUNCATED_FILE')
return issues
def main():
rows = []
for cat, path, meta in load_all():
try:
iss = check(cat, path, meta)
except Exception as e:
iss = [f'CHECK_ERROR {type(e).__name__}: {e}']
rows.append((cat, path, meta, iss))
clean = [r for r in rows if not r[3]]
dirty = [r for r in rows if r[3]]
print(f'{len(rows)} videos audited: {len(clean)} clean, {len(dirty)} with issues\n')
for cat, path, meta, iss in dirty:
print(f'{meta.get("title","?")[:52]}')
print(f' {cat} | {meta.get("duration_display","?")} | {meta.get("video_id")}')
for i in iss:
print(f' - {i}')
counts = {}
for _, _, _, iss in dirty:
for i in iss:
k = i.split(' ')[0]
counts[k] = counts.get(k, 0) + 1
if counts:
print('\nISSUE TOTALS:')
for k, v in sorted(counts.items(), key=lambda x: -x[1]):
print(f' {v:>3} {k}')
if '--json' in sys.argv:
io.open('audit_report.json', 'w', encoding='utf-8').write(
json.dumps([
{'category': c, 'path': p, 'video_id': m.get('video_id'),
'title': m.get('title'), 'issues': i}
for c, p, m, i in rows
], indent=2)
)
print('\nwrote audit_report.json')
if __name__ == '__main__':
main()