Repository navigation
Expand file tree
/
Copy pathgrapher.py
More file actions
899 lines (778 loc) · 34.9 KB
/
Copy pathgrapher.py
File metadata and controls
899 lines (778 loc) · 34.9 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
# This Python file uses the following encoding: utf-8
# Copyright (C) 2017-present, Raphael Halff
#
# This program is free software: you can redistribute it and/or modify
# it under the terms of the GNU Affero General Public License as published
# by the Free Software Foundation, either version 3 of the License, or
# (at your option) any later version.
#
# This program is distributed in the hope that it will be useful,
# but WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
# GNU Affero General Public License for more details.
#
# This source code is licensed under the license found in the
# LICENSE file in the root directory of this source tree,
# but you can also find it here: <http://www.gnu.org/licenses/>.
#
# =====================================================================
# WHAT CHANGED, AND WHY
#
# The graphs are the same five Bokeh figures. What feeds them is different.
#
# 1. COUNTING WAS A SUBSTRING TEST. The old condition was
#
# if w.find(target) != -1
#
# which counts every word that CONTAINS the search word. Searching
# דער — one of the built-in defaults — also counted יעדער, קינדער,
# װידער and לידער, reporting 1,045 occurrences where the true count is
# 542. The default גוט was inflated 2.6x by גוטער, גוטע, גוטן.
#
# This affected selectors 0, 1, 3 and 4. Selector 2, the dispersion
# plot, already compared with == and was correct.
#
# Now: exact matches against the `token` index, so דער means דער.
#
# 2. THE WORD COUNT SKIPPED WORDS. w_count() did
#
# for t in tks:
# if re.search(pat, t) == None:
# tks.remove(t)
#
# Removing from a list while iterating it advances the iterator past the
# following element, so roughly half the non-Yiddish tokens survived the
# filter. That count was the denominator of every "normalized" graph.
#
# Now: SUM(poem_stat.tokens), counted once at index time.
#
# 3. NLTK TOKENISED YIDDISH WITH AN ENGLISH MODEL. nltk.word_tokenize has
# no notion that ס'איז is one word, and treats װאָס and וואָס — the
# ligature and the two-letter spelling of the same word — as unrelated.
# The index folds those together deliberately. See admin/tokenize.php.
#
# 4. LEGEND LABELS WERE REVERSED. rtl() returned text[::-1]. Reversing a
# Unicode string puts each combining vowel point BEFORE the letter it
# belongs to, so אַ came out as a bare א followed by a stray patakh.
# Browsers lay out Hebrew right-to-left on their own; Bokeh renders into
# the browser. The function is gone.
#
# 5. IT RE-READ AND RE-TOKENISED ALL 211 POEMS ON EVERY PAGE VIEW. Now it
# runs a few indexed aggregate queries.
#
# The index must exist. Build it with:
# php admin/reindex.php
# =====================================================================
# Hebrew appears in string literals throughout this file. Under Python 2
# those would be byte strings, and folding a byte string against unicode
# text fails or silently mismatches. Harmless under Python 3.
from __future__ import unicode_literals
import codecs
import os
import pickle
import sys
import unicodedata
import mysql.connector
from bokeh.embed import file_html
from bokeh.models import ColumnDataSource, HoverTool, FactorRange
from bokeh.palettes import Set2
from bokeh.plotting import figure
from bokeh.resources import INLINE
from bokeh.transform import dodge, jitter
from math import pi
MAX_TOKENS = 8 # Set2 has 8 colours; more series is unreadable anyway
# INLINE embeds the whole Bokeh JS bundle in every generated page, which is
# why a graph weighs about 1.3 MB. That is what the site does today, so it
# stays the default. Switching to CDN takes the same page to roughly 20 KB
# at the cost of a request to cdn.bokeh.org:
# from bokeh.resources import CDN
# RESOURCES = CDN
RESOURCES = INLINE
with open('/home/raphi/config/mysql.p', 'rb') as f:
CONFIG = pickle.load(f)
# ---------------------------------------------------------------------
# Normalisation. MUST match admin/tokenize.php::yiNormalize and
# corpus.php::corpusNormalize exactly — a search folded differently from
# the way the corpus was indexed silently finds nothing.
# ---------------------------------------------------------------------
LIGATURES = {
'װ': 'וו', # װ -> וו
'ױ': 'וי', # ױ -> וי
'ײ': 'יי', # ײ -> יי
"'": '׳',
'’': '׳',
'"': '״',
}
FINALS = {
'ך': 'כ', # ך -> כ
'ם': 'מ', # ם -> מ
'ן': 'נ', # ן -> נ
'ף': 'פ', # ף -> פ
'ץ': 'צ', # ץ -> צ
}
# Typographic gaps inside a word whose letters are set apart for effect.
# Must match admin/tokenize.php::YI_GAP. U+00A0 is excluded on purpose — it
# is used as an ordinary space and would fuse separate words.
GAPS = '\u202f\u2009\u200a'
def normalize(word):
"""Fold a word to its match key. Never displayed — see display_form()."""
word = ''.join(c for c in word if c not in GAPS)
for src, dst in LIGATURES.items():
word = word.replace(src, dst)
return ''.join(FINALS.get(c, c) for c in word)
def clean(word):
"""Keep only Yiddish letters, points and word-internal joiners."""
out = []
for c in word:
o = ord(c)
if 0x05D0 <= o <= 0x05EA or 0x05F0 <= o <= 0x05F2:
out.append(c)
elif unicodedata.combining(c):
out.append(c)
elif c in "־׳״'’-" or c in GAPS:
out.append(c)
return ''.join(out)[:64]
# ---------------------------------------------------------------------
# Queries
# ---------------------------------------------------------------------
class Corpus(object):
"""Thin wrapper over the token index. One connection per request."""
def __init__(self, genre='poem'):
self.cnx = mysql.connector.connect(**CONFIG)
self.genre = genre
def close(self):
try:
self.cnx.close()
except Exception:
pass
@staticmethod
def _text(v):
"""Decode what the driver hands back for a binary-collated column.
token.surface, token.norm and form.disp are utf8mb4_bin on purpose —
a case-insensitive collation would make א and אַ equal and undo the
whole point of the index. The cost is that MySQL reports those
columns as BINARY, so mysql-connector returns bytearray rather than
str. A bytearray reaching Bokeh is not a visible mistake: it fails at
JSON serialisation, the process exits non-zero, and graph.php shows a
404 with no clue why. Decode at the boundary, once.
"""
if isinstance(v, (bytearray, bytes)):
return v.decode('utf-8', 'replace')
return v
def _rows(self, sql, args=()):
cur = self.cnx.cursor()
cur.execute(sql, args)
rows = [tuple(self._text(c) for c in r) for r in cur.fetchall()]
cur.close()
return rows
def display_form(self, key):
"""The spelling a reader should see for a match key.
`norm` is a lookup string, not a word: because the key rewrites
final letters, the key for אין is אינ, which is not a legal
spelling. Labels come from `form`, never from `norm`.
Falls back to the commonest surface form in `token`, and finally to
the key itself. A graph with a slightly wrong legend is worth having;
a 404 because one lookup table has not been created yet is not. This
exact case took every graph on the site down.
"""
for sql in ("SELECT disp FROM `form` WHERE norm = %s",
"SELECT surface FROM token WHERE norm = %s "
"GROUP BY surface ORDER BY COUNT(*) DESC, surface LIMIT 1"):
try:
rows = self._rows(sql, (key,))
if rows:
return rows[0][0]
except Exception:
continue
return key
def totals_by_year(self):
"""Words and poems per year — the denominator for normalising."""
return self._rows(
"SELECT YEAR(p.date) yr, SUM(s.tokens), COUNT(*) "
" FROM poem p JOIN poem_stat s ON s.poem = p.poem "
" WHERE p.public IS TRUE AND p.genre = %s AND YEAR(p.date) > 0 "
" GROUP BY yr ORDER BY yr", (self.genre,))
def totals_by_poet(self):
return self._rows(
"SELECT p.poet, SUM(s.tokens), COUNT(*) "
" FROM poem p JOIN poem_stat s ON s.poem = p.poem "
" WHERE p.public IS TRUE AND p.genre = %s "
" GROUP BY p.poet ORDER BY p.poet", (self.genre,))
def counts_by_year(self, key):
return dict(self._rows(
"SELECT YEAR(p.date) yr, COUNT(*) "
" FROM token t JOIN poem p ON p.poem = t.poem "
" WHERE t.norm = %s AND p.public IS TRUE AND p.genre = %s "
" AND YEAR(p.date) > 0 "
" GROUP BY yr", (key, self.genre)))
def counts_by_poet(self, key):
return dict(self._rows(
"SELECT p.poet, COUNT(*) "
" FROM token t JOIN poem p ON p.poem = t.poem "
" WHERE t.norm = %s AND p.public IS TRUE AND p.genre = %s "
" GROUP BY p.poet", (key, self.genre)))
def poet_of(self, code):
rows = self._rows("SELECT poet FROM poem WHERE poem = %s", (code,))
return rows[0][0] if rows else None
def poems_of(self, poet):
"""Every public poem by one poet: code, title, year, word count."""
return self._rows(
"SELECT p.poem, p.title_y, YEAR(p.date), s.tokens "
" FROM poem p JOIN poem_stat s ON s.poem = p.poem "
" WHERE p.poet = %s AND p.public IS TRUE AND p.genre = %s "
" ORDER BY p.date, p.poem", (poet, self.genre))
def counts_in_poems(self, key, codes):
if not codes:
return {}
marks = ','.join(['%s'] * len(codes))
return dict(self._rows(
"SELECT poem, COUNT(*) FROM token "
" WHERE norm = %s AND poem IN (" + marks + ") GROUP BY poem",
tuple([key] + list(codes))))
def counts_by_line(self, key, code):
"""Occurrences per line — the shape of one poem."""
return dict(self._rows(
"SELECT line_no, COUNT(*) FROM token "
" WHERE norm = %s AND poem = %s GROUP BY line_no", (key, code)))
def poem_lines(self, code):
rows = self._rows(
"SELECT lines_n FROM poem_stat WHERE poem = %s", (code,))
return int(rows[0][0]) if rows else 0
def poem_title(self, code):
rows = self._rows("SELECT title_y FROM poem WHERE poem = %s", (code,))
return rows[0][0] if rows else code
def poem_words(self, code):
"""Every token of one poem in order: (pos, line_no, surface, norm).
Both the dispersion marks and their context tooltips come from this,
so the context is always the poem's own spelling.
"""
return self._rows(
"SELECT pos, line_no, surface, norm FROM token "
" WHERE poem = %s ORDER BY pos", (code,))
def poet_yiddish(self, code):
"""The poet's name in Yiddish, for the chart title.
The old version built this query by concatenating the poet's name
into the SQL string. The name comes from the database rather than
the URL, so it was not directly reachable, but a poet named with an
apostrophe would have broken the query outright. Bound instead.
"""
rows = self._rows(
"SELECT t.name_y FROM poet t JOIN poem p ON p.poet = t.name_e "
" WHERE p.poem = %s", (code,))
return rows[0][0] if rows else None
# ---------------------------------------------------------------------
# Shared chart furniture
# ---------------------------------------------------------------------
TOOLS = "pan,wheel_zoom,box_zoom,save,reset"
# Bokeh renamed things between 1.x and 3.x and this file has to run against
# whatever is installed on the server. Probe once rather than pin a version.
import bokeh as _bokeh
_BK_MAJOR = int(_bokeh.__version__.split('.')[0])
def _fig(**kw):
"""figure() with the height argument this Bokeh version understands."""
h = kw.pop('height', None)
if h is not None:
kw['height' if _BK_MAJOR >= 2 else 'plot_height'] = h
return figure(**kw)
def _legend(label):
"""legend_label= on Bokeh 2+, legend= on 1.x."""
return {'legend_label': label} if _BK_MAJOR >= 2 else {'legend': label}
def _dots(p, **kw):
"""Sized circle markers. circle(size=...) is deprecated from Bokeh 3.4."""
if _BK_MAJOR >= 3 and hasattr(p, 'scatter'):
return p.scatter(marker='circle', **kw)
return p.circle(**kw)
def _factors(names):
"""A categorical range this Bokeh understands.
Bokeh 2 and 3 want a FactorRange. Bokeh 0.12 — which is what is actually
installed here — takes a plain list of strings, and that is what the
original grapher.py passed. Building a FactorRange on 0.12 is not
reliably equivalent, so give each version the form it was written for.
"""
return FactorRange(*names) if _BK_MAJOR >= 2 else list(names)
def _bar_geometry(n_series):
"""Widths and offsets for n side-by-side bars in one categorical slot."""
width = 1.0 / (n_series + 1)
space = 0.2 * width
width -= space
return {
'width': width,
'step': width + space,
'start': -((width * (n_series - 1)) - (n_series * space)) / 1.0 + space / 2.0,
}
def _trim_ends(slots, found_by_key):
"""Drop the empty run at each end of a series, keep the interior.
For a time axis only. Interior zeros are evidence — a word absent for a
decade is something the chart should show — but years before the first
occurrence and after the last are just blank margin that squashes the
part anyone is looking at.
"""
hit = [any(int(f.get(s, 0)) for f in found_by_key) for s in slots]
if not any(hit):
return []
return slots[hit.index(True): len(hit) - hit[::-1].index(True)]
def _empty(message):
"""A figure that says why it is empty, rather than an empty figure."""
p = _fig(height=200, tools="", toolbar_location=None,
x_range=(0, 1), y_range=(0, 1))
p.axis.visible = False
p.grid.visible = False
p.text(x=[0.5], y=[0.5], text=[message], text_align='center',
text_baseline='middle', text_font_size='13px', text_color='#666')
return p
def _hover(pairs, mode=None):
return HoverTool(tooltips=list(pairs), **({'mode': mode} if mode else {}))
# ---------------------------------------------------------------------
# The five graphs
# ---------------------------------------------------------------------
def by_date(corpus, keys, labels, normalize_y):
"""func=0 — every poem, occurrences by year."""
totals = corpus.totals_by_year()
if not totals:
return _empty("No dated poems in the index.")
years = [int(r[0]) for r in totals]
words = {int(r[0]): int(r[1] or 0) for r in totals}
npoems = {int(r[0]): int(r[2]) for r in totals}
# A time series keeps its interior zeros: a year where a word is absent
# is a real observation, and dropping it would join 1939 straight to
# 1961 as though nothing lay between. Only the empty run at each END is
# trimmed, since that is margin rather than evidence.
found_by_key = [corpus.counts_by_year(k) for k in keys]
years = _trim_ends(years, found_by_key)
if not years:
return _empty("No dated poem uses " + ", ".join(labels) + ".")
p = _fig(height=280, tools=TOOLS, active_scroll="wheel_zoom",
title="Word frequency by year"
+ (" (per 1,000 words)" if normalize_y else " (raw count)"))
p.xaxis.axis_label = "Year"
p.yaxis.axis_label = "Per 1,000 words" if normalize_y else "Occurrences"
p.add_tools(_hover([("Year", "@x"), ("Poems", "@n_poems"),
("Words that year", "@wc{0,0}"),
("Occurrences", "@count"),
("Per 1,000 words", "@rate{0.00}")]))
for i, key in enumerate(keys):
found = found_by_key[i]
counts = [int(found.get(y, 0)) for y in years]
# Rate per 1,000 words, not per cent: at these frequencies a
# percentage is all leading zeros. 542 occurrences of דער in
# 32,557 words is 16.6 per 1,000, which reads; 1.66% does not.
rates = [(c / words[y] * 1000.0) if words.get(y) else 0.0
for c, y in zip(counts, years)]
src = ColumnDataSource(dict(
x=years, count=counts, rate=rates,
y=rates if normalize_y else counts,
n_poems=[npoems[y] for y in years],
wc=[words[y] for y in years]))
p.line('x', 'y', source=src, line_width=2,
line_color=Set2[8][i % 8], **_legend(labels[i]))
_dots(p, x='x', y='y', source=src, size=5, fill_color="white",
line_color=Set2[8][i % 8])
p.legend.location = "top_left"
p.legend.click_policy = "hide"
return p
def by_poets(corpus, keys, labels, normalize_y):
"""func=1 — every poem, occurrences by poet."""
totals = corpus.totals_by_poet()
if not totals:
return _empty("No poems in the index.")
names = [r[0] for r in totals]
words = {r[0]: int(r[1] or 0) for r in totals}
npoems = {r[0]: int(r[2]) for r in totals}
# Only poets who actually use one of the words. Every poet in the corpus
# used to get a slot, so a search for an uncommon word produced forty
# empty columns with two short bars lost among them. A category with
# nothing in it carries no information on a bar chart — unlike a gap in
# a time series, which says the word fell out of use.
found_by_key = [corpus.counts_by_poet(k) for k in keys]
hits = [n for n in names
if any(int(f.get(n, 0)) for f in found_by_key)]
hidden = len(names) - len(hits)
if not hits:
return _empty("No poet uses " + ", ".join(labels) + ".")
names = hits
p = _fig(x_range=_factors(names), height=380, tools=TOOLS,
active_scroll="wheel_zoom",
title="Word frequency by poet"
+ (" (per 1,000 words)" if normalize_y else " (raw count)")
+ (" — %d poet%s with no occurrences not shown"
% (hidden, '' if hidden == 1 else 's') if hidden else ""))
p.xaxis.axis_label = "Poet"
p.yaxis.axis_label = "Per 1,000 words" if normalize_y else "Occurrences"
p.xaxis.major_label_orientation = pi / 4
p.x_range.range_padding = 0.1
p.add_tools(_hover([("Poet", "@x"), ("Poems", "@n_poems"),
("Words", "@wc{0,0}"), ("Occurrences", "@count"),
("Per 1,000 words", "@rate{0.00}")]))
geo = _bar_geometry(len(keys))
offset = geo['start']
for i, key in enumerate(keys):
found = found_by_key[i]
counts = [int(found.get(n, 0)) for n in names]
rates = [(c / words[n] * 1000.0) if words.get(n) else 0.0
for c, n in zip(counts, names)]
src = ColumnDataSource(dict(
x=names, count=counts, rate=rates,
y=rates if normalize_y else counts,
n_poems=[npoems[n] for n in names],
wc=[words[n] for n in names]))
p.vbar(x=dodge('x', offset, range=p.x_range), top='y', source=src,
width=geo['width'], color=Set2[8][i % 8], **_legend(labels[i]))
offset += geo['step']
p.legend.location = "top_right"
p.legend.click_policy = "hide"
return p
def in_poem(corpus, code, keys, labels, normalize_y):
"""func=2 — dispersion: where each word falls through one poem.
Kept as the original dot plot rather than converted to bars. It is the
most informative of the five: it shows clustering and absence, not just
a total. The context tooltip now comes from the index, so it respects
word boundaries and shows the poem's own spelling.
"""
words = corpus.poem_words(code)
if not words:
return _empty("This poem is not in the index yet.")
total = len(words)
poet_y = corpus.poet_yiddish(code) or ''
title = corpus.poem_title(code)
xs, ys, cons, lns = [], [], [], []
for i, key in enumerate(keys):
for pos, line_no, surface, norm in words:
if norm != key:
continue
xs.append(pos)
ys.append(labels[i])
lns.append(line_no)
# Seven words of context, in the poem's own spelling.
lo, hi = max(0, pos - 3), min(total, pos + 4)
cons.append('… ' + ' '.join(w[2] for w in words[lo:hi]) + ' …')
if not xs:
return _empty("None of these words occur in this poem.")
p = _fig(height=340, y_range=_factors(labels), tools=TOOLS,
active_scroll="wheel_zoom",
title="Dispersion: " + title + (" פֿון " + poet_y if poet_y else ""))
p.xaxis.axis_label = "Word position (of %d)" % total
p.yaxis.axis_label = "Search word"
p.x_range.range_padding = 0
p.ygrid.grid_line_color = None
p.add_tools(_hover([("Word", "@y"), ("Position", "@x of %d" % total),
("Line", "@line"), ("Context", "@con")]))
src = ColumnDataSource(dict(x=xs, y=ys, con=cons, line=lns))
_dots(p, x='x', y=jitter('y', width=0.5, range=p.y_range), source=src,
alpha=0.65, size=7, fill_color=Set2[8][0], line_color=None)
return p
def by_poem_of_poet(corpus, code, keys, labels, normalize_y):
"""func=3 — one poet, occurrences per poem."""
poet = corpus.poet_of(code)
if not poet:
return _empty("Unknown poem.")
rows = corpus.poems_of(poet)
if not rows:
return _empty("No indexed poems for " + poet + ".")
codes = [r[0] for r in rows]
words = {r[0]: int(r[3] or 0) for r in rows}
years = {r[0]: (int(r[2]) if r[2] else None) for r in rows}
title_of = {r[0]: (r[1] or r[0]) for r in rows}
# Same as the poet chart: drop poems where none of the words occur. A
# prolific poet otherwise fills the axis with empty columns.
found_by_key = [corpus.counts_in_poems(k, codes) for k in keys]
hits = [c for c in codes
if any(int(f.get(c, 0)) for f in found_by_key)]
hidden = len(codes) - len(hits)
if not hits:
return _empty(poet + " does not use " + ", ".join(labels) + ".")
codes = hits
# Titles repeat across a poet's work; the x axis needs unique factors.
seen = {}
factors = []
for c in codes:
t = title_of[c]
seen[t] = seen.get(t, 0) + 1
factors.append(t if seen[t] == 1 else "%s (%d)" % (t, seen[t]))
p = _fig(x_range=_factors(factors), height=380, tools=TOOLS,
active_scroll="wheel_zoom",
title="Word frequency in the poems of " + poet
+ (" (per 1,000 words)" if normalize_y else " (raw count)")
+ (" — %d poem%s with no occurrences not shown"
% (hidden, '' if hidden == 1 else 's') if hidden else ""))
p.xaxis.axis_label = "Poem"
p.yaxis.axis_label = "Per 1,000 words" if normalize_y else "Occurrences"
p.xaxis.major_label_orientation = pi / 4
p.x_range.range_padding = 0.1
p.add_tools(_hover([("Poem", "@x"), ("Year", "@year"),
("Words", "@wc{0,0}"), ("Occurrences", "@count"),
("Per 1,000 words", "@rate{0.00}")]))
geo = _bar_geometry(len(keys))
offset = geo['start']
for i, key in enumerate(keys):
found = found_by_key[i]
counts = [int(found.get(c, 0)) for c in codes]
rates = [(n / words[c] * 1000.0) if words.get(c) else 0.0
for n, c in zip(counts, codes)]
src = ColumnDataSource(dict(
x=factors, count=counts, rate=rates,
y=rates if normalize_y else counts,
wc=[words[c] for c in codes],
year=[years[c] if years[c] else "n.d." for c in codes]))
p.vbar(x=dodge('x', offset, range=p.x_range), top='y', source=src,
width=geo['width'], color=Set2[8][i % 8], **_legend(labels[i]))
offset += geo['step']
p.legend.location = "top_right"
p.legend.click_policy = "hide"
return p
def by_date_of_poet(corpus, code, keys, labels, normalize_y):
"""func=4 — one poet, occurrences over time."""
poet = corpus.poet_of(code)
if not poet:
return _empty("Unknown poem.")
rows = [r for r in corpus.poems_of(poet) if r[2]]
if not rows:
return _empty("No dated poems by " + poet + ".")
by_year = {}
for pcode, _title, yr, tok in rows:
yr = int(yr)
slot = by_year.setdefault(yr, {'codes': [], 'words': 0, 'poems': 0})
slot['codes'].append(pcode)
slot['words'] += int(tok or 0)
slot['poems'] += 1
years = sorted(by_year)
# Counts per year, computed once. Same rule as the corpus-wide time
# series: keep interior zeros, trim the empty ends.
per_key = []
for key in keys:
acc = {}
for yr in years:
found = corpus.counts_in_poems(key, by_year[yr]['codes'])
acc[yr] = sum(int(v) for v in found.values())
per_key.append(acc)
years = _trim_ends(years, per_key)
if not years:
return _empty(poet + " does not use " + ", ".join(labels) + ".")
p = _fig(height=280, tools=TOOLS, active_scroll="wheel_zoom",
title="Word frequency over time: " + poet
+ (" (per 1,000 words)" if normalize_y else " (raw count)"))
p.xaxis.axis_label = "Year"
p.yaxis.axis_label = "Per 1,000 words" if normalize_y else "Occurrences"
p.add_tools(_hover([("Year", "@x"), ("Poems", "@n_poems"),
("Words", "@wc{0,0}"), ("Occurrences", "@count"),
("Per 1,000 words", "@rate{0.00}")]))
for i, key in enumerate(keys):
counts, rates = [], []
for yr in years:
slot = by_year[yr]
n = per_key[i][yr]
counts.append(n)
rates.append(n / slot['words'] * 1000.0 if slot['words'] else 0.0)
src = ColumnDataSource(dict(
x=years, count=counts, rate=rates,
y=rates if normalize_y else counts,
n_poems=[by_year[y]['poems'] for y in years],
wc=[by_year[y]['words'] for y in years]))
p.line('x', 'y', source=src, line_width=2,
line_color=Set2[8][i % 8], **_legend(labels[i]))
_dots(p, x='x', y='y', source=src, size=5, fill_color="white",
line_color=Set2[8][i % 8])
p.legend.location = "top_left"
p.legend.click_policy = "hide"
return p
# ---------------------------------------------------------------------
# Entry point. grapher.py OUTFILE FUNC [poem_code] TOKEN...
# ---------------------------------------------------------------------
DEFAULT_TOKENS = ['דער', 'וואָס'] # דער, װאָס
def selftest(genre='poem'):
"""grapher.py --selftest — say exactly what is missing.
graph.php can only ever answer "it worked" or "it did not". Run this by
hand to get the reason.
"""
ok = True
print("section : %s" % genre)
print("python : %s" % sys.version.split()[0])
try:
import bokeh
print("bokeh : %s" % bokeh.__version__)
except Exception as e:
print("bokeh : MISSING (%s)" % e); ok = False
try:
import jinja2
print("jinja2 : %s" % jinja2.__version__)
except Exception as e:
print("jinja2 : missing (%s) — template skipped, graphs still work" % e)
try:
import mysql.connector as _mc
print("mysql conn. : present")
except Exception as e:
print("mysql conn. : MISSING (%s)" % e); ok = False
cfgpath = '/home/raphi/config/mysql.p'
print("config : %s" % ("found" if os.path.isfile(cfgpath) else "MISSING " + cfgpath))
if not os.path.isfile(cfgpath):
return 1
tmpl = os.path.join(os.path.dirname(os.path.abspath(__file__)), 'graph.html')
print("graph.html : %s" % ("found" if os.path.isfile(tmpl) else "missing (optional)"))
try:
c = Corpus(genre)
except Exception as e:
print("database : CANNOT CONNECT (%s)" % e)
return 1
print("database : connected")
for tbl, need in (('token', True), ('poem_stat', True), ('form', False)):
try:
n = c._rows("SELECT COUNT(*) FROM `%s`" % tbl)[0][0]
print(" %-10s %s rows" % (tbl, n))
if not n and need:
print(" ^ empty — run: php admin/reindex.php"); ok = False
except Exception as e:
if need:
print(" %-10s MISSING — %s" % (tbl, e)); ok = False
else:
print(" %-10s missing (labels fall back to token)" % tbl)
try:
key = normalize('דער')
print("normalise : דער -> %s, label %s" % (key, c.display_form(key)))
n = c._rows("SELECT COUNT(*) FROM token WHERE norm = %s", (key,))[0][0]
print("lookup : %s occurrences of דער" % n)
if not n:
print(" ^ zero: the index does not match this folding"); ok = False
except Exception as e:
print("normalise : FAILED — %s" % e); ok = False
# Actually render. Everything above can pass while the render fails, and
# that is exactly what happened: the driver returns bytearray for the
# binary-collated columns, which only breaks at JSON serialisation. A
# check that stops short of the real work reports "all good" about a
# page that 404s. So build each figure and write the HTML.
print("\nrender")
codes = c._rows("SELECT poem FROM poem WHERE public IS TRUE "
"AND genre = %s LIMIT 1", (c.genre,))
code = codes[0][0] if codes else None
keys, labels = [], []
for w in DEFAULT_TOKENS:
k = normalize(w)
keys.append(k)
labels.append(c.display_form(k))
for lab in labels:
if not isinstance(lab, str):
print(" label is %s, not text — it will break Bokeh"
% type(lab).__name__)
ok = False
jobs = [('0 by date', lambda: by_date(c, keys, labels, False)),
('1 by poet', lambda: by_poets(c, keys, labels, False))]
if code:
jobs += [('2 dispersion', lambda: in_poem(c, code, keys, labels, False)),
('3 by poem', lambda: by_poem_of_poet(c, code, keys, labels, False)),
('4 poet dates', lambda: by_date_of_poet(c, code, keys, labels, False))]
for name, build in jobs:
try:
html = file_html(build(), RESOURCES, "test")
print(" %-14s %s bytes" % (name, len(html)))
except Exception as e:
print(" %-14s FAILED — %s: %s" % (name, type(e).__name__, e))
ok = False
c.close()
print("\n%s" % ("All good — graph.php should work." if ok
else "Fix the items marked above."))
return 0 if ok else 1
def main(argv):
if '--selftest' in argv:
g = 'poem'
if '--genre' in argv:
i = argv.index('--genre')
if i + 1 < len(argv) and argv[i + 1] in ('poem', 'story'):
g = argv[i + 1]
return selftest(g)
if len(argv) < 3:
sys.stderr.write("usage: grapher.py OUTFILE FUNC [ARGS...]\n")
sys.stderr.write(" grapher.py --selftest\n")
return 1
outfile = argv[1]
try:
func = int(argv[2])
except ValueError:
return 1
if func < 0 or func > 4:
return 1
# Which section to read: poetry, or the prose site at mer-vi.
#
# mer-vi used to keep its own copy of this file with genre='story'
# hardcoded. Sharing one copy means the caller has to say which section
# it wants — otherwise the prose site silently graphs poetry, which looks
# perfectly plausible and is simply the wrong corpus.
rest = argv[3:]
genre = 'poem'
if '--genre' in rest:
i = rest.index('--genre')
if i + 1 < len(rest) and rest[i + 1] in ('poem', 'story'):
genre = rest[i + 1]
rest = rest[:i] + rest[i + 2:]
# Selectors 2, 3 and 4 take the poem code as their first argument, the
# convention the old script and poem.php already use.
code = None
if func in (2, 3, 4):
if not rest:
return 1
code = rest[0]
rest = rest[1:]
words = []
for token in rest:
w = clean(token)
if w:
words.append(w)
if not words:
words = DEFAULT_TOKENS
words = words[:MAX_TOKENS]
corpus = Corpus(genre)
try:
# Deduplicate by match key: searching װאָס and וואָס is one series,
# not two identical ones. Label from `form`, never from the key.
keys, labels, seen = [], [], set()
for w in words:
k = normalize(w)
if k in seen:
continue
seen.add(k)
keys.append(k)
labels.append(corpus.display_form(k))
# As before, the count and the rate are shown one above the other
# rather than behind a toggle. They answer different questions and
# can disagree: a poet with few poems can have the highest rate and
# the lowest count, and seeing only one of those misleads.
if func == 0:
figs = [by_date(corpus, keys, labels, True),
by_date(corpus, keys, labels, False)]
elif func == 1:
figs = [by_poets(corpus, keys, labels, True),
by_poets(corpus, keys, labels, False)]
elif func == 2:
figs = [in_poem(corpus, code, keys, labels, False)]
elif func == 3:
figs = [by_poem_of_poet(corpus, code, keys, labels, True),
by_poem_of_poet(corpus, code, keys, labels, False)]
else:
figs = [by_date_of_poet(corpus, code, keys, labels, True),
by_date_of_poet(corpus, code, keys, labels, False)]
# graph.html carries the site header, stylesheet and analytics tag,
# so the plot page still looks like the rest of the site.
html = None
try:
import codecs
from jinja2 import Template
with codecs.open(os.path.join(os.path.dirname(
os.path.abspath(__file__)), 'graph.html'),
'r', encoding='utf8') as fh:
html = file_html(figs, RESOURCES, template=Template(fh.read()))
except Exception:
# A missing or incompatible template must not lose the graph.
html = file_html(figs, RESOURCES, "Word frequency")
finally:
corpus.close()
with codecs.open(outfile, 'w', encoding='utf8') as fh:
fh.write(html)
# graph.php treats any non-zero stdout as failure and serves vos.html.
print(0)
return 0
if __name__ == '__main__':
try:
sys.exit(main(sys.argv))
except Exception:
# graph.php treats any non-zero exit as "show vos.html". The
# traceback goes to the error log, never to the browser.
import traceback
traceback.print_exc(file=sys.stderr)
sys.exit(1)