Repository navigation
Expand file tree
/
Copy pathsolution.py
More file actions
86 lines (60 loc) · 2.45 KB
/
Copy pathsolution.py
File metadata and controls
86 lines (60 loc) · 2.45 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
import re
from collections import Counter, defaultdict
import math
def reg_tokenize(sent):
fin = []
for line in iter(sent.splitlines()):
fin.append(' '.join(re.findall(r'[\U00010000-\U0010ffff]'\
r'|[A-Z0-9a-z]+[A-Z0-9a-z._%+-]*@[A-Z0-9a-z]+(?:\.[A-Z0-9a-z]+)+'\
r'|(?<= )[$€£¥₹]?[0-9]+(?:[,.][0-9]+)*[$€£¥₹]?'\
r'|(?:(?:https?:\/\/(?:www.)?)|www.)[A-Z0-9a-z_-]+(?:\.[A-Z0-9a-z_\/-]+)+'\
r'|(?:(?<=[^A-Za-z0-9])|^)@[A-Z0-9a-z._+]+[A-Za-z0-9_]'\
r'|#[A-Za-z0-9]+(?:[\._-][A-Za-z0-9]+)*'\
r'|\.{3,}'\
r'|[!"#$%\&\'()*+,\-.:;<=>?@\[\\\/\]\^_`{\|}~]'\
r'|[A-Z]\.'\
r'|\w+', line)))
return fin
def ngrams(sentence, n):
tokens = reg_tokenize(sentence)
ngrams = zip(*[tokens[i:] for i in range(n)])
return [ngram for ngram in ngrams]
cn = defaultdict(lambda: 0)
def create_model(file="../corpus3.txt", n=3):
'''
Takes a file to train on, an 'n'
Returns predictive n-gram model
'''
global cn
# Create a placeholder for model
model = defaultdict(lambda: defaultdict(lambda: 0))
incv = defaultdict(lambda: defaultdict(lambda: 0))
invc = defaultdict(lambda: defaultdict(lambda: 0))
with open(file, 'r') as f:
linet = f.read()
# for line in f:
for i in range(1, n+1):
for ngram in ngrams(linet, n=i):
cn[i] += 1
wprec = tuple(ngram[:i-1])
waft = ngram[i-1]
model[wprec][waft] += 1
if i >= 2:
incv[waft][wprec] += 1
if incv[waft][wprec] == 1:
invc[waft][len(wprec)] += 1
f.close()
return model, incv, invc
model, incv, invc = create_model(n=3)
def perplexity(file='../corpus3.txt', n=2):
text = ''
with open(file, 'r') as f:
text = f.read()
summed = 0
seqs = ngrams(text, n=n)
for seq in seqs:
total_count = float(sum(model[seq[:-1]].values()))
prob = model[seq[:-1]][seq[-1]]/total_count
summed += math.log(prob, 2)
summed = (summed*-1)/len(seqs)
return pow(2, summed)