-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathdostoyevsky_2grams.py
More file actions
52 lines (38 loc) · 1.51 KB
/
Copy pathdostoyevsky_2grams.py
File metadata and controls
52 lines (38 loc) · 1.51 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
from intros import (read_txt_file_to_list,
write_dict_to_csvfile)
import string
inputfile = "brothers_karamazov.txt"
search_term = 'suddenly'
def strip_punctuation(text):
# strip out all punctuation characters
exclude_characters = set(string.punctuation)
# some add additional characters we want to strip
exclude_characters = exclude_characters.union(set("“”"))
for character in exclude_characters:
text = text.replace(character, "")
return text
# read our text file into a list
lines = read_txt_file_to_list(inputfile)
# Join all the separate lines of text into one string
text_combined = ''.join(lines)
# make everything lowercase
text_combined = text_combined.lower()
# strip punctuation
text_combined = strip_punctuation(text_combined)
# Create text tokens - that is, a list of all the words in th text
text_tokens = text_combined.split()
# create an empty list
tokens_with_search_term_preceding = []
for i in range(0, len(text_tokens) - 2):
tokenpair = text_tokens[i:i + 2]
if search_term == tokenpair[0]:
tokens_with_search_term_preceding.append(tokenpair[1])
# find unique terms in our group of terms:
tokens_set = set(tokens_with_search_term_preceding)
token_counts_dict = {}
# count the number of occurences for each item in set
for token in tokens_set:
occurrences = tokens_with_search_term_preceding.count(token)
token_counts_dict[token] = occurrences
outputfile = "suddenly_2grams.csv"
write_dict_to_csvfile(token_counts_dict, outputfile)