-
Notifications
You must be signed in to change notification settings - Fork 1
Expand file tree
/
Copy pathapp.py
More file actions
231 lines (206 loc) · 8.11 KB
/
Copy pathapp.py
File metadata and controls
231 lines (206 loc) · 8.11 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
from flask import Flask, render_template, url_for, request
from bs4 import BeautifulSoup
import requests
import os
from cmath import cos
import json
import requests
from bs4 import BeautifulSoup
import requests_random_user_agent
import os
from nltk.tokenize import RegexpTokenizer
from nltk.corpus import stopwords
from nltk.stem import WordNetLemmatizer
import nltk
nltk.download('omw-1.4')
from gensim import corpora
from gensim.models import TfidfModel
import matplotlib.pyplot as plt
from wordcloud import WordCloud
from collections import Counter
import pyLDAvis.gensim_models
import re
from sklearn.metrics.pairwise import cosine_similarity
import numpy as np
from textblob import TextBlob
import seaborn as sns
import pandas as pd
app = Flask(__name__)
@app.route("/")
def homepage():
"""View function for Home Page."""
return render_template("index.html")
@app.route("/10k-analysis", methods=['GET', 'POST'])
def analysisPage():
def mapTickerCIK():
path = "/Users/anish/Sentient/"
os.chdir(path)
"""
Map stock tickers to CIKs.
"""
f = open("company_tickers.json")
data = json.load(f)
tickerCIKs = {}
for i, l in data.items():
currTicker = l['ticker']
currCIK = l['cik_str']
tickerCIKs[currTicker] = currCIK
return tickerCIKs
def fetchDocuments(tickerCIKs, company):
"""
Fetch all the documents and store them into our local folder.
"""
path = "/Users/anish/Sentient/10Ks/"
os.chdir(path)
try:
os.mkdir(str(tickerCIKs[company]))
os.chdir(str(tickerCIKs[company]))
except OSError:
print(company + "'s CIK has been previously utilized")
return
s = requests.Session()
url = "https://www.sec.gov/cgi-bin/browse-edgar?action=getcompany&CIK=0000"+ str(tickerCIKs[company]) +"&type=10-K%25&dateb=&owner=exclude&start=0&count=40&output=atom"
xml = requests.get(url)
soup = BeautifulSoup(xml.content, 'html.parser')
tenKs = soup.find_all('filing-href')
# Fetch the documents and save into our local folder
documentNames = [] # the filenames of all the documents
counter = 0
for tenK in tenKs:
kurl = tenK.text
file = requests.get(kurl)
soup = BeautifulSoup(file.content, 'html.parser')
kurls = [i['href'] for i in soup.find_all('a', href=True)]
docURL = "https://www.sec.gov/" + kurls[9]
if("ex" not in docURL and "k" in docURL):
kfile = requests.get(docURL)
if("ix?" not in docURL):
soup = BeautifulSoup(kfile.content, 'html')
text = soup.get_text()
fileName = company + "_" + str(counter) + ".txt"
file = open(fileName, 'a')
file.write(text)
file.close()
documentNames.append(fileName)
counter += 1
return documentNames
def processData(documentNames):
"""
Data cleaning + preprocessing by tokenization, removing stopwords, and lemmatization.
"""
alphabet = ['a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't'
, 'u', 'v', 'w', 'x', 'y', 'z']
# Data cleaning + preprocessing
dataset = []
for documentName in documentNames:
filtered = []
text = open(documentName, "r").read()
text = BeautifulSoup(text, 'html.parser').get_text() # remove HTML tags
tokenizer = RegexpTokenizer(r'\w+')
text = tokenizer.tokenize(text) # Tokenization
for word in text:
word = word.lower()
if(word not in stopwords.words('english') and word not in alphabet and word.isalpha()): # remove stopwords
word = WordNetLemmatizer().lemmatize(word) # lemmatization
filtered.append(word)
dataset.append(filtered)
return dataset
def generateBoW(dataset):
"""
Generates the BoW of the current dataset.
"""
dict = corpora.Dictionary(dataset)
BoW_corpus = [dict.doc2bow(file, allow_update=True) for file in dataset]
# id_words = [[(dict[id], count) for id, count in line] for line in BoW_corpus]
return BoW_corpus
def wordCloud(dataset):
"""
Generates + displays the wordcloud of the latest report. Can be some fun feature.
"""
wordcloud = WordCloud(max_font_size=50, max_words=50, background_color="white", width=800, height=400).generate(" ".join(dataset[0]))
plt.figure( figsize=(20,10), facecolor='k' )
plt.imshow(wordcloud, interpolation='bilinear')
plt.axis("off")
plt.show()
def top20(dataset):
counter=Counter(dataset[0]) #first documents tokens from docs(which contains many tokens from different docs)
most=counter.most_common()
x, y=[], []
for word,count in most[:20]:
x.append(word)
y.append(count)
plt.figure(figsize=(16,6))
sns.barplot(x=y,y=x)
def CosSim(A, B):
"""
Calculate the cosine similarity between two sets A and B.
"""
vec_A = []
vec_B = []
rvector = list(A.union(B))
for w in rvector:
if w in A:
vec_A.append(1)
elif w not in A:
vec_A.append(0)
if w in B:
vec_B.append(1)
elif w not in B:
vec_B.append(0)
mul = 0
for i in range(len(rvector)):
mul += vec_A[i] * vec_B[i]
return mul / float((sum(vec_A) * sum(vec_B)) ** 0.5)
def JaccardSim(A, B):
"""
Calculate the Jaccard similarities between two sets A and B.
"""
return len(A.intersection(B)) / len(A.union(B))
def computeSim(dataset):
"""
Generate CosSim and JaccardSim for every two years, starting from the latest.
"""
l = 0 # 0 is the latest year
thisYearLastYear = {}
A = set(dataset[0])
for r in range(1, len(dataset)):
B = set(dataset[r])
cos_score = CosSim(A, B)
jaccard_score = JaccardSim(A, B)
scores = []
scores.append(cos_score)
scores.append(jaccard_score)
thisYearLastYear["(" + str(l) + "-" + str(r) + ")"] = scores
l += 1
return thisYearLastYear
def getPositivity(dataset):
"""
Get the positivity of the current report.
"""
positivity = TextBlob(" ".join(dataset[0])) # assuming for the latest 10K.
return positivity.sentiment
"""View function for About Page."""
if request.method == 'POST':
"""
The driver of the program
"""
company = request.form.get('javascript_data')
tickerCIKs = mapTickerCIK()
documentNames = fetchDocuments(tickerCIKs, company)
dataset = processData(documentNames)
thisYearLastYear = computeSim(dataset)
#print(thisYearLastYear)
central_df_dict = {'Pair': [], 'Cosine Similarity': [], 'Jaccard Similarity': []}
for pair, vals in thisYearLastYear.items():
central_df_dict['Pair'].append(pair)
central_df_dict['Cosine Similarity'].append(vals[0])
central_df_dict['Jaccard Similarity'].append(vals[1])
central_df = pd.DataFrame(central_df_dict)
central_df = central_df.set_index('Pair')
central_df = central_df.fillna(0)
central_df = pd.concat([central_df, central_df.pct_change()], axis=1, sort=False)
central_df.columns = ['CosSim', 'JaccSim', 'CosSim % Change', 'Jacc % Change']
print(central_df)
return render_template("10k.html")
if __name__ == "__main__":
app.run(debug=True)