-
1
-
2
-
3
-
4
-
5
-
6
-
7
-
8
-
9
-
10
-
11
-
12
-
13
-
14
-
15
-
16
-
17
-
18
-
19
-
20
-
21
-
22
-
23
-
24
-
25
-
26
-
27
-
28
-
29
#!/usr/bin/python3
import re
import psycopg2
# Fetch messages from database since it's way faster than using the API
conn = psycopg2.connect(dbname="mastodon_production")
cur = conn.cursor()
cur.execute('SELECT * FROM statuses')
statuses = cur.fetchall()
# Extract all words from statuses
# Use regex to remove HTML stuff
text = [re.sub(r'<[^>]*>', ' ', status[2]) for status in statuses]
# print(text[0:100])
words = [word for message in text for word in message.split()]
# Remove URLs and special characters and convert to lowercase
words = [re.sub(r'[^a-z0-9]', '', word.lower()) for word in words if word.find('://') == -1]
# Remove empty strings
words = [word for word in words if word != '']
with open('/tmp/text', 'w') as f:
for word in words:
f.write(word + '\n')