Changes
2 changed files (+33/-29)
-
data.py (new)
-
@@ -0,0 +1,33 @@from re import sub from html import unescape from argparse import ArgumentParser from psycopg2 import connect parser = ArgumentParser() parser.add_argument('-d', '--database', help='database connection string') parser.add_argument('-o', '--output', help='Output file', default='data') args = parser.parse_args() # Fetch messages from database since it's way faster than using the API conn = connect(args.database) cur = conn.cursor() cur.execute('SELECT * FROM statuses') statuses = cur.fetchall() # Use regex to remove HTML stuff text = [unescape(sub(r'<[^>]*>', ' ', status[2])) for status in statuses] # Extract all words from statuses words = [word for message in text for word in message.split()] # Remove URLs and special characters and convert to lowercase words = [sub(r'[^a-z0-9]', '', word.lower()) for word in words if word.find('://') == -1] # Remove empty strings words = [word for word in words if word != ''] with open(args.output, 'w') as f: for word in words: f.write(word + '\n')
-
-
db.py (deleted)
-
@@ -1,29 +0,0 @@#!/usr/bin/python3 import re import psycopg2 # Fetch messages from database since it's way faster than using the API conn = psycopg2.connect(dbname="mastodon_production") cur = conn.cursor() cur.execute('SELECT * FROM statuses') statuses = cur.fetchall() # Extract all words from statuses # Use regex to remove HTML stuff text = [re.sub(r'<[^>]*>', ' ', status[2]) for status in statuses] # print(text[0:100]) words = [word for message in text for word in message.split()] # Remove URLs and special characters and convert to lowercase words = [re.sub(r'[^a-z0-9]', '', word.lower()) for word in words if word.find('://') == -1] # Remove empty strings words = [word for word in words if word != ''] with open('/tmp/text', 'w') as f: for word in words: f.write(word + '\n')
-