Not a member of Pastebin yet?
Sign Up,
it unlocks many cool features!
- import os
- import re
- import csv
- import string
- from bs4 import BeautifulSoup
- #full path to folder containing the html files
- mydir = "C:/downloads/Yuki/vg.tar/extracted"
- #REGEX of the thread subject
- regex_subject = re.compile("(?i)example videogames? general")
- #A post replying to two or more posts will be split according to the regex below with each reply consisting of a separate post
- regex_split_post = re.compile("^(?:(?:>{2,}[\d ]+)+(?:\n>\".*\")*\n+)+", re.MULTILINE)
- #REGEX to strip away backquotes such as >10000 and >"text like this"
- regex_clean_post = re.compile("(^(?:>{2,}[\d ]+)+)(?:\n>\".*\")*\n*", re.MULTILINE)
- with open('posts.csv', mode='w', newline='') as csv_file:
- csv_writer = csv.writer(csv_file, delimiter=',', quotechar='"', quoting=csv.QUOTE_MINIMAL)
- filelist = [ f for f in os.listdir(mydir) if f.endswith(".html") ]
- fcur = 0
- for f in filelist:
- fcur = fcur+1
- print(fcur, len(filelist), f)
- with open(os.path.join(mydir,f),encoding="utf8") as fc:
- soup = BeautifulSoup(fc.read(), 'html.parser')
- res_subj = soup.find(class_='subject')
- if not regex_subject.search(res_subj.text): # If thread subject is not a match loop again
- continue
- posts = soup.find_all('blockquote', class_='postMessage') # Find all posts
- for p in posts:
- p = p.get_text()
- p = regex_split_post.split(p) #separate p into a list of all "sub posts" found within the post
- for sp in p:
- sp = regex_clean_post.sub('', sp) #remove quote to post numbers and >"posts like this"
- sp = sp.strip() #remove whitespaces and newlines at the beginning and end
- sp = "".join(filter(lambda char: char in string.printable, sp)) #remove characters that are not digits, ascii_letters, punctuation, or whitespace
- if not sp: # Skip empty posts
- continue
- csv_writer.writerow([str(sp)])
Advertisement
Add Comment
Please, Sign In to add comment