Guest User

Untitled

a guest
Jul 9th, 2021
345
0
Never
Not a member of Pastebin yet? Sign Up, it unlocks many cool features!
Python 2.05 KB | None | 0 0
  1. import os
  2. import re
  3. import csv
  4. import string
  5. from bs4 import BeautifulSoup
  6.  
  7. #full path to folder containing the html files
  8. mydir = "C:/downloads/Yuki/vg.tar/extracted"
  9.  
  10. #REGEX of the thread subject
  11. regex_subject = re.compile("(?i)example videogames? general")
  12.  
  13. #A post replying to two or more posts will be split according to the regex below with each reply consisting of a separate post
  14. regex_split_post = re.compile("^(?:(?:>{2,}[\d ]+)+(?:\n>\".*\")*\n+)+", re.MULTILINE)
  15.  
  16. #REGEX to strip away backquotes such as >10000 and >"text like this"
  17. regex_clean_post = re.compile("(^(?:>{2,}[\d ]+)+)(?:\n>\".*\")*\n*", re.MULTILINE)
  18.  
  19. with open('posts.csv', mode='w', newline='') as csv_file:
  20.     csv_writer = csv.writer(csv_file, delimiter=',', quotechar='"', quoting=csv.QUOTE_MINIMAL)
  21.  
  22.     filelist = [ f for f in os.listdir(mydir) if f.endswith(".html") ]
  23.     fcur = 0
  24.     for f in filelist:
  25.         fcur = fcur+1
  26.         print(fcur, len(filelist),  f)
  27.         with open(os.path.join(mydir,f),encoding="utf8") as fc:
  28.             soup = BeautifulSoup(fc.read(), 'html.parser')
  29.             res_subj = soup.find(class_='subject')
  30.             if not regex_subject.search(res_subj.text): # If thread subject is not a match loop again
  31.                 continue
  32.             posts = soup.find_all('blockquote', class_='postMessage') # Find all posts
  33.             for p in posts:
  34.                 p = p.get_text()
  35.                 p = regex_split_post.split(p) #separate p into a list of all "sub posts" found within the post
  36.  
  37.                 for sp in p:
  38.                     sp = regex_clean_post.sub('', sp) #remove quote to post numbers and >"posts like this"
  39.                     sp = sp.strip() #remove whitespaces and newlines at the beginning and end
  40.                     sp = "".join(filter(lambda char: char in string.printable, sp)) #remove characters that are not digits, ascii_letters, punctuation, or whitespace
  41.                     if not sp: # Skip empty posts
  42.                         continue
  43.                     csv_writer.writerow([str(sp)])
  44.        
  45.  
Advertisement
Add Comment
Please, Sign In to add comment