Not a member of Pastebin yet?
Sign Up,
it unlocks many cool features!
- """
- Author: [.2C52.69AD.]
- Key (auth= fng^k): 10101 01010 - then -> 01010
- :)
- just automates the manual labor of testing the JE-files that are given as .PDFs but are actually media
- like mp4 and avi files. Save for later investigation if desired.
- You can change headless to True down a little to not disp browser.
- """
- from playwright.sync_api import sync_playwright, TimeoutError as PlaywrightTimeoutError
- import time
- from urllib.parse import urlencode
- import random
- SEARCH_TERM = "no images produced"
- BASE_URL = "https://www.justice.gov"
- SEARCH_PATH = "/multimedia-search"
- START_PAGE = 1
- MAX_PAGES = 10
- SKIP_DATASETS = ['DataSet 9'] # skip media testing on these
- SKIP_COLLECTING_SKIP_DATASETS = True
- MIN_VALID_FILE = 2000 # bytes — filter out junk? probably just rm this + associated code.
- all_links = [] # collected PDF links
- valid_media = [] # discovered media URLs
- # JSON capture
- captured_json = None
- with sync_playwright() as p:
- browser = p.chromium.launch(headless=False, slow_mo=150)
- context = browser.new_context(
- viewport={"width": 620, "height": 400},
- user_agent="Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/132.0.0.0 Safari/537.36",
- locale="en-US",
- timezone_id="America/New_York",
- )
- page = context.new_page()
- print("Visiting main page to establish session / pass Akamai + gates")
- page.goto(BASE_URL + "/epstein", wait_until="networkidle", timeout=60000)
- # Robot challenge
- try:
- robot = page.wait_for_selector(
- 'input[type="button"][class="usa-button"][value="I am not a robot"][onclick*="reauth()"]',
- timeout=12000
- )
- robot.click()
- page.wait_for_load_state("networkidle", timeout=15000)
- print("Robot challenge passed")
- except PlaywrightTimeoutError:
- pass
- except Exception as e:
- print(f"Robot button failed: {e}")
- # Age gate
- try:
- page.wait_for_selector('text=/are you 18|age verification/i', timeout=12000)
- page.click('text=Yes', timeout=8000)
- page.wait_for_load_state("networkidle", timeout=25000)
- print("Age gate passed?")
- except PlaywrightTimeoutError:
- pass
- except Exception as e:
- print(f"Age gate failed: {e}")
- time.sleep(2)
- def on_response(response):
- global captured_json
- ct = response.headers.get("content-type", "").lower()
- if "application/json" in ct:
- try:
- data = response.json()
- if "hits" in data and "hits" in data["hits"]:
- captured_json = data
- print(f"Captured JSON — {len(data['hits']['hits'])} hits")
- except:
- pass
- page.on("response", on_response)
- # Collect links
- for pg in range(START_PAGE, START_PAGE + MAX_PAGES):
- captured_json = None
- params = {"keys": SEARCH_TERM, "page": pg}
- url = f"{BASE_URL}{SEARCH_PATH}?{urlencode(params)}"
- print(f"Loading page {pg}: {url}")
- page.goto(url, wait_until="networkidle", timeout=60000)
- time.sleep(0.2)
- if captured_json:
- for hit in captured_json["hits"]["hits"]:
- src = hit.get("_source", {})
- uri = src.get("ORIGIN_FILE_URI")
- if uri:
- if SKIP_COLLECTING_SKIP_DATASETS and not any(ds in uri for ds in SKIP_DATASETS):
- clean = uri.replace("\\/", "/")
- all_links.append(clean)
- print(" ->", clean)
- else:
- print(f"Ignoring skip-list item? -> {uri}")
- else:
- name = src.get("ORIGIN_FILE_NAME")
- if name:
- print(f"Warning: No URI for {name} — skipped")
- else:
- print("No JSON — snippet:", page.content()[:300])
- # Media variant testing
- print("\nTesting media variants...")
- EXTENSIONS = ['.mp4', '.avi', '.wav', '.mov']
- for pdf_url in all_links:
- if any(ds in pdf_url for ds in SKIP_DATASETS):
- continue
- base = pdf_url.rsplit('.', 1)[0]
- found = False
- for ext in EXTENSIONS:
- test_url = base + ext
- try:
- result = page.evaluate('''async (url) => {
- try {
- const res = await fetch(url, {method: 'HEAD', redirect: 'follow'});
- return {
- ok: res.ok,
- status: res.status,
- contentType: res.headers.get('content-type') || '',
- length: parseInt(res.headers.get('content-length') || '0', 10)
- };
- } catch (e) {
- return {ok: false};
- }
- }''', test_url)
- if result.get('ok') and result['status'] == 200:
- ct = result['contentType'].lower()
- length = result['length']
- if length > MIN_VALID_FILE and ('video/' in ct or 'audio/' in ct or 'octet-stream' in ct):
- print(f"✓ VALID: {test_url} ({ct}, {length:,} bytes)")
- valid_media.append(test_url)
- found = True
- break
- except:
- pass
- if not found:
- print(f"No media: {pdf_url}")
- time.sleep(0.8 + random.random() * 1.1)
- browser.close()
- # Save results
- if all_links:
- with open("epstein_no_images_links.txt", "w", encoding="utf-8") as f:
- for link in sorted(set(all_links)):
- f.write(link + "\n")
- print(f"\nSaved {len(all_links)} PDF links → epstein_no_images_links.txt")
- if valid_media:
- with open("epstein_valid_media.txt", "w", encoding="utf-8") as f:
- for url in sorted(set(valid_media)):
- f.write(url + "\n")
- print(f"Saved {len(valid_media)} media files → epstein_valid_media.txt")
- else:
- print("\nNo valid media found.")
- print("Done.")
Advertisement
Add Comment
Please, Sign In to add comment