370 lines
15 KiB
Python
Executable File
370 lines
15 KiB
Python
Executable File
import re
|
|
import cloudscraper # Changed from requests
|
|
from bs4 import BeautifulSoup
|
|
import os
|
|
import time
|
|
import json
|
|
from urllib.parse import urlparse, urljoin
|
|
|
|
class SimPCityExtractor:
|
|
def __init__(self, cookies=None):
|
|
# Use cloudscraper instead of requests.Session()
|
|
self.session = cloudscraper.create_scraper(
|
|
browser={
|
|
'browser': 'chrome',
|
|
'platform': 'windows',
|
|
'desktop': True
|
|
}
|
|
)
|
|
|
|
# Optional: Add any custom headers (cloudscraper handles most automatically)
|
|
self.session.headers.update({
|
|
'Accept-Language': 'en-US,en;q=0.9',
|
|
})
|
|
|
|
if cookies:
|
|
for name, value in cookies.items():
|
|
self.session.cookies.set(name, value, domain='.f95zone.to')
|
|
print("Cookies loaded for authenticated session.")
|
|
print(f"Total cookies loaded: {len(cookies)}")
|
|
|
|
# Test the session
|
|
self._test_session()
|
|
|
|
def _test_session(self):
|
|
"""Test if the session is properly authenticated"""
|
|
try:
|
|
print("\nTesting session authentication...")
|
|
test_url = "https://f95zone.to/"
|
|
response = self.session.get(test_url, timeout=30, allow_redirects=True)
|
|
|
|
print(f"Status Code: {response.status_code}")
|
|
print(f"Final URL: {response.url}")
|
|
|
|
if response.status_code == 403:
|
|
print("\n⚠ WARNING: Still receiving 403 Forbidden even with cloudscraper")
|
|
print("This suggests:")
|
|
print("1. Your cookies are definitely expired/invalid")
|
|
print("2. Your IP may be blocked")
|
|
print("3. The site requires CAPTCHA solving")
|
|
|
|
# Check for specific protection
|
|
if 'cloudflare' in response.text.lower() or 'cf-ray' in response.headers:
|
|
print("→ Cloudflare challenge page detected")
|
|
print(" Cloudscraper may need a delay or the challenge is too advanced")
|
|
|
|
if 'captcha' in response.text.lower():
|
|
print("→ CAPTCHA detected - cloudscraper cannot solve CAPTCHAs")
|
|
print(" You'll need to use Selenium with CAPTCHA solving service")
|
|
|
|
# Save response for debugging
|
|
with open('403_response_cloudscraper.html', 'w', encoding='utf-8') as f:
|
|
f.write(response.text)
|
|
print("→ Saved response to '403_response_cloudscraper.html'")
|
|
|
|
elif response.status_code == 200:
|
|
if 'login' in response.url.lower():
|
|
print("⚠ Redirected to login page - cookies are invalid or expired")
|
|
print(" Export fresh cookies from your browser")
|
|
else:
|
|
print("✓ Cloudscraper successfully bypassed protection!")
|
|
|
|
# Check if logged in
|
|
if 'logout' in response.text.lower() or 'account' in response.text.lower():
|
|
print("✓ Authentication confirmed - you're logged in")
|
|
else:
|
|
print("⚠ Not logged in - update your cookies")
|
|
|
|
except Exception as e:
|
|
print(f"✗ Error testing session: {e}")
|
|
|
|
def get_thread_pages(self, base_url, max_pages=50):
|
|
"""Discover all pages in a thread by scanning links containing '/page-'."""
|
|
pages = []
|
|
try:
|
|
print(f"\nDiscovering pages for thread: {base_url}")
|
|
|
|
response = self.session.get(base_url, timeout=30, allow_redirects=True)
|
|
|
|
print(f"Status: {response.status_code}, URL: {response.url}")
|
|
|
|
if response.status_code == 403:
|
|
print("\n✗ 403 Forbidden - Cannot access thread")
|
|
print("Even with cloudscraper, getting 403. This means:")
|
|
print("1. Cookies are expired - export fresh ones")
|
|
print("2. IP is blocked - try different IP/VPN")
|
|
print("3. CAPTCHA required - need Selenium solution")
|
|
|
|
with open('thread_403_cloudscraper.html', 'w', encoding='utf-8') as f:
|
|
f.write(response.text)
|
|
print("→ Saved response to 'thread_403_cloudscraper.html'")
|
|
return pages
|
|
|
|
response.raise_for_status()
|
|
|
|
soup = BeautifulSoup(response.text, 'html.parser')
|
|
page_links = soup.find_all('a', href=True)
|
|
|
|
page_numbers = set()
|
|
for link in page_links:
|
|
href = link['href']
|
|
match = re.search(r'/page-(\d+)', href)
|
|
if match:
|
|
page_numbers.add(int(match.group(1)))
|
|
|
|
if page_numbers:
|
|
max_page = min(max(page_numbers), max_pages)
|
|
print(f"Found {max_page} pages in thread")
|
|
pages.append(base_url)
|
|
for page_num in range(2, max_page + 1):
|
|
if base_url.endswith('/'):
|
|
pages.append(f"{base_url}page-{page_num}")
|
|
else:
|
|
pages.append(f"{base_url}/page-{page_num}")
|
|
else:
|
|
print("No pagination found, defaulting to single page")
|
|
pages.append(base_url)
|
|
|
|
except Exception as e:
|
|
print(f"Error discovering pages: {e}")
|
|
if pages:
|
|
return pages
|
|
pages.append(base_url)
|
|
|
|
return pages
|
|
|
|
def fetch_all_pages_to_file(self, page_urls, output_file):
|
|
"""Fetch all thread pages and append them into a single HTML file."""
|
|
if not page_urls:
|
|
print("No pages to fetch")
|
|
return
|
|
|
|
print(f"\nSaving all pages into: {output_file}")
|
|
|
|
successful_pages = 0
|
|
with open(output_file, 'w', encoding='utf-8') as out_file:
|
|
for i, url in enumerate(page_urls, 1):
|
|
print(f"\nFetching page {i}/{len(page_urls)}: {url}")
|
|
|
|
try:
|
|
response = self.session.get(url, timeout=30, allow_redirects=True)
|
|
|
|
print(f" Status: {response.status_code}")
|
|
|
|
if response.status_code == 403:
|
|
print(f" ✗ 403 Forbidden on page {i}")
|
|
continue
|
|
|
|
response.raise_for_status()
|
|
|
|
if 'login' in response.url.lower():
|
|
print(" ✗ Redirected to login page")
|
|
continue
|
|
|
|
out_file.write(f"\n<!-- PAGE {i}: {url} -->\n")
|
|
out_file.write(response.text + "\n\n")
|
|
successful_pages += 1
|
|
print(f" ✓ Successfully saved page {i}")
|
|
|
|
except Exception as e:
|
|
print(f" ✗ Error fetching page {i}: {e}")
|
|
|
|
# Be polite with delays
|
|
if i < len(page_urls):
|
|
time.sleep(2)
|
|
|
|
print(f"\n✓ Successfully saved {successful_pages}/{len(page_urls)} pages")
|
|
|
|
def resolve_turbovid(self, url):
|
|
"""Resolve Turbovid URLs using their sign API endpoint."""
|
|
try:
|
|
match = re.search(r'/(v|d|embed)/([a-zA-Z0-9]+)', url)
|
|
if not match:
|
|
print(f" ✗ Could not extract video ID from: {url}")
|
|
return None
|
|
|
|
video_id = match.group(2)
|
|
print(f" Resolving Turbovid ID: {video_id}")
|
|
|
|
embed_url = f"https://turbo.cr/embed/{video_id}"
|
|
|
|
sign_urls = [
|
|
f"https://turbo.cr/api/sign?v={video_id}",
|
|
f"https://turbo.cr/sign?v={video_id}",
|
|
]
|
|
|
|
for sign_url in sign_urls:
|
|
try:
|
|
response = self.session.get(
|
|
sign_url,
|
|
headers={'Referer': embed_url},
|
|
timeout=30
|
|
)
|
|
|
|
if response.status_code == 200:
|
|
data = response.json()
|
|
|
|
if data.get('success') and data.get('url'):
|
|
signed_url = data['url']
|
|
|
|
original_name = data.get('original_filename')
|
|
if original_name and 'fn=' not in signed_url:
|
|
delimiter = '&' if '?' in signed_url else '?'
|
|
signed_url += f"{delimiter}fn={original_name}"
|
|
|
|
print(f" ✓ Resolved to: {signed_url[:80]}...")
|
|
return signed_url
|
|
|
|
except Exception as e:
|
|
continue
|
|
|
|
try:
|
|
response = self.session.get(embed_url, headers={'Referer': embed_url}, timeout=30)
|
|
soup = BeautifulSoup(response.text, 'html.parser')
|
|
|
|
source_tag = soup.find('source', src=True)
|
|
if source_tag:
|
|
return source_tag['src']
|
|
|
|
video_tag = soup.find('video', src=True)
|
|
if video_tag:
|
|
return video_tag['src']
|
|
|
|
except Exception:
|
|
pass
|
|
|
|
return None
|
|
|
|
except Exception as e:
|
|
print(f" ✗ Error resolving Turbovid: {e}")
|
|
return None
|
|
|
|
# Rest of the functions remain the same...
|
|
def extract_links(file_path, resolve_turbovid=False, extractor=None):
|
|
"""Extract links from an HTML file using regex."""
|
|
if not os.path.exists(file_path):
|
|
print(f"File not found: {file_path}")
|
|
return []
|
|
|
|
with open(file_path, 'r', encoding='utf-8') as f:
|
|
html = f.read()
|
|
|
|
if len(html) < 1000 and '403' in html:
|
|
print("⚠ Warning: HTML file appears to contain 403 error page")
|
|
return []
|
|
|
|
patterns = [
|
|
r'https?://\S*bunkr\.\S*',
|
|
r'https?://[a-zA-Z0-9\-]+\.jpg5\.su/images3/[^\s"\'<>]+?\.md\.jpg',
|
|
r'https?://jpg6\.su/img/[a-zA-Z0-9]+',
|
|
r'https?://jpg6\.su/a/[^\s"\'<>]+',
|
|
r'https?://saint2\.cr/embed/[a-zA-Z0-9]+',
|
|
r'https?://(?:www\.)?redgifs\.com/watch/[a-zA-Z0-9]+',
|
|
r'//redgifs\.com/ifr/[a-zA-Z0-9]+',
|
|
r'https?://gofile\.io/d/[a-zA-Z0-9]+',
|
|
r'https?://pixeldrain\.com/u/[a-zA-Z0-9]+',
|
|
r'https?://cyberdrop\.me/a/[a-zA-Z0-9]+',
|
|
r'https?://cyberdrop\.cr/f/[a-zA-Z0-9]+',
|
|
r'https?://(?:[\w-]+\.)?turbo(?:vid)?\.cr/(?:v|d|embed)/[a-zA-Z0-9]+',
|
|
r'https?://goonbox\.cr/img/[a-zA-Z0-9]+',
|
|
r'https?://goonbox\.ce/a/[^\s"\'<>]+',
|
|
r'https?://attachments\.f95zone\.to(?:/[a-zA-Z0-9]+)+'
|
|
r'https?://mega\.nz/file/[a-zA-Z0-9]+',
|
|
r'https?://drive\.google\.come/file/d/[a-zA-Z0-9]+',
|
|
r'https?://f95zone\.to/attachments/[a-zA-Z0-9]+',
|
|
]
|
|
|
|
clean_links = []
|
|
for pattern in patterns:
|
|
matches = re.findall(pattern, html)
|
|
for match in matches:
|
|
clean_link = match.strip().rstrip('"\'<>')
|
|
clean_links.append(clean_link)
|
|
|
|
unique_links = list(set(clean_links))
|
|
print(f"\nFound {len(unique_links)} unique links total")
|
|
|
|
if resolve_turbovid and extractor:
|
|
print("\n--- Resolving Turbovid Links ---")
|
|
resolved_links = []
|
|
turbovid_count = 0
|
|
failed_count = 0
|
|
for link in unique_links:
|
|
if 'turbo' in link.lower() and '.cr' in link.lower():
|
|
turbovid_count += 1
|
|
print(f"\nProcessing Turbovid link {turbovid_count}: {link}")
|
|
resolved = extractor.resolve_turbovid(link)
|
|
if resolved:
|
|
resolved_links.append(resolved)
|
|
else:
|
|
failed_count += 1
|
|
time.sleep(1)
|
|
else:
|
|
resolved_links.append(link)
|
|
print(f"\nResolved {turbovid_count - failed_count}/{turbovid_count} Turbovid links")
|
|
if failed_count > 0:
|
|
print(f"Failed: {failed_count} Turbovid link(s)")
|
|
return resolved_links
|
|
|
|
return unique_links
|
|
|
|
def save_links_to_file(links, output_file):
|
|
if not links:
|
|
print("No links found to save.")
|
|
return
|
|
|
|
existing = set()
|
|
if os.path.exists(output_file):
|
|
with open(output_file, 'r', encoding='utf-8') as f:
|
|
existing = set(line.strip() for line in f if line.strip())
|
|
|
|
new_links = [l for l in links if l not in existing]
|
|
|
|
if not new_links:
|
|
print("No new unique links found.")
|
|
return
|
|
|
|
with open(output_file, 'a', encoding='utf-8') as f:
|
|
for link in new_links:
|
|
f.write(link + "\n")
|
|
|
|
print(f"✓ Saved {len(new_links)} new links to {output_file}")
|
|
|
|
if __name__ == "__main__":
|
|
print("="*70)
|
|
print("SimPCity Link Extractor - Cloudscraper Version")
|
|
print("="*70)
|
|
print("\nMake sure you have cloudscraper installed:")
|
|
print(" pip install cloudscraper\n")
|
|
|
|
base_thread_url = "https://f95zone.to/threads/qiandai-collection-2026-01-04-qiandaiyiyu-qiandai.246056/"
|
|
output_html = "page_source.html"
|
|
output_links = "links.txt"
|
|
|
|
# CRITICAL: Export FRESH cookies from your browser while logged in
|
|
cookies = {
|
|
"__ddg1_": "ztu3bAWXt1yNTQLx4anD",
|
|
"__ddg2_": "XITPIP1OKNx5qhiT",
|
|
"__ddg8_": "oWldvDW5w2JAjymf",
|
|
"__ddg9_": "217.138.213.52",
|
|
"__ddg10_": "1781612477",
|
|
"__ddgid_": "yhSli2A9nG5Xkwcu",
|
|
"__ddgmark_": "pVXsbrKqDemwPHiu",
|
|
"oMased": "b4",
|
|
"oMasid": "9e19ba410603e03b1286baca7c2917ef56b9c760415db5e2b9574c87eb092e7e",
|
|
"ogaddgmetaprof_csrf": "",
|
|
"yMziCv8BrCZz1o7_csrf": "vGZEMX5-P-XSKuG_",
|
|
"yMziCv8BrCZz1o7_session": "upTVlbK2kspfKlRGyD74rOh_PacNF7DI",
|
|
"yMziCv8BrCZz1o7_user": "3944332%2Cr8xH2AcnE_hg0kkeNSSsi3luzSkW6odfKX21CHdk",
|
|
"yMziCv8BrCZz1o7_dbWriteForced": "1781612424"
|
|
}
|
|
|
|
extractor = SimPCityExtractor(cookies=cookies)
|
|
pages = extractor.get_thread_pages(base_thread_url)
|
|
|
|
if pages:
|
|
extractor.fetch_all_pages_to_file(pages, output_html)
|
|
links = extract_links(output_html, resolve_turbovid=True, extractor=extractor)
|
|
save_links_to_file(links, output_links)
|
|
print("\n" + "="*70)
|
|
print("Process completed!") |