bartoo revised this gist 1 month ago. Go to revision
1 file changed, 173 insertions
htb_scraper.py(file created)
| @@ -0,0 +1,173 @@ | |||
| 1 | + | import os | |
| 2 | + | import sys | |
| 3 | + | import time | |
| 4 | + | import random | |
| 5 | + | import re | |
| 6 | + | import requests | |
| 7 | + | from urllib.parse import urljoin, urlparse | |
| 8 | + | from dotenv import load_dotenv | |
| 9 | + | ||
| 10 | + | def rewrite_image_paths(content, session, output_dir, base_url): | |
| 11 | + | def replace_html_img(match): | |
| 12 | + | return match.group(1) + download_image(match.group(2), session, output_dir, base_url) + match.group(3) | |
| 13 | + | ||
| 14 | + | def replace_markdown_img(match): | |
| 15 | + | return match.group(1) + download_image(match.group(2), session, output_dir, base_url) + match.group(3) | |
| 16 | + | ||
| 17 | + | content = re.sub(r'(<img\b[^>]*\bsrc=["\'])([^"\']+)(["\'][^>]*>)', replace_html_img, content, flags=re.I) | |
| 18 | + | return re.sub(r'(!\[[^\]]*\]\()([^)\s]+)(\))', replace_markdown_img, content) | |
| 19 | + | ||
| 20 | + | def print_status(current, total, section, sleep_duration=0): | |
| 21 | + | percent = int(current / total * 100) if total else 0 | |
| 22 | + | filled = int(20 * current / total) if total else 0 | |
| 23 | + | bar = "#" * filled + "-" * (20 - filled) | |
| 24 | + | print(f"\r\033[K[{bar}] {current}/{total} - {percent}% | {section} | sleep {sleep_duration}s", end="", flush=True) | |
| 25 | + | ||
| 26 | + | ||
| 27 | + | def download_image(url, session, output_dir, base_url): | |
| 28 | + | try: | |
| 29 | + | if not url.startswith('http'): | |
| 30 | + | if url.startswith('/'): | |
| 31 | + | url = urljoin(base_url, url) | |
| 32 | + | else: | |
| 33 | + | return url | |
| 34 | + | ||
| 35 | + | parsed_url = urlparse(url) | |
| 36 | + | filename = os.path.basename(parsed_url.path) | |
| 37 | + | if not filename: | |
| 38 | + | filename = f"image_{int(time.time())}.png" | |
| 39 | + | ||
| 40 | + | local_path = os.path.join(output_dir, filename) | |
| 41 | + | if os.path.exists(local_path): | |
| 42 | + | return local_path | |
| 43 | + | ||
| 44 | + | # Keep terminal output to the single progress line; failures print below. | |
| 45 | + | response = session.get(url, stream=True) | |
| 46 | + | response.raise_for_status() | |
| 47 | + | with open(local_path, 'wb') as f: | |
| 48 | + | for chunk in response.iter_content(chunk_size=8192): | |
| 49 | + | f.write(chunk) | |
| 50 | + | return local_path | |
| 51 | + | except Exception as e: | |
| 52 | + | print(f"\n[-] Failed to download {url}: {e}") | |
| 53 | + | return url | |
| 54 | + | ||
| 55 | + | def main(): | |
| 56 | + | load_dotenv() | |
| 57 | + | cookie_string = os.getenv('HTB_COOKIE') | |
| 58 | + | user_agent = os.getenv('HTB_USER_AGENT', 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36') | |
| 59 | + | ||
| 60 | + | if not cookie_string: | |
| 61 | + | print("Error: HTB_COOKIE not found in .env file.") | |
| 62 | + | sys.exit(1) | |
| 63 | + | ||
| 64 | + | if len(sys.argv) < 2: | |
| 65 | + | print("Usage: python htb_scraper.py https://academy.hackthebox.com/app/module/144") | |
| 66 | + | sys.exit(1) | |
| 67 | + | ||
| 68 | + | module_url = sys.argv[1] | |
| 69 | + | ||
| 70 | + | # Extract module ID | |
| 71 | + | # URLs typically: /app/module/144 or /module/144 | |
| 72 | + | parts = module_url.strip('/').split('/') | |
| 73 | + | try: | |
| 74 | + | module_idx = parts.index('module') | |
| 75 | + | module_id = parts[module_idx + 1] | |
| 76 | + | except (ValueError, IndexError): | |
| 77 | + | print("Error: Could not extract module ID from URL. Make sure it contains '/module/<id>'") | |
| 78 | + | sys.exit(1) | |
| 79 | + | ||
| 80 | + | session = requests.Session() | |
| 81 | + | headers = { | |
| 82 | + | 'accept': 'application/json', | |
| 83 | + | 'referer': module_url, | |
| 84 | + | 'user-agent': user_agent, | |
| 85 | + | 'sec-ch-ua': '"Brave";v="149", "Chromium";v="149", "Not)A;Brand";v="24"', | |
| 86 | + | 'sec-ch-ua-mobile': '?0', | |
| 87 | + | 'sec-ch-ua-platform': '"Linux"', | |
| 88 | + | 'sec-fetch-dest': 'empty', | |
| 89 | + | 'sec-fetch-mode': 'cors', | |
| 90 | + | 'sec-fetch-site': 'same-origin', | |
| 91 | + | 'sec-gpc': '1', | |
| 92 | + | 'Cookie': cookie_string | |
| 93 | + | } | |
| 94 | + | ||
| 95 | + | session.headers.update(headers) | |
| 96 | + | ||
| 97 | + | print_status(0, 0, f"module {module_id}", 0) | |
| 98 | + | api_base = "https://academy.hackthebox.com" | |
| 99 | + | sections_url = f"{api_base}/api/v3/modules/{module_id}/sections" | |
| 100 | + | ||
| 101 | + | res = session.get(sections_url) | |
| 102 | + | if res.status_code != 200: | |
| 103 | + | print(f"\n[-] API returned status {res.status_code}. Cookie invalid or expired.") | |
| 104 | + | sys.exit(1) | |
| 105 | + | ||
| 106 | + | data = res.json() | |
| 107 | + | # API v3 returns list of groups containing sections | |
| 108 | + | sections = [] | |
| 109 | + | if isinstance(data, dict) and "data" in data: | |
| 110 | + | groups = data["data"] | |
| 111 | + | elif isinstance(data, list): | |
| 112 | + | groups = data | |
| 113 | + | else: | |
| 114 | + | groups = [] | |
| 115 | + | ||
| 116 | + | for group in groups: | |
| 117 | + | if isinstance(group, dict) and "sections" in group: | |
| 118 | + | for s in group["sections"]: | |
| 119 | + | if isinstance(s, dict) and "id" in s: | |
| 120 | + | sections.append(s) | |
| 121 | + | ||
| 122 | + | if not sections: | |
| 123 | + | print("\n[-] No sections found in the API response.") | |
| 124 | + | sys.exit(1) | |
| 125 | + | ||
| 126 | + | print_status(0, len(sections), f"module {module_id}", 0) | |
| 127 | + | ||
| 128 | + | output_file = f"module_{module_id}.md" | |
| 129 | + | images_dir = f"images_module_{module_id}" | |
| 130 | + | os.makedirs(images_dir, exist_ok=True) | |
| 131 | + | ||
| 132 | + | with open(output_file, "w", encoding="utf-8") as f: | |
| 133 | + | f.write(f"# HTB Academy Module {module_id}\n\n") | |
| 134 | + | f.write(f"Source: {module_url}\n\n") | |
| 135 | + | ||
| 136 | + | for i, sec in enumerate(sections, 1): | |
| 137 | + | sec_id = sec['id'] | |
| 138 | + | sec_title = sec.get('title', f"Section {sec_id}") | |
| 139 | + | ||
| 140 | + | delay = random.randint(3, 8) | |
| 141 | + | for remaining in range(delay, 0, -1): | |
| 142 | + | print_status(i, len(sections), sec_title, remaining) | |
| 143 | + | time.sleep(1) | |
| 144 | + | ||
| 145 | + | print_status(i, len(sections), sec_title, 0) | |
| 146 | + | content_url = f"{api_base}/api/v2/modules/{module_id}/sections/{sec_id}" | |
| 147 | + | ||
| 148 | + | sec_res = session.get(content_url) | |
| 149 | + | if sec_res.status_code != 200: | |
| 150 | + | print(f"\n[-] Failed to fetch {sec_title}. Status: {sec_res.status_code}") | |
| 151 | + | continue | |
| 152 | + | ||
| 153 | + | sec_data = sec_res.json() | |
| 154 | + | if isinstance(sec_data, dict) and "data" in sec_data: | |
| 155 | + | sec_data = sec_data["data"] | |
| 156 | + | ||
| 157 | + | html_content = sec_data.get("content", "") | |
| 158 | + | if not html_content: | |
| 159 | + | print(f"\n[-] No content found for {sec_title}.") | |
| 160 | + | continue | |
| 161 | + | ||
| 162 | + | # HTB's API content is already Markdown. Do not run it through an | |
| 163 | + | # HTML parser; XML examples inside fenced code blocks are real content. | |
| 164 | + | markdown_text = rewrite_image_paths(html_content, session, images_dir, api_base) | |
| 165 | + | ||
| 166 | + | with open(output_file, "a", encoding="utf-8") as f: | |
| 167 | + | f.write(f"\n\n---\n\n## {sec_title}\n\n") | |
| 168 | + | f.write(markdown_text) | |
| 169 | + | ||
| 170 | + | print(f"\r\033[K[+] Done! Saved to {output_file}") | |
| 171 | + | ||
| 172 | + | if __name__ == "__main__": | |
| 173 | + | main() | |
Newer
Older