import os import re import gzip SRC_DIR = "src" def strip_js_comments(js: str) -> str: """Remove JS comments while preserving string literals and URLs.""" result = [] i = 0 length = len(js) while i < length: # String literals — pass through unchanged if js[i] in ('"', "'", "`"): quote = js[i] result.append(js[i]) i += 1 while i < length: if js[i] == "\\" and i + 1 < length: result.append(js[i : i + 2]) i += 2 elif js[i] == quote: result.append(js[i]) i += 1 break else: result.append(js[i]) i += 1 # Block comment /* ... */ elif js[i] == "/" and i + 1 < length and js[i + 1] == "*": end = js.find("*/", i + 2) i = end + 2 if end != -1 else length # Line comment // ... elif js[i] == "/" and i + 1 < length and js[i + 1] == "/": end = js.find("\n", i) if end == -1: i = length else: # Keep the newline to preserve line structure result.append("\n") i = end + 1 # Regex literal — pass through unchanged # Heuristic: / after = ( , ; ! & | ? : [ { } ~ ^ or line start elif js[i] == "/" and i > 0: # Look back for operator context (skip whitespace) j = i - 1 while j >= 0 and js[j] in " \t": j -= 1 if j >= 0 and js[j] in "=(!,;:&|?[{}>~^+-*%": result.append(js[i]) i += 1 while i < length: if js[i] == "\\" and i + 1 < length: result.append(js[i : i + 2]) i += 2 elif js[i] == "/": result.append(js[i]) i += 1 # Regex flags while i < length and js[i].isalpha(): result.append(js[i]) i += 1 break elif js[i] == "[": # Character class — / doesn't end regex inside [] result.append(js[i]) i += 1 while i < length and js[i] != "]": if js[i] == "\\" and i + 1 < length: result.append(js[i : i + 2]) i += 2 else: result.append(js[i]) i += 1 else: result.append(js[i]) i += 1 else: result.append(js[i]) i += 1 else: result.append(js[i]) i += 1 return "".join(result) def minify_html(html: str) -> str: # Tags where whitespace should be preserved preserve_tags = ["pre", "code", "textarea"] script_style_tags = ["script", "style"] preserve_regex = "|".join(preserve_tags) script_style_regex = "|".join(script_style_tags) # Protect preserve blocks (pre/code/textarea) with placeholders preserve_blocks = [] def preserve(match): preserve_blocks.append(match.group(0)) return f"__PRESERVE_BLOCK_{len(preserve_blocks) - 1}__" html = re.sub( rf"<({preserve_regex})[\s\S]*?", preserve, html, flags=re.IGNORECASE ) # Strip JS/CSS comments inside