Something went wrong. Try again.
A fork of https://github.com/crosspoint-reader/crosspoint-reader
Something went wrong. Try again.
4.5 kB · 108 lines
Python
123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109import osimport reimport gzipimport hashlib
SRC_DIR = "src"HTML_EXT = ".html"JS_EXT = ".js"HEADER_PREAMBLE = "// THIS FILE IS AUTOGENERATED, DO NOT EDIT MANUALLY\n\n"HEADER_PRAGMA = "#pragma once\n"HEADER_INCLUDE = "#include <cstddef>\n\n"
def minify_html(html: str) -> str: # Tags where whitespace should be preserved preserve_tags = ['pre', 'code', 'textarea', 'script', 'style'] preserve_regex = '|'.join(preserve_tags)
# Protect preserve blocks with placeholders preserve_blocks = [] def preserve(match): preserve_blocks.append(match.group(0)) return f"__PRESERVE_BLOCK_{len(preserve_blocks)-1}__"
html = re.sub(rf'<({preserve_regex})[\s\S]*?</\1>', preserve, html, flags=re.IGNORECASE)
# Remove HTML comments html = re.sub(r'<!--.*?-->', '', html, flags=re.DOTALL)
# Collapse all whitespace between tags html = re.sub(r'>\s+<', '><', html)
# Collapse multiple spaces inside tags html = re.sub(r'\s+', ' ', html)
# Restore preserved blocks for i, block in enumerate(preserve_blocks): html = html.replace(f"__PRESERVE_BLOCK_{i}__", block)
return html.strip()
def sanitize_identifier(name: str) -> str: """Sanitize a filename to create a valid C identifier.
C identifiers must: - Start with a letter or underscore - Contain only letters, digits, and underscores """ # Replace non-alphanumeric characters (including hyphens) with underscores sanitized = re.sub(r'\W', '_', name, flags=re.ASCII) # Prefix with underscore if starts with a digit if sanitized and sanitized[0].isdigit(): sanitized = f"_{sanitized}" return sanitized
for root, _, files in os.walk(SRC_DIR): for file in files: if file.endswith((HTML_EXT, JS_EXT)): file_path = os.path.join(root, file) with open(file_path, "r", encoding="utf-8") as f: content = f.read()
# Only minify HTML files; JS files are typically pre-minified (e.g., jszip.min.js) is_html = file.endswith(HTML_EXT) if is_html: processed = minify_html(content) else: processed = content
# Compress with gzip (compresslevel 9 is maximum compression) # mtime=0 keeps the output reproducible across builds # IMPORTANT: we don't use brotli because Firefox doesn't support brotli with insecured context (only supported on HTTPS) compressed = gzip.compress(processed.encode('utf-8'), compresslevel=9, mtime=0)
# Create valid C identifier from filename # Use appropriate suffix based on file type suffix = "Html" if is_html else "Js" base_name = sanitize_identifier(f"{os.path.splitext(file)[0]}{suffix}") header_path = os.path.join(root, f"{base_name}.generated.h")
with open(header_path, "w", encoding="utf-8") as h: h.write(HEADER_PREAMBLE) h.write(HEADER_PRAGMA) h.write(HEADER_INCLUDE)
# Write the compressed data as a byte array h.write(f"constexpr char {base_name}[] PROGMEM = {{\n")
# Write bytes in rows of 16 for i in range(0, len(compressed), 16): chunk = compressed[i:i+16] hex_values = ', '.join(f'0x{b:02x}' for b in chunk) h.write(f" {hex_values},\n")
h.write("};\n\n") h.write(f"constexpr size_t {base_name}CompressedSize = {len(compressed)};\n") h.write(f"constexpr size_t {base_name}OriginalSize = {len(processed)};\n")
# ETag derived from the compressed payload. The content is # immutable at runtime (baked into flash at build time), so a # strong ETag keyed on the bytes is safe and stable. Browsers # echo it back as If-None-Match, enabling 304 responses. etag = hashlib.sha256(compressed).hexdigest()[:16] h.write(f'constexpr const char* {base_name}ETag = "\\"{etag}\\"";\n')
print(f"Generated: {header_path}") print(f" Original: {len(content)} bytes") print(f" Minified: {len(processed)} bytes ({100*len(processed)/len(content):.1f}%)") print(f" Compressed: {len(compressed)} bytes ({100*len(compressed)/len(content):.1f}%)")