Files
MeshCore-mqtt-observer/scripts/generate_webconfig_html.py
agessaman 8cbc5520c9 build(webconfig): strip comments before embedding the portal page
The generator gzipped webui/index.html verbatim, so the page's comments — and
this file is commented heavily by house style — were paying flash rent. A
line-based pass now drops comments, indentation and blank lines before
compressing. The source stays as readable as it was.

Conservative on purpose: only a comment that starts its own line is removed, so
a `//` inside a URL or a `/*` inside a regex can never be mistaken for one.
Line breaks survive, which leaves JS statement boundaries (and the space a
newline contributes between HTML inline elements) exactly as written.

This ships to thousands of devices, so it is not taken on trust:
  - check_stripped() fails the build if the page's structure changed or the
    output shrank implausibly
  - the pass lives in its own module, shared with the mock backend's new
    --minify flag, so the bytes exercised in a browser are the bytes that get
    embedded rather than a second implementation that could drift
  - webconfig_minify.py joins the generator in the freshness hash, so editing
    the stripper forces a regenerate

Today's page: 22,678 -> 17,671 bytes gzipped.
2026-08-07 22:35:00 -07:00

144 lines
5.1 KiB
Python

#!/usr/bin/env python
#
# Pre-build script: gzip webui/index.html into a PROGMEM C header so the
# webconfig portal can serve the page straight from flash with
# Content-Encoding: gzip. The generated header is .gitignored; this script
# regenerates it whenever the source page (or this script) content changes.
#
# Freshness is decided by a content hash of the build inputs (not timestamps):
# a generated header written with a skewed/future mtime would otherwise mask a
# later source edit and embed stale UI. gzip output is deterministic (mtime=0),
# so an unchanged source produces an unchanged header.
#
# The generator is idempotent and cheap: when the inputs are unchanged it only
# hashes two small files and returns, so running it from esp32_base on every
# ESP32 build is negligible even for targets that don't compile the portal.
#
# Comments and indentation are stripped before compressing (see strip_source).
# The source page is heavily commented by house style and none of it is worth
# flash, so the page ships smaller than it reads.
#
# Output: src/helpers/esp32/WebConfigHtml.h
# WEBCONFIG_HTML_GZ[] - gzipped page (PROGMEM)
# WEBCONFIG_HTML_GZ_LEN - byte length
# WEBCONFIG_HTML_ETAG - quoted strong ETag (sha256 prefix of the gz body)
#
# Runnable outside SCons to inspect exactly what gets shipped:
# python3 scripts/generate_webconfig_html.py --emit /tmp/shipped.html
# or served directly by the mock backend with its --minify flag.
import gzip
import hashlib
import os
import sys
try:
Import("env") # noqa: F821
except NameError:
pass # running standalone (--emit), not as a PIO extra_script
SOURCE = os.path.join("webui", "index.html")
OUTPUT = os.path.join("src", "helpers", "esp32", "WebConfigHtml.h")
# __file__ is not defined inside PIO/SCons-executed extra_scripts
SCRIPT = os.path.join("scripts", "generate_webconfig_html.py")
MINIFIER = os.path.join("scripts", "webconfig_minify.py")
HASH_MARKER = "// build-inputs-sha256: "
sys.path.insert(0, os.path.join(os.getcwd(), "scripts"))
from webconfig_minify import check_stripped, strip_source # noqa: E402
def status(msg):
sys.stderr.write("WebConfig HTML: %s\n" % msg)
def content_hash():
# Hash the source page, this generator and the minifier so any change to
# any of them forces a regenerate, independent of file timestamps.
h = hashlib.sha256()
with open(SOURCE, "rb") as f:
h.update(f.read())
for path in (SCRIPT, MINIFIER):
if os.path.isfile(path):
h.update(b"\0")
with open(path, "rb") as f:
h.update(f.read())
return h.hexdigest()
def stored_hash():
try:
with open(OUTPUT, "r") as f:
for line in f:
if line.startswith(HASH_MARKER):
return line[len(HASH_MARKER):].strip()
if line.startswith("#"): # reached the C preprocessor lines
break
except OSError:
return None
return None
def shipped_page():
"""The exact bytes the device serves: the source page, stripped."""
with open(SOURCE, "r", encoding="utf-8") as f:
raw = f.read()
stripped = strip_source(raw)
problem = check_stripped(raw, stripped)
if problem:
status("ERROR: comment stripping corrupted the page (%s)" % problem)
sys.exit(2)
return raw, stripped
def main():
if not os.path.isfile(SOURCE):
status("ERROR: %s not found" % SOURCE)
sys.exit(2)
if "--emit" in sys.argv:
dest = sys.argv[sys.argv.index("--emit") + 1]
src, stripped = shipped_page()
with open(dest, "w", encoding="utf-8") as f:
f.write(stripped)
status("%s -> %s (%d -> %d bytes)" % (SOURCE, dest, len(src), len(stripped)))
return
src_hash = content_hash()
if os.path.isfile(OUTPUT) and stored_hash() == src_hash:
return
src, stripped = shipped_page()
raw = stripped.encode("utf-8")
# mtime=0 keeps the gzip output (and therefore the ETag) deterministic
gz = gzip.compress(raw, compresslevel=9, mtime=0)
etag = hashlib.sha256(gz).hexdigest()[:16]
lines = []
lines.append("// Auto-generated by scripts/generate_webconfig_html.py from %s" % SOURCE.replace(os.sep, "/"))
lines.append("// DO NOT EDIT - edit webui/index.html instead.")
lines.append("%s%s" % (HASH_MARKER, src_hash))
lines.append("#pragma once")
lines.append("#include <stdint.h>")
lines.append("#include <pgmspace.h>")
lines.append("")
lines.append("const uint32_t WEBCONFIG_HTML_GZ_LEN = %d;" % len(gz))
lines.append('const char WEBCONFIG_HTML_ETAG[] = "\\"%s\\"";' % etag)
lines.append("const uint8_t WEBCONFIG_HTML_GZ[] PROGMEM = {")
for i in range(0, len(gz), 16):
chunk = gz[i:i + 16]
lines.append(" " + "".join("0x%02x," % b for b in chunk))
lines.append("};")
lines.append("")
os.makedirs(os.path.dirname(OUTPUT), exist_ok=True)
with open(OUTPUT, "w") as f:
f.write("\n".join(lines))
status("%s -> %s (%d bytes source, %d stripped, %d gzipped)"
% (SOURCE, OUTPUT, len(src.encode("utf-8")), len(raw), len(gz)))
main()