Build and push server image / build-and-push (push) Successful in 39s
/api/frames/{id}/firmware/check could silently stage new firmware as a
side effect (the auto-apply path, when firmware_auto_update is on and
a newer release exists) but was gated by require_frame_view instead of
require_frame_control like its sibling firmware routes, and being a
GET, was exempt from the app's CSRF check (which only applies to
non-GET/HEAD/OPTIONS). A linked viewer without control -- or a
cross-site page riding a control-holding victim's session via a plain
GET -- could trigger an unreviewed firmware install. Now POST +
require_frame_control, matching /firmware/apply-latest; the frontend's
two callers (passive poll on page load, "Check now" button) both
already handle a 409 from a non-controller gracefully via the existing
apiError()/control-banner pattern, so this doesn't change UX for a
frame's actual controller.
Separately: Immich has been observed to return a face detection entry
with a null bounding-box field (a still-pending or otherwise
incomplete detection). Both places that do arithmetic on those fields
-- image_pipeline._face_aware_crop_box (crop_faces display mode) and
face_labels.compute_face_labels (manage-menu name labels) -- crashed
with an unhandled TypeError on such an entry, taking down that frame's
whole photo instead of the intended graceful fallback. Both now skip
any face missing a bounding-box field via a shared _has_bounding_box()
check; a face list with zero valid entries already degrades cleanly to
the plain center crop (the existing inf/-inf sentinel math already
handled "no faces" correctly, it just couldn't tell "none passed
Immich" apart from "one broken entry" before).
391 lines
17 KiB
Python
391 lines
17 KiB
Python
"""Resize, quantize, and pack a photo into the panel's raw 4bpp format."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
|
|
from PIL import Image, ImageEnhance, ImageOps
|
|
|
|
EPD_WIDTH = 800
|
|
EPD_HEIGHT = 480
|
|
|
|
# How each orientation maps the logically-composed image onto the native
|
|
# 800x480 panel. "portrait"/"portrait_flipped" compose at 480x800 (so the
|
|
# crop ratio matches how the frame actually hangs) and rotate into native
|
|
# space afterwards -- rotation happens after dithering, which is lossless
|
|
# (a pure pixel permutation). Which of 90/270 is "portrait" vs
|
|
# "portrait_flipped" is a convention pick; whichever way the frame is
|
|
# hung, one of the two is right.
|
|
ORIENTATION_TRANSPOSE = {
|
|
"landscape": None,
|
|
"landscape_flipped": Image.Transpose.ROTATE_180,
|
|
"portrait": Image.Transpose.ROTATE_90,
|
|
"portrait_flipped": Image.Transpose.ROTATE_270,
|
|
}
|
|
|
|
|
|
def logical_render_size(orientation: str) -> tuple[int, int]:
|
|
"""(width, height) the photo is composed/cropped at for this
|
|
orientation, before rotating into native panel space."""
|
|
if orientation in ("portrait", "portrait_flipped"):
|
|
return EPD_HEIGHT, EPD_WIDTH
|
|
return EPD_WIDTH, EPD_HEIGHT
|
|
|
|
|
|
def logical_to_native(x: float, y: float, orientation: str) -> tuple[int, int]:
|
|
"""Maps a point in logical (pre-rotation) frame space to native
|
|
800x480 panel space, applying the same rotation ORIENTATION_TRANSPOSE
|
|
applies to the pixels -- anything positioned in logical coordinates
|
|
(e.g. face labels) needs this to stay attached to the rotated
|
|
content. PIL's ROTATE_90 is counterclockwise; ROTATE_270 clockwise."""
|
|
logical_w, logical_h = logical_render_size(orientation)
|
|
if orientation == "landscape_flipped":
|
|
return int(logical_w - 1 - x), int(logical_h - 1 - y)
|
|
if orientation == "portrait": # ROTATE_90 (CCW)
|
|
return int(y), int(logical_w - 1 - x)
|
|
if orientation == "portrait_flipped": # ROTATE_270 (CW)
|
|
return int(logical_h - 1 - y), int(x)
|
|
return int(x), int(y)
|
|
|
|
# Approximate sRGB for each of the panel's 6 ink colors -- reasonable
|
|
# placeholders, not measured values (Waveshare doesn't publish exact
|
|
# color primaries for this panel). This is the fallback for any frame
|
|
# that hasn't tuned its own (Frame.palette_rgb, set from a frame's
|
|
# Configuration tab -- "Advanced configuration" -- once you can compare
|
|
# a rendered test image against the real panel; different panel units
|
|
# can vary enough to be worth calibrating per frame).
|
|
DEFAULT_PALETTE_RGB = [
|
|
(0, 0, 0), # BLACK
|
|
(255, 255, 255), # WHITE
|
|
(255, 219, 0), # YELLOW
|
|
(207, 0, 15), # RED
|
|
(0, 39, 133), # BLUE
|
|
(0, 133, 55), # GREEN
|
|
]
|
|
|
|
PALETTE_LABELS = ["Black", "White", "Yellow", "Red", "Blue", "Green"]
|
|
|
|
# The panel's actual 4-bit color codes (see firmware/components/epd7in3e),
|
|
# in the same order as DEFAULT_PALETTE_RGB/PALETTE_LABELS -- fixed by the
|
|
# hardware protocol, never user-configurable. 0x4 is intentionally unused
|
|
# upstream.
|
|
PANEL_CODES = [0x0, 0x1, 0x2, 0x3, 0x5, 0x6]
|
|
|
|
|
|
def palette_to_hex(palette_rgb: list) -> list[str]:
|
|
"""[(0,0,0), ...] -> ["#000000", ...], for pre-filling the Advanced
|
|
configuration color pickers."""
|
|
return ["#%02x%02x%02x" % tuple(c) for c in palette_rgb]
|
|
|
|
|
|
def hex_to_rgb(hex_str: str) -> tuple[int, int, int] | None:
|
|
""""#1a2b3c" -> (26, 43, 60), or None for anything that isn't exactly
|
|
a 6-hex-digit color (what <input type="color"> always sends, but a
|
|
direct API call might not)."""
|
|
hex_str = hex_str.strip().lstrip("#")
|
|
if len(hex_str) != 6:
|
|
return None
|
|
try:
|
|
return (int(hex_str[0:2], 16), int(hex_str[2:4], 16), int(hex_str[4:6], 16))
|
|
except ValueError:
|
|
return None
|
|
|
|
|
|
def _build_palette_image(palette_rgb: list) -> Image.Image:
|
|
pal_img = Image.new("P", (1, 1))
|
|
pal_img.putpalette([channel for rgb in palette_rgb for channel in rgb])
|
|
return pal_img
|
|
|
|
|
|
def _plain_center_crop_box(
|
|
img_width: int, img_height: int, target_width: int, target_height: int
|
|
) -> tuple[float, float, int, int]:
|
|
"""The largest target_width:target_height window centered in the
|
|
source image -- the same box ImageOps.fit() computes internally when
|
|
there's no face-aware shift to apply. Returns (left, top, crop_w,
|
|
crop_h); left/top are floats (not yet rounded) since callers that go
|
|
on to face-shift this box need the unrounded center point."""
|
|
target_ratio = target_width / target_height
|
|
if img_width / img_height > target_ratio:
|
|
crop_h = img_height
|
|
crop_w = int(crop_h * target_ratio)
|
|
else:
|
|
crop_w = img_width
|
|
crop_h = int(crop_w / target_ratio)
|
|
|
|
left = (img_width - crop_w) / 2
|
|
top = (img_height - crop_h) / 2
|
|
return left, top, crop_w, crop_h
|
|
|
|
|
|
def _has_bounding_box(face: dict) -> bool:
|
|
"""Immich has occasionally been observed to return a face entry with
|
|
a still-pending or otherwise incomplete bounding box (a null field)
|
|
-- treat it as undetected rather than crash on arithmetic with None."""
|
|
return all(
|
|
face.get(k) is not None
|
|
for k in ("boundingBoxX1", "boundingBoxX2", "boundingBoxY1", "boundingBoxY2")
|
|
)
|
|
|
|
|
|
def _face_aware_crop_box(
|
|
img_width: int, img_height: int, target_width: int, target_height: int, faces: list[dict]
|
|
) -> tuple[int, int, int, int]:
|
|
"""Largest crop window matching target_width:target_height that fits
|
|
inside the source image. Starts from the plain center crop and only
|
|
shifts it the minimum amount needed to bring any faces that would
|
|
otherwise be cut off back on screen -- an already-fine composition
|
|
(faces already fully inside the center crop) is left untouched rather
|
|
than re-centered on the faces. If the faces themselves span wider than
|
|
the crop window allows, centers on their midpoint as best-effort,
|
|
since there's no shift that fits them all regardless.
|
|
|
|
Each face's box is given relative to its own imageWidth/imageHeight
|
|
(the resolution Immich ran detection on), which may differ from the
|
|
downloaded preview's resolution passed in here, so each box is scaled
|
|
into img_width/img_height space before use.
|
|
"""
|
|
min_x = min_y = float("inf")
|
|
max_x = max_y = float("-inf")
|
|
for face in faces:
|
|
if not _has_bounding_box(face):
|
|
continue
|
|
face_w = face.get("imageWidth") or img_width
|
|
face_h = face.get("imageHeight") or img_height
|
|
scale_x = img_width / face_w
|
|
scale_y = img_height / face_h
|
|
min_x = min(min_x, face["boundingBoxX1"] * scale_x)
|
|
max_x = max(max_x, face["boundingBoxX2"] * scale_x)
|
|
min_y = min(min_y, face["boundingBoxY1"] * scale_y)
|
|
max_y = max(max_y, face["boundingBoxY2"] * scale_y)
|
|
|
|
left, top, crop_w, crop_h = _plain_center_crop_box(img_width, img_height, target_width, target_height)
|
|
|
|
if max_x - min_x <= crop_w:
|
|
if min_x < left:
|
|
left = min_x
|
|
elif max_x > left + crop_w:
|
|
left = max_x - crop_w
|
|
else:
|
|
left = (min_x + max_x) / 2 - crop_w / 2
|
|
|
|
if max_y - min_y <= crop_h:
|
|
if min_y < top:
|
|
top = min_y
|
|
elif max_y > top + crop_h:
|
|
top = max_y - crop_h
|
|
else:
|
|
top = (min_y + max_y) / 2 - crop_h / 2
|
|
|
|
left = max(0, min(left, img_width - crop_w))
|
|
top = max(0, min(top, img_height - crop_h))
|
|
|
|
return (int(left), int(top), int(left) + crop_w, int(top) + crop_h)
|
|
|
|
|
|
# Display modes: how a photo's aspect ratio gets reconciled with the
|
|
# panel's. "crop_faces" falls back to "crop_fill" behavior when no faces
|
|
# were detected/passed. DEFAULT_DISPLAY_MODE matches this project's old
|
|
# always-on smart_crop_faces=True default.
|
|
DISPLAY_MODES = ["crop_fill", "crop_faces", "stretch_fill", "letterbox"]
|
|
DISPLAY_MODE_LABELS = {
|
|
"crop_fill": "Crop to fill",
|
|
"crop_faces": "Crop to faces",
|
|
"stretch_fill": "Stretch to fill",
|
|
"letterbox": "Shrink to fit",
|
|
}
|
|
DEFAULT_DISPLAY_MODE = "crop_faces"
|
|
LETTERBOX_BG = (255, 255, 255)
|
|
|
|
|
|
def _placement_transform(
|
|
img_width: int, img_height: int, target_w: int, target_h: int,
|
|
display_mode: str, faces: list[dict] | None = None,
|
|
) -> tuple[float, float, float, float]:
|
|
"""Returns (scale_x, scale_y, offset_x, offset_y) mapping a point in
|
|
source-image pixel space to a point in target logical space, for the
|
|
given display_mode. Shared by render_frame (which also does the
|
|
actual pixel crop/resize/pad) and face_labels.py (label position
|
|
math) -- they must stay in exact agreement or overlay labels drift
|
|
off the people they're meant to point at."""
|
|
if display_mode == "stretch_fill":
|
|
return target_w / img_width, target_h / img_height, 0.0, 0.0
|
|
if display_mode == "letterbox":
|
|
scale = min(target_w / img_width, target_h / img_height)
|
|
return scale, scale, (target_w - img_width * scale) / 2, (target_h - img_height * scale) / 2
|
|
if display_mode == "crop_faces" and faces:
|
|
left, top, right, bottom = _face_aware_crop_box(img_width, img_height, target_w, target_h, faces)
|
|
crop_w, crop_h = right - left, bottom - top
|
|
else:
|
|
left, top, crop_w, crop_h = _plain_center_crop_box(img_width, img_height, target_w, target_h)
|
|
scale_x, scale_y = target_w / crop_w, target_h / crop_h
|
|
return scale_x, scale_y, -left * scale_x, -top * scale_y
|
|
|
|
|
|
def _compose(source: Image.Image, faces: list[dict] | None, orientation: str, display_mode: str) -> Image.Image:
|
|
"""Crop/resize/letterbox `source` per display_mode -- returns an RGB
|
|
image at logical_render_size(orientation), before enhancement or
|
|
quantization. See render_frame for what each display_mode does."""
|
|
logical_w, logical_h = logical_render_size(orientation)
|
|
fitted = ImageOps.exif_transpose(source.convert("RGB"))
|
|
|
|
if display_mode == "stretch_fill":
|
|
return fitted.resize((logical_w, logical_h), Image.LANCZOS)
|
|
if display_mode == "letterbox":
|
|
scale = min(logical_w / fitted.width, logical_h / fitted.height)
|
|
new_w, new_h = max(1, round(fitted.width * scale)), max(1, round(fitted.height * scale))
|
|
resized = fitted.resize((new_w, new_h), Image.LANCZOS)
|
|
canvas = Image.new("RGB", (logical_w, logical_h), LETTERBOX_BG)
|
|
canvas.paste(resized, ((logical_w - new_w) // 2, (logical_h - new_h) // 2))
|
|
return canvas
|
|
if display_mode == "crop_faces" and faces:
|
|
box = _face_aware_crop_box(fitted.width, fitted.height, logical_w, logical_h, faces)
|
|
return fitted.crop(box).resize((logical_w, logical_h), Image.LANCZOS)
|
|
return ImageOps.fit(fitted, (logical_w, logical_h), method=Image.LANCZOS) # crop_fill, or crop_faces w/ no faces
|
|
|
|
|
|
def _enhance(img: Image.Image, color_boost: float, contrast_boost: float) -> Image.Image:
|
|
if color_boost != 1.0:
|
|
img = ImageEnhance.Color(img).enhance(color_boost)
|
|
if contrast_boost != 1.0:
|
|
img = ImageEnhance.Contrast(img).enhance(contrast_boost)
|
|
return img
|
|
|
|
|
|
def _quantize(img: Image.Image, palette_rgb: list | None, dither_strength: float) -> Image.Image:
|
|
"""RGB -> palette-quantized P-mode image, same size/orientation as
|
|
`img` (no rotation here). dither_strength blends `img` toward its own
|
|
flat (undithered) quantization before running Floyd-Steinberg on the
|
|
blend: at 0 there's zero quantization error left to diffuse (so the
|
|
result IS the flat quantization, no dithering texture at all); at 1
|
|
it's `img` unchanged (full-strength dithering, this project's
|
|
original always-on behavior); values between give a smooth continuum
|
|
of dithering intensity rather than an on/off toggle."""
|
|
palette_image = _build_palette_image(palette_rgb or DEFAULT_PALETTE_RGB)
|
|
if dither_strength >= 1.0:
|
|
return img.quantize(palette=palette_image, dither=Image.Dither.FLOYDSTEINBERG)
|
|
if dither_strength <= 0.0:
|
|
return img.quantize(palette=palette_image, dither=Image.Dither.NONE)
|
|
flat = img.quantize(palette=palette_image, dither=Image.Dither.NONE).convert("RGB")
|
|
blended = Image.blend(flat, img, dither_strength)
|
|
return blended.quantize(palette=palette_image, dither=Image.Dither.FLOYDSTEINBERG)
|
|
|
|
|
|
def _transpose_and_pack(quantized: Image.Image, orientation: str) -> bytes:
|
|
"""Rotates a logical-space quantized image into native panel space
|
|
and packs it 2 pixels/byte the way epd7in3e.c expects. Always
|
|
returns exactly EPD_WIDTH*EPD_HEIGHT/2 bytes."""
|
|
transpose = ORIENTATION_TRANSPOSE.get(orientation)
|
|
if transpose is not None:
|
|
quantized = quantized.transpose(transpose)
|
|
pixels = quantized.load()
|
|
|
|
out = bytearray(EPD_WIDTH * EPD_HEIGHT // 2)
|
|
i = 0
|
|
for y in range(EPD_HEIGHT):
|
|
for x in range(0, EPD_WIDTH, 2):
|
|
left = PANEL_CODES[pixels[x, y]]
|
|
right = PANEL_CODES[pixels[x + 1, y]]
|
|
out[i] = (left << 4) | right
|
|
i += 1
|
|
|
|
return bytes(out)
|
|
|
|
|
|
def render_frame(source: Image.Image, faces: list[dict] | None = None,
|
|
orientation: str = "landscape", palette_rgb: list | None = None,
|
|
display_mode: str = DEFAULT_DISPLAY_MODE, color_boost: float = 1.0,
|
|
contrast_boost: float = 1.0, dither_strength: float = 1.0) -> bytes:
|
|
"""Fits `source` to the panel's resolution, applies color/contrast
|
|
enhancement, quantizes it to the 6-color palette, and packs 2
|
|
pixels/byte the way epd7in3e.c expects. Always returns exactly
|
|
EPD_WIDTH*EPD_HEIGHT/2 bytes.
|
|
|
|
`display_mode` (see DISPLAY_MODES) picks how the photo's aspect ratio
|
|
is reconciled with the panel's: crop_fill (center-crop to fill,
|
|
excess trimmed), crop_faces (as crop_fill, but shifts the crop to
|
|
keep `faces` on screen -- falls back to crop_fill if none), stretch_fill
|
|
(fills exactly, aspect ratio not preserved), letterbox (whole photo
|
|
visible, letterboxed with LETTERBOX_BG where it doesn't fill).
|
|
|
|
`color_boost`/`contrast_boost` are PIL ImageEnhance factors (1.0 =
|
|
unchanged, matching PIL's own convention); `dither_strength` is
|
|
0.0-1.0 (see _quantize).
|
|
|
|
`orientation` (see ORIENTATION_TRANSPOSE) composes the photo for how
|
|
the frame physically hangs, then rotates into native panel space --
|
|
the output byte layout is identical either way.
|
|
|
|
`palette_rgb` overrides DEFAULT_PALETTE_RGB (a frame's tuned colors,
|
|
see Frame.palette_rgb) -- None uses the default.
|
|
"""
|
|
fitted = _enhance(_compose(source, faces, orientation, display_mode), color_boost, contrast_boost)
|
|
quantized = _quantize(fitted, palette_rgb, dither_strength)
|
|
return _transpose_and_pack(quantized, orientation)
|
|
|
|
|
|
def render_preview_png(source: Image.Image, faces: list[dict] | None = None,
|
|
orientation: str = "landscape", palette_rgb: list | None = None,
|
|
display_mode: str = DEFAULT_DISPLAY_MODE, color_boost: float = 1.0,
|
|
contrast_boost: float = 1.0, dither_strength: float = 1.0) -> bytes:
|
|
"""Identical composition/enhancement/quantization pipeline as
|
|
render_frame, but returned as a normal browser-viewable PNG in
|
|
logical (upright, as-the-frame-actually-hangs) orientation rather
|
|
than packed native-panel bytes and rotation -- what the web UI's
|
|
"how it will look on the frame" preview shows."""
|
|
fitted = _enhance(_compose(source, faces, orientation, display_mode), color_boost, contrast_boost)
|
|
quantized = _quantize(fitted, palette_rgb, dither_strength)
|
|
buf = io.BytesIO()
|
|
quantized.convert("RGB").save(buf, format="PNG")
|
|
return buf.getvalue()
|
|
|
|
|
|
def render_placeholder(lines: list[str], qr_url: str | None = None,
|
|
orientation: str = "landscape", palette_rgb: list | None = None) -> bytes:
|
|
"""A readable full-panel message (plus an optional QR code) in the
|
|
same packed format as render_frame -- what /frame/image serves for a
|
|
frame that isn't claimed or configured yet, so a fresh device shows
|
|
instructions instead of an error screen and never error-loops."""
|
|
from PIL import ImageDraw, ImageFont
|
|
|
|
logical_w, logical_h = logical_render_size(orientation)
|
|
img = Image.new("RGB", (logical_w, logical_h), (255, 255, 255))
|
|
draw = ImageDraw.Draw(img)
|
|
|
|
title_font = ImageFont.load_default(size=34)
|
|
body_font = ImageFont.load_default(size=24)
|
|
|
|
qr_img = None
|
|
if qr_url:
|
|
import qrcode
|
|
|
|
qr = qrcode.QRCode(border=1, box_size=1)
|
|
qr.add_data(qr_url)
|
|
qr.make(fit=True)
|
|
raw = qr.make_image().get_image().convert("RGB")
|
|
# Integer upscale with NEAREST keeps modules crisp on the panel.
|
|
target = 220
|
|
scale = max(1, target // raw.width)
|
|
qr_img = raw.resize((raw.width * scale, raw.height * scale), Image.NEAREST)
|
|
|
|
# Vertical layout: text block, then QR under it, centered as a group.
|
|
line_heights = []
|
|
for i, line in enumerate(lines):
|
|
font = title_font if i == 0 else body_font
|
|
bbox = draw.textbbox((0, 0), line, font=font)
|
|
line_heights.append((line, font, bbox[2] - bbox[0], bbox[3] - bbox[1]))
|
|
gap = 14
|
|
text_h = sum(h for _, _, _, h in line_heights) + gap * (len(line_heights) - 1 if line_heights else 0)
|
|
total_h = text_h + (qr_img.height + 28 if qr_img else 0)
|
|
y = max(20, (logical_h - total_h) // 2)
|
|
|
|
for line, font, w, h in line_heights:
|
|
draw.text(((logical_w - w) // 2, y), line, fill=(0, 0, 0), font=font)
|
|
y += h + gap
|
|
|
|
if qr_img:
|
|
img.paste(qr_img, ((logical_w - qr_img.width) // 2, y + 14))
|
|
|
|
quantized = _quantize(img, palette_rgb, dither_strength=1.0)
|
|
return _transpose_and_pack(quantized, orientation)
|