379 lines
17 KiB
Python
379 lines
17 KiB
Python
"""Resize, quantize, and pack a photo into the panel's raw 4bpp format."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import io
|
|
|
|
from PIL import Image, ImageEnhance, ImageOps
|
|
|
|
EPD_WIDTH = 800
|
|
EPD_HEIGHT = 480
|
|
|
|
# How each orientation maps the logically-composed image onto the native
|
|
# 800x480 panel. "portrait"/"portrait_flipped" compose at 480x800 (so the
|
|
# crop ratio matches how the frame actually hangs) and rotate into native
|
|
# space afterwards -- rotation happens after dithering, which is lossless
|
|
# (a pure pixel permutation). Which of 90/270 is "portrait" vs
|
|
# "portrait_flipped" is a convention pick; whichever way the frame is
|
|
# hung, one of the two is right.
|
|
ORIENTATION_TRANSPOSE = {
|
|
"landscape": None,
|
|
"landscape_flipped": Image.Transpose.ROTATE_180,
|
|
"portrait": Image.Transpose.ROTATE_90,
|
|
"portrait_flipped": Image.Transpose.ROTATE_270,
|
|
}
|
|
|
|
|
|
def logical_render_size(orientation: str) -> tuple[int, int]:
|
|
"""(width, height) the photo is composed/cropped at for this
|
|
orientation, before rotating into native panel space."""
|
|
if orientation in ("portrait", "portrait_flipped"):
|
|
return EPD_HEIGHT, EPD_WIDTH
|
|
return EPD_WIDTH, EPD_HEIGHT
|
|
|
|
|
|
def logical_to_native(x: float, y: float, orientation: str) -> tuple[int, int]:
|
|
"""Maps a point in logical (pre-rotation) frame space to native
|
|
800x480 panel space, applying the same rotation ORIENTATION_TRANSPOSE
|
|
applies to the pixels -- anything positioned in logical coordinates
|
|
(e.g. face labels) needs this to stay attached to the rotated
|
|
content. PIL's ROTATE_90 is counterclockwise; ROTATE_270 clockwise."""
|
|
logical_w, logical_h = logical_render_size(orientation)
|
|
if orientation == "landscape_flipped":
|
|
return int(logical_w - 1 - x), int(logical_h - 1 - y)
|
|
if orientation == "portrait": # ROTATE_90 (CCW)
|
|
return int(y), int(logical_w - 1 - x)
|
|
if orientation == "portrait_flipped": # ROTATE_270 (CW)
|
|
return int(logical_h - 1 - y), int(x)
|
|
return int(x), int(y)
|
|
|
|
# Approximate sRGB for each of the panel's 6 ink colors -- reasonable
|
|
# placeholders, not measured values (Waveshare doesn't publish exact
|
|
# color primaries for this panel). This is the fallback for any frame
|
|
# that hasn't tuned its own (Frame.palette_rgb, set from a frame's
|
|
# Configuration tab -- "Advanced configuration" -- once you can compare
|
|
# a rendered test image against the real panel; different panel units
|
|
# can vary enough to be worth calibrating per frame).
|
|
DEFAULT_PALETTE_RGB = [
|
|
(0, 0, 0), # BLACK
|
|
(255, 255, 255), # WHITE
|
|
(255, 219, 0), # YELLOW
|
|
(207, 0, 15), # RED
|
|
(0, 39, 133), # BLUE
|
|
(0, 133, 55), # GREEN
|
|
]
|
|
|
|
PALETTE_LABELS = ["Black", "White", "Yellow", "Red", "Blue", "Green"]
|
|
|
|
# The panel's actual 4-bit color codes (see firmware/components/epd7in3e),
|
|
# in the same order as DEFAULT_PALETTE_RGB/PALETTE_LABELS -- fixed by the
|
|
# hardware protocol, never user-configurable. 0x4 is intentionally unused
|
|
# upstream.
|
|
PANEL_CODES = [0x0, 0x1, 0x2, 0x3, 0x5, 0x6]
|
|
|
|
|
|
def palette_to_hex(palette_rgb: list) -> list[str]:
|
|
"""[(0,0,0), ...] -> ["#000000", ...], for pre-filling the Advanced
|
|
configuration color pickers."""
|
|
return ["#%02x%02x%02x" % tuple(c) for c in palette_rgb]
|
|
|
|
|
|
def hex_to_rgb(hex_str: str) -> tuple[int, int, int] | None:
|
|
""""#1a2b3c" -> (26, 43, 60), or None for anything that isn't exactly
|
|
a 6-hex-digit color (what <input type="color"> always sends, but a
|
|
direct API call might not)."""
|
|
hex_str = hex_str.strip().lstrip("#")
|
|
if len(hex_str) != 6:
|
|
return None
|
|
try:
|
|
return (int(hex_str[0:2], 16), int(hex_str[2:4], 16), int(hex_str[4:6], 16))
|
|
except ValueError:
|
|
return None
|
|
|
|
|
|
def _build_palette_image(palette_rgb: list) -> Image.Image:
|
|
pal_img = Image.new("P", (1, 1))
|
|
pal_img.putpalette([channel for rgb in palette_rgb for channel in rgb])
|
|
return pal_img
|
|
|
|
|
|
def _plain_center_crop_box(
|
|
img_width: int, img_height: int, target_width: int, target_height: int
|
|
) -> tuple[float, float, int, int]:
|
|
"""The largest target_width:target_height window centered in the
|
|
source image -- the same box ImageOps.fit() computes internally when
|
|
there's no face-aware shift to apply. Returns (left, top, crop_w,
|
|
crop_h); left/top are floats (not yet rounded) since callers that go
|
|
on to face-shift this box need the unrounded center point."""
|
|
target_ratio = target_width / target_height
|
|
if img_width / img_height > target_ratio:
|
|
crop_h = img_height
|
|
crop_w = int(crop_h * target_ratio)
|
|
else:
|
|
crop_w = img_width
|
|
crop_h = int(crop_w / target_ratio)
|
|
|
|
left = (img_width - crop_w) / 2
|
|
top = (img_height - crop_h) / 2
|
|
return left, top, crop_w, crop_h
|
|
|
|
|
|
def _face_aware_crop_box(
|
|
img_width: int, img_height: int, target_width: int, target_height: int, faces: list[dict]
|
|
) -> tuple[int, int, int, int]:
|
|
"""Largest crop window matching target_width:target_height that fits
|
|
inside the source image. Starts from the plain center crop and only
|
|
shifts it the minimum amount needed to bring any faces that would
|
|
otherwise be cut off back on screen -- an already-fine composition
|
|
(faces already fully inside the center crop) is left untouched rather
|
|
than re-centered on the faces. If the faces themselves span wider than
|
|
the crop window allows, centers on their midpoint as best-effort,
|
|
since there's no shift that fits them all regardless.
|
|
|
|
Each face's box is given relative to its own imageWidth/imageHeight
|
|
(the resolution Immich ran detection on), which may differ from the
|
|
downloaded preview's resolution passed in here, so each box is scaled
|
|
into img_width/img_height space before use.
|
|
"""
|
|
min_x = min_y = float("inf")
|
|
max_x = max_y = float("-inf")
|
|
for face in faces:
|
|
face_w = face.get("imageWidth") or img_width
|
|
face_h = face.get("imageHeight") or img_height
|
|
scale_x = img_width / face_w
|
|
scale_y = img_height / face_h
|
|
min_x = min(min_x, face["boundingBoxX1"] * scale_x)
|
|
max_x = max(max_x, face["boundingBoxX2"] * scale_x)
|
|
min_y = min(min_y, face["boundingBoxY1"] * scale_y)
|
|
max_y = max(max_y, face["boundingBoxY2"] * scale_y)
|
|
|
|
left, top, crop_w, crop_h = _plain_center_crop_box(img_width, img_height, target_width, target_height)
|
|
|
|
if max_x - min_x <= crop_w:
|
|
if min_x < left:
|
|
left = min_x
|
|
elif max_x > left + crop_w:
|
|
left = max_x - crop_w
|
|
else:
|
|
left = (min_x + max_x) / 2 - crop_w / 2
|
|
|
|
if max_y - min_y <= crop_h:
|
|
if min_y < top:
|
|
top = min_y
|
|
elif max_y > top + crop_h:
|
|
top = max_y - crop_h
|
|
else:
|
|
top = (min_y + max_y) / 2 - crop_h / 2
|
|
|
|
left = max(0, min(left, img_width - crop_w))
|
|
top = max(0, min(top, img_height - crop_h))
|
|
|
|
return (int(left), int(top), int(left) + crop_w, int(top) + crop_h)
|
|
|
|
|
|
# Display modes: how a photo's aspect ratio gets reconciled with the
|
|
# panel's. "crop_faces" falls back to "crop_fill" behavior when no faces
|
|
# were detected/passed. DEFAULT_DISPLAY_MODE matches this project's old
|
|
# always-on smart_crop_faces=True default.
|
|
DISPLAY_MODES = ["crop_fill", "crop_faces", "stretch_fill", "letterbox"]
|
|
DISPLAY_MODE_LABELS = {
|
|
"crop_fill": "Crop to fill",
|
|
"crop_faces": "Crop to faces",
|
|
"stretch_fill": "Stretch to fill",
|
|
"letterbox": "Shrink to fit",
|
|
}
|
|
DEFAULT_DISPLAY_MODE = "crop_faces"
|
|
LETTERBOX_BG = (255, 255, 255)
|
|
|
|
|
|
def _placement_transform(
|
|
img_width: int, img_height: int, target_w: int, target_h: int,
|
|
display_mode: str, faces: list[dict] | None = None,
|
|
) -> tuple[float, float, float, float]:
|
|
"""Returns (scale_x, scale_y, offset_x, offset_y) mapping a point in
|
|
source-image pixel space to a point in target logical space, for the
|
|
given display_mode. Shared by render_frame (which also does the
|
|
actual pixel crop/resize/pad) and face_labels.py (label position
|
|
math) -- they must stay in exact agreement or overlay labels drift
|
|
off the people they're meant to point at."""
|
|
if display_mode == "stretch_fill":
|
|
return target_w / img_width, target_h / img_height, 0.0, 0.0
|
|
if display_mode == "letterbox":
|
|
scale = min(target_w / img_width, target_h / img_height)
|
|
return scale, scale, (target_w - img_width * scale) / 2, (target_h - img_height * scale) / 2
|
|
if display_mode == "crop_faces" and faces:
|
|
left, top, right, bottom = _face_aware_crop_box(img_width, img_height, target_w, target_h, faces)
|
|
crop_w, crop_h = right - left, bottom - top
|
|
else:
|
|
left, top, crop_w, crop_h = _plain_center_crop_box(img_width, img_height, target_w, target_h)
|
|
scale_x, scale_y = target_w / crop_w, target_h / crop_h
|
|
return scale_x, scale_y, -left * scale_x, -top * scale_y
|
|
|
|
|
|
def _compose(source: Image.Image, faces: list[dict] | None, orientation: str, display_mode: str) -> Image.Image:
|
|
"""Crop/resize/letterbox `source` per display_mode -- returns an RGB
|
|
image at logical_render_size(orientation), before enhancement or
|
|
quantization. See render_frame for what each display_mode does."""
|
|
logical_w, logical_h = logical_render_size(orientation)
|
|
fitted = ImageOps.exif_transpose(source.convert("RGB"))
|
|
|
|
if display_mode == "stretch_fill":
|
|
return fitted.resize((logical_w, logical_h), Image.LANCZOS)
|
|
if display_mode == "letterbox":
|
|
scale = min(logical_w / fitted.width, logical_h / fitted.height)
|
|
new_w, new_h = max(1, round(fitted.width * scale)), max(1, round(fitted.height * scale))
|
|
resized = fitted.resize((new_w, new_h), Image.LANCZOS)
|
|
canvas = Image.new("RGB", (logical_w, logical_h), LETTERBOX_BG)
|
|
canvas.paste(resized, ((logical_w - new_w) // 2, (logical_h - new_h) // 2))
|
|
return canvas
|
|
if display_mode == "crop_faces" and faces:
|
|
box = _face_aware_crop_box(fitted.width, fitted.height, logical_w, logical_h, faces)
|
|
return fitted.crop(box).resize((logical_w, logical_h), Image.LANCZOS)
|
|
return ImageOps.fit(fitted, (logical_w, logical_h), method=Image.LANCZOS) # crop_fill, or crop_faces w/ no faces
|
|
|
|
|
|
def _enhance(img: Image.Image, color_boost: float, contrast_boost: float) -> Image.Image:
|
|
if color_boost != 1.0:
|
|
img = ImageEnhance.Color(img).enhance(color_boost)
|
|
if contrast_boost != 1.0:
|
|
img = ImageEnhance.Contrast(img).enhance(contrast_boost)
|
|
return img
|
|
|
|
|
|
def _quantize(img: Image.Image, palette_rgb: list | None, dither_strength: float) -> Image.Image:
|
|
"""RGB -> palette-quantized P-mode image, same size/orientation as
|
|
`img` (no rotation here). dither_strength blends `img` toward its own
|
|
flat (undithered) quantization before running Floyd-Steinberg on the
|
|
blend: at 0 there's zero quantization error left to diffuse (so the
|
|
result IS the flat quantization, no dithering texture at all); at 1
|
|
it's `img` unchanged (full-strength dithering, this project's
|
|
original always-on behavior); values between give a smooth continuum
|
|
of dithering intensity rather than an on/off toggle."""
|
|
palette_image = _build_palette_image(palette_rgb or DEFAULT_PALETTE_RGB)
|
|
if dither_strength >= 1.0:
|
|
return img.quantize(palette=palette_image, dither=Image.Dither.FLOYDSTEINBERG)
|
|
if dither_strength <= 0.0:
|
|
return img.quantize(palette=palette_image, dither=Image.Dither.NONE)
|
|
flat = img.quantize(palette=palette_image, dither=Image.Dither.NONE).convert("RGB")
|
|
blended = Image.blend(flat, img, dither_strength)
|
|
return blended.quantize(palette=palette_image, dither=Image.Dither.FLOYDSTEINBERG)
|
|
|
|
|
|
def _transpose_and_pack(quantized: Image.Image, orientation: str) -> bytes:
|
|
"""Rotates a logical-space quantized image into native panel space
|
|
and packs it 2 pixels/byte the way epd7in3e.c expects. Always
|
|
returns exactly EPD_WIDTH*EPD_HEIGHT/2 bytes."""
|
|
transpose = ORIENTATION_TRANSPOSE.get(orientation)
|
|
if transpose is not None:
|
|
quantized = quantized.transpose(transpose)
|
|
pixels = quantized.load()
|
|
|
|
out = bytearray(EPD_WIDTH * EPD_HEIGHT // 2)
|
|
i = 0
|
|
for y in range(EPD_HEIGHT):
|
|
for x in range(0, EPD_WIDTH, 2):
|
|
left = PANEL_CODES[pixels[x, y]]
|
|
right = PANEL_CODES[pixels[x + 1, y]]
|
|
out[i] = (left << 4) | right
|
|
i += 1
|
|
|
|
return bytes(out)
|
|
|
|
|
|
def render_frame(source: Image.Image, faces: list[dict] | None = None,
|
|
orientation: str = "landscape", palette_rgb: list | None = None,
|
|
display_mode: str = DEFAULT_DISPLAY_MODE, color_boost: float = 1.0,
|
|
contrast_boost: float = 1.0, dither_strength: float = 1.0) -> bytes:
|
|
"""Fits `source` to the panel's resolution, applies color/contrast
|
|
enhancement, quantizes it to the 6-color palette, and packs 2
|
|
pixels/byte the way epd7in3e.c expects. Always returns exactly
|
|
EPD_WIDTH*EPD_HEIGHT/2 bytes.
|
|
|
|
`display_mode` (see DISPLAY_MODES) picks how the photo's aspect ratio
|
|
is reconciled with the panel's: crop_fill (center-crop to fill,
|
|
excess trimmed), crop_faces (as crop_fill, but shifts the crop to
|
|
keep `faces` on screen -- falls back to crop_fill if none), stretch_fill
|
|
(fills exactly, aspect ratio not preserved), letterbox (whole photo
|
|
visible, letterboxed with LETTERBOX_BG where it doesn't fill).
|
|
|
|
`color_boost`/`contrast_boost` are PIL ImageEnhance factors (1.0 =
|
|
unchanged, matching PIL's own convention); `dither_strength` is
|
|
0.0-1.0 (see _quantize).
|
|
|
|
`orientation` (see ORIENTATION_TRANSPOSE) composes the photo for how
|
|
the frame physically hangs, then rotates into native panel space --
|
|
the output byte layout is identical either way.
|
|
|
|
`palette_rgb` overrides DEFAULT_PALETTE_RGB (a frame's tuned colors,
|
|
see Frame.palette_rgb) -- None uses the default.
|
|
"""
|
|
fitted = _enhance(_compose(source, faces, orientation, display_mode), color_boost, contrast_boost)
|
|
quantized = _quantize(fitted, palette_rgb, dither_strength)
|
|
return _transpose_and_pack(quantized, orientation)
|
|
|
|
|
|
def render_preview_png(source: Image.Image, faces: list[dict] | None = None,
|
|
orientation: str = "landscape", palette_rgb: list | None = None,
|
|
display_mode: str = DEFAULT_DISPLAY_MODE, color_boost: float = 1.0,
|
|
contrast_boost: float = 1.0, dither_strength: float = 1.0) -> bytes:
|
|
"""Identical composition/enhancement/quantization pipeline as
|
|
render_frame, but returned as a normal browser-viewable PNG in
|
|
logical (upright, as-the-frame-actually-hangs) orientation rather
|
|
than packed native-panel bytes and rotation -- what the web UI's
|
|
"how it will look on the frame" preview shows."""
|
|
fitted = _enhance(_compose(source, faces, orientation, display_mode), color_boost, contrast_boost)
|
|
quantized = _quantize(fitted, palette_rgb, dither_strength)
|
|
buf = io.BytesIO()
|
|
quantized.convert("RGB").save(buf, format="PNG")
|
|
return buf.getvalue()
|
|
|
|
|
|
def render_placeholder(lines: list[str], qr_url: str | None = None,
|
|
orientation: str = "landscape", palette_rgb: list | None = None) -> bytes:
|
|
"""A readable full-panel message (plus an optional QR code) in the
|
|
same packed format as render_frame -- what /frame/image serves for a
|
|
frame that isn't claimed or configured yet, so a fresh device shows
|
|
instructions instead of an error screen and never error-loops."""
|
|
from PIL import ImageDraw, ImageFont
|
|
|
|
logical_w, logical_h = logical_render_size(orientation)
|
|
img = Image.new("RGB", (logical_w, logical_h), (255, 255, 255))
|
|
draw = ImageDraw.Draw(img)
|
|
|
|
title_font = ImageFont.load_default(size=34)
|
|
body_font = ImageFont.load_default(size=24)
|
|
|
|
qr_img = None
|
|
if qr_url:
|
|
import qrcode
|
|
|
|
qr = qrcode.QRCode(border=1, box_size=1)
|
|
qr.add_data(qr_url)
|
|
qr.make(fit=True)
|
|
raw = qr.make_image().get_image().convert("RGB")
|
|
# Integer upscale with NEAREST keeps modules crisp on the panel.
|
|
target = 220
|
|
scale = max(1, target // raw.width)
|
|
qr_img = raw.resize((raw.width * scale, raw.height * scale), Image.NEAREST)
|
|
|
|
# Vertical layout: text block, then QR under it, centered as a group.
|
|
line_heights = []
|
|
for i, line in enumerate(lines):
|
|
font = title_font if i == 0 else body_font
|
|
bbox = draw.textbbox((0, 0), line, font=font)
|
|
line_heights.append((line, font, bbox[2] - bbox[0], bbox[3] - bbox[1]))
|
|
gap = 14
|
|
text_h = sum(h for _, _, _, h in line_heights) + gap * (len(line_heights) - 1 if line_heights else 0)
|
|
total_h = text_h + (qr_img.height + 28 if qr_img else 0)
|
|
y = max(20, (logical_h - total_h) // 2)
|
|
|
|
for line, font, w, h in line_heights:
|
|
draw.text(((logical_w - w) // 2, y), line, fill=(0, 0, 0), font=font)
|
|
y += h + gap
|
|
|
|
if qr_img:
|
|
img.paste(qr_img, ((logical_w - qr_img.width) // 2, y + 14))
|
|
|
|
quantized = _quantize(img, palette_rgb, dither_strength=1.0)
|
|
return _transpose_and_pack(quantized, orientation)
|