361 files
This commit is contained in:
@@ -2,8 +2,10 @@ from __future__ import annotations
|
||||
|
||||
import io
|
||||
import logging
|
||||
from dataclasses import dataclass
|
||||
from typing import NamedTuple
|
||||
|
||||
from PIL import Image, ImageColor, ImageFilter, ImageOps
|
||||
from PIL import Image, ImageColor, ImageDraw, ImageFilter, ImageFont, ImageOps
|
||||
|
||||
from .coordinator import MediaItem
|
||||
|
||||
@@ -14,6 +16,42 @@ FILL_COVER = "cover"
|
||||
FILL_CONTAIN = "contain"
|
||||
FILL_BLUR = "blur"
|
||||
|
||||
class FaceBox(NamedTuple):
|
||||
left: float
|
||||
top: float
|
||||
right: float
|
||||
bottom: float
|
||||
weight: float
|
||||
selected: bool = False
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class CropHints:
|
||||
"""Per-photo inputs for face-aware cropping and the debug overlay.
|
||||
|
||||
``faces`` is None when no face data is known for the photo (not an
|
||||
Immich source, or not scanned yet) and an empty tuple when the photo
|
||||
was scanned and has no faces.
|
||||
"""
|
||||
|
||||
faces: tuple[FaceBox, ...] | None = None
|
||||
debug: bool = False
|
||||
name: str | None = None
|
||||
|
||||
|
||||
# Padding around a face box, as a fraction of the face's own size, so the
|
||||
# crop keeps the whole head rather than just the tight face box.
|
||||
_FACE_PAD_SIDE = 0.3
|
||||
_FACE_PAD_TOP = 0.6
|
||||
_FACE_PAD_BOTTOM = 0.3
|
||||
# A face that is cut by the crop edge counts this much worse than one that is
|
||||
# left out entirely: half a face looks like a mistake, a missing face doesn't.
|
||||
_PARTIAL_PENALTY = 1.5
|
||||
|
||||
FACE_KEPT = "kept"
|
||||
FACE_CUT = "cut"
|
||||
FACE_DROPPED = "dropped"
|
||||
|
||||
# Absolute pixel ceiling. A 20000x20000 JPEG decodes to ~1.2 GB of RGB; Pillow
|
||||
# raises DecompressionBombError above MAX_IMAGE_PIXELS. We set this high enough
|
||||
# that 4K+ sources still decode, but reject anything absurd to protect
|
||||
@@ -120,13 +158,20 @@ def resolve_output_size(
|
||||
return (width, height)
|
||||
|
||||
|
||||
def render_image(img: Image.Image, fill_mode: str, width: int, height: int) -> Image.Image:
|
||||
def render_image(
|
||||
img: Image.Image,
|
||||
fill_mode: str,
|
||||
width: int,
|
||||
height: int,
|
||||
hints: CropHints | None = None,
|
||||
) -> Image.Image:
|
||||
"""Render img into a (width x height) canvas using the given fill mode."""
|
||||
hints = hints or CropHints()
|
||||
if fill_mode == FILL_CONTAIN:
|
||||
return _resize_contain(img, width, height)
|
||||
return _resize_contain(img, width, height, hints=hints)
|
||||
if fill_mode == FILL_BLUR:
|
||||
return _blur_fill(img, width, height)
|
||||
return _resize_cover(img, width, height)
|
||||
return _blur_fill(img, width, height, hints=hints)
|
||||
return _resize_cover(img, width, height, hints)
|
||||
|
||||
|
||||
def pair_images(
|
||||
@@ -139,6 +184,8 @@ def pair_images(
|
||||
divider: int,
|
||||
divider_fill: tuple[int, int, int] | tuple[int, int, int, int],
|
||||
transparent_divider: bool,
|
||||
hints1: CropHints | None = None,
|
||||
hints2: CropHints | None = None,
|
||||
) -> Image.Image:
|
||||
canvas_mode = "RGBA" if transparent_divider else "RGB"
|
||||
canvas = Image.new(canvas_mode, (target_w, target_h), divider_fill)
|
||||
@@ -146,8 +193,8 @@ def pair_images(
|
||||
if portrait_canvas:
|
||||
top_h = max(1, (target_h - divider) // 2)
|
||||
bottom_h = max(1, target_h - divider - top_h)
|
||||
top_img = render_image(img1, fill_mode, target_w, top_h)
|
||||
bottom_img = render_image(img2, fill_mode, target_w, bottom_h)
|
||||
top_img = render_image(img1, fill_mode, target_w, top_h, hints1)
|
||||
bottom_img = render_image(img2, fill_mode, target_w, bottom_h, hints2)
|
||||
canvas.paste(top_img.convert(canvas_mode), (0, 0))
|
||||
canvas.paste(bottom_img.convert(canvas_mode), (0, top_h + divider))
|
||||
safe_close(top_img)
|
||||
@@ -156,8 +203,8 @@ def pair_images(
|
||||
|
||||
left_w = max(1, (target_w - divider) // 2)
|
||||
right_w = max(1, target_w - divider - left_w)
|
||||
left_img = render_image(img1, fill_mode, left_w, target_h)
|
||||
right_img = render_image(img2, fill_mode, right_w, target_h)
|
||||
left_img = render_image(img1, fill_mode, left_w, target_h, hints1)
|
||||
right_img = render_image(img2, fill_mode, right_w, target_h, hints2)
|
||||
canvas.paste(left_img.convert(canvas_mode), (0, 0))
|
||||
canvas.paste(right_img.convert(canvas_mode), (left_w + divider, 0))
|
||||
safe_close(left_img)
|
||||
@@ -252,7 +299,131 @@ def _parse_aspect_ratio(ratio: str) -> tuple[int, int]:
|
||||
return (16, 9)
|
||||
|
||||
|
||||
def _resize_cover(img: Image.Image, target_w: int, target_h: int) -> Image.Image:
|
||||
def _padded_face(face: FaceBox) -> tuple[float, float, float, float]:
|
||||
x1, y1, x2, y2 = face[:4]
|
||||
fw, fh = x2 - x1, y2 - y1
|
||||
return (
|
||||
max(0.0, x1 - fw * _FACE_PAD_SIDE),
|
||||
max(0.0, y1 - fh * _FACE_PAD_TOP),
|
||||
min(1.0, x2 + fw * _FACE_PAD_SIDE),
|
||||
min(1.0, y2 + fh * _FACE_PAD_BOTTOM),
|
||||
)
|
||||
|
||||
|
||||
def _span_status(a: float, b: float, offset: float, window: float) -> str:
|
||||
# One pixel of slack absorbs rounding when the image is scaled to fit
|
||||
# an axis exactly.
|
||||
if a >= offset - 1 and b <= offset + window + 1:
|
||||
return FACE_KEPT
|
||||
if b <= offset or a >= offset + window:
|
||||
return FACE_DROPPED
|
||||
return FACE_CUT
|
||||
|
||||
|
||||
def choose_crop_offset(
|
||||
spans: list[tuple[float, float, float]],
|
||||
src_len: float,
|
||||
window: float,
|
||||
*,
|
||||
selected: list[bool] | None = None,
|
||||
padded_spans: list[tuple[float, float]] | None = None,
|
||||
) -> float:
|
||||
"""Pick where a crop window of ``window`` px starts along one axis.
|
||||
|
||||
``spans`` are unpadded ``(start, end, weight)`` face intervals. Selected
|
||||
people rank ahead of all bystanders. Keep whole faces where possible,
|
||||
retain visible face area otherwise, and use padding as a preference.
|
||||
"""
|
||||
max_offset = max(0.0, src_len - window)
|
||||
centred = max_offset / 2
|
||||
if max_offset <= 0 or not spans:
|
||||
return centred
|
||||
|
||||
def clamp(value: float) -> float:
|
||||
return max(0.0, min(max_offset, value))
|
||||
|
||||
priorities = selected if selected is not None else [False] * len(spans)
|
||||
padding = padded_spans if padded_spans is not None else [span[:2] for span in spans]
|
||||
|
||||
def visible_fraction(start: float, end: float, offset: float) -> float:
|
||||
overlap = max(0.0, min(end, offset + window) - max(start, offset))
|
||||
return overlap / (end - start)
|
||||
|
||||
def score(offset: float) -> tuple[float, ...]:
|
||||
tiers = []
|
||||
for priority in (True, False):
|
||||
kept_weight = cut_weight = visible_weight = 0.0
|
||||
for (start, end, weight), is_selected in zip(spans, priorities):
|
||||
if is_selected != priority:
|
||||
continue
|
||||
state = _span_status(start, end, offset, window)
|
||||
if state == FACE_KEPT:
|
||||
kept_weight += weight
|
||||
elif state == FACE_CUT:
|
||||
cut_weight += weight
|
||||
visible_weight += weight * visible_fraction(start, end, offset)
|
||||
tiers.extend((
|
||||
kept_weight,
|
||||
-_PARTIAL_PENALTY * cut_weight if kept_weight else visible_weight,
|
||||
))
|
||||
tiers.append(sum(
|
||||
span[2] * visible_fraction(start, end, offset)
|
||||
for span, (start, end) in zip(spans, padding)
|
||||
))
|
||||
return tuple(round(value, 12) for value in tiers)
|
||||
|
||||
candidates = {0.0, centred, max_offset}
|
||||
for start, end in [span[:2] for span in spans] + padding:
|
||||
for offset in (start, end - window, end, start - window, (start + end - window) / 2):
|
||||
candidates.add(clamp(offset))
|
||||
best = max(candidates, key=lambda offset: (score(offset), -abs(offset - centred), -offset))
|
||||
|
||||
kept = [
|
||||
index for index, (start, end, _weight) in enumerate(spans)
|
||||
if _span_status(start, end, best, window) == FACE_KEPT
|
||||
]
|
||||
if not kept:
|
||||
return best
|
||||
group_start = min(padding[index][0] for index in kept)
|
||||
group_end = max(padding[index][1] for index in kept)
|
||||
if group_end - group_start > window:
|
||||
group_start = min(spans[index][0] for index in kept)
|
||||
group_end = max(spans[index][1] for index in kept)
|
||||
target = clamp((group_start + group_end - window) / 2)
|
||||
candidates.add(target)
|
||||
return max(candidates, key=lambda offset: (score(offset), -abs(offset - target), -offset))
|
||||
|
||||
|
||||
def _face_statuses(
|
||||
faces: tuple[FaceBox, ...],
|
||||
new_w: int,
|
||||
new_h: int,
|
||||
left: int,
|
||||
top: int,
|
||||
target_w: int,
|
||||
target_h: int,
|
||||
) -> list[str]:
|
||||
statuses = []
|
||||
for face in faces:
|
||||
px1, py1, px2, py2 = face[:4]
|
||||
x = _span_status(px1 * new_w, px2 * new_w, left, target_w)
|
||||
y = _span_status(py1 * new_h, py2 * new_h, top, target_h)
|
||||
if FACE_DROPPED in (x, y):
|
||||
statuses.append(FACE_DROPPED)
|
||||
elif FACE_CUT in (x, y):
|
||||
statuses.append(FACE_CUT)
|
||||
else:
|
||||
statuses.append(FACE_KEPT)
|
||||
return statuses
|
||||
|
||||
|
||||
def _resize_cover(
|
||||
img: Image.Image,
|
||||
target_w: int,
|
||||
target_h: int,
|
||||
hints: CropHints | None = None,
|
||||
) -> Image.Image:
|
||||
hints = hints or CropHints()
|
||||
src_w, src_h = img.size
|
||||
if src_w <= 0 or src_h <= 0:
|
||||
return img.resize((target_w, target_h))
|
||||
@@ -260,15 +431,137 @@ def _resize_cover(img: Image.Image, target_w: int, target_h: int) -> Image.Image
|
||||
new_w = max(1, int(round(src_w * scale)))
|
||||
new_h = max(1, int(round(src_h * scale)))
|
||||
resized = img.resize((new_w, new_h), Image.Resampling.LANCZOS)
|
||||
left = max(0, int(round((new_w - target_w) / 2)))
|
||||
top = max(0, int(round((new_h - target_h) / 2)))
|
||||
|
||||
faces = tuple(FaceBox(*face) for face in hints.faces or ())
|
||||
padded = [_padded_face(face) for face in faces]
|
||||
left = choose_crop_offset(
|
||||
[(face.left * new_w, face.right * new_w, face.weight) for face in faces],
|
||||
new_w,
|
||||
target_w,
|
||||
selected=[face.selected for face in faces],
|
||||
padded_spans=[(box[0] * new_w, box[2] * new_w) for box in padded],
|
||||
)
|
||||
top = choose_crop_offset(
|
||||
[(face.top * new_h, face.bottom * new_h, face.weight) for face in faces],
|
||||
new_h,
|
||||
target_h,
|
||||
selected=[face.selected for face in faces],
|
||||
padded_spans=[(box[1] * new_h, box[3] * new_h) for box in padded],
|
||||
)
|
||||
left = max(0, min(max(0, new_w - target_w), int(round(left))))
|
||||
top = max(0, min(max(0, new_h - target_h), int(round(top))))
|
||||
|
||||
statuses = _face_statuses(faces, new_w, new_h, left, top, target_w, target_h)
|
||||
summary = _face_summary(hints.faces, statuses)
|
||||
if faces:
|
||||
_LOGGER.debug("Crop %s: %s", hints.name or "photo", summary)
|
||||
if hints.debug:
|
||||
_draw_debug_marks(resized, faces, statuses)
|
||||
|
||||
cropped = resized.crop((left, top, left + target_w, top + target_h))
|
||||
if cropped is not resized:
|
||||
safe_close(resized)
|
||||
if hints.debug:
|
||||
_draw_offscreen_centre(cropped, new_w / 2 - left, new_h / 2 - top)
|
||||
_draw_debug_label(cropped, summary)
|
||||
return cropped
|
||||
|
||||
|
||||
def _resize_contain(img: Image.Image, target_w: int, target_h: int, bg=(0, 0, 0)) -> Image.Image:
|
||||
def _face_summary(faces: tuple[FaceBox, ...] | None, statuses: list[str]) -> str:
|
||||
if faces is None:
|
||||
return "no face data"
|
||||
return (
|
||||
f"faces {len(statuses)} · kept {statuses.count(FACE_KEPT)} · "
|
||||
f"cut {statuses.count(FACE_CUT)} · dropped {statuses.count(FACE_DROPPED)}"
|
||||
)
|
||||
|
||||
|
||||
_DEBUG_COLORS = {
|
||||
FACE_KEPT: (0, 220, 0),
|
||||
FACE_CUT: (255, 40, 40),
|
||||
FACE_DROPPED: (160, 160, 160),
|
||||
}
|
||||
_CROSSHAIR_COLOR = (255, 220, 0)
|
||||
|
||||
|
||||
def _draw_debug_marks(
|
||||
img: Image.Image, faces: tuple[FaceBox, ...], statuses: list[str]
|
||||
) -> None:
|
||||
"""Draw face boxes and a crosshair on the photo's own centre.
|
||||
|
||||
Drawn before cropping, so the marks get cut exactly like the photo: a
|
||||
crosshair that is off-centre or missing shows how much was cropped away.
|
||||
"""
|
||||
w, h = img.size
|
||||
draw = ImageDraw.Draw(img)
|
||||
line = max(2, round(min(w, h) / 200))
|
||||
for face, status in zip(faces, statuses):
|
||||
color = _DEBUG_COLORS[status]
|
||||
x1, y1, x2, y2 = face[:4]
|
||||
draw.rectangle((x1 * w, y1 * h, x2 * w, y2 * h), outline=color, width=line)
|
||||
px1, py1, px2, py2 = _padded_face(face)
|
||||
draw.rectangle(
|
||||
(px1 * w, py1 * h, px2 * w, py2 * h),
|
||||
outline=color,
|
||||
width=max(1, line // 2),
|
||||
)
|
||||
cx, cy = w / 2, h / 2
|
||||
arm = max(12, round(min(w, h) / 12))
|
||||
for width, color in ((line + 2, (0, 0, 0)), (line, _CROSSHAIR_COLOR)):
|
||||
draw.line((cx - arm, cy, cx + arm, cy), fill=color, width=width)
|
||||
draw.line((cx, cy - arm, cx, cy + arm), fill=color, width=width)
|
||||
radius = arm / 3
|
||||
draw.ellipse(
|
||||
(cx - radius, cy - radius, cx + radius, cy + radius),
|
||||
outline=_CROSSHAIR_COLOR,
|
||||
width=line,
|
||||
)
|
||||
|
||||
|
||||
def _draw_offscreen_centre(img: Image.Image, cx: float, cy: float) -> None:
|
||||
"""When the photo's centre was cropped away, point at it from the edge."""
|
||||
w, h = img.size
|
||||
if 0 <= cx <= w and 0 <= cy <= h:
|
||||
return
|
||||
size = max(10, round(min(w, h) / 20))
|
||||
x = max(size, min(w - size, cx))
|
||||
y = max(size, min(h - size, cy))
|
||||
if cx < 0:
|
||||
points = [(0, y), (size, y - size), (size, y + size)]
|
||||
elif cx > w:
|
||||
points = [(w, y), (w - size, y - size), (w - size, y + size)]
|
||||
elif cy < 0:
|
||||
points = [(x, 0), (x - size, size), (x + size, size)]
|
||||
else:
|
||||
points = [(x, h), (x - size, h - size), (x + size, h - size)]
|
||||
ImageDraw.Draw(img).polygon(points, fill=_CROSSHAIR_COLOR, outline=(0, 0, 0))
|
||||
|
||||
|
||||
def _debug_font(size: int):
|
||||
try:
|
||||
return ImageFont.load_default(size=size)
|
||||
except TypeError: # Pillow < 10.1 has a fixed-size default font
|
||||
return ImageFont.load_default()
|
||||
|
||||
|
||||
def _draw_debug_label(img: Image.Image, text: str) -> None:
|
||||
w, h = img.size
|
||||
draw = ImageDraw.Draw(img)
|
||||
size = max(12, round(min(w, h) / 30))
|
||||
font = _debug_font(size)
|
||||
pad = max(4, size // 3)
|
||||
x1, y1, x2, y2 = draw.textbbox((pad, pad), text, font=font)
|
||||
draw.rectangle((x1 - pad, y1 - pad, x2 + pad, y2 + pad), fill=(0, 0, 0))
|
||||
draw.text((pad, pad), text, fill=(255, 255, 255), font=font)
|
||||
|
||||
|
||||
def _resize_contain(
|
||||
img: Image.Image,
|
||||
target_w: int,
|
||||
target_h: int,
|
||||
bg=(0, 0, 0),
|
||||
hints: CropHints | None = None,
|
||||
) -> Image.Image:
|
||||
src_w, src_h = img.size
|
||||
if src_w <= 0 or src_h <= 0:
|
||||
return img.resize((target_w, target_h))
|
||||
@@ -276,6 +569,7 @@ def _resize_contain(img: Image.Image, target_w: int, target_h: int, bg=(0, 0, 0)
|
||||
new_w = max(1, int(src_w * scale))
|
||||
new_h = max(1, int(src_h * scale))
|
||||
resized = img.resize((new_w, new_h), Image.Resampling.LANCZOS)
|
||||
_debug_uncropped(resized, hints)
|
||||
canvas = Image.new("RGB", (target_w, target_h), bg)
|
||||
rgb_resized = resized if resized.mode == "RGB" else resized.convert("RGB")
|
||||
canvas.paste(rgb_resized, ((target_w - new_w) // 2, (target_h - new_h) // 2))
|
||||
@@ -285,7 +579,22 @@ def _resize_contain(img: Image.Image, target_w: int, target_h: int, bg=(0, 0, 0)
|
||||
return canvas
|
||||
|
||||
|
||||
def _blur_fill(img: Image.Image, target_w: int, target_h: int) -> Image.Image:
|
||||
def _debug_uncropped(img: Image.Image, hints: CropHints | None) -> None:
|
||||
"""Debug marks for fill modes that never crop: every face is kept."""
|
||||
if hints is None or not hints.debug:
|
||||
return
|
||||
faces = hints.faces or ()
|
||||
statuses = [FACE_KEPT] * len(faces)
|
||||
_draw_debug_marks(img, faces, statuses)
|
||||
_draw_debug_label(img, _face_summary(hints.faces, statuses))
|
||||
|
||||
|
||||
def _blur_fill(
|
||||
img: Image.Image,
|
||||
target_w: int,
|
||||
target_h: int,
|
||||
hints: CropHints | None = None,
|
||||
) -> Image.Image:
|
||||
bg = _resize_cover(img, target_w, target_h).filter(ImageFilter.GaussianBlur(radius=24))
|
||||
src_w, src_h = img.size
|
||||
if src_w <= 0 or src_h <= 0:
|
||||
@@ -294,10 +603,10 @@ def _blur_fill(img: Image.Image, target_w: int, target_h: int) -> Image.Image:
|
||||
new_w = max(1, int(src_w * scale))
|
||||
new_h = max(1, int(src_h * scale))
|
||||
fg = img.resize((new_w, new_h), Image.Resampling.LANCZOS)
|
||||
_debug_uncropped(fg, hints)
|
||||
rgb_fg = fg if fg.mode == "RGB" else fg.convert("RGB")
|
||||
bg.paste(rgb_fg, ((target_w - new_w) // 2, (target_h - new_h) // 2))
|
||||
if rgb_fg is not fg:
|
||||
safe_close(rgb_fg)
|
||||
safe_close(fg)
|
||||
return bg
|
||||
|
||||
|
||||
Reference in New Issue
Block a user