Files
wordcloud/backend/core/pipeline.py
T

1066 lines
46 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
import logging
import math
import os
import sqlite3
import sys
import time
from pathlib import Path
import numpy as np
from PIL import Image, ImageDraw, ImageFont
import pandas as pd
from . import config
from .fonts import get_cached_font
from .layout import OptimizedEfficientWordCloud
from .mask import analyze_mask, apply_safe_padding, calculate_dynamic_dimensions, prepare_mask
from .render import (
compute_coverage_score,
compute_fill_ratio_fast,
count_layout_overlap_pixels,
largest_empty_square_size,
append_layout_with_hd_clearance,
refine_layout_with_hd_clearance,
render_layout_occupancy,
scale_layout_for_hd,
)
from .weights import (
calculate_font_by_area_model,
extract_weights_from_df,
get_stroke_complexity_batch,
merge_weight_maps,
)
log = logging.getLogger("core.pipeline")
def run_generation_pass(
names,
frequencies_data,
name_weights_map,
mask_hd,
real_hd_w,
real_hd_h,
_collision_margin=0,
):
log.info("[run_generation_pass] 开始")
log.info(" 输入: %d 词 | HD尺寸: %dx%d", len(names), real_hd_w, real_hd_h)
w_small = max(1, int(real_hd_w * config.WORK_SCALE))
h_small = max(1, int(real_hd_h * config.WORK_SCALE))
img_small = Image.fromarray(mask_hd).resize((w_small, h_small), Image.NEAREST)
mask_small = np.array(img_small)
apply_safe_padding(mask_small)
log.info(" 运算网格: %dx%d (WORK_SCALE=%.4f)", w_small, h_small, config.WORK_SCALE)
log.info(" mask_small 统计: 总像素=%d, 空闲像素=%d, 空闲率=%.4f",
mask_small.size, int(np.sum(mask_small == 0)),
int(np.sum(mask_small == 0)) / mask_small.size if mask_small.size else 0)
if config.SAVE_DEBUG_IMAGES:
debug_dir = Path(config.DEBUG_OUTPUT_DIR)
debug_dir.mkdir(parents=True, exist_ok=True)
Image.fromarray(mask_hd).save(str(debug_dir / "mask_hd.png"))
Image.fromarray(mask_small).save(str(debug_dir / "mask_small.png"))
print(f"最终输出: {real_hd_w}x{real_hd_h} | 运算网格: {w_small}x{h_small}")
readable_min_font = max(
config.MIN_FONT_FLOOR,
int(math.ceil(config.MIN_READABLE_HEIGHT_PX * config.WORK_SCALE)),
)
total_target = len(names) * config.N_REPETITIONS
final_wc = None
final_scale = 1.0
base_layout_seed = config.LAYOUT_SEED if config.LAYOUT_SEED is not None else config.SEED
final_layout_seed = base_layout_seed
base_min_font, base_max_font = calculate_font_by_area_model(
mask_small,
names,
name_weights_map,
config.TARGET_FILL_RATIO,
config.SIZE_RATIO,
config.PACKING_EFFICIENCY,
config.N_REPETITIONS,
)
explicit_min_font = config.USER_MIN_FONT_SIZE is not None
hard_min_font = readable_min_font
if explicit_min_font:
hard_min_font = max(config.MIN_FONT_FLOOR, int(round(config.USER_MIN_FONT_SIZE)))
hard_max_font = None
if config.USER_MAX_FONT_SIZE is not None:
hard_max_font = max(config.MIN_FONT_FLOOR, int(round(config.USER_MAX_FONT_SIZE)))
if explicit_min_font and hard_max_font < hard_min_font:
raise ValueError(
f"字号硬约束冲突: USER_MAX_FONT_SIZE={hard_max_font} "
f"小于最小允许字号 {hard_min_font}"
)
if not explicit_min_font:
# A user-specified maximum outranks the automatic readability
# suggestion. It remains an exact ceiling rather than causing an
# artificial conflict with a value the user never requested.
hard_min_font = min(hard_min_font, hard_max_font)
equal_size_mode = math.isclose(float(config.SIZE_RATIO), 1.0, rel_tol=0.0, abs_tol=1e-9)
base_min_font = max(hard_min_font, int(base_min_font))
base_max_font = max(base_min_font, int(base_max_font))
if hard_max_font is not None:
base_min_font = min(base_min_font, hard_max_font)
base_max_font = min(base_max_font, hard_max_font)
if equal_size_mode:
# Keep one scalar throughout every retry. This is what makes
# SIZE_RATIO=1 exact even after automatic batch scaling.
equal_font = min(base_min_font, base_max_font)
base_min_font = equal_font
base_max_font = equal_font
def scaled_bounds(scale):
if equal_size_mode:
size = max(hard_min_font, int(round(base_min_font * scale)))
if hard_max_font is not None:
size = min(size, hard_max_font)
return size, size
min_font = max(hard_min_font, int(round(base_min_font * scale)))
max_font = max(min_font, int(round(base_max_font * scale)))
if hard_max_font is not None:
min_font = min(min_font, hard_max_font)
max_font = min(max_font, hard_max_font)
return min_font, max(min_font, max_font)
def try_place(scale, layout_seed=base_layout_seed, probe=False):
min_font, max_font = scaled_bounds(scale)
wc = OptimizedEfficientWordCloud(
width=w_small,
height=h_small,
mask=mask_small,
font_path=config.WC_FONT_PATH,
max_words=total_target,
min_font_size=min_font,
max_font_size=max_font,
background_color=config.get_output_background(),
# Each word independently draws horizontal vs vertical, so the mix
# is scattered rather than banded. VERTICAL_RATIO is the chance of
# a vertical word; mixing orientations is one of the cheapest ways
# to break up an over-regular grid-like look.
prefer_horizontal=1.0 - float(config.VERTICAL_RATIO),
# HD clearance is applied after scaling. Keeping the coarse-grid
# margin at zero avoids turning 1 HD pixel into 5-6 output pixels.
margin=_collision_margin,
)
wc.layout_seed = layout_seed
# A probe only needs to answer "does every word fit at this scale?", so
# it stops at the first word that cannot be placed. The answer is exact
# -- a word is only reported unplaced once an exhaustive scan has ruled
# out every position -- and it avoids paying for a doomed batch's
# remaining failures, each of which is far more expensive than a
# successful placement.
wc.max_failures = 1 if probe else None
wc.generate_from_frequencies(frequencies_data)
return wc, len(wc.layout_), min_font, max_font
print(f"--- 5. 启动生成 (目标: {total_target} 词) ---")
log.info("--- 5. 启动生成 ---")
log.info(" 目标词数: %d (names=%d * N_REPETITIONS=%d)", total_target, len(names), config.N_REPETITIONS)
log.info(
" 基准字号: [%d, %d], 硬边界: [%d, %s], 等字号=%s",
base_min_font,
base_max_font,
hard_min_font,
hard_max_font if hard_max_font is not None else "∞",
equal_size_mode,
)
# Every attempt is a fresh, whole-cloud layout. No word may silently
# receive a smaller fallback size. Usually the area model succeeds on
# attempt one; two adaptive retries cover fragmentation-heavy masks.
best_wc = None
best_count = 0
best_scale = 1.0
best_coverage = -1.0 # shape-aware coverage of the current best candidate
best_occ = None # occupancy raster of the current best candidate
tried_layouts = set()
failed_scales = []
attempt = 0
def probe_scale(scale):
"""Lay out every word at `scale`; return (wc, placed_count, complete)."""
nonlocal attempt, best_wc, best_count, best_scale
nonlocal best_coverage, best_occ
attempt += 1
bounds = scaled_bounds(scale)
tried_layouts.add((bounds, base_layout_seed))
wc, placed_count, min_font, max_font = try_place(scale, probe=True)
complete = placed_count >= total_target
print(
f" 整批布局 #{attempt}: scale={scale:.3f}, "
f"字号=[{min_font}, {max_font}] -> "
f"{'完整' if complete else '不足'} ({placed_count}/{total_target})"
)
log.info(
" 整批布局 #%d scale=%.3f 字号=[%d,%d] -> %d/%d complete=%s",
attempt, scale, min_font, max_font, placed_count, total_target, complete,
)
if complete:
# Among complete layouts prefer the one whose ink reaches furthest
# into the mask shape, not merely the largest font scale. A Fermat
# spiral packs words into a disc around the fillable centroid; with
# few words or large fonts that disc stays small and never reaches
# the mask's protrusions, so the cloud reads as a circle instead of
# the intended silhouette. Coverage rewards layouts that spread into
# those deep regions, letting a slightly smaller scale win when it
# trades font size for a recognisable outline. Scale is the tie-
# breaker so an equal-coverage denser picture is still preferred.
occ = render_layout_occupancy(wc.layout_, mask_small.shape, config.WC_FONT_PATH)
coverage = compute_coverage_score(occ, mask_small)
log.info(
" 整批布局 #%d coverage=%.4f (best=%.4f)",
attempt, coverage, best_coverage,
)
if coverage > best_coverage or (
coverage == best_coverage and scale > best_scale
):
best_wc, best_count, best_scale = wc, placed_count, scale
best_coverage, best_occ = coverage, occ
if not complete:
failed_scales.append(scale)
return wc, placed_count, complete
# Find the largest scale at which every word still fits. Bigger is strictly
# better here: the same names drawn larger leave less blank space. A probe
# answers feasibility exactly and stops at the first unplaceable word, so
# searching for the best scale costs little more than accepting the first
# one that happens to work.
lo = None # largest scale known to fit everything
hi = None # smallest scale known to be too big
scale = 1.0
for _ in range(2 if config.FAST_MODE else 4):
wc, placed_count, complete = probe_scale(scale)
if complete:
lo = scale
break
hi = scale
placed_ratio = placed_count / max(1, total_target)
# Area scales with size², so linear size scales with sqrt(ratio). The
# probe stops early, which understates how many words would have fit,
# so this deliberately undershoots and the bisection below climbs back.
shrink = 0.62 if placed_ratio <= 0 else min(0.92, max(0.55, math.sqrt(placed_ratio) * 0.92))
scale *= shrink
# Close the gap between the largest failing scale and the smallest passing
# one. Each step recovers font size that the shrink above gave away.
if lo is not None and hi is not None:
for _ in range(1 if config.FAST_MODE else 3):
mid = (lo + hi) / 2.0
if hi - lo < 0.02 or scaled_bounds(mid) == scaled_bounds(lo):
break
_wc, _placed, complete = probe_scale(mid)
if complete:
lo = mid
else:
hi = mid
if best_wc is not None:
final_wc = best_wc
final_scale = best_scale
final_layout_seed = base_layout_seed
if final_wc is None:
return {
"wc": None,
"fill_ratio": 0.0,
"coverage": 0.0,
"occ_fast": None,
"w_small": w_small,
"h_small": h_small,
"base_min_font": base_min_font,
"base_max_font": base_max_font,
"mask_small": mask_small,
"size_scale": final_scale,
"layout_seed": final_layout_seed,
"collision_margin": _collision_margin,
"hd_overlap_pixels": 0,
"hd_layout": [],
"hd_clearance": None,
}
fill_ratio, occ_fast = compute_fill_ratio_fast(final_wc.layout_, mask_small, config.WC_FONT_PATH)
largest_empty_square = largest_empty_square_size(occ_fast, mask_small)
def has_character_sized_hole(wc, hole_size):
if not equal_size_mode or wc is None or not wc.layout_:
return False
font_size = int(wc.layout_[0][1])
return hole_size >= max(2, int(math.ceil(font_size * 1.25)))
complete_candidates = []
if len(final_wc.layout_) >= total_target:
complete_candidates.append(
(final_wc, final_scale, final_layout_seed, fill_ratio, occ_fast)
)
print(f"填充率: {fill_ratio:.3f}")
log.info("[填充率] 初始填充率: %.4f", fill_ratio)
log.info(" layout_ 词数: %d", len(final_wc.layout_))
if config.SAVE_DEBUG_IMAGES and occ_fast is not None:
debug_dir = Path(config.DEBUG_OUTPUT_DIR)
debug_dir.mkdir(parents=True, exist_ok=True)
Image.fromarray((occ_fast * 255).astype(np.uint8)).save(str(debug_dir / "occ_fast.png"))
# Probe whole-cloud layouts in BOTH font-size directions and keep the one
# whose ink reaches furthest into the mask shape. The original search only
# grew the font (chasing a higher pixel fill ratio), but a Fermat spiral
# packs words into a disc around the centroid: growing the font shrinks that
# disc, so an under-filled silhouette gets *more* circular, not less.
# Shrinking the font lets the spiral walk further out and reach the mask's
# protrusions, which raises shape coverage even when the raw fill ratio
# drops a little. Both directions are probed each round and the higher-
# coverage candidate wins; the loop stops when neither improves coverage.
# Every probe rebuilds the entire cloud with one shared scale.
if (
len(final_wc.layout_) >= total_target
and fill_ratio > 0
and (
fill_ratio < config.TARGET_FILL_RATIO * 0.98
or has_character_sized_hole(final_wc, largest_empty_square)
)
):
current_coverage = compute_coverage_score(occ_fast, mask_small)
free_px = max(1, int(np.sum(mask_small == 0)))
upper_scale = min(
(failed for failed in failed_scales if failed > final_scale),
default=None,
)
def _fill_and_coverage(wc):
occ = render_layout_occupancy(wc.layout_, mask_small.shape, config.WC_FONT_PATH)
new_fill = float(np.sum((mask_small == 0) & (occ == 1))) / free_px
cov = compute_coverage_score(occ, mask_small)
return new_fill, cov, occ
for density_attempt in range(1, 2 if config.FAST_MODE else 5):
# Grow direction (larger font): bisect toward a known-too-big scale,
# or nudge up by the fill-ratio deficit, exactly as before.
if upper_scale is not None:
grow_scale = (final_scale + upper_scale) / 2.0
elif equal_size_mode:
current_size, _ = scaled_bounds(final_scale)
grow_scale = (current_size + 1) / max(1, base_min_font)
else:
desired_growth = min(
1.12,
math.sqrt(config.TARGET_FILL_RATIO / fill_ratio) * 0.98,
)
grow_scale = final_scale * desired_growth if desired_growth > 1.005 else None
# Shrink direction (smaller font): the inverse nudge. Letting the
# spiral walk further out costs font size but can reach protrusions
# the grow direction abandons. Cap the shrink so one round cannot
# collapse the font to the floor.
shrink_scale = None
if not equal_size_mode and fill_ratio > 0:
shrink_factor = 1.0 / max(1.02, min(1.20, math.sqrt(fill_ratio / max(0.05, config.TARGET_FILL_RATIO)) * 1.02))
cand = final_scale * shrink_factor
if scaled_bounds(cand) != scaled_bounds(final_scale):
shrink_scale = cand
elif equal_size_mode:
current_size, _ = scaled_bounds(final_scale)
if current_size > hard_min_font:
shrink_scale = (current_size - 1) / max(1, base_min_font)
candidates = []
for direction, scale in (("grow", grow_scale), ("shrink", shrink_scale)):
if scale is None or scaled_bounds(scale) == scaled_bounds(final_scale):
continue
bounds = scaled_bounds(scale)
d_min, d_max = bounds
key = (bounds, base_layout_seed)
wc, placed_count, selected_seed = None, -1, base_layout_seed
if key not in tried_layouts:
tried_layouts.add(key)
wc, placed_count, _, _ = try_place(scale)
# Bounded seed retry to recover a complete layout, as before.
if placed_count < total_target and base_layout_seed is not None:
candidate_seed = (int(base_layout_seed) * 3 + 3) % (2**31 - 1)
retry_key = (bounds, candidate_seed)
if retry_key not in tried_layouts:
tried_layouts.add(retry_key)
retry_wc, retry_count, _, _ = try_place(scale, candidate_seed)
print(
f" 密度优化 #{density_attempt} {direction} 整批重排: "
f"seed={candidate_seed}, 字号=[{d_min}, {d_max}] -> "
f"{retry_count}/{total_target}"
)
if retry_count > placed_count:
wc, placed_count = retry_wc, retry_count
selected_seed = candidate_seed
if placed_count < total_target:
if direction == "grow":
upper_scale = scale
print(
f" 密度优化 #{density_attempt} {direction}: scale={scale:.3f}, "
f"字号=[{d_min}, {d_max}] -> {placed_count}/{total_target} (不完整)"
)
continue
new_fill, cov, occ = _fill_and_coverage(wc)
print(
f" 密度优化 #{density_attempt} {direction}: scale={scale:.3f}, "
f"字号=[{d_min}, {d_max}] -> {placed_count}/{total_target} "
f"fill={new_fill:.3f} coverage={cov:.4f}"
)
candidates.append((direction, scale, selected_seed, wc, placed_count, new_fill, cov, occ))
if not candidates:
break
# Pick the higher-coverage candidate; tie-break on fill ratio so a
# genuinely denser picture still wins when coverage is equal.
candidates.sort(key=lambda c: (c[6], c[5]))
direction, scale, selected_seed, wc, placed_count, new_fill, cov, occ = candidates[-1]
if cov <= current_coverage and new_fill <= fill_ratio:
break
final_wc = wc
final_scale = scale
final_layout_seed = selected_seed
fill_ratio = new_fill
occ_fast = occ
current_coverage = cov
largest_empty_square = largest_empty_square_size(occ_fast, mask_small)
complete_candidates.append(
(final_wc, final_scale, final_layout_seed, fill_ratio, occ_fast)
)
if (
fill_ratio >= config.TARGET_FILL_RATIO * 0.98
and not has_character_sized_hole(final_wc, largest_empty_square)
and current_coverage >= 0.98
):
break
# At the largest complete equal-size tier, compare a small deterministic
# set of whole-cloud reorderings and retain the one with the smallest
# character-scale void. This never changes an individual font size.
if (
equal_size_mode
and len(final_wc.layout_) >= total_target
and has_character_sized_hole(final_wc, largest_empty_square)
and base_layout_seed is not None
and not config.FAST_MODE
):
if total_target < 100:
hole_attempt_budget = 3
elif total_target <= 300:
hole_attempt_budget = 2
else:
hole_attempt_budget = 1
modulus = 2**31 - 1
for hole_attempt in range(1, hole_attempt_budget + 1):
candidate_seed = (
int(base_layout_seed) ^ ((0x9E3779B9 * hole_attempt) & 0x7FFFFFFF)
) % modulus
bounds = scaled_bounds(final_scale)
layout_key = (bounds, candidate_seed)
if layout_key in tried_layouts:
continue
tried_layouts.add(layout_key)
candidate_wc, placed_count, _, _ = try_place(final_scale, candidate_seed)
if placed_count < total_target:
continue
candidate_fill, candidate_occ = compute_fill_ratio_fast(
candidate_wc.layout_, mask_small, config.WC_FONT_PATH
)
candidate_hole = largest_empty_square_size(candidate_occ, mask_small)
print(
f" 空洞优化 #{hole_attempt}: seed={candidate_seed}, "
f"最大空白={candidate_hole}px -> {placed_count}/{total_target}"
)
if candidate_hole >= largest_empty_square:
continue
final_wc = candidate_wc
final_layout_seed = candidate_seed
fill_ratio = candidate_fill
occ_fast = candidate_occ
largest_empty_square = candidate_hole
complete_candidates.append(
(final_wc, final_scale, final_layout_seed, fill_ratio, occ_fast)
)
# ── 工作网格增量填充:原名字号不变,用更小字号追加副本填轮廓 ──
# 已填区域标记为阻挡,新词只能进空白间隙。每一轮只生成一份名单,
# 并把新占用合并回工作网格;因此自动填充没有固定的重复数量,
# 只在轮廓仍未覆盖且还有合法位置时继续追加。
fill_work_layout = []
if (
config.AUTO_REPEAT_TO_FILL
and final_wc is not None
and len(final_wc.layout_) >= total_target
and occ_fast is not None
):
current_cov = compute_coverage_score(occ_fast, mask_small)
if current_cov < 0.95 and names:
# 原名字号范围
original_sizes = [size for _, size, *_ in final_wc.layout_]
orig_min_font = int(min(original_sizes))
orig_max_font = int(max(original_sizes))
span = orig_max_font - orig_min_font
# 追加用缩小字号:原最大字号的 50~60%
fill_min_font = max(config.MIN_FONT_FLOOR, int(round(orig_min_font * 0.50)))
fill_max_font = max(fill_min_font, int(round(orig_min_font + span * 0.60)))
fill_seed = (final_layout_seed ^ 0x9E3779B9) & 0x7FFFFFFF
if fill_seed == 0:
fill_seed = 1
# AUTO_REPEAT_MAX 只是防止异常掩膜导致无限循环,不是目标重复次数。
max_fill_rounds = max(1, int(config.AUTO_REPEAT_MAX))
for fill_round in range(max_fill_rounds):
if current_cov >= 0.95:
break
# 融合 mask:原阻挡 + 已填区域都标为 255
fill_mask = mask_small.copy()
fill_mask[(occ_fast == 1)] = 255
fill_wc = OptimizedEfficientWordCloud(
width=w_small, height=h_small,
mask=fill_mask,
font_path=config.WC_FONT_PATH,
# 一轮只追加一份名单;需要更多时由下一轮按需追加。
max_words=len(names),
min_font_size=fill_min_font,
max_font_size=fill_max_font,
background_color=config.get_output_background(),
prefer_horizontal=1.0 - float(config.VERTICAL_RATIO),
margin=_collision_margin,
)
fill_wc.layout_seed = (fill_seed + fill_round) & 0x7FFFFFFF or 1
fill_freq = {name: 0.3 for name in names}
fill_wc.generate_from_frequencies(fill_freq)
fill_placed = len(fill_wc.layout_)
print(
f"[增量填充#{fill_round + 1}] 工作网格追加放置 "
f"{fill_placed}/{len(names)} 词,字号=[{fill_min_font}, {fill_max_font}]"
)
log.info(
"[增量填充#%d] 工作网格追加放置 %d/%d 词, 字号=[%d,%d]",
fill_round + 1, fill_placed, len(names), fill_min_font, fill_max_font,
)
if fill_placed <= 0:
break
fill_work_layout.extend(fill_wc.layout_)
fill_occ = render_layout_occupancy(
fill_wc.layout_, mask_small.shape, config.WC_FONT_PATH
)
occ_fast = np.maximum(occ_fast, fill_occ)
current_cov = compute_coverage_score(occ_fast, mask_small)
print(f"[增量填充#{fill_round + 1}] 工作网格覆盖度={current_cov:.4f}")
hd_layout = None
hd_clearance = None
raw_hd_layout = []
for candidate_index, candidate in enumerate(reversed(complete_candidates), start=1):
candidate_wc, candidate_scale, candidate_seed, candidate_fill, candidate_occ = candidate
raw_hd_layout = scale_layout_for_hd(candidate_wc.layout_, config.WORK_SCALE)
refined_layout = None
clearance_stats = None
clearance_modes = (1, 0)
if total_target >= 100 and candidate_index > 1:
clearance_modes = (0,)
for clearance_px in clearance_modes:
priority_layout = raw_hd_layout
priority_restarts = 0
max_priority_attempts = 2 if clearance_px == 0 or total_target < 100 else 1
for priority_attempt in range(max_priority_attempts):
refined_layout, clearance_stats = refine_layout_with_hd_clearance(
priority_layout,
mask_hd,
config.WC_FONT_PATH,
clearance=clearance_px,
)
if refined_layout is not None:
break
failed_word = clearance_stats["failed_word"]
failed_index = next(
(index for index, item in enumerate(priority_layout) if item[0] == failed_word),
None,
)
if (
failed_index is None
or failed_index == 0
or priority_attempt + 1 >= max_priority_attempts
):
break
failed_item = priority_layout[failed_index]
priority_layout = [failed_item, *priority_layout[:failed_index], *priority_layout[failed_index + 1:]]
priority_restarts += 1
if refined_layout is not None:
clearance_stats["clearance_px"] = clearance_px
clearance_stats["priority_restarts"] = priority_restarts
break
clearance_stats["clearance_px"] = clearance_px
clearance_stats["priority_restarts"] = priority_restarts
if refined_layout is None:
log.warning(
"高清候选 #%d 隔离精修失败: word=%s, shifted=%d, max_shift=%d",
candidate_index,
clearance_stats["failed_word"],
clearance_stats["shifted_words"],
clearance_stats["max_shift"],
)
hd_clearance = clearance_stats
continue
final_wc = candidate_wc
final_scale = candidate_scale
final_layout_seed = candidate_seed
fill_ratio = candidate_fill
occ_fast = candidate_occ
largest_empty_square = largest_empty_square_size(occ_fast, mask_small)
hd_layout = refined_layout
hd_clearance = clearance_stats
break
if hd_layout is None:
hd_overlap_pixels = -1
hd_layout = raw_hd_layout
if hd_clearance is None:
hd_clearance = {
"shifted_words": 0,
"max_shift": 0,
"clearance_px": None,
"priority_restarts": 0,
"failed_word": None,
}
else:
hd_overlap_pixels = count_layout_overlap_pixels(
hd_layout,
(real_hd_h, real_hd_w),
config.WC_FONT_PATH,
)
# ── 增量填充(工作网格 → HD) ──
# 工作网格上的合法位置经过放大后可能因取整发生碰撞,因此填充词必须
# 与基础布局一起再次做高清精修。若整批填充无法通过,则二分保留最多
# 的追加词;绝不能让自动填充破坏原本已经成功的基础布局。
if (
fill_work_layout
and hd_layout is not None
):
old_count = len(hd_layout)
fill_hd = scale_layout_for_hd(fill_work_layout, config.WORK_SCALE)
base_hd_layout = list(hd_layout)
accepted_additions, clearance_stats = append_layout_with_hd_clearance(
base_hd_layout,
fill_hd,
mask_hd,
config.WC_FONT_PATH,
clearance=0,
allow_global_search=False,
)
accepted_count = len(accepted_additions)
accepted_clearance = 0
hd_layout = base_hd_layout + accepted_additions
hd_overlap_pixels = count_layout_overlap_pixels(
hd_layout, (real_hd_h, real_hd_w), config.WC_FONT_PATH
)
new_fill, new_occ = compute_fill_ratio_fast(
hd_layout, mask_hd, config.WC_FONT_PATH
)
new_cov = compute_coverage_score(new_occ, mask_hd) if new_occ is not None else 0.0
clearance_stats = dict(clearance_stats or {})
clearance_stats["clearance_px"] = accepted_clearance
hd_clearance = clearance_stats
print(
f"[增量填充] 高清验收: {old_count}+{accepted_count}="
f"{len(hd_layout)} 词, fill={new_fill:.3f}, "
f"coverage={new_cov:.4f}, overlap={hd_overlap_pixels}"
)
print(
f"[增量填充] 工作网格候选 {len(fill_work_layout)} 词,"
f"高清接受 {len(hd_layout) - old_count} 词"
)
log.info(
"[增量填充] HD 验收: base=%d candidate=%d accepted=%d fill=%.4f coverage=%.4f overlap=%d",
old_count, len(fill_hd), len(hd_layout) - old_count, new_fill, new_cov, hd_overlap_pixels,
)
fill_ratio = new_fill
occ_fast = new_occ
largest_empty_square = largest_empty_square_size(occ_fast, mask_hd)
print(
f"最终填充率: {fill_ratio:.3f} | 高清重叠像素: {hd_overlap_pixels} | "
f"精修位移: {hd_clearance['shifted_words']} 词, 最大 {hd_clearance['max_shift']}px | "
f"隔离带: {hd_clearance['clearance_px']}px"
)
# occ_fast 可能是工作网格或 HD 网格形状(增量填充后),按形状匹配计算覆盖度
if occ_fast is not None and occ_fast.shape == mask_small.shape:
coverage = compute_coverage_score(occ_fast, mask_small)
elif occ_fast is not None and occ_fast.shape == mask_hd.shape:
coverage = compute_coverage_score(occ_fast, mask_hd)
else:
coverage = 0.0
print(f"轮廓覆盖度: {coverage:.4f}")
return {
"wc": final_wc,
"fill_ratio": fill_ratio,
"coverage": coverage,
"occ_fast": occ_fast,
"w_small": w_small,
"h_small": h_small,
"base_min_font": base_min_font,
"base_max_font": base_max_font,
"mask_small": mask_small,
"size_scale": final_scale,
"layout_seed": final_layout_seed,
"collision_margin": _collision_margin,
"hd_overlap_pixels": hd_overlap_pixels,
"hd_layout": hd_layout,
"hd_clearance": hd_clearance,
"largest_empty_square_work_px": largest_empty_square,
"largest_empty_square_font_ratio": (
largest_empty_square / max(1, int(final_wc.layout_[0][1]))
if final_wc.layout_ and equal_size_mode
else None
),
}
def main():
t_start = time.time()
print("--- 1. 读取数据 ---")
log.info("=" * 60)
log.info("[Pipeline] main() 开始")
log.info(" EXCEL_PATH = %s", config.EXCEL_PATH)
log.info(" DATA_COL = %d", config.DATA_COL_INDEX)
log.info(" MODE = %s", config.MODE)
log.info(" FILL_ON = %s", config.FILL_ON)
log.info(" WORK_SCALE = %.4f", config.WORK_SCALE)
log.info(" SEED = %s", config.SEED)
names = []
df = None
if os.path.exists(config.EXCEL_PATH):
try:
df = pd.read_excel(config.EXCEL_PATH)
raw_names = df.iloc[:, config.DATA_COL_INDEX].dropna().astype(str)
if config.REMOVE_DUPLICATES:
names = raw_names.unique().tolist()
print(f"模式: 去重 | 数量: {len(names)}")
else:
names = raw_names.tolist()
print(f"模式: 保留重复 | 数量: {len(names)}")
except Exception as e:
print(f"读取 Excel 失败: {e}")
log.error("读取 Excel 失败: %s", e)
sys.exit(1)
else:
count = 12000
print(f"未找到Excel,使用测试数据: {count}条")
log.info("未找到 Excel,使用测试数据: %d 条", count)
names = [f"测试_{i % 100}" for i in range(count)]
input_count = len(names)
log.info("[阶段1] 读取完成: input_count=%d, 去重=%s", input_count, config.REMOVE_DUPLICATES)
if names:
sample = names[:min(10, len(names))]
log.info(" 前10个名字: %s", sample)
print("--- 2. 智能画幅计算 ---")
log.info("--- 阶段2: 智能画幅计算 ---")
avg_len = sum(len(n) for n in names) / len(names) if names else 3
log.info(" 平均名字长度: %.2f 字符", avg_len)
log.info(" BASE_HD: %dx%d", config.BASE_HD_WIDTH, config.BASE_HD_HEIGHT)
# 只生成一次掩膜:先 probe 尺寸,再按需扩展后复用(避免二次 LANCZOS)
probe_mask_hd, (probe_w, probe_h), _ = prepare_mask(config.BASE_HD_WIDTH, config.BASE_HD_HEIGHT)
probe_stats = analyze_mask(probe_mask_hd)
log.info(" Probe mask: %dx%d, free_ratio=%.4f, bbox_fill_ratio=%.4f",
probe_w, probe_h, probe_stats['free_ratio'], probe_stats['bbox_fill_ratio'])
if probe_stats.get('bbox'):
log.info(" Probe bbox: %s", probe_stats['bbox'])
print(f"[Mask Probe] 可填充比例={probe_stats['free_ratio']:.3f}")
hd_w, hd_h = calculate_dynamic_dimensions(probe_w, probe_h, len(names), avg_len, probe_stats)
log.info(" 动态画幅计算结果: %dx%d", hd_w, hd_h)
print("--- 3. 生成掩膜 (High Quality & Edge Fix) ---")
log.info("--- 阶段3: 生成掩膜 ---")
if (hd_w, hd_h) == (probe_w, probe_h):
mask_hd, real_hd_w, real_hd_h = probe_mask_hd, probe_w, probe_h
mask_stats = probe_stats
else:
mask_hd, (real_hd_w, real_hd_h), _ = prepare_mask(hd_w, hd_h)
mask_stats = analyze_mask(mask_hd)
log.info(" mask_hd: %dx%d", real_hd_w, real_hd_h)
log.info(" free_area=%d, free_ratio=%.6f", mask_stats['free_area'], mask_stats['free_ratio'])
log.info(" bbox_fill_ratio=%.6f", mask_stats['bbox_fill_ratio'])
if mask_stats.get('bbox'):
log.info(" bbox=%s", mask_stats['bbox'])
print(f"[Mask Final] 可填充比例={mask_stats['free_ratio']:.3f}")
print("--- 4. 计算权重 ---")
log.info("--- 阶段4: 计算权重 ---")
t_weights = time.time()
if config.ENABLE_STROKE_WEIGHTS:
stroke_weights_map = get_stroke_complexity_batch(names, config.WC_FONT_PATH)
log.info(" 笔画权重计算完成: %d 个词, 耗时=%.3fs", len(stroke_weights_map), time.time() - t_weights)
else:
stroke_weights_map = {}
print("笔画权重已关闭")
log.info(" 笔画权重已关闭")
# 打印权重分布统计
if stroke_weights_map:
w_vals = list(stroke_weights_map.values())
log.info(" 笔画权重分布: min=%.1f, max=%.1f, avg=%.1f, median=%.1f",
min(w_vals), max(w_vals), sum(w_vals)/len(w_vals),
sorted(w_vals)[len(w_vals)//2])
sample_items = list(stroke_weights_map.items())[:5]
log.info(" 笔画权重样本: %s", sample_items)
excel_weights_map = extract_weights_from_df(df, names) if df is not None else {}
if excel_weights_map:
print(f"Excel 权重生效: {len(excel_weights_map)} 个词")
log.info(" Excel 权重生效: %d 个词", len(excel_weights_map))
ew_vals = list(excel_weights_map.values())
log.info(" Excel 权重分布: min=%.1f, max=%.1f, avg=%.1f",
min(ew_vals), max(ew_vals), sum(ew_vals)/len(ew_vals))
elif config.WEIGHT_COL_NAME is not None or config.WEIGHT_COL_INDEX is not None:
fallback = "笔画权重" if config.ENABLE_STROKE_WEIGHTS else "均等权重"
print(f"Excel 权重不可用,已回退{fallback}")
log.info(" Excel 权重不可用,已回退%s", fallback)
name_weights_map = merge_weight_maps(
names,
stroke_weights_map if config.ENABLE_STROKE_WEIGHTS else {},
excel_weights_map,
)
frequencies_data = name_weights_map if config.REMOVE_DUPLICATES else [(name, name_weights_map.get(name, 10)) for name in names]
canvas_retry_round = 0
generation_result = None
t_gen = time.time()
canvas_retry_round = 0
generation_result = None
t_gen = time.time()
while True:
log.info("[画布] 第%d轮生成 pass, 当前画布: %dx%d", canvas_retry_round + 1, real_hd_w, real_hd_h)
generation_result = run_generation_pass(
names,
frequencies_data,
name_weights_map,
mask_hd,
real_hd_w,
real_hd_h,
)
wc = generation_result.get("wc")
placed = len(wc.layout_) if wc is not None else 0
target = len(names) * config.N_REPETITIONS
# 名单完整性是硬约束。填充率优化不得以漏掉姓名为代价。
hd_overlap_pixels = int(generation_result.get("hd_overlap_pixels", 0))
placement_ok = wc is not None and placed >= target and hd_overlap_pixels == 0
if placement_ok or canvas_retry_round >= config.CANVAS_RETRY_MAX_ROUNDS:
if not placement_ok:
print(
f"生成失败:放置 {placed}/{target},高清重叠像素 {hd_overlap_pixels}"
"未满足完整名单与零碰撞约束"
)
log.error(
"生成失败:放置 %d/%d, 高清重叠像素=%d (已重试 %d 轮)",
placed,
target,
hd_overlap_pixels,
canvas_retry_round,
)
sys.exit(1)
log.info(" 生成完成 placed=%d/%d fill=%.4f coverage=%.4f retry=%d",
placed, target, generation_result["fill_ratio"],
generation_result.get("coverage", 0.0), canvas_retry_round)
break
canvas_retry_round += 1
next_w = int(real_hd_w * config.CANVAS_RETRY_GROWTH)
next_h = int(real_hd_h * config.CANVAS_RETRY_GROWTH)
print(f"[画布重试#{canvas_retry_round}] {real_hd_w}x{real_hd_h} -> {next_w}x{next_h} (放置 {placed}/{target})")
log.info("[画布重试#%d] %dx%d -> %dx%d (growth=%.2f)",
canvas_retry_round, real_hd_w, real_hd_h, next_w, next_h, config.CANVAS_RETRY_GROWTH)
mask_hd, (real_hd_w, real_hd_h), _ = prepare_mask(next_w, next_h)
mask_stats = analyze_mask(mask_hd)
final_wc = generation_result["wc"]
fill_ratio = generation_result["fill_ratio"]
w_small = generation_result["w_small"]
h_small = generation_result["h_small"]
log.info("[阶段5完成] 生成耗时=%.2fs, fill_ratio=%.4f, size_scale=%.4f",
time.time() - t_gen, fill_ratio, generation_result["size_scale"])
print("--- 6. 高清渲染 ---")
log.info("--- 阶段6: 高清渲染 ---")
t_render = time.time()
hd_layout = generation_result["hd_layout"]
log.info(" HD layout 词数: %d", len(hd_layout))
log.info(" HD 画布: %dx%d", real_hd_w, real_hd_h)
if hd_layout:
sample = hd_layout[:3]
for s in sample:
log.info(" 样本: text='%s', size=%d, pos=(%d,%d), orient=%s, color=%s",
s[0], s[1], s[2][1], s[2][0], s[3], s[4])
final_wc.layout_ = hd_layout
final_wc.width = real_hd_w
final_wc.height = real_hd_h
base_img = final_wc.to_image().convert("RGB")
# 更快的 PNG 写出(压缩等级 1,视觉无损)
base_img.save(config.OUTPUT_PNG, compress_level=1)
print(f"已保存: {config.OUTPUT_PNG}")
log.info(" PNG 已保存: %s (%.2f MB)", config.OUTPUT_PNG,
Path(config.OUTPUT_PNG).stat().st_size / 1024 / 1024 if Path(config.OUTPUT_PNG).exists() else 0)
# 一次构建路径,同时写出 fill / stroke 两份 SVG
stroke_svg = final_wc.export_svgs(config.OUTPUT_SVG)
print(f"已保存: {config.OUTPUT_SVG}")
print(f"已保存: {stroke_svg}")
log.info(" SVG 已保存: %s", config.OUTPUT_SVG)
log.info(" SVG(stroke) 已保存: %s", stroke_svg)
log.info(" 渲染耗时: %.2fs", time.time() - t_render)
log.info("--- 阶段7: 写入数据库 ---")
t_db = time.time()
try:
conn = sqlite3.connect(config.DB_PATH)
cursor = conn.cursor()
cursor.execute("DROP TABLE IF EXISTS word_locations")
cursor.execute("""
CREATE TABLE word_locations (
id INTEGER PRIMARY KEY AUTOINCREMENT,
name TEXT,
x INTEGER,
y INTEGER,
font_size INTEGER,
color TEXT,
orientation TEXT,
box_x INTEGER,
box_y INTEGER,
box_width INTEGER,
box_height INTEGER
)
""")
# Cache full font bearings as well as dimensions so search highlights
# match the actual PIL-rendered glyph position.
bbox_cache = {}
_measure = ImageDraw.Draw(Image.new("L", (1, 1)))
db_data = []
for name, font_size, (y, x), orient, color in final_wc.layout_:
orientation = "vertical" if orient else "horizontal"
key = (name, int(font_size), bool(orient))
box = bbox_cache.get(key)
if box is None:
font = get_cached_font(config.WC_FONT_PATH, max(1, int(font_size)))
if orient:
font = ImageFont.TransposedFont(font, orientation=orient)
bb = _measure.textbbox((0, 0), name, font=font)
box = (bb[0], bb[1], bb[2] - bb[0], bb[3] - bb[1])
bbox_cache[key] = box
bx, by, bw, bh = box
db_data.append(
(
name,
x + bx,
y + by,
font_size,
color,
orientation,
x,
y,
bw,
bh,
)
)
cursor.executemany(
"""
INSERT INTO word_locations
(name, x, y, font_size, color, orientation, box_x, box_y, box_width, box_height)
VALUES (?,?,?,?,?,?,?,?,?,?)
""",
db_data,
)
conn.commit()
conn.close()
log.info(" DB 写入完成: %s, %d 行, 耗时=%.3fs", config.DB_PATH, len(db_data), time.time() - t_db)
except sqlite3.Error as e:
print(f"DB Error: {e}")
log.error(" DB 写入失败: %s", e)
sys.exit(1)
elapsed = time.time() - t_start
placed_count = len(final_wc.layout_)
metrics = {
"seed": config.SEED,
"layout_seed": generation_result.get("layout_seed", config.LAYOUT_SEED),
"input_count": input_count,
"placed_count": placed_count,
"fill_ratio": fill_ratio,
"largest_empty_square_work_px": int(
generation_result.get("largest_empty_square_work_px", 0)
),
"largest_empty_square_font_ratio": generation_result.get(
"largest_empty_square_font_ratio"
),
"hd_overlap_pixels": int(generation_result.get("hd_overlap_pixels", -1)),
"hd_clearance": generation_result.get("hd_clearance"),
"elapsed_seconds": round(elapsed, 4),
"font_info": {
"layout_font_path": config.WC_FONT_PATH,
"mask_font_path": config.MASK_FONT_PATH,
"palette": list(config.get_output_palette()),
"background": config.get_output_background(),
},
"mask_info": {
"free_ratio": round(mask_stats["free_ratio"], 6),
"bbox_fill_ratio": round(mask_stats["bbox_fill_ratio"], 6),
"canvas_retry_rounds": canvas_retry_round,
},
"canvas_info": {
"hd_width": real_hd_w,
"hd_height": real_hd_h,
"work_width": w_small,
"work_height": h_small,
"work_scale": config.WORK_SCALE,
},
"output_paths": {
"png": config.OUTPUT_PNG,
"svg": config.OUTPUT_SVG,
"db": config.DB_PATH,
"metrics": config.METRICS_FILE,
"debug_dir": config.DEBUG_OUTPUT_DIR,
},
"config_snapshot": {
"mode": config.MODE,
"excel_path": config.EXCEL_PATH,
"mask_image_path": config.MASK_IMAGE_PATH,
"output_dir": config.OUTPUT_DIR,
"output_prefix": config.OUTPUT_PREFIX,
"min_font_size": config.MIN_FONT_SIZE,
"fill_on": config.FILL_ON,
"layout_seed": generation_result.get("layout_seed", config.LAYOUT_SEED),
}
}
config.write_metrics(metrics)
print(f"\n✅ 完成! 总耗时: {elapsed:.2f}s")
log.info("=" * 60)
log.info("[Pipeline] 全流程完成!")
log.info(" 总耗时: %.2fs", elapsed)
log.info(" 输入: %d 词 -> 放置: %d 词", input_count, placed_count)
log.info(" 填充率: %.4f", fill_ratio)
log.info(" 画布: %dx%d (运算: %dx%d)", real_hd_w, real_hd_h, w_small, h_small)
log.info(" 输出: PNG=%s", config.OUTPUT_PNG)
log.info(" 输出: SVG=%s", config.OUTPUT_SVG)
log.info(" 输出: DB=%s", config.DB_PATH)
log.info("=" * 60)