Rework layout engine around exact-glyph collision, add tests and docs sync

Replace the old bbox/heuristic placement (scale search rounds, large-font
capping, stratified sampling, fill-retry ladders) with an area-model font
sizing pass feeding a C++ exact-glyph collision engine (centroid-biased
spiral + random probing, HD clearance refinement, density/hole
optimization). Simplify the frontend advanced-params panel and JobParams
type to match the surviving config surface, add a layout-constraints test
suite and a repeatable benchmark tool, and bring docs/*.md back in sync
with current code (plus new TESTING.md and DEPLOYMENT.md).

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
2026-07-26 18:32:25 +08:00
co-authored by Claude Sonnet 5
parent bf2b138007
commit 1d17b5e20d
24 changed files with 2600 additions and 1283 deletions
+53 -5
View File
@@ -73,13 +73,55 @@ def extract_weights_from_df(df, names):
return dict(zip(grouped["name"], grouped["weight"]))
def merge_weight_maps(names, stroke_weights=None, excel_weights=None):
"""Combine optional manual weights with normalized stroke complexity.
Manual Excel weights remain the base signal. When both sources exist,
stroke complexity is normalized around the median and applied as a
multiplicative factor, so an all-ones Excel column still enables visibly
different stroke-driven font sizes while meaningful manual ratios remain.
"""
stroke_weights = stroke_weights or {}
excel_weights = excel_weights or {}
if not stroke_weights:
return dict(excel_weights)
if not excel_weights:
return dict(stroke_weights)
positive_strokes = sorted(
max(float(stroke_weights[name]), 1.0)
for name in names
if name in stroke_weights
)
if not positive_strokes:
return dict(excel_weights)
midpoint = positive_strokes[len(positive_strokes) // 2]
merged = {}
for name in names:
manual = excel_weights.get(name)
stroke = stroke_weights.get(name)
if manual is None:
if stroke is not None:
merged[name] = float(stroke)
continue
if stroke is None:
merged[name] = float(manual)
continue
merged[name] = max(float(manual), 1e-6) * max(float(stroke), 1.0) / midpoint
return merged
def calculate_font_by_area_model(mask, names, weights_map, fill_ratio, size_ratio, packing_efficiency, n_rep):
free_area = int(np.sum(mask == 0))
if free_area <= 0:
free_area = int(mask.size)
effective_fill = fill_ratio if fill_ratio > 0 else max(config.MIN_ACCEPT_FILL_RATIO, 0.82)
target_area = free_area * effective_fill * packing_efficiency
# 中文实心笔画约占字形包围盒 35–55%;目标取 0.42 附近,略保守以便一次放满。
effective_fill = fill_ratio if fill_ratio > 0 else 0.42
# 包围盒面积 vs 实际笔画:面积模型用包围盒估算,需额外 ink_factor 校正
ink_factor = 0.42
target_area = free_area * effective_fill * packing_efficiency / ink_factor
weights = [max(float(weights_map.get(name, 10)), 1.0) for name in names]
if not weights:
@@ -89,13 +131,19 @@ def calculate_font_by_area_model(mask, names, weights_map, fill_ratio, size_rati
char_mass = 0.0
for name, score in zip(names, log_scores):
length = max(1, len(name))
char_mass += length * (0.9 + 0.9 * score)
char_mass += length * (0.85 + 0.7 * score)
char_mass *= max(1, n_rep)
if char_mass <= 0:
return max(config.MIN_FONT_SIZE, 10), max(config.MIN_FONT_SIZE + 4, 20)
nominal_size = math.sqrt(target_area / char_mass)
min_f = max(config.MIN_FONT_SIZE, int(nominal_size * 0.72))
max_f = max(min_f + 1, int(min_f * max(1.4, size_ratio)))
min_f = max(config.MIN_FONT_SIZE, int(round(nominal_size * 0.78)))
ratio = max(1.0, float(size_ratio))
if math.isclose(ratio, 1.0, rel_tol=0.0, abs_tol=1e-9):
# A ratio of exactly one is a hard semantic guarantee: every word
# receives the same target size. Do not inject an artificial span.
max_f = min_f
else:
max_f = max(min_f, int(round(min_f * ratio)))
return min_f, max_f