Initial project baseline

This commit is contained in:
2026-07-04 02:40:45 +08:00
commit d5d8caef2f
86 changed files with 15590 additions and 0 deletions
+101
View File
@@ -0,0 +1,101 @@
import math
import numpy as np
import pandas as pd
from PIL import Image, ImageDraw
from . import config
from .fonts import get_cached_font
from .layout import normalize_relative_scores
def get_stroke_complexity_batch(names, font_p, test_size=64):
font = get_cached_font(font_p, test_size)
img = Image.new("L", (test_size, test_size), 255)
draw = ImageDraw.Draw(img)
char_complexity_cache = {}
weights = {}
all_chars = set("".join(names))
for char in all_chars:
draw.rectangle([0, 0, test_size, test_size], fill=255)
draw.text((0, 0), char, font=font, fill=0)
char_complexity_cache[char] = np.sum(np.array(img) < 200)
unique_names = set(names)
for name in unique_names:
if not name:
weights[name] = 10
continue
complexities = [char_complexity_cache.get(c, 10) for c in name]
weights[name] = max(complexities)
return weights
def extract_weights_from_df(df, names):
series = None
if config.WEIGHT_COL_NAME is not None:
if config.WEIGHT_COL_NAME in df.columns:
series = df[config.WEIGHT_COL_NAME]
else:
config._warn(f"权重列名不存在: {config.WEIGHT_COL_NAME},尝试使用权重列索引")
if series is None and config.WEIGHT_COL_INDEX is not None:
if 0 <= config.WEIGHT_COL_INDEX < len(df.columns):
series = df.iloc[:, config.WEIGHT_COL_INDEX]
else:
config._warn(f"权重列索引越界: {config.WEIGHT_COL_INDEX},将回退到笔画权重")
if series is None:
return {}
name_series = df.iloc[:, config.DATA_COL_INDEX]
numeric = pd.to_numeric(series, errors='coerce')
pairs = pd.DataFrame({"name": name_series, "weight": numeric})
pairs = pairs[pairs["name"].notna()]
pairs["name"] = pairs["name"].astype(str)
pairs = pairs[pairs["weight"].notna() & (pairs["weight"] > 0)]
if pairs.empty:
config._warn("Excel 权重列没有可用正数,全部回退到笔画权重")
return {}
if config.REMOVE_DUPLICATES:
grouped = pairs.groupby("name", as_index=False)["weight"].max()
return dict(zip(grouped["name"], grouped["weight"]))
valid_name_set = set(names)
pairs = pairs[pairs["name"].isin(valid_name_set)]
if pairs.empty:
config._warn("Excel 权重与名称列未形成有效映射,全部回退到笔画权重")
return {}
grouped = pairs.groupby("name", as_index=False)["weight"].max()
return dict(zip(grouped["name"], grouped["weight"]))
def calculate_font_by_area_model(mask, names, weights_map, fill_ratio, size_ratio, packing_efficiency, n_rep):
free_area = int(np.sum(mask == 0))
if free_area <= 0:
free_area = int(mask.size)
effective_fill = fill_ratio if fill_ratio > 0 else max(config.MIN_ACCEPT_FILL_RATIO, 0.82)
target_area = free_area * effective_fill * packing_efficiency
weights = [max(float(weights_map.get(name, 10)), 1.0) for name in names]
if not weights:
return max(config.MIN_FONT_SIZE, 10), max(config.MIN_FONT_SIZE + 4, 20)
log_scores = normalize_relative_scores([math.log1p(weight) for weight in weights])
char_mass = 0.0
for name, score in zip(names, log_scores):
length = max(1, len(name))
char_mass += length * (0.9 + 0.9 * score)
char_mass *= max(1, n_rep)
if char_mass <= 0:
return max(config.MIN_FONT_SIZE, 10), max(config.MIN_FONT_SIZE + 4, 20)
nominal_size = math.sqrt(target_area / char_mass)
min_f = max(config.MIN_FONT_SIZE, int(nominal_size * 0.72))
max_f = max(min_f + 1, int(min_f * max(1.4, size_ratio)))
return min_f, max_f