This commit is contained in:
2026-05-09 20:45:14 +08:00
commit 46c2ad1f36
5 changed files with 849 additions and 0 deletions

209
autoCut.py Normal file
View File

@@ -0,0 +1,209 @@
import cv2
import numpy as np
import os
from pathlib import Path
# ========== 基础配置(只用填这几个,不用改其他) ==========
INPUT_FOLDER = "pages"
OUTPUT_FOLDER = "chars_auto"
# 第一个字中心点、最后一个字中心点
CX0, CY0 = 1027, 1257
CX1, CY1 = 4029, 5673
CHAR_SIZE = 715
ROWS = 5
COLS = 5
# 搜索范围扩展(像素)
SEARCH_PAD = 150
# 偏移修正最大幅度(像素)
MAX_SHIFT = 100
# ==========================================================
Path(OUTPUT_FOLDER).mkdir(exist_ok=True)
# 兼容中文路径读图
def imread_unicode(path):
return cv2.imdecode(np.fromfile(path, dtype=np.uint8), cv2.IMREAD_COLOR)
# 兼容中文路径存图
def imwrite_unicode(path, img):
cv2.imencode('.png', img)[1].tofile(path)
def get_char_center_from_bbox(
img_gray,
roi_x1,
roi_y1,
roi_x2,
roi_y2,
target_cx=None,
target_cy=None,
max_shift=MAX_SHIFT,
):
"""
在理论区域内,自动计算文字的真实边界框中心
返回:真实中心点 (cx, cy)
"""
# 截取局部ROI
roi = img_gray[roi_y1:roi_y2, roi_x1:roi_x2]
roi_h, roi_w = roi.shape[:2]
blur = cv2.GaussianBlur(roi, (5, 5), 0)
_, binary = cv2.threshold(blur, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)
kernel = cv2.getStructuringElement(cv2.MORPH_RECT, (5, 5))
binary = cv2.morphologyEx(binary, cv2.MORPH_CLOSE, kernel, iterations=1)
num_labels, _, stats, centroids = cv2.connectedComponentsWithStats(binary, connectivity=8)
if num_labels <= 1:
if target_cx is not None and target_cy is not None:
return float(target_cx), float(target_cy)
cx = (roi_x1 + roi_x2) / 2.0
cy = (roi_y1 + roi_y2) / 2.0
return cx, cy
roi_area = float(roi_h * roi_w)
min_area = max(300.0, roi_area * 0.002)
if target_cx is None:
tx = roi_w / 2.0
else:
tx = float(target_cx - roi_x1)
if target_cy is None:
ty = roi_h / 2.0
else:
ty = float(target_cy - roi_y1)
gate = float(max(40.0, float(max_shift) + 20.0))
best_score = None
best_center = None
for i in range(1, num_labels):
x, y, w, h, area = stats[i]
area = float(area)
if area < min_area:
continue
box_area = float(w * h) if w > 0 and h > 0 else 1.0
fill = area / box_area
if (w > 0.85 * roi_w or h > 0.85 * roi_h) and fill < 0.08:
continue
aspect = float(w) / (float(h) + 1e-6)
if (aspect > 12.0 or aspect < (1.0 / 12.0)) and fill < 0.15:
continue
cx, cy = centroids[i]
if abs(float(cx) - tx) > gate or abs(float(cy) - ty) > gate:
continue
dx = (float(cx) - tx) / (roi_w + 1e-6)
dy = (float(cy) - ty) / (roi_h + 1e-6)
dist2 = dx * dx + dy * dy
score = area * (fill + 0.2) * np.exp(-16.0 * dist2)
if best_score is None or score > best_score:
best_score = score
best_center = (float(cx), float(cy))
if best_center is None:
min_area_relaxed = max(100.0, roi_area * 0.0005)
for i in range(1, num_labels):
x, y, w, h, area = stats[i]
area = float(area)
if area < min_area_relaxed:
continue
box_area = float(w * h) if w > 0 and h > 0 else 1.0
fill = area / box_area
if (w > 0.90 * roi_w or h > 0.90 * roi_h) and fill < 0.06:
continue
cx, cy = centroids[i]
if abs(float(cx) - tx) > gate or abs(float(cy) - ty) > gate:
continue
score = area * (fill + 0.2)
if best_score is None or score > best_score:
best_score = score
best_center = (float(cx), float(cy))
if best_center is None:
if target_cx is not None and target_cy is not None:
return float(target_cx), float(target_cy)
cx = (roi_x1 + roi_x2) / 2.0
cy = (roi_y1 + roi_y2) / 2.0
return cx, cy
return roi_x1 + best_center[0], roi_y1 + best_center[1]
def process_page(img_path, page_idx):
img = imread_unicode(str(img_path))
if img is None:
print(f"跳过无法读取:{img_path.name}")
return
h, w = img.shape[:2]
gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)
# 生成理论网格行列间隔
step_x = (CX1 - CX0) / (COLS - 1)
step_y = (CY1 - CY0) / (ROWS - 1)
for row in range(ROWS):
for col in range(COLS):
# 理论中心点
theo_cx = CX0 + col * step_x
theo_cy = CY0 + row * step_y
# 理论区域ROI范围
r_x1 = int(theo_cx - CHAR_SIZE // 2 - SEARCH_PAD)
r_y1 = int(theo_cy - CHAR_SIZE // 2 - SEARCH_PAD)
r_x2 = int(theo_cx + CHAR_SIZE // 2 + SEARCH_PAD)
r_y2 = int(theo_cy + CHAR_SIZE // 2 + SEARCH_PAD)
# 边界保护
r_x1 = max(0, r_x1)
r_y1 = max(0, r_y1)
r_x2 = min(w, r_x2)
r_y2 = min(h, r_y2)
# 自动获取真实文字边界框中心(只允许小幅修正)
detect_cx, detect_cy = get_char_center_from_bbox(
gray, r_x1, r_y1, r_x2, r_y2, theo_cx, theo_cy, max_shift=MAX_SHIFT
)
dx = float(detect_cx - theo_cx)
dy = float(detect_cy - theo_cy)
dx = float(np.clip(dx, -MAX_SHIFT, MAX_SHIFT))
dy = float(np.clip(dy, -MAX_SHIFT, MAX_SHIFT))
real_cx = float(theo_cx + dx)
real_cy = float(theo_cy + dy)
# 用真实中心裁切 715x715
x1 = int(round(real_cx - CHAR_SIZE / 2.0))
y1 = int(round(real_cy - CHAR_SIZE / 2.0))
x2 = x1 + CHAR_SIZE
y2 = y1 + CHAR_SIZE
canvas = np.full((CHAR_SIZE, CHAR_SIZE, 3), 255, dtype=np.uint8)
sx1 = max(0, x1)
sy1 = max(0, y1)
sx2 = min(w, x2)
sy2 = min(h, y2)
src = img[sy1:sy2, sx1:sx2]
dx = max(0, -x1)
dy = max(0, -y1)
canvas[dy:dy + src.shape[0], dx:dx + src.shape[1]] = src
crop = canvas
fname = f"page{page_idx}_r{row}_c{col}.png"
save_path = os.path.join(OUTPUT_FOLDER, fname)
imwrite_unicode(save_path, crop)
if __name__ == "__main__":
print("开始 5×5 逐字自动边界检测中心定位切割...")
img_list = sorted(Path(INPUT_FOLDER).glob("*.png"))
for idx, f in enumerate(img_list):
process_page(f, idx + 1)
print(f"已完成:{f.name}")
print("✅ 全部自动切割+自动边界检测定位完成!")