# -*- coding: utf-8 -*- # ============================================================ # tool_crop_mall_logos.py - 宝龙美食打卡 lab · 导视牌 logo 重裁脚本(方案 A) # # 病根:原 images/ 是按"等分网格"从斜拍照片硬裁的,透视变形导致越靠边错位越大, # 裁切图带白底店名标签、串邻格、切边。 # # 算法流程(经典 CV,离线确定可重复跑): # 1) 展板四边形检测(亮板 vs 暗背景 OTSU + 最大轮廓 approxPolyDP)→ 四点透视矫正拉正 # 2) 置信照片块检测:HSV 颜色掩膜 (饱和度高 OR 暗) → 闭/开运算 → 连通域, # 只保留尺寸落在照片先验区间内的"置信块"(过曝区检不出就检不出,不强求) # 3) 全局网格拟合:导视牌是印刷规整网格,矫正后行/列等距 —— # 用置信块的最小二乘拟合 行顶线 row_top(r) 与 列中心线 col_center(c), # 过曝检不出的格子直接按几何矩形取,天然不错位 # 4) 几何兜底格先做"内容密度"校验:空白板面(如 r2c10「待确认」牌上不存在)跳过, # 有边框/文字/图案的(如商业街白框)保留 # 5) 输出统一宽度 jpg + 拼版预览图 + report.json,供人工 / qwen-vl 抽检 # # 用法: # python tool_crop_mall_logos.py # 默认参数直接跑 # python tool_crop_mall_logos.py --pad 4 # 调整裁切外扩像素 # python tool_crop_mall_logos.py --dry-run # 只出预览和报告,不写 images/ # ============================================================ import argparse import json import os import sys import cv2 import numpy as np DEFAULT_SRC = r'D:\Temp文件\baolong-mall\照片.jpg' DEFAULT_OUT = r'd:\Trae_Files\TRAE-Toolbox\public\tools\thought_lab\labs\mall_food_checkin\images' DEFAULT_SEED = r'd:\Trae_Files\TRAE-Toolbox\src\server\thought_lab\labs\mall_food_checkin\shops_seed.json' DEFAULT_PREVIEW = r'd:\Trae_Files\TRAE-Toolbox\dev_test_scripts\debug\mall_crop_contact.jpg' DEFAULT_REPORT = r'd:\Trae_Files\TRAE-Toolbox\dev_test_scripts\debug\mall_crop_report.json' ROWS, COLS = 7, 11 def imread_u(path): # Windows 下 cv2.imread 不支持中文路径,用 np.fromfile + imdecode 兜底 data = np.fromfile(path, dtype=np.uint8) return cv2.imdecode(data, cv2.IMREAD_COLOR) def imwrite_u(path, img, params=None): ext = os.path.splitext(path)[1] or '.jpg' ok, buf = cv2.imencode(ext, img, params or []) if not ok: return False buf.tofile(path) return True def order_points(pts): # 排序为 tl, tr, br, bl pts = np.array(pts, dtype='float32') s = pts.sum(axis=1) d = np.diff(pts, axis=1).ravel() tl = pts[np.argmin(s)] br = pts[np.argmax(s)] tr = pts[np.argmin(d)] bl = pts[np.argmax(d)] return np.array([tl, tr, br, bl], dtype='float32') def detect_board_quad(img): gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY) blur = cv2.GaussianBlur(gray, (7, 7), 0) _, th = cv2.threshold(blur, 0, 255, cv2.THRESH_BINARY + cv2.THRESH_OTSU) th = cv2.morphologyEx(th, cv2.MORPH_CLOSE, np.ones((25, 25), np.uint8)) cnts, _ = cv2.findContours(th, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) if not cnts: raise RuntimeError('未找到展板轮廓') c = max(cnts, key=cv2.contourArea) peri = cv2.arcLength(c, True) approx = cv2.approxPolyDP(c, 0.02 * peri, True) if len(approx) == 4: return order_points(approx.reshape(4, 2)) rect = cv2.minAreaRect(c) return order_points(cv2.boxPoints(rect)) def warp_board(img, quad): (tl, tr, br, bl) = quad w = int(max(np.linalg.norm(tr - tl), np.linalg.norm(br - bl))) h = int(max(np.linalg.norm(bl - tl), np.linalg.norm(br - tr))) m = cv2.getPerspectiveTransform(quad, np.array([[0, 0], [w - 1, 0], [w - 1, h - 1], [0, h - 1]], dtype='float32')) return cv2.warpPerspective(img, m, (w, h)) def detect_confident_blocks(warped): """颜色掩膜 + 尺寸先验,只返回高置信照片块(过曝区检不出不强求)""" hsv = cv2.cvtColor(warped, cv2.COLOR_BGR2HSV) s = hsv[:, :, 1] v = hsv[:, :, 2] mask = (((s > 45) | (v < 140)).astype(np.uint8)) * 255 mask[:60, :] = 0 mask[-60:, :] = 0 mask[:, :60] = 0 mask[:, -60:] = 0 mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, np.ones((9, 9), np.uint8)) mask = cv2.morphologyEx(mask, cv2.MORPH_OPEN, np.ones((7, 7), np.uint8)) n, labels, stats, cents = cv2.connectedComponentsWithStats(mask, 8) h, w = warped.shape[:2] blocks = [] for i in range(1, n): x, y, bw, bh, area = stats[i] cx, cy = cents[i] if area < 30000: continue if cy < 0.17 * h or cx < 0.06 * w: continue if not (180 <= bw <= 360 and 150 <= bh <= 300): continue blocks.append({'x': int(x), 'y': int(y), 'w': int(bw), 'h': int(bh), 'cx': float(cx), 'cy': float(cy), 'area': int(area)}) return blocks def fit_grid(blocks): """用置信块拟合全局网格:行顶线 / 列中心线 / 照片宽高 列:全局 cx 聚类(不依赖"检满 11 格的行",过曝行缺块也不错位) 行:cy 聚类 + 行距推算行号(缺行不影响编号)""" # ---- 列:全局 cx 聚类 ---- by_cx = sorted(blocks, key=lambda b: b['cx']) col_clusters = [] for b in by_cx: if col_clusters and b['cx'] - col_clusters[-1][-1]['cx'] < 120: col_clusters[-1].append(b) else: col_clusters.append([b]) if len(col_clusters) != COLS: raise RuntimeError('列聚类=%d(期望 %d)' % (len(col_clusters), COLS)) col_center = [float(np.mean([b['cx'] for b in c])) for c in col_clusters] # ---- 行:cy 聚类 + 行距推号 ---- by_cy = sorted(blocks, key=lambda b: b['cy']) row_clusters = [] for b in by_cy: if row_clusters and b['cy'] - row_clusters[-1][-1]['cy'] < 120: row_clusters[-1].append(b) else: row_clusters.append([b]) cys = [float(np.mean([b['cy'] for b in c])) for c in row_clusters] diffs = [cys[i + 1] - cys[i] for i in range(len(cys) - 1)] step = float(np.median([d for d in diffs if d < 400])) if diffs else 290.0 row_idx = [int(round((cy - cys[0]) / step)) for cy in cys] if len(set(row_idx)) != len(row_idx) or max(row_idx) >= ROWS or min(row_idx) < 0: raise RuntimeError('行聚类异常: %s' % row_idx) # ---- 行顶线最小二乘拟合(检出的行 -> 预测全部 7 行)---- pts = [(float(ri), float(np.mean([b['y'] for b in c]))) for ri, c in zip(row_idx, row_clusters)] if len(pts) >= 2: ra, rb = np.polyfit([p[0] for p in pts], [p[1] for p in pts], 1) row_top = [float(rb + ra * r) for r in range(ROWS)] else: row_top = [pts[0][1] + step * r for r in range(ROWS)] pw = float(np.median([b['w'] for b in blocks])) ph = float(np.median([b['h'] for b in blocks])) grid = {} for ri, c in zip(row_idx, row_clusters): for b in c: ci = int(np.argmin([abs(cc - b['cx']) for cc in col_center])) key = (ri, ci) if key not in grid or b['area'] > grid[key]['area']: grid[key] = b return {'row_top': row_top, 'col_center': col_center, 'pw': pw, 'ph': ph, 'grid': grid, 'rows_found': sorted(set(row_idx)), 'step': step} def trim_label(crop): """裁掉底部白底店名标签条 —— 间隙定位法(逐行扫描对死区/剖面重叠太脆弱): 1) 向量化行剖面:gap 行=全宽白(dark<0.03 且 mean>=180);text 行=dark 0.08~0.7 且 mean 100~215 2) 在底部 50% 内找连续 gap 段,自底向上取第一个同时满足以下条件的段作为裁切线: a. 段下方有 >=8 个 text 行(标签文字) b. 最后一个 text 行距裁切图底边 <=20 行(标签贴着底边;商业街白框的文字在格子中部,被排除) c. 段上方 10 行内 gap 行 <5(上面是照片内容,不是另一段白) 找不到合格间隙 → 不裁。""" g = cv2.cvtColor(crop, cv2.COLOR_BGR2GRAY) h = g.shape[0] if h < 40: return crop gf = g.astype(np.float32) dark = (gf < 160).mean(axis=1) mean = gf.mean(axis=1) # 列结构指标:文字行 dark 集中在中部(字),间隙/阴影行全宽均匀分布 w = g.shape[1] c0, c1 = int(w * 0.2), int(w * 0.8) dark_c = (gf[:, c0:c1] < 160).mean(axis=1) dark_o = ((gf[:, :c0] < 160).sum(axis=1) + (gf[:, c1:] < 160).sum(axis=1)) / float(w - (c1 - c0)) ratio = (dark_c + 0.004) / (dark_o + 0.004) # 阈值按实测剖面放宽:角落阴影区标签白底 mean 仅 120~180、间隙行 dark 到 0.09; # 稀疏字行(DQ 两字 dark~0.05)靠 ratio 与阴影白区分 is_text = (dark >= 0.04) & (mean <= 220) & (ratio > 2.5) is_gap = (dark < 0.12) & (mean >= 165) & (~is_text) lo = h - 1 - int(h * 0.5) runs = [] y = h - 1 while y > lo: if is_gap[y]: y2 = y while y2 > lo and is_gap[y2]: y2 -= 1 runs.append((y2 + 1, y)) y = y2 else: y -= 1 for (start, end) in runs: # runs 自底向上收集,天然从最低段开始 if end - start + 1 < 4: # 标签内部笔画间的假间隙通常只有 1~3 行 continue below = is_text[end + 1:h] if end + 1 < h else np.zeros(0, bool) if below.sum() < 8: continue last_text = end + 1 + int(np.max(np.nonzero(below))) below_any = (is_text | is_gap)[end + 1:h] if end + 1 < h else np.zeros(0, bool) if not below_any.any(): continue last_below = end + 1 + int(np.max(np.nonzero(below_any))) # 底边允许一段阴影带(既非 text 也非 gap);标签文字与底边之间只允许留白/阴影 if (h - 1) - last_below > 20: continue # 文字与底边内容之间允许留白/阴影:实测最大 32 行(r6c0 兜底矩形探到板面); # 商业街白框的文字距底边 60 行,仍被排除 if last_below - last_text > 35: continue above = is_gap[max(0, start - 10):start] if above.sum() >= 5: continue return crop[:start] return crop def region_has_content(warped, x, y, w, h): # 空白板面平滑(边缘密度/标准差低);有 logo/边框/文字的区域高 hh, ww = warped.shape[:2] x0, y0 = max(0, x), max(0, y) x1, y1 = min(ww, x + w), min(hh, y + h) if x1 - x0 < 20 or y1 - y0 < 20: return False roi = cv2.cvtColor(warped[y0:y1, x0:x1], cv2.COLOR_BGR2GRAY) edges = cv2.Canny(roi, 60, 160) density = float(np.count_nonzero(edges)) / float(edges.size) return density > 0.006 or float(np.std(roi)) > 24 def main(): ap = argparse.ArgumentParser() ap.add_argument('--src', default=DEFAULT_SRC) ap.add_argument('--out', default=DEFAULT_OUT) ap.add_argument('--seed', default=DEFAULT_SEED) ap.add_argument('--preview', default=DEFAULT_PREVIEW) ap.add_argument('--report', default=DEFAULT_REPORT) ap.add_argument('--pad', type=int, default=2) ap.add_argument('--width', type=int, default=440) ap.add_argument('--dry-run', action='store_true') args = ap.parse_args() img = imread_u(args.src) if img is None: print('读取源照片失败:', args.src) sys.exit(1) seed = json.load(open(args.seed, encoding='utf-8')) ids = [s['id'] for s in seed['shops']] quad = detect_board_quad(img) warped = warp_board(img, quad) wh, ww = warped.shape[:2] print('透视矫正完成: %dx%d' % (ww, wh)) blocks = detect_confident_blocks(warped) print('置信照片块: %d 个' % len(blocks)) if len(blocks) < 20: print('置信块太少,终止(避免误裁覆盖好图)') sys.exit(2) g = fit_grid(blocks) print('网格拟合: 行=%s 行距=%.1f 照片=%.0fx%.0f' % (g['rows_found'], g['step'], g['pw'], g['ph'])) report = {'warped_size': [ww, wh], 'quad': np.round(quad).astype(int).tolist(), 'confident': len(blocks), 'detected': [], 'fallback': [], 'skipped': []} os.makedirs(args.out, exist_ok=True) crops = {} half_w = g['pw'] / 2.0 for ri in range(ROWS): for ci in range(COLS): sid = 'r%dc%d' % (ri, ci) b = g['grid'].get((ri, ci)) if b is not None: x0 = max(0, b['x'] - args.pad) y0 = max(0, b['y'] - args.pad) x1 = min(ww, b['x'] + b['w'] + args.pad) y1 = min(wh, b['y'] + b['h'] + args.pad) report['detected'].append(sid) else: cx = g['col_center'][ci] if cx is None: report['skipped'].append(sid) continue x0 = int(max(0, cx - half_w - args.pad)) x1 = int(min(ww, cx + half_w + args.pad)) y0 = int(max(0, g['row_top'][ri] - args.pad)) y1 = int(min(wh, g['row_top'][ri] + g['ph'] + args.pad)) if not region_has_content(warped, x0, y0, x1 - x0, y1 - y0): report['skipped'].append(sid) continue report['fallback'].append(sid) crop = warped[y0:y1, x0:x1] if crop.size == 0: report['skipped'].append(sid) continue crop = trim_label(crop) scale = args.width / float(crop.shape[1]) crop = cv2.resize(crop, (args.width, max(1, int(crop.shape[0] * scale))), interpolation=cv2.INTER_AREA) crops[sid] = crop if not args.dry_run: imwrite_u(os.path.join(args.out, sid + '.jpg'), crop, [cv2.IMWRITE_JPEG_QUALITY, 88]) # 种子中存在但牌子上不存在的 id:清掉旧误裁图,前端显示占位符 if not args.dry_run: for sid in ids: if sid not in crops: p = os.path.join(args.out, sid + '.jpg') if os.path.exists(p): os.remove(p) print('清除旧误裁图:', sid) # 拼版预览(11 列 x 7 行,带 id 标注) cw, ch = 160, 118 lab_h = 18 sheet = np.full((ROWS * (ch + lab_h), COLS * cw, 3), 255, np.uint8) for ri in range(ROWS): for ci in range(COLS): sid = 'r%dc%d' % (ri, ci) x, y = ci * cw, ri * (ch + lab_h) cv2.putText(sheet, sid, (x + 4, y + 13), cv2.FONT_HERSHEY_SIMPLEX, 0.42, (0, 0, 0), 1) c = crops.get(sid) if c is None: cv2.putText(sheet, 'MISS', (x + 40, y + 70), cv2.FONT_HERSHEY_SIMPLEX, 0.6, (0, 0, 255), 2) continue sc = min((cw - 4) / float(c.shape[1]), (ch - 4) / float(c.shape[0])) cc = cv2.resize(c, (max(1, int(c.shape[1] * sc)), max(1, int(c.shape[0] * sc)))) sheet[y + lab_h:y + lab_h + cc.shape[0], x + 2:x + 2 + cc.shape[1]] = cc os.makedirs(os.path.dirname(args.preview), exist_ok=True) imwrite_u(args.preview, sheet, [cv2.IMWRITE_JPEG_QUALITY, 90]) report['written'] = sorted(crops.keys()) os.makedirs(os.path.dirname(args.report), exist_ok=True) json.dump(report, open(args.report, 'w', encoding='utf-8'), ensure_ascii=False, indent=1) print('裁切完成: 直检 %d / 兜底 %d / 跳过 %d -> %s' % ( len(report['detected']), len(report['fallback']), len(report['skipped']), args.preview)) if report['fallback']: print('兜底裁切:', report['fallback']) if report['skipped']: print('跳过(牌子上不存在):', report['skipped']) if __name__ == '__main__': main()