from PIL import Image from PIL.Image import Image import cv2 import numpy as np from PIL import Image def split_image(image_array: np.ndarray, dirt: int) -> (np.ndarray, np.ndarray): # 确保图像是二维的灰度图 if len(image_array.shape) > 2: # 如果是彩色图,转换为灰度图 (取平均值或者只取一个通道) image_array = image_array.mean(axis=2) # 或者直接取第一个通道:image_array = image_array[:, :, 0] # 将图片转为二值化(假设白色背景值为255,黑色文字值为0) # 你可以根据图片实际情况调整阈值 _, binary_image = cv2.threshold(image_array, 127, 255, cv2.THRESH_BINARY) # 寻找切割线 split_line = find_split_line(binary_image, dirt) if dirt == 1: # 竖着切 split1 = image_array[:, :split_line] split2 = image_array[:, split_line:] else: # 横着切 split1 = image_array[:split_line, :] split2 = image_array[split_line:, :] return split1, split2 def find_split_line(image_array, dirt): height, width = image_array.shape # 现在应该是 (height, width) # 根据dirt值选择切割方向 if dirt == 1: # 竖着切 middle = width // 2 for x in range(middle, width): # 检查竖直方向的列是否全为空白 if np.all(image_array[:, x] > 200): # 假设200是空白阈值 return x return middle # 如果找不到合适的切割线,返回中间 else: # 横着切 middle = height // 2 for y in range(middle, height): # 检查水平方向的行是否全为空白 if np.all(image_array[y, :] > 200): # 假设200是空白阈值 return y return middle # 如果找不到合适的切割线,返回中间 async def image_split(image: np.ndarray, dirt: int): part1, part2 = split_image(image, dirt) # 将 NumPy 数组转换为 Pillow 图像对象 pil_image1 = Image.fromarray(part1) pil_image2 = Image.fromarray(part2) # 显示图像 pil_image1.show() pil_image2.show() return pil_image1, pil_image2 async def get_able_area(image, coords, text): """ 根据给定的文本区域坐标,从图像中截取文本区域。 :param text: :param image: 原始图像 (使用cv2.imread读取的图像) :param coords: 文本区域的四个角坐标,格式为: [[x1_top_left, y1_top_left], [x2_top_right, y2_top_right], [x3_bottom_right, y3_bottom_right], [x4_bottom_left, y4_bottom_left]] :return: 截取的文本区域图像 """ # 将坐标转换为numpy数组并重新整理为 [[x1, y1], [x2, y2], [x3, y3], [x4, y4]] 的形式 pts = np.float32([coords[0], coords[1], coords[2], coords[3]]) # 计算变换后的目标矩形的宽和高 top_width = np.linalg.norm(np.array(coords[1]) - np.array(coords[0])) # 顶边长度 bottom_width = np.linalg.norm(np.array(coords[2]) - np.array(coords[3])) # 底边长度 height = np.linalg.norm(np.array(coords[2]) - np.array(coords[1])) # 高 # 取最大宽度作为目标宽度 max_width = max(top_width, bottom_width) # 目标矩形的四个角点 dst = np.float32([[0, 0], [max_width - 1, 0], [max_width - 1, height - 1], [0, height - 1]]) # 计算透视变换矩阵 M = cv2.getPerspectiveTransform(pts, dst) # 进行透视变换,获取裁剪后的图像 cropped_image = cv2.warpPerspective(image, M, (int(max_width), int(height))) await get_box_size(cropped_image, text) async def get_box_size(image, text): # 预处理图像 gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY) _, binary = cv2.threshold(gray, 128, 255, cv2.THRESH_BINARY_INV) # 查找轮廓 contours, _ = cv2.findContours(binary, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) # 存储所有文字大小 text_sizes = [] # 遍历每个轮廓 for cnt in contours: # 获取包围盒的坐标和尺寸 x, y, w, h = cv2.boundingRect(cnt) # 计算文字区域的面积 area = cv2.contourArea(cnt) # 假设每个文字区域有一个平均字符数 (可以根据实际情况调整) avg_char_width = w // len(text) # 假设每行有10个字符 avg_char_height = h // 1 # 假设每行有1个字符 # 计算字符的大小 character_size = (avg_char_width, avg_char_height) text_sizes.append(character_size) # 可视化结果 cv2.rectangle(image, (x, y), (x + h, y + h), (0, 255, 0), 2) # 计算平均字符大小 if len(text_sizes) > 0: avg_char_width = sum(size[0] for size in text_sizes) / len(text_sizes) avg_char_height = sum(size[1] for size in text_sizes) / len(text_sizes) print(f"平均字符宽度: {avg_char_width}, 平均字符高度: {avg_char_height}") else: print("未检测到文字区域") # 显示结果图像 im = cv2.cvtColor(image, cv2.COLOR_BGR2RGB) il_image1 = Image.fromarray(im) il_image1.show() #增强图片 async def positive_image(image): # 1. 灰度化(注意:输入为 RGB,须用 COLOR_RGB2GRAY;误用 BGR 会翻转红蓝通道,彩色文字/背景灰度失真) gray_img = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY) # 2. 对比度增强(CLAHE) clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8)) enhanced = clahe.apply(gray_img) # 3. 锐化(可选) blurred = cv2.GaussianBlur(enhanced, (0, 0), 2) sharpened = cv2.addWeighted(enhanced, 1.5, blurred, -0.5, 0) return sharpened def draw_text_box(image, coords): pos1 = coords[0] pos2 = coords[2] x1 = int(pos1[0]) y1 = int(pos1[1]) x2 = int(pos2[0]) y2 = int(pos2[1]) # 可视化结果 cv2.rectangle(image, (x1, y1), (x2, y2), (0, 255, 0), 1) # 倾斜矫正 async def tilt_correction(image): # 转为灰度图 gray_img = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY) # 二值化 _, binary_img = cv2.threshold(gray_img, 128, 255, cv2.THRESH_BINARY) # 边缘检测 edges = cv2.Canny(binary_img, 50, 150) # 提取轮廓 contours, _ = cv2.findContours(edges, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE) if len(contours) == 0: print("No contours found.") return image # 没有找到轮廓,返回原图 # 找到最大轮廓 largest_contour = max(contours, key=cv2.contourArea) # 计算最小外接矩形的角度 rect = cv2.minAreaRect(largest_contour) angle = rect[-1] # 修正角度 if angle < -45: angle = -(90 + angle) # 修正角度,使其在 [-90, 90] 范围内 else: angle = -angle # 直接取反号 # 如果角度接近0度,则不进行旋转 if abs(angle) < 0.5: return image # 角度非常小,几乎不需要旋转 # 旋转图像 (h, w) = image.shape[:2] # 使用原始图像的形状 center = (w // 2, h // 2) M = cv2.getRotationMatrix2D(center, angle, 1.0) rotated_img = cv2.warpAffine(image, M, (w, h), flags=cv2.INTER_CUBIC, borderMode=cv2.BORDER_REPLICATE) return rotated_img