Files
ocr-server/utils/image_utils.py
T

213 lines
7.1 KiB
Python
Raw Normal View History

2026-08-31 16:41:12 +08:00
from PIL import Image
from PIL.Image import Image
import cv2
import numpy as np
from PIL import Image
def split_image(image_array: np.ndarray, dirt: int) -> (np.ndarray, np.ndarray):
# 确保图像是二维的灰度图
if len(image_array.shape) > 2:
# 如果是彩色图,转换为灰度图 (取平均值或者只取一个通道)
image_array = image_array.mean(axis=2) # 或者直接取第一个通道:image_array = image_array[:, :, 0]
# 将图片转为二值化(假设白色背景值为255,黑色文字值为0)
# 你可以根据图片实际情况调整阈值
_, binary_image = cv2.threshold(image_array, 127, 255, cv2.THRESH_BINARY)
# 寻找切割线
split_line = find_split_line(binary_image, dirt)
if dirt == 1:
# 竖着切
split1 = image_array[:, :split_line]
split2 = image_array[:, split_line:]
else:
# 横着切
split1 = image_array[:split_line, :]
split2 = image_array[split_line:, :]
return split1, split2
def find_split_line(image_array, dirt):
height, width = image_array.shape # 现在应该是 (height, width)
# 根据dirt值选择切割方向
if dirt == 1:
# 竖着切
middle = width // 2
for x in range(middle, width):
# 检查竖直方向的列是否全为空白
if np.all(image_array[:, x] > 200): # 假设200是空白阈值
return x
return middle # 如果找不到合适的切割线,返回中间
else:
# 横着切
middle = height // 2
for y in range(middle, height):
# 检查水平方向的行是否全为空白
if np.all(image_array[y, :] > 200): # 假设200是空白阈值
return y
return middle # 如果找不到合适的切割线,返回中间
async def image_split(image: np.ndarray, dirt: int):
part1, part2 = split_image(image, dirt)
# 将 NumPy 数组转换为 Pillow 图像对象
pil_image1 = Image.fromarray(part1)
pil_image2 = Image.fromarray(part2)
# 显示图像
pil_image1.show()
pil_image2.show()
return pil_image1, pil_image2
async def get_able_area(image, coords, text):
"""
根据给定的文本区域坐标,从图像中截取文本区域。
:param text:
:param image: 原始图像 (使用cv2.imread读取的图像)
:param coords: 文本区域的四个角坐标,格式为:
[[x1_top_left, y1_top_left],
[x2_top_right, y2_top_right],
[x3_bottom_right, y3_bottom_right],
[x4_bottom_left, y4_bottom_left]]
:return: 截取的文本区域图像
"""
# 将坐标转换为numpy数组并重新整理为 [[x1, y1], [x2, y2], [x3, y3], [x4, y4]] 的形式
pts = np.float32([coords[0], coords[1], coords[2], coords[3]])
# 计算变换后的目标矩形的宽和高
top_width = np.linalg.norm(np.array(coords[1]) - np.array(coords[0])) # 顶边长度
bottom_width = np.linalg.norm(np.array(coords[2]) - np.array(coords[3])) # 底边长度
height = np.linalg.norm(np.array(coords[2]) - np.array(coords[1])) # 高
# 取最大宽度作为目标宽度
max_width = max(top_width, bottom_width)
# 目标矩形的四个角点
dst = np.float32([[0, 0], [max_width - 1, 0], [max_width - 1, height - 1], [0, height - 1]])
# 计算透视变换矩阵
M = cv2.getPerspectiveTransform(pts, dst)
# 进行透视变换,获取裁剪后的图像
cropped_image = cv2.warpPerspective(image, M, (int(max_width), int(height)))
await get_box_size(cropped_image, text)
async def get_box_size(image, text):
# 预处理图像
gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
_, binary = cv2.threshold(gray, 128, 255, cv2.THRESH_BINARY_INV)
# 查找轮廓
contours, _ = cv2.findContours(binary, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
# 存储所有文字大小
text_sizes = []
# 遍历每个轮廓
for cnt in contours:
# 获取包围盒的坐标和尺寸
x, y, w, h = cv2.boundingRect(cnt)
# 计算文字区域的面积
area = cv2.contourArea(cnt)
# 假设每个文字区域有一个平均字符数 (可以根据实际情况调整)
avg_char_width = w // len(text) # 假设每行有10个字符
avg_char_height = h // 1 # 假设每行有1个字符
# 计算字符的大小
character_size = (avg_char_width, avg_char_height)
text_sizes.append(character_size)
# 可视化结果
cv2.rectangle(image, (x, y), (x + h, y + h), (0, 255, 0), 2)
# 计算平均字符大小
if len(text_sizes) > 0:
avg_char_width = sum(size[0] for size in text_sizes) / len(text_sizes)
avg_char_height = sum(size[1] for size in text_sizes) / len(text_sizes)
print(f"平均字符宽度: {avg_char_width}, 平均字符高度: {avg_char_height}")
else:
print("未检测到文字区域")
# 显示结果图像
im = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)
il_image1 = Image.fromarray(im)
il_image1.show()
#增强图片
async def positive_image(image):
# 1. 灰度化(注意:输入为 RGB,须用 COLOR_RGB2GRAY;误用 BGR 会翻转红蓝通道,彩色文字/背景灰度失真)
gray_img = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)
# 2. 对比度增强(CLAHE)
clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))
enhanced = clahe.apply(gray_img)
# 3. 锐化(可选)
blurred = cv2.GaussianBlur(enhanced, (0, 0), 2)
sharpened = cv2.addWeighted(enhanced, 1.5, blurred, -0.5, 0)
return sharpened
def draw_text_box(image, coords):
pos1 = coords[0]
pos2 = coords[2]
x1 = int(pos1[0])
y1 = int(pos1[1])
x2 = int(pos2[0])
y2 = int(pos2[1])
# 可视化结果
cv2.rectangle(image, (x1, y1), (x2, y2), (0, 255, 0), 1)
# 倾斜矫正
async def tilt_correction(image):
# 转为灰度图
gray_img = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
# 二值化
_, binary_img = cv2.threshold(gray_img, 128, 255, cv2.THRESH_BINARY)
# 边缘检测
edges = cv2.Canny(binary_img, 50, 150)
# 提取轮廓
contours, _ = cv2.findContours(edges, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
if len(contours) == 0:
print("No contours found.")
return image # 没有找到轮廓,返回原图
# 找到最大轮廓
largest_contour = max(contours, key=cv2.contourArea)
# 计算最小外接矩形的角度
rect = cv2.minAreaRect(largest_contour)
angle = rect[-1]
# 修正角度
if angle < -45:
angle = -(90 + angle) # 修正角度,使其在 [-90, 90] 范围内
else:
angle = -angle # 直接取反号
# 如果角度接近0度,则不进行旋转
if abs(angle) < 0.5:
return image # 角度非常小,几乎不需要旋转
# 旋转图像
(h, w) = image.shape[:2] # 使用原始图像的形状
center = (w // 2, h // 2)
M = cv2.getRotationMatrix2D(center, angle, 1.0)
rotated_img = cv2.warpAffine(image, M, (w, h), flags=cv2.INTER_CUBIC, borderMode=cv2.BORDER_REPLICATE)
return rotated_img