首次提交
This commit is contained in:
@@ -0,0 +1,212 @@
|
||||
from PIL import Image
|
||||
from PIL.Image import Image
|
||||
import cv2
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
|
||||
|
||||
def split_image(image_array: np.ndarray, dirt: int) -> (np.ndarray, np.ndarray):
|
||||
# 确保图像是二维的灰度图
|
||||
if len(image_array.shape) > 2:
|
||||
# 如果是彩色图,转换为灰度图 (取平均值或者只取一个通道)
|
||||
image_array = image_array.mean(axis=2) # 或者直接取第一个通道:image_array = image_array[:, :, 0]
|
||||
|
||||
# 将图片转为二值化(假设白色背景值为255,黑色文字值为0)
|
||||
# 你可以根据图片实际情况调整阈值
|
||||
_, binary_image = cv2.threshold(image_array, 127, 255, cv2.THRESH_BINARY)
|
||||
|
||||
# 寻找切割线
|
||||
split_line = find_split_line(binary_image, dirt)
|
||||
|
||||
if dirt == 1:
|
||||
# 竖着切
|
||||
split1 = image_array[:, :split_line]
|
||||
split2 = image_array[:, split_line:]
|
||||
else:
|
||||
# 横着切
|
||||
split1 = image_array[:split_line, :]
|
||||
split2 = image_array[split_line:, :]
|
||||
|
||||
return split1, split2
|
||||
|
||||
|
||||
def find_split_line(image_array, dirt):
|
||||
height, width = image_array.shape # 现在应该是 (height, width)
|
||||
|
||||
# 根据dirt值选择切割方向
|
||||
if dirt == 1:
|
||||
# 竖着切
|
||||
middle = width // 2
|
||||
for x in range(middle, width):
|
||||
# 检查竖直方向的列是否全为空白
|
||||
if np.all(image_array[:, x] > 200): # 假设200是空白阈值
|
||||
return x
|
||||
return middle # 如果找不到合适的切割线,返回中间
|
||||
else:
|
||||
# 横着切
|
||||
middle = height // 2
|
||||
for y in range(middle, height):
|
||||
# 检查水平方向的行是否全为空白
|
||||
if np.all(image_array[y, :] > 200): # 假设200是空白阈值
|
||||
return y
|
||||
return middle # 如果找不到合适的切割线,返回中间
|
||||
|
||||
|
||||
async def image_split(image: np.ndarray, dirt: int):
|
||||
part1, part2 = split_image(image, dirt)
|
||||
# 将 NumPy 数组转换为 Pillow 图像对象
|
||||
pil_image1 = Image.fromarray(part1)
|
||||
pil_image2 = Image.fromarray(part2)
|
||||
# 显示图像
|
||||
pil_image1.show()
|
||||
pil_image2.show()
|
||||
return pil_image1, pil_image2
|
||||
|
||||
|
||||
async def get_able_area(image, coords, text):
|
||||
"""
|
||||
根据给定的文本区域坐标,从图像中截取文本区域。
|
||||
|
||||
:param text:
|
||||
:param image: 原始图像 (使用cv2.imread读取的图像)
|
||||
:param coords: 文本区域的四个角坐标,格式为:
|
||||
[[x1_top_left, y1_top_left],
|
||||
[x2_top_right, y2_top_right],
|
||||
[x3_bottom_right, y3_bottom_right],
|
||||
[x4_bottom_left, y4_bottom_left]]
|
||||
:return: 截取的文本区域图像
|
||||
"""
|
||||
# 将坐标转换为numpy数组并重新整理为 [[x1, y1], [x2, y2], [x3, y3], [x4, y4]] 的形式
|
||||
pts = np.float32([coords[0], coords[1], coords[2], coords[3]])
|
||||
|
||||
# 计算变换后的目标矩形的宽和高
|
||||
top_width = np.linalg.norm(np.array(coords[1]) - np.array(coords[0])) # 顶边长度
|
||||
bottom_width = np.linalg.norm(np.array(coords[2]) - np.array(coords[3])) # 底边长度
|
||||
height = np.linalg.norm(np.array(coords[2]) - np.array(coords[1])) # 高
|
||||
|
||||
# 取最大宽度作为目标宽度
|
||||
max_width = max(top_width, bottom_width)
|
||||
|
||||
# 目标矩形的四个角点
|
||||
dst = np.float32([[0, 0], [max_width - 1, 0], [max_width - 1, height - 1], [0, height - 1]])
|
||||
|
||||
# 计算透视变换矩阵
|
||||
M = cv2.getPerspectiveTransform(pts, dst)
|
||||
|
||||
# 进行透视变换,获取裁剪后的图像
|
||||
cropped_image = cv2.warpPerspective(image, M, (int(max_width), int(height)))
|
||||
await get_box_size(cropped_image, text)
|
||||
|
||||
|
||||
async def get_box_size(image, text):
|
||||
# 预处理图像
|
||||
gray = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
|
||||
_, binary = cv2.threshold(gray, 128, 255, cv2.THRESH_BINARY_INV)
|
||||
|
||||
# 查找轮廓
|
||||
contours, _ = cv2.findContours(binary, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
|
||||
|
||||
# 存储所有文字大小
|
||||
text_sizes = []
|
||||
|
||||
# 遍历每个轮廓
|
||||
for cnt in contours:
|
||||
# 获取包围盒的坐标和尺寸
|
||||
x, y, w, h = cv2.boundingRect(cnt)
|
||||
|
||||
# 计算文字区域的面积
|
||||
area = cv2.contourArea(cnt)
|
||||
|
||||
# 假设每个文字区域有一个平均字符数 (可以根据实际情况调整)
|
||||
avg_char_width = w // len(text) # 假设每行有10个字符
|
||||
avg_char_height = h // 1 # 假设每行有1个字符
|
||||
|
||||
# 计算字符的大小
|
||||
character_size = (avg_char_width, avg_char_height)
|
||||
text_sizes.append(character_size)
|
||||
|
||||
# 可视化结果
|
||||
cv2.rectangle(image, (x, y), (x + h, y + h), (0, 255, 0), 2)
|
||||
|
||||
# 计算平均字符大小
|
||||
if len(text_sizes) > 0:
|
||||
avg_char_width = sum(size[0] for size in text_sizes) / len(text_sizes)
|
||||
avg_char_height = sum(size[1] for size in text_sizes) / len(text_sizes)
|
||||
print(f"平均字符宽度: {avg_char_width}, 平均字符高度: {avg_char_height}")
|
||||
else:
|
||||
print("未检测到文字区域")
|
||||
|
||||
# 显示结果图像
|
||||
im = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)
|
||||
il_image1 = Image.fromarray(im)
|
||||
il_image1.show()
|
||||
|
||||
#增强图片
|
||||
async def positive_image(image):
|
||||
# 1. 灰度化(注意:输入为 RGB,须用 COLOR_RGB2GRAY;误用 BGR 会翻转红蓝通道,彩色文字/背景灰度失真)
|
||||
gray_img = cv2.cvtColor(image, cv2.COLOR_RGB2GRAY)
|
||||
|
||||
# 2. 对比度增强(CLAHE)
|
||||
clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))
|
||||
enhanced = clahe.apply(gray_img)
|
||||
|
||||
# 3. 锐化(可选)
|
||||
blurred = cv2.GaussianBlur(enhanced, (0, 0), 2)
|
||||
sharpened = cv2.addWeighted(enhanced, 1.5, blurred, -0.5, 0)
|
||||
|
||||
return sharpened
|
||||
|
||||
|
||||
def draw_text_box(image, coords):
|
||||
pos1 = coords[0]
|
||||
pos2 = coords[2]
|
||||
x1 = int(pos1[0])
|
||||
y1 = int(pos1[1])
|
||||
x2 = int(pos2[0])
|
||||
y2 = int(pos2[1])
|
||||
# 可视化结果
|
||||
cv2.rectangle(image, (x1, y1), (x2, y2), (0, 255, 0), 1)
|
||||
|
||||
|
||||
# 倾斜矫正
|
||||
async def tilt_correction(image):
|
||||
# 转为灰度图
|
||||
gray_img = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)
|
||||
|
||||
# 二值化
|
||||
_, binary_img = cv2.threshold(gray_img, 128, 255, cv2.THRESH_BINARY)
|
||||
|
||||
# 边缘检测
|
||||
edges = cv2.Canny(binary_img, 50, 150)
|
||||
|
||||
# 提取轮廓
|
||||
contours, _ = cv2.findContours(edges, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)
|
||||
|
||||
if len(contours) == 0:
|
||||
print("No contours found.")
|
||||
return image # 没有找到轮廓,返回原图
|
||||
|
||||
# 找到最大轮廓
|
||||
largest_contour = max(contours, key=cv2.contourArea)
|
||||
|
||||
# 计算最小外接矩形的角度
|
||||
rect = cv2.minAreaRect(largest_contour)
|
||||
angle = rect[-1]
|
||||
|
||||
# 修正角度
|
||||
if angle < -45:
|
||||
angle = -(90 + angle) # 修正角度,使其在 [-90, 90] 范围内
|
||||
else:
|
||||
angle = -angle # 直接取反号
|
||||
|
||||
# 如果角度接近0度,则不进行旋转
|
||||
if abs(angle) < 0.5:
|
||||
return image # 角度非常小,几乎不需要旋转
|
||||
|
||||
# 旋转图像
|
||||
(h, w) = image.shape[:2] # 使用原始图像的形状
|
||||
center = (w // 2, h // 2)
|
||||
M = cv2.getRotationMatrix2D(center, angle, 1.0)
|
||||
rotated_img = cv2.warpAffine(image, M, (w, h), flags=cv2.INTER_CUBIC, borderMode=cv2.BORDER_REPLICATE)
|
||||
|
||||
return rotated_img
|
||||
+116
@@ -0,0 +1,116 @@
|
||||
import math
|
||||
from PIL import Image as pli
|
||||
from PIL.Image import Image
|
||||
import cv2
|
||||
import numpy as np
|
||||
from PIL import Image
|
||||
|
||||
|
||||
def distance(p1, p2):
|
||||
return math.sqrt((p1[0] - p2[0]) ** 2 + (p1[1] - p2[1]) ** 2)
|
||||
|
||||
|
||||
def find_edge_lengths(points):
|
||||
# 计算所有可能的边的长度
|
||||
edge_lengths = []
|
||||
for i in range(len(points)):
|
||||
for j in range(i + 1, len(points)):
|
||||
edge_lengths.append(distance(points[i], points[j]))
|
||||
return edge_lengths
|
||||
|
||||
|
||||
def find_opposite_edges(edge_lengths):
|
||||
# 对边的长度应该是成对的
|
||||
edge_lengths.sort()
|
||||
return edge_lengths
|
||||
|
||||
|
||||
async def determine_orientation(points):
|
||||
# 提取所有点的x和y坐标
|
||||
x_coords = [p[0] for p in points]
|
||||
y_coords = [p[1] for p in points]
|
||||
|
||||
# 计算x和y坐标的最大值和最小值
|
||||
x_max, x_min = max(x_coords), min(x_coords)
|
||||
y_max, y_min = max(y_coords), min(y_coords)
|
||||
|
||||
# 计算宽度和高度
|
||||
width = x_max - x_min
|
||||
height = y_max - y_min
|
||||
|
||||
edge_lengths = find_edge_lengths(points)
|
||||
# 计算对边长度
|
||||
opposite_edges = find_opposite_edges(edge_lengths)
|
||||
|
||||
# 判断竖着还是横着
|
||||
if height > width:
|
||||
return 1, opposite_edges
|
||||
else:
|
||||
return 0, opposite_edges
|
||||
|
||||
|
||||
async def split_text_box(direction, opposite_edges, text_content, points):
|
||||
# 左上,右上,右下,左下
|
||||
point_lu_x, point_lu_y = points[0]
|
||||
point_ru_x, point_ru_y = points[1]
|
||||
point_rd_x, point_rd_y = points[2]
|
||||
point_ld_x, point_ld_y = points[3]
|
||||
if direction == 0:
|
||||
# 文本框 长度
|
||||
box_length_l = point_ru_x - point_lu_x
|
||||
box_length_r = point_ld_x - point_rd_x
|
||||
text_length = len(text_content)
|
||||
each_text_size = (box_length_l + box_length_r) / 2 / text_length
|
||||
min_size = min(opposite_edges)
|
||||
cha = min_size / 6
|
||||
print(f'差值:{cha}')
|
||||
word_box = []
|
||||
point_lu_x_sp = point_lu_x
|
||||
point_ru_x_sp = point_ru_x
|
||||
point_rd_x_sp = point_rd_x
|
||||
point_ld_x_sp = point_ld_x
|
||||
for char in text_content:
|
||||
word_const = {
|
||||
"text": char,
|
||||
"box": [[point_lu_x_sp, point_lu_y],
|
||||
[point_ru_x_sp, point_ru_y],
|
||||
[point_rd_x_sp, point_rd_y],
|
||||
[point_ld_x_sp, point_ld_y]]
|
||||
}
|
||||
point_lu_x_sp += min_size + cha
|
||||
point_ru_x_sp += min_size + cha
|
||||
point_rd_x_sp += min_size + cha
|
||||
point_ld_x_sp += min_size + cha
|
||||
word_box.append(word_const)
|
||||
return word_box
|
||||
elif direction == 1:
|
||||
# 文本框 长度
|
||||
box_length_l = point_ru_y - point_lu_y
|
||||
box_length_r = point_ld_y - point_rd_y
|
||||
text_length = len(text_content)
|
||||
each_text_size = (box_length_l + box_length_r) / 2 / text_length
|
||||
min_size = min(opposite_edges)
|
||||
cha = min_size / 5
|
||||
print(f'差值:{cha}')
|
||||
word_box = []
|
||||
point_lu_y_sp = point_lu_y
|
||||
point_ru_y_sp = point_ru_y
|
||||
point_rd_y_sp = point_rd_y
|
||||
point_ld_y_sp = point_ld_y
|
||||
for char in text_content:
|
||||
word_const = {
|
||||
"text": char,
|
||||
"box": [[point_lu_x, point_lu_y_sp],
|
||||
[point_ru_x, point_ru_y_sp],
|
||||
[point_rd_x, point_rd_y_sp],
|
||||
[point_ld_x, point_ld_y_sp]]
|
||||
}
|
||||
point_lu_y_sp += min_size + cha
|
||||
point_ru_y_sp += min_size + cha
|
||||
point_rd_y_sp += min_size + cha
|
||||
point_ld_y_sp += min_size + cha
|
||||
word_box.append(word_const)
|
||||
return word_box
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user