diff --git a/core/parser.py b/core/parser.py index 5a66934..c3d3411 100644 --- a/core/parser.py +++ b/core/parser.py @@ -195,9 +195,9 @@ class K3VoucherParser: result['period'] = int(m.group(1)) elif '合计' in t and any(c in t for c in '零壹贰叁肆伍陆柒捌玖拾佰仟万亿'): result['total_str'] = t - m = re.search(r'([\d,]+\.\d{2})', t.replace(',', '')) + m = re.search(r'(\d+\.\d{2})', re.sub(r'[^\d.]', '', t)) if m: - result['total_amount'] = float(m.group(1).replace(',', '')) + result['total_amount'] = float(m.group(1)) elif t.startswith('制单') or t.startswith('制单人'): result['preparer'] = t.split(':')[-1].split(':')[-1].strip() elif t.startswith('审核'): @@ -233,8 +233,8 @@ class K3VoucherParser: continue # x坐标:科目代码通常在表格中间区域 - if cx < 1500 or cx > 2600: - continue + # 注:不设硬x过滤器,正则 \d{3,4}\.\d 和首位数 1000-9999 校验已足够精准 + # 设 x 过滤器会误杀 PDF 版本(排版偏左)和少部分照片(x=1297 也有代码) t = text.strip() # 排除表头文字 @@ -415,11 +415,13 @@ class K3VoucherParser: if any(k in text for k in ['Manager', 'HONOR', 'Magic6', 'RMB']): continue - m = re.search(r'([\d,]+\.\d{2})', text.replace(',', '')) + # 提取金额:去掉所有非数字字符(逗号/空格等),避免"11, 075.68"被解析为75.68 + cleaned = re.sub(r'[^\d.]', '', text) + m = re.search(r'(\d+\.\d{2})', cleaned) if not m: continue - - val = float(m.group(1).replace(',', '')) + + val = float(m.group(1)) # 排除年份 if 1900 < val < 2100: continue diff --git a/kocr.py b/kocr.py index 31bcf05..a7eff3f 100644 --- a/kocr.py +++ b/kocr.py @@ -220,6 +220,33 @@ def extract_pdf_pages(pdf_path, output_dir): new_name = f"_{pdf_name}_p{page_num}.jpg" new_path = os.path.join(output_dir, new_name) os.rename(old_path, new_path) + + # 统一图片尺寸:缩放+白边到 3072×4096(对齐照片坐标) + # 解析器硬编码了 x/y 坐标阈值(适配 3072×4096 的照片) + # PDF 图片尺寸和比例不同,必须统一到照片尺寸才不走样 + try: + import cv2 + img = cv2.imread(new_path) + if img is not None and (img.shape[1] != 3072 or img.shape[0] != 4096): + h, w = img.shape[:2] + # 计算缩放比例(保持宽高比,适配目标尺寸) + scale = min(3072 / w, 4096 / h) + new_w = int(w * scale) + new_h = int(h * scale) + # 缩放 + resized = cv2.resize(img, (new_w, new_h), interpolation=cv2.INTER_CUBIC) + # 中心对齐,加白边补齐到 3072×4096 + pad_top = (4096 - new_h) // 2 + pad_bottom = 4096 - new_h - pad_top + pad_left = (3072 - new_w) // 2 + pad_right = 3072 - new_w - pad_left + padded = cv2.copyMakeBorder(resized, pad_top, pad_bottom, + pad_left, pad_right, + cv2.BORDER_CONSTANT, value=[255, 255, 255]) + cv2.imwrite(new_path, padded) + except Exception as e: + pass # 缩放失败不影响主流程 + page_label = f"第{page_num}页" if m else os.path.basename(pdf_path) result_list.append((new_path, page_label))