"""image_service.py — 图片提取和处理服务""" import base64 import json from typing import Dict, List from docx import Document from docx.enum.text import WD_ALIGN_PARAGRAPH def extract_images_from_word(doc: Document) -> List[Dict]: """从 Word 文档提取图片(paragraph_index, data_url, style: 宽高/对齐/段落样式)""" images = [] # 建立 rel_id -> 图片数据映射 image_parts = {} for rel in doc.part.rels.values(): if "image" in rel.target_ref: image_parts[rel.rId] = { "blob": rel.target_part.blob, "content_type": rel.target_part.content_type } # 遍历段落查找图片 for para_idx, paragraph in enumerate(doc.paragraphs): for run in paragraph.runs: # 查找 inline 图片(w:drawing 元素) for drawing in run._element.findall('.//{http://schemas.openxmlformats.org/wordprocessingml/2006/main}drawing'): # 提取图片引用 blip = drawing.find('.//{http://schemas.openxmlformats.org/drawingml/2006/main}blip') if blip is None: continue rel_id = blip.get('{http://schemas.openxmlformats.org/officeDocument/2006/relationships}embed') if not rel_id or rel_id not in image_parts: continue # 提取尺寸(EMU 转厘米,1厘米 = 360000 EMU) extent = drawing.find('.//{http://schemas.openxmlformats.org/drawingml/2006/wordprocessingDrawing}extent') width_emu = int(extent.get('cx')) if extent is not None else 360000 * 10 # 默认 10cm height_emu = int(extent.get('cy')) if extent is not None else 360000 * 7 # 默认 7cm width_cm = round(width_emu / 360000, 2) height_cm = round(height_emu / 360000, 2) # 获取对齐方式 align = 'left' if paragraph.alignment == WD_ALIGN_PARAGRAPH.CENTER: align = 'center' elif paragraph.alignment == WD_ALIGN_PARAGRAPH.RIGHT: align = 'right' # 获取段落样式名称 para_style = paragraph.style.name if paragraph.style else "Normal" # Base64 编码 image_blob = image_parts[rel_id]['blob'] content_type = image_parts[rel_id]['content_type'] b64_data = base64.b64encode(image_blob).decode('utf-8') data_url = f"data:{content_type};base64,{b64_data}" images.append({ "paragraph_index": para_idx, "data_url": data_url, "style": { "width": width_cm, "height": height_cm, "align": align, "unit": "cm", "para_style": para_style # 新增:段落样式 } }) return images def create_image_markdown(data_url: str, style: Dict, alt: str = "图片") -> str: """生成带样式注释的图片 Markdown( ![alt](data_url))""" style_json = json.dumps(style, ensure_ascii=False) return f'\n![{alt}]({data_url})\n'