"""研发报告 PPT 生成:markdown 报告正文 → .pptx(专业排版,可下载查证)。 触发时机:opp_submit_report 提交人工确认时生成/更新,存项目工作空间 deliverables/。 设计系统(2026-09-11 重做——旧版纯白底+平铺要点,用户反馈"没有排版和风格,很难看"): - 统一配色:深海军蓝主色 + 亮蓝强调 + 中性灰阶,全文一致不混色; - 封面页:深色满版背景 + 大标题 + 方向标签 + 日期 + 数据来源署名; - 章节页:四节模板(研发场景/输入/输出/成果)各一张深色分隔页,带节序号; - 内容页:顶部强调条 + 标题 + 正文区,多级要点缩进,粗体引导词单独着色; - **表格真渲染**:markdown 表格转 pptx Table(表头深蓝白字、隔行浅底),不再拍平成文本行; - 页脚:页码 + 署名;长内容自动分页(≤7 要点/页); - python-pptx 纯后端,无外部模板依赖;import 失败如实降级报错(不静默产出丑文件)。 """ import logging import os import re logger = logging.getLogger("pipeline.opportunity.ppt") # ── 设计令牌(配色/字号/版式,集中一处便于统一调风格)── NAVY = (0x0E, 0x2A, 0x5C) # 主色:深海军蓝(封面/章节页背景、表头) NAVY_DARK = (0x08, 0x1C, 0x3E) # 封面渐变下段 ACCENT = (0x2F, 0x6F, 0xED) # 强调亮蓝(顶部条、要点符、引导词) ACCENT_SOFT = (0xE8, 0xF0, 0xFE) # 强调浅底(要点卡片底) INK = (0x1F, 0x29, 0x37) # 正文墨色 SLATE = (0x47, 0x55, 0x69) # 次级正文 MUTED = (0x94, 0xA3, 0xB8) # 弱化(署名/页码) LINE = (0xE2, 0xE8, 0xF0) # 分隔线 ROW_ALT = (0xF6, 0xF8, 0xFC) # 表格隔行 WHITE = (0xFF, 0xFF, 0xFF) FONT = "Microsoft YaHei" # 中文优先(缺失时 PowerPoint 自动回退) SW, SH = 13.333, 7.5 # 16:9 英寸 MARGIN_X = 0.72 CONTENT_TOP = 1.42 CONTENT_H = 5.18 MAX_BULLETS_PER_PAGE = 7 # 四节模板顺序(与 opp_report_capability.REPORT_SECTIONS 一致) SECTION_ORDER = ("研发场景", "输入", "输出", "成果") def _rgb(t): from pptx.dml.color import RGBColor return RGBColor(t[0], t[1], t[2]) # ══════════════════════ markdown 解析 ══════════════════════ def _split_blocks(md): """把 markdown 段切成结构化块:[('h', lvl, text) | ('p', text) | ('ul', lvl, text) | ('ol', num, text) | ('table', [rows]) | ('quote', text)]。""" blocks = [] lines = (md or "").split("\n") i = 0 while i < len(lines): raw = lines[i].rstrip() s = raw.strip() if not s: i += 1 continue # 表格:当前行是 | 开头且下一行是分隔线 if s.startswith("|") and i + 1 < len(lines) and re.match(r"^\s*\|?[\s:|-]+\|?\s*$", lines[i + 1]) \ and "-" in lines[i + 1]: rows = [_split_row(s)] i += 2 while i < len(lines) and lines[i].strip().startswith("|"): rows.append(_split_row(lines[i].strip())) i += 1 blocks.append(("table", rows)) continue m = re.match(r"^(#{1,4})\s+(.+)$", s) if m: blocks.append(("h", len(m.group(1)), m.group(2).strip())) i += 1 continue m = re.match(r"^>\s*(.+)$", s) if m: blocks.append(("quote", m.group(1).strip())) i += 1 continue m = re.match(r"^[-*+]\s+(.+)$", s) if m: indent = len(raw) - len(raw.lstrip()) lvl = 1 if indent < 2 else (2 if indent < 4 else 3) blocks.append(("ul", lvl, m.group(1).strip())) i += 1 continue m = re.match(r"^(\d+)[.、)]\s*(.+)$", s) if m: blocks.append(("ol", m.group(1), m.group(2).strip())) i += 1 continue if re.match(r"^[-=*_]{3,}$", s): # 水平线 → 忽略 i += 1 continue # 普通段落:合并连续行 para = [s] i += 1 while i < len(lines): ns = lines[i].strip() if not ns or ns.startswith(("#", "|", ">", "- ", "* ", "+ ")) \ or re.match(r"^\d+[.、)]", ns) or re.match(r"^[-=*_]{3,}$", ns): break para.append(ns) i += 1 blocks.append(("p", " ".join(para))) return blocks def _split_row(line): """markdown 表格行 → 单元格列表。""" s = line.strip() if s.startswith("|"): s = s[1:] if s.endswith("|"): s = s[:-1] return [_clean_inline(c.strip()) for c in s.split("|")] def _clean_inline(s): """去除行内 markdown 标记(粗体/斜体/行内码/链接)。""" s = s or "" s = re.sub(r"\*\*(.+?)\*\*", r"\1", s) s = re.sub(r"__(.+?)__", r"\1", s) s = re.sub(r"(? 0x2E7F else 0.55 for ch in (s or "")) def _add_table(slide, rows, top, width=None): """渲染 markdown 表格为 pptx Table;返回占用高度(英寸)。超宽自动缩字号。 行高按内容估算逐行设置(2026-09-11 实测教训:统一行高时 LibreOffice/WPS 不按换行撑行 → 长文本在行底被截断)。总高超页则截行加提示(不跨页拆表, 保持单表完整可读)。 """ from pptx.util import Inches, Pt import math if not rows or not rows[0]: return 0.0 width = width or (SW - 2 * MARGIN_X) ncol = max(len(r) for r in rows) rows = [r + [""] * (ncol - len(r)) for r in rows] nrow = len(rows) avail = SH - 0.95 - top # 列宽:首列略宽(多为名称),其余均分 first = min(width * 0.30, 4.6) rest = (width - first) / max(1, ncol - 1) col_w = [first if ci == 0 else rest for ci in range(ncol)] fs = 11 if ncol <= 5 else (10 if ncol <= 7 else 9) line_h = fs * 1.45 / 72.0 # 每行文本高度(英寸,含行距余量) pad = 0.12 # 单元格上下边距+余量 def row_height(row_vals): lines = 1 for ci, val in enumerate(row_vals): cap = max(2.0, (col_w[ci] - 0.16) * 72.0 / fs) # 每行可容纳字符单位 n = 0 for seg in (val or "")[:180].split("\n"): # 显式换行强制断行 n += max(1, math.ceil(_disp_w(seg) / cap)) lines = max(lines, n) return lines * line_h + pad heights = [row_height(r) for r in rows] total_h = sum(heights) if total_h > avail: # 从尾行往前砍到放得下(至少留表头+1 数据行+提示行) keep = nrow while keep > 3 and sum(heights[:keep]) + row_height( ["…"] + [""] * (ncol - 1)) > avail: keep -= 1 cut = nrow - keep rows = rows[:keep] + [["… 其余 %d 行见平台详情页" % cut] + [""] * (ncol - 1)] heights = heights[:keep] + [row_height(rows[-1])] nrow = len(rows) total_h = sum(heights) gf = slide.shapes.add_table(nrow, ncol, Inches(MARGIN_X), Inches(top), Inches(width), Inches(total_h)) tbl = gf.table for ci in range(ncol): tbl.columns[ci].width = Inches(col_w[ci]) for ri, row in enumerate(rows): tbl.rows[ri].height = Inches(heights[ri]) for ci, val in enumerate(row): cell = tbl.cell(ri, ci) cell.margin_left = cell.margin_right = Inches(0.07) cell.margin_top = cell.margin_bottom = Inches(0.03) cell.vertical_anchor = 3 # MIDDLE cell.fill.solid() if ri == 0: cell.fill.fore_color.rgb = _rgb(NAVY) else: cell.fill.fore_color.rgb = _rgb(WHITE if ri % 2 else ROW_ALT) tf = cell.text_frame tf.word_wrap = True p = tf.paragraphs[0] r = p.add_run() r.text = (val or "")[:180] r.font.size = Pt(fs if ri else fs) r.font.bold = (ri == 0) r.font.name = FONT r.font.color.rgb = _rgb(WHITE if ri == 0 else (INK if ci == 0 else SLATE)) return total_h + 0.18 # ══════════════════════ 页面构建 ══════════════════════ def _cover(prs, title, software, date_str): from pptx.util import Inches, Pt slide = prs.slides.add_slide(prs.slide_layouts[6]) _rect(slide, 0, 0, SW, SH, NAVY) _rect(slide, 0, SH * 0.62, SW, SH * 0.38, NAVY_DARK) _rect(slide, MARGIN_X, 2.28, 1.5, 0.075, ACCENT) # 强调短线 _text(slide, MARGIN_X, 1.62, SW - 2 * MARGIN_X, 0.4, "RESEARCH REPORT · 研发商机报告", size=13, color=(0x9F, 0xC0, 0xFF)) _text(slide, MARGIN_X, 2.62, SW - 2 * MARGIN_X - 0.8, 1.9, _clean_inline(title) or "研发报告", size=38, color=WHITE, bold=True, line_spacing=1.12) if software: _rect(slide, MARGIN_X, 4.62, 0.055, 0.42, ACCENT) _text(slide, MARGIN_X + 0.22, 4.66, 8.0, 0.4, "方向:%s" % _clean_inline(software), size=16, color=(0xC7, 0xD7, 0xF5)) _text(slide, MARGIN_X, SH - 1.16, SW - 2 * MARGIN_X, 0.34, date_str, size=12, color=(0x8F, 0xA6, 0xCC)) _text(slide, MARGIN_X, SH - 0.82, SW - 2 * MARGIN_X, 0.34, "商机产线 · 全部数据来自数据爬取平台(内网采集、已去重、逐条可查证)", size=11, color=(0x7B, 0x92, 0xBB)) return slide def _section_divider(prs, idx, name, total_sections): slide = prs.slides.add_slide(prs.slide_layouts[6]) _rect(slide, 0, 0, SW, SH, NAVY) _rect(slide, 0, 0, 4.35, SH, NAVY_DARK) from pptx.util import Inches _text(slide, 0.85, 2.55, 2.9, 1.6, "%02d" % idx, size=76, color=ACCENT, bold=True) _rect(slide, 5.05, 3.02, 0.9, 0.06, ACCENT) _text(slide, 5.05, 3.28, SW - 5.05 - 0.8, 1.1, _clean_inline(name), size=32, color=WHITE, bold=True) _text(slide, 5.05, 4.42, SW - 5.05 - 0.8, 0.4, "SECTION %d OF %d" % (idx, total_sections), size=11, color=(0x8F, 0xA6, 0xCC)) return slide def _bullet_slide(prs, title, kicker, items, page_no): """要点页:items = [(level, lead, text)]。""" from pptx.util import Inches, Pt slide = prs.slides.add_slide(prs.slide_layouts[6]) _rect(slide, 0, 0, SW, SH, WHITE) _rect(slide, 0, 0, SW, 0.14, ACCENT) _content_header(slide, title, kicker) top = CONTENT_TOP + (0.26 if kicker else 0.0) y = top for lvl, lead, text in items: indent = 0.0 if lvl <= 1 else (0.34 if lvl == 2 else 0.68) marker = "▪" if lvl <= 1 else "–" mcolor = ACCENT if lvl <= 1 else MUTED _text(slide, MARGIN_X + indent, y, 0.26, 0.3, marker, size=12 if lvl <= 1 else 11, color=mcolor, bold=True) runs = [] if lead: runs.append((lead, {"size": 14, "bold": True, "color": NAVY})) runs.append((" ", {})) if text: runs.append((text, {"size": 14 if lvl <= 1 else 13, "color": INK if lvl <= 1 else SLATE})) if not runs: runs = [("(空)", {"size": 13, "color": MUTED})] _text(slide, MARGIN_X + indent + 0.26, y - 0.02, SW - 2 * MARGIN_X - indent - 0.26, 0.72, runs, line_spacing=1.18) est = max(1, int(len((lead or "") + (text or "")) / 46) + 1) y += 0.30 * est + 0.16 _footer(slide, page_no) return slide def _table_slide(prs, title, kicker, table_rows, extra_items, page_no): from pptx.util import Inches slide = prs.slides.add_slide(prs.slide_layouts[6]) _rect(slide, 0, 0, SW, SH, WHITE) _rect(slide, 0, 0, SW, 0.14, ACCENT) _content_header(slide, title, kicker) top = CONTENT_TOP + (0.26 if kicker else 0.0) used = _add_table(slide, table_rows, top) if extra_items: y = top + used + 0.05 for lvl, lead, text in extra_items: _text(slide, MARGIN_X, y, 0.26, 0.3, "▪", size=12, color=ACCENT, bold=True) runs = [] if lead: runs.append((lead, {"size": 13, "bold": True, "color": NAVY})) runs.append((" ", {})) if text: runs.append((text, {"size": 13, "color": SLATE})) _text(slide, MARGIN_X + 0.26, y - 0.02, SW - 2 * MARGIN_X - 0.26, 0.5, runs) y += 0.30 * max(1, int(len((lead or "") + (text or "")) / 52) + 1) + 0.10 _footer(slide, page_no) return slide def _closing(prs, page_no): from pptx.util import Inches slide = prs.slides.add_slide(prs.slide_layouts[6]) _rect(slide, 0, 0, SW, SH, NAVY) _rect(slide, MARGIN_X, 2.0, 1.2, 0.06, ACCENT) _text(slide, MARGIN_X, 2.28, SW - 2 * MARGIN_X, 0.8, "说明与查证", size=30, color=WHITE, bold=True) notes = [ "本报告全部数据来自数据爬取平台(内网采集、已去重),未做人工编造。", "每条引用均带公告原文 / 采集来源,可在产线平台「📋 数据参考」逐条查证。", "报告状态以平台门禁为准:人工确认 → 研发审批;未经审批不得视为立项依据。", "功能点与成本估算为模型基于公开数据的推算,落地前需结合团队实际复核。", ] y = 3.28 for n in notes: _text(slide, MARGIN_X, y, 0.3, 0.3, "▪", size=12, color=ACCENT, bold=True) _text(slide, MARGIN_X + 0.3, y - 0.02, SW - 2 * MARGIN_X - 0.3, 0.6, n, size=14, color=(0xC7, 0xD7, 0xF5), line_spacing=1.2) y += 0.62 return slide # ══════════════════════ 主入口 ══════════════════════ def build_report_ppt(title, content, software="", out_path=""): """生成 PPT 并保存。返回 (True, 文件路径) 或 (False, 错误)。 签名保持不变(opp_report_capability._build_report_ppt 调用方零改动)。 """ try: from pptx import Presentation from pptx.util import Inches except ImportError: return False, "缺少 python-pptx(pip install python-pptx)" if not out_path: return False, "缺少输出路径" import datetime date_str = datetime.date.today().strftime("%Y 年 %m 月 %d 日") prs = Presentation() prs.slide_width = Inches(SW) prs.slide_height = Inches(SH) _cover(prs, title, software, date_str) # 预扫描:算出要渲染的页面,便于一次性算总页数(页脚 x/y) pages = [] # [('divider', idx, name) | ('bullets', title, kicker, items) # | ('table', title, kicker, rows, extra)] sections = _parse_report(content) sec_idx = 0 for sec_name, sec_md in sections: if sec_name in SECTION_ORDER: sec_idx += 1 pages.append(("divider", sec_idx, sec_name)) blocks = _split_blocks(sec_md) cur_title = sec_name if sec_name not in SECTION_ORDER else sec_name buf_items = [] # [(lvl, lead, text)] pending_table = None def flush(kicker=""): if not buf_items and pending_table is None: return if pending_table is not None: pages.append(("table", cur_title, kicker, pending_table, list(buf_items))) else: # 按页容量切分要点 chunk = [] for it in buf_items: chunk.append(it) if len(chunk) >= MAX_BULLETS_PER_PAGE: pages.append(("bullets", cur_title, kicker, chunk)) chunk = [] if chunk: pages.append(("bullets", cur_title, kicker, chunk)) del buf_items[:] for b in blocks: kind = b[0] if kind == "h": flush() cur_title = b[2] elif kind == "table": # 表格独立成页(带上此前累积的要点作为补充说明) flush() pending_table = b[1] pages.append(("table", cur_title, "", pending_table, [])) pending_table = None elif kind == "ul": lead, text = _split_lead(b[2]) buf_items.append((b[1], lead, text)) elif kind == "ol": lead, text = _split_lead(b[2]) buf_items.append((1, "%s. %s" % (b[1], lead) if lead else "%s." % b[1], text)) elif kind == "quote": buf_items.append((1, "", _clean_inline(b[1]))) else: # p lead, text = _split_lead(b[1]) buf_items.append((1, lead, text)) flush() # 渲染(带总页数) total_pages = len(pages) + 2 # 封面 + 内容 + 尾页 page_no = 1 for pg in pages: page_no += 1 if pg[0] == "divider": _section_divider(prs, pg[1], pg[2], sec_idx or 1) elif pg[0] == "bullets": _bullet_slide(prs, pg[1], pg[2], pg[3], page_no) else: _table_slide(prs, pg[1], pg[2], pg[3], pg[4], page_no) _closing(prs, total_pages) os.makedirs(os.path.dirname(out_path), exist_ok=True) prs.save(out_path) try: npages = len(prs.slides._sldIdLst) except Exception: npages = -1 logger.info("报告 PPT 已生成: %s (%d 页)", out_path, npages) return True, out_path