162 lines
6.1 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""研发报告 PPT 生成:markdown 报告正文 → .pptx 文件(可下载查证)。
触发时机:opp_submit_report 提交人工确认时生成/更新,存到
项目工作空间 deliverables/ 目录(与报告同源,人工确认时可直接下载审阅)。
设计取舍:
- python-pptx 纯后端生成,无外部模板依赖(模板缺失时降级纯文本版式)。
- markdown 按 # / ## 切页;列表/表格转要点;来源链接保留为文本(PPT 超链接
兼容性差,查证走界面按钮)。
"""
import logging
import os
import re
logger = logging.getLogger("pipeline.opportunity.ppt")
def _md_to_sections(content):
"""markdown 正文切分为 [(标题, [要点行], 原文块)] 段。
# 一级标题视为章节页;无标题的连续文本归入"正文"页。
"""
sections = []
cur_title = ""
cur_lines = []
for raw in (content or "").split("\n"):
line = raw.rstrip()
m = re.match(r"^(#{1,3})\s+(.+)$", line)
if m and len(m.group(1)) <= 2:
if cur_title or cur_lines:
sections.append((cur_title, cur_lines))
cur_title = m.group(2).strip()
cur_lines = []
continue
if line.strip():
cur_lines.append(line)
if cur_title or cur_lines:
sections.append((cur_title, cur_lines))
return sections
def _line_to_bullet(line):
"""markdown 行 → PPT 要点文本(去标记、保留链接为 文字(链接))。"""
s = line.strip()
s = re.sub(r"^\s*[-*+]\s+", "", s) # 列表符
s = re.sub(r"^\s*\d+[.、)]\s*", "", s) # 有序列表
s = re.sub(r"\*\*(.+?)\*\*", r"\1", s) # 粗体
s = re.sub(r"\*(.+?)\*", r"\1", s) # 斜体
s = re.sub(r"^\|.*\|$", lambda m: " | ".join(
c.strip() for c in m.group(0).strip("|").split("|") if c.strip()), s) # 表格行
if re.match(r"^[-: =]+$", s):
return "" # 表格分隔线
s = re.sub(r"\[([^\]]+)\]\(([^)]+)\)", r"\1(\2)", s) # 链接 → 文字(链接)
return s.strip()
def build_report_ppt(title, content, software="", out_path=""):
"""生成 PPT 并保存。返回 (True, 文件路径) 或 (False, 错误)。"""
try:
from pptx import Presentation
from pptx.util import Inches, Pt
from pptx.dml.color import RGBColor
except ImportError:
return False, "缺少 python-pptx(pip install python-pptx)"
if not out_path:
return False, "缺少输出路径"
prs = Presentation()
prs.slide_width = Inches(13.333)
prs.slide_height = Inches(7.5)
blank_layout = prs.slide_layouts[6]
def _add_title_bar(slide, text, color=(0x1E, 0x40, 0xAF)):
box = slide.shapes.add_textbox(Inches(0.5), Inches(0.3), Inches(12.3), Inches(0.8))
tf = box.text_frame
p = tf.paragraphs[0]
run = p.add_run()
run.text = text
run.font.size = Pt(26)
run.font.bold = True
run.font.color.rgb = RGBColor(*color)
# ── 封面页 ──
slide = prs.slides.add_slide(blank_layout)
box = slide.shapes.add_textbox(Inches(0.8), Inches(2.6), Inches(11.7), Inches(2))
tf = box.text_frame
tf.word_wrap = True
p = tf.paragraphs[0]
run = p.add_run()
run.text = title or "研发报告"
run.font.size = Pt(36)
run.font.bold = True
run.font.color.rgb = RGBColor(0x0F, 0x17, 0x2A)
sub = software or ""
if sub:
p2 = tf.add_paragraph()
r2 = p2.add_run()
r2.text = "方向:%s" % sub
r2.font.size = Pt(18)
r2.font.color.rgb = RGBColor(0x64, 0x74, 0x8B)
p3 = tf.add_paragraph()
r3 = p3.add_run()
r3.text = "商机产线 · 数据来自数据爬取平台(可查证)"
r3.font.size = Pt(14)
r3.font.color.rgb = RGBColor(0x94, 0xA3, 0xB8)
# ── 内容页 ──
sections = _md_to_sections(content)
for sec_title, lines in sections:
bullets = [b for b in (_line_to_bullet(l) for l in lines) if b]
if not sec_title and not bullets:
continue
# 要点过多 → 分页(每页 ≤ 8 条,防溢出)
chunks = [bullets[i:i + 8] for i in range(0, max(len(bullets), 1), 8)] or [[]]
for ci, chunk in enumerate(chunks):
slide = prs.slides.add_slide(blank_layout)
t = sec_title or "正文"
if len(chunks) > 1:
t += "(%d/%d)" % (ci + 1, len(chunks))
_add_title_bar(slide, t)
box = slide.shapes.add_textbox(Inches(0.7), Inches(1.3), Inches(12), Inches(5.8))
tf = box.text_frame
tf.word_wrap = True
first = True
for b in chunk:
p = tf.paragraphs[0] if first else tf.add_paragraph()
first = False
run = p.add_run()
run.text = "• " + b
run.font.size = Pt(15)
run.font.color.rgb = RGBColor(0x33, 0x41, 0x55)
p.space_after = Pt(8)
# ── 尾页 ──
slide = prs.slides.add_slide(blank_layout)
_add_title_bar(slide, "说明", color=(0x64, 0x74, 0x8B))
box = slide.shapes.add_textbox(Inches(0.7), Inches(1.4), Inches(12), Inches(5))
tf = box.text_frame
tf.word_wrap = True
for i, note in enumerate([
"本报告全部数据来自数据爬取平台(内网采集、已去重)。",
"每条引用均带公告原文/采集来源,可在产线平台「数据参考」逐条查证。",
"报告状态以平台门禁为准(人工确认 → 研发审批),未经审批不得视为立项依据。",
]):
p = tf.paragraphs[0] if i == 0 else tf.add_paragraph()
run = p.add_run()
run.text = note
run.font.size = Pt(15)
run.font.color.rgb = RGBColor(0x47, 0x55, 0x69)
p.space_after = Pt(10)
os.makedirs(os.path.dirname(out_path), exist_ok=True)
prs.save(out_path)
try:
npages = len(prs.slides._sldIdLst)
except Exception:
npages = -1
logger.info("报告 PPT 已生成: %s (%d 页)", out_path, npages)
return True, out_path