| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556 |
- """
- proposal_writer.py - 投标文件DOCX输出模块
- 将 ProposalDocument 格式化为结构化的 Word 文档(.docx),
- 支持封面、目录、标题层级、表格、列表、页脚页码等排版要素。
- """
- import logging
- import re
- from typing import Optional
- from docx import Document
- from docx.enum.table import WD_TABLE_ALIGNMENT
- from docx.enum.text import WD_ALIGN_PARAGRAPH, WD_LINE_SPACING
- from docx.oxml.ns import qn, nsdecls
- from docx.oxml import parse_xml
- from docx.shared import Cm, Pt, RGBColor
- from docx.table import _Cell
- from .models import ProposalDocument, ProposalSection
- logger = logging.getLogger(__name__)
- # ---- 常量 ----
- PAGE_WIDTH_CM = 21.0
- PAGE_HEIGHT_CM = 29.7
- MARGIN_CM = 2.0
- BODY_WIDTH_CM = PAGE_WIDTH_CM - 2 * MARGIN_CM # 17.0 cm
- FONT_NAME = "宋体"
- FONT_NAME_ASCII = "Times New Roman"
- HEADING_FONTS = {1: 16, 2: 14, 3: 12}
- BODY_FONT_SIZE = 12 # pt
- def _set_cell_border(cell: _Cell, **kwargs):
- """设置单元格边框。"""
- tc = cell._tc
- tcPr = tc.get_or_add_tcPr()
- tcBorders = parse_xml(
- f'<w:tcBorders {nsdecls("w")}>'
- ' <w:top w:val="single" w:sz="4" w:space="0" w:color="000000"/>'
- ' <w:left w:val="single" w:sz="4" w:space="0" w:color="000000"/>'
- ' <w:bottom w:val="single" w:sz="4" w:space="0" w:color="000000"/>'
- ' <w:right w:val="single" w:sz="4" w:space="0" w:color="000000"/>'
- '</w:tcBorders>'
- )
- tcPr.append(tcBorders)
- def _set_paragraph_spacing(paragraph, line_spacing: float = 1.5, space_before: float = 0.5, space_after: float = 0.5):
- """设置段落间距。"""
- pf = paragraph.paragraph_format
- pf.line_spacing_rule = WD_LINE_SPACING.MULTIPLE
- pf.line_spacing = line_spacing
- pf.space_before = Pt(space_before * BODY_FONT_SIZE)
- pf.space_after = Pt(space_after * BODY_FONT_SIZE)
- def _run_format(run, font_name: str = FONT_NAME, font_size: int = BODY_FONT_SIZE,
- bold: bool = False, color: Optional[str] = None):
- """设置 run 的字体格式。"""
- run.font.name = font_name
- run.font.size = Pt(font_size)
- run.font.bold = bold
- run._element.rPr.rFonts.set(qn("w:eastAsia"), font_name)
- if color:
- run.font.color.rgb = RGBColor(*bytes.fromhex(color.lstrip("#")))
- def _setup_page(doc: Document):
- """设置页面格式:A4、2cm页边距。"""
- section = doc.sections[0]
- section.page_width = Cm(PAGE_WIDTH_CM)
- section.page_height = Cm(PAGE_HEIGHT_CM)
- section.top_margin = Cm(MARGIN_CM)
- section.bottom_margin = Cm(MARGIN_CM)
- section.left_margin = Cm(MARGIN_CM)
- section.right_margin = Cm(MARGIN_CM)
- def _add_footer_page_number(doc: Document):
- """添加页脚页码。"""
- section = doc.sections[0]
- footer = section.footer
- footer.is_linked_to_previous = False
- p = footer.paragraphs[0]
- p.alignment = WD_ALIGN_PARAGRAPH.CENTER
- # "第 X 页 / 共 Y 页"
- run1 = p.add_run("第 ")
- _run_format(run1, font_size=10)
- fld_char1 = parse_xml(f'<w:fldChar {nsdecls("w")} w:fldCharType="begin"/>')
- run2 = p.add_run()
- run2._element.append(fld_char1)
- instr = parse_xml(f'<w:instrText {nsdecls("w")} xml:space="preserve"> PAGE </w:instrText>')
- run3 = p.add_run()
- run3._element.append(instr)
- fld_char2 = parse_xml(f'<w:fldChar {nsdecls("w")} w:fldCharType="end"/>')
- run4 = p.add_run()
- run4._element.append(fld_char2)
- run5 = p.add_run(" 页 / 共 ")
- _run_format(run5, font_size=10)
- fld_char3 = parse_xml(f'<w:fldChar {nsdecls("w")} w:fldCharType="begin"/>')
- run6 = p.add_run()
- run6._element.append(fld_char3)
- instr2 = parse_xml(f'<w:instrText {nsdecls("w")} xml:space="preserve"> NUMPAGES </w:instrText>')
- run7 = p.add_run()
- run7._element.append(instr2)
- fld_char4 = parse_xml(f'<w:fldChar {nsdecls("w")} w:fldCharType="end"/>')
- run8 = p.add_run()
- run8._element.append(fld_char4)
- run9 = p.add_run(" 页")
- _run_format(run9, font_size=10)
- def _create_cover_page(doc: Document, title: str, date_str: str = ""):
- """创建封面页。
- 封面排版:居中,从上到下依次为:
- - "投标文件"(大号标题)
- - 项目名称
- - 编制日期
- """
- import datetime
- if not date_str:
- date_str = datetime.date.today().strftime("%Y年%m月%d日")
- # 上部留白
- for _ in range(6):
- doc.add_paragraph()
- # 大标题:投标文件
- p_title = doc.add_paragraph()
- p_title.alignment = WD_ALIGN_PARAGRAPH.CENTER
- run = p_title.add_run("投 标 文 件")
- _run_format(run, font_name=FONT_NAME, font_size=28, bold=True)
- # 装饰线
- p_line = doc.add_paragraph()
- p_line.alignment = WD_ALIGN_PARAGRAPH.CENTER
- run_line = p_line.add_run("━" * 20)
- _run_format(run_line, font_size=14, color="333333")
- doc.add_paragraph()
- # 项目名称
- p_proj = doc.add_paragraph()
- p_proj.alignment = WD_ALIGN_PARAGRAPH.CENTER
- run_proj = p_proj.add_run(f"项目名称:{title}")
- _run_format(run_proj, font_size=16)
- doc.add_paragraph()
- doc.add_paragraph()
- # 编制日期
- p_date = doc.add_paragraph()
- p_date.alignment = WD_ALIGN_PARAGRAPH.CENTER
- run_date = p_date.add_run(f"编制日期:{date_str}")
- _run_format(run_date, font_size=14)
- # 分页
- doc.add_page_break()
- def _add_heading(doc: Document, text: str, level: int = 1):
- """添加格式化标题。
- 使用预定义的字体大小:Heading1=16pt, Heading2=14pt, Heading3=12pt。
- """
- heading = doc.add_heading(text, level=min(level, 3))
- font_size = HEADING_FONTS.get(min(level, 3), 12)
- for run in heading.runs:
- run.font.name = FONT_NAME
- run.font.size = Pt(font_size)
- run._element.rPr.rFonts.set(qn("w:eastAsia"), FONT_NAME)
- return heading
- def _add_body_paragraph(doc: Document, text: str):
- """添加格式化正文段落。"""
- p = doc.add_paragraph()
- p.alignment = WD_ALIGN_PARAGRAPH.JUSTIFY
- _set_paragraph_spacing(p)
- # 处理内联格式:**粗体**
- parts = re.split(r'(\*\*.+?\*\*)', text)
- for part in parts:
- if part.startswith("**") and part.endswith("**"):
- run = p.add_run(part[2:-2])
- _run_format(run, bold=True)
- else:
- run = p.add_run(part)
- _run_format(run)
- p.style = doc.styles["Normal"]
- return p
- def _add_bullet_list(doc: Document, items: list[str]):
- """添加无序列表。"""
- for item in items:
- p = doc.add_paragraph(style="List Bullet")
- _set_paragraph_spacing(p, line_spacing=1.3, space_before=0, space_after=0)
- p.alignment = WD_ALIGN_PARAGRAPH.JUSTIFY
- run = p.add_run(item.strip())
- _run_format(run)
- def _add_numbered_list(doc: Document, items: list[str]):
- """添加有序列表。"""
- for item in items:
- p = doc.add_paragraph(style="List Number")
- _set_paragraph_spacing(p, line_spacing=1.3, space_before=0, space_after=0)
- p.alignment = WD_ALIGN_PARAGRAPH.JUSTIFY
- run = p.add_run(item.strip())
- _run_format(run)
- def _parse_markdown_table(md_table: str) -> list[list[str]]:
- """解析Markdown表格为二维列表。
- Args:
- md_table: Markdown表格文本
- Returns:
- list[list[str]]: 二维字符串列表,第一行为表头
- """
- lines = [line.strip() for line in md_table.strip().split("\n")]
- lines = [line for line in lines if line and not line.startswith("| ---") and not line.startswith("|---")]
- if not lines:
- return []
- table_data = []
- for line in lines:
- line = line.strip()
- if line.startswith("|"):
- line = line[1:]
- if line.endswith("|"):
- line = line[:-1]
- cells = [cell.strip() for cell in line.split("|")]
- table_data.append(cells)
- return table_data
- def _write_table(doc: Document, table_lines: list[str]):
- """将Markdown表格写入DOCX表格。"""
- table_text = "\n".join(table_lines)
- data = _parse_markdown_table(table_text)
- if not data or len(data) < 2:
- logger.warning("表格数据不足,跳过表格写入")
- return
- rows = len(data)
- cols = max(len(row) for row in data)
- doc_table = doc.add_table(rows=rows, cols=cols)
- doc_table.autofit = True
- doc_table.alignment = WD_TABLE_ALIGNMENT.CENTER
- # 设置表格边框
- tbl = doc_table._tbl
- tblPr = tbl.tblPr if tbl.tblPr is not None else parse_xml(
- f'<w:tblPr {nsdecls("w")}/>'
- )
- borders = parse_xml(
- f'<w:tblBorders {nsdecls("w")}>'
- ' <w:top w:val="single" w:sz="4" w:space="0" w:color="000000"/>'
- ' <w:left w:val="single" w:sz="4" w:space="0" w:color="000000"/>'
- ' <w:bottom w:val="single" w:sz="4" w:space="0" w:color="000000"/>'
- ' <w:right w:val="single" w:sz="4" w:space="0" w:color="000000"/>'
- ' <w:insideH w:val="single" w:sz="4" w:space="0" w:color="000000"/>'
- ' <w:insideV w:val="single" w:sz="4" w:space="0" w:color="000000"/>'
- '</w:tblBorders>'
- )
- tblPr.append(borders)
- for r_idx, row_data in enumerate(data):
- for c_idx in range(cols):
- cell_text = row_data[c_idx] if c_idx < len(row_data) else ""
- cell = doc_table.cell(r_idx, c_idx)
- # 清空并重新设置内容
- cell.text = ""
- p = cell.paragraphs[0]
- p.alignment = WD_ALIGN_PARAGRAPH.CENTER
- run = p.add_run(cell_text)
- is_header = r_idx == 0
- _run_format(run, font_size=10, bold=is_header)
- doc.add_paragraph() # 表后空行
- def _write_markdown_content(doc, markdown_text: str, base_level: int = 2):
- """将Markdown格式的内容写入DOCX。
- 支持的Markdown元素:
- - # 标题 → Heading样式
- - **粗体** → 粗体
- - 普通段落 → Normal样式
- - - 列表项 → 无序列表
- - 1. 列表项 → 有序列表
- - | 表格 | → DOCX表格
- - 空行 → 段落分隔
- Args:
- doc: Document对象
- markdown_text: Markdown格式的文本
- base_level: 基础标题层级偏移(默认2,即# -> Heading 2)
- """
- lines = markdown_text.split("\n")
- i = 0
- in_table = False
- table_lines = []
- in_code_block = False
- code_lines = []
- while i < len(lines):
- line = lines[i]
- stripped = line.strip()
- # 代码块处理
- if stripped.startswith("```"):
- if in_code_block:
- # 结束代码块,作为普通段落写入
- for code_line in code_lines:
- _add_body_paragraph(doc, code_line)
- code_lines = []
- in_code_block = False
- else:
- in_code_block = True
- i += 1
- continue
- if in_code_block:
- code_lines.append(line)
- i += 1
- continue
- # 表格处理:连续以 | 开头的行
- # 判断是否为 Markdown 表格分隔行(如 |---|---| 或 |:---|:---:|)
- _is_separator = bool(re.match(r'^\|[\s\-:]+\|?$', stripped))
- if stripped.startswith("|") and not _is_separator:
- in_table = True
- table_lines.append(line)
- i += 1
- continue
- elif _is_separator:
- # 分隔行,跳过
- i += 1
- continue
- else:
- if in_table:
- # 表格结束,写入表格
- _write_table(doc, table_lines)
- table_lines = []
- in_table = False
- # 空行
- if not stripped:
- i += 1
- continue
- # 标题
- heading_match = re.match(r'^(#{1,6})\s+(.+)$', stripped)
- if heading_match:
- level = min(len(heading_match.group(1)) + base_level - 1, 3)
- # level最小为1
- level = max(level, 1)
- title_text = heading_match.group(2)
- # 去掉可能的内联格式标记(标题中保留文字)
- title_text = re.sub(r'\*\*(.+?)\*\*', r'\1', title_text)
- _add_heading(doc, title_text, level)
- i += 1
- continue
- # 无序列表
- if stripped.startswith("- ") or stripped.startswith("* "):
- items = []
- while i < len(lines):
- s = lines[i].strip()
- if s.startswith("- ") or s.startswith("* "):
- items.append(s[2:])
- i += 1
- else:
- break
- _add_bullet_list(doc, items)
- continue
- # 有序列表
- ordered_match = re.match(r'^\d+\.\s+(.+)$', stripped)
- if ordered_match:
- items = []
- while i < len(lines):
- s = lines[i].strip()
- if re.match(r'^\d+\.\s+.+$', s):
- items.append(re.match(r'^\d+\.\s+(.+)$', s).group(1))
- i += 1
- else:
- break
- _add_numbered_list(doc, items)
- continue
- # 普通段落
- paragraph_text = stripped
- i += 1
- # 合并后续非空、非特殊行
- while i < len(lines):
- next_line = lines[i]
- next_stripped = next_line.strip()
- if not next_stripped:
- break
- if next_stripped.startswith("#"):
- break
- if next_stripped.startswith("- ") or next_stripped.startswith("* "):
- break
- if re.match(r'^\d+\.\s+', next_stripped):
- break
- if next_stripped.startswith("|") and not next_stripped.startswith("| ---"):
- break
- paragraph_text += " " + next_stripped
- i += 1
- _add_body_paragraph(doc, paragraph_text)
- # 如果文档以表格结尾
- if in_table and table_lines:
- _write_table(doc, table_lines)
- def write_proposal_docx(content: ProposalDocument, output_path: str) -> str:
- """将投标内容写入格式化的DOCX文件。
- 文档结构:
- 1. 封面("投标文件"大标题 + 项目名称 + 日期)
- 2. 目录
- 3. 投标函(从 sections 中识别)
- 4. 投标概要(content.summary)
- 5. 正文章节(按层级组织)
- 6. 页脚页码
- Args:
- content: 投标文档内容(ProposalDocument)
- output_path: 输出文件路径(应以 .docx 结尾)
- Returns:
- str: 输出文件路径
- Raises:
- ValueError: 如果 content 为空或 output_path 不合法
- IOError: 如果文件写入失败
- """
- if not content:
- raise ValueError("投标内容不能为空")
- if not output_path:
- raise ValueError("输出路径不能为空")
- logger.info("开始生成DOCX文档:%s", output_path)
- doc = Document()
- _setup_page(doc)
- # 从 sections 中分离出投标函和内容章节
- cover_letter_section = None
- content_sections: list[ProposalSection] = []
- for s in content.sections:
- if s.title == "投标函" and cover_letter_section is None:
- cover_letter_section = s
- else:
- content_sections.append(s)
- # 1. 封面
- _create_cover_page(doc, content.title or content.project_name or "投标文件")
- # 2. 目录
- _add_heading(doc, "目 录", 1)
- toc_num = 1
- if cover_letter_section:
- p = doc.add_paragraph(f"{_to_chinese_number(toc_num)}、{cover_letter_section.title}")
- _set_paragraph_spacing(p, line_spacing=1.3, space_before=0, space_after=0)
- toc_num += 1
- if content.summary:
- p = doc.add_paragraph(f"{_to_chinese_number(toc_num)}、编制说明")
- _set_paragraph_spacing(p, line_spacing=1.3, space_before=0, space_after=0)
- toc_num += 1
- for s in content_sections:
- indent = " " * (s.level - 1) if s.level > 1 else ""
- p = doc.add_paragraph(f"{indent}{s.title}")
- _set_paragraph_spacing(p, line_spacing=1.3, space_before=0, space_after=0)
- doc.add_page_break()
- # 3. 投标函
- section_counter = 1
- if cover_letter_section:
- _add_heading(doc, f"{_to_chinese_number(section_counter)}、{cover_letter_section.title}", 1)
- section_counter += 1
- _write_markdown_content(doc, cover_letter_section.content, base_level=2)
- doc.add_page_break()
- # 4. 投标概要
- if content.summary:
- _add_heading(doc, f"{_to_chinese_number(section_counter)}、编制说明", 1)
- section_counter += 1
- _write_markdown_content(doc, content.summary, base_level=2)
- doc.add_page_break()
- # 5. 各章节
- for section in content_sections:
- numbered_title = f"{_to_chinese_number(section_counter)}、{section.title}"
- section_counter += 1
- _add_heading(doc, numbered_title, 1)
- _write_markdown_content(doc, section.content, base_level=2)
- # 6. 页脚页码
- _add_footer_page_number(doc)
- try:
- doc.save(output_path)
- except (OSError, IOError) as e:
- raise OSError(
- f"无法保存DOCX文件到路径: {output_path}\n"
- f"请检查目录是否存在、磁盘空间是否充足。\n"
- f"原始错误: {e}"
- ) from e
- logger.info("DOCX文档已保存至: %s", output_path)
- return output_path
- def _to_chinese_number(n: int) -> str:
- """将阿拉伯数字转为中文数字(一、二、三...)。
- Args:
- n: 阿拉伯数字
- Returns:
- str: 中文数字
- """
- chinese_nums = ["〇", "一", "二", "三", "四", "五", "六", "七", "八", "九", "十",
- "十一", "十二", "十三", "十四", "十五", "十六", "十七", "十八", "十九", "二十"]
- if 1 <= n <= len(chinese_nums):
- return chinese_nums[n]
- return str(n)
|