fix: 保留上传模板的文本对齐(居中/右对齐/两端)
- docx_to_html: Mammoth转换后用BeautifulSoup注入text-align CSS - html_to_docx: 解析style属性还原WD_ALIGN_PARAGRAPH对齐 - 支持 center/right/justify 三种对齐
This commit is contained in:
parent
d155a00e51
commit
81cb155e3f
|
|
@ -1,14 +1,44 @@
|
||||||
import mammoth
|
import mammoth
|
||||||
from io import BytesIO
|
from io import BytesIO
|
||||||
|
from bs4 import BeautifulSoup
|
||||||
from docx import Document
|
from docx import Document
|
||||||
from docx.shared import Pt, Inches
|
from docx.shared import Pt, Inches
|
||||||
from docx.enum.text import WD_ALIGN_PARAGRAPH
|
from docx.enum.text import WD_ALIGN_PARAGRAPH
|
||||||
from app.services.file_storage import get_file_content
|
from app.services.file_storage import get_file_content
|
||||||
|
|
||||||
|
ALIGN_MAP = {
|
||||||
|
WD_ALIGN_PARAGRAPH.CENTER: "center",
|
||||||
|
WD_ALIGN_PARAGRAPH.RIGHT: "right",
|
||||||
|
WD_ALIGN_PARAGRAPH.JUSTIFY: "justify",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
async def docx_to_html(file_content: bytes) -> str:
|
async def docx_to_html(file_content: bytes) -> str:
|
||||||
result = mammoth.convert_to_html(BytesIO(file_content))
|
result = mammoth.convert_to_html(BytesIO(file_content))
|
||||||
return result.value
|
html = result.value
|
||||||
|
|
||||||
|
try:
|
||||||
|
doc = Document(BytesIO(file_content))
|
||||||
|
soup = BeautifulSoup(html, "html.parser")
|
||||||
|
paragraphs = doc.paragraphs
|
||||||
|
html_paras = soup.find_all("p")
|
||||||
|
|
||||||
|
for i, para in enumerate(paragraphs):
|
||||||
|
if i >= len(html_paras):
|
||||||
|
break
|
||||||
|
if para.alignment and para.alignment in ALIGN_MAP:
|
||||||
|
css = ALIGN_MAP[para.alignment]
|
||||||
|
existing_style = html_paras[i].get("style", "")
|
||||||
|
styles = f"text-align:{css}"
|
||||||
|
if existing_style:
|
||||||
|
styles = existing_style.rstrip(";") + ";" + styles
|
||||||
|
html_paras[i]["style"] = styles
|
||||||
|
|
||||||
|
html = str(soup)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
return html
|
||||||
|
|
||||||
|
|
||||||
def html_to_docx_bytes(html_content: str) -> bytes:
|
def html_to_docx_bytes(html_content: str) -> bytes:
|
||||||
|
|
@ -25,7 +55,7 @@ def html_to_docx_bytes(html_content: str) -> bytes:
|
||||||
def __init__(self):
|
def __init__(self):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.paragraphs: list[dict] = []
|
self.paragraphs: list[dict] = []
|
||||||
self.current = {"runs": []}
|
self.current = {"runs": [], "align": None}
|
||||||
self.in_paragraph = False
|
self.in_paragraph = False
|
||||||
self.current_run = {"text": "", "bold": False, "italic": False, "underline": False}
|
self.current_run = {"text": "", "bold": False, "italic": False, "underline": False}
|
||||||
self.tag_stack: list[str] = []
|
self.tag_stack: list[str] = []
|
||||||
|
|
@ -33,9 +63,18 @@ def html_to_docx_bytes(html_content: str) -> bytes:
|
||||||
|
|
||||||
def handle_starttag(self, tag, attrs):
|
def handle_starttag(self, tag, attrs):
|
||||||
tag_lower = tag.lower()
|
tag_lower = tag.lower()
|
||||||
|
attrs_dict = dict(attrs)
|
||||||
if tag_lower in ("p", "div", "li"):
|
if tag_lower in ("p", "div", "li"):
|
||||||
self.in_paragraph = True
|
self.in_paragraph = True
|
||||||
|
self.current = {"runs": [], "align": None}
|
||||||
self.current_run = {"text": "", "bold": False, "italic": False, "underline": False}
|
self.current_run = {"text": "", "bold": False, "italic": False, "underline": False}
|
||||||
|
style = attrs_dict.get("style", "")
|
||||||
|
if "text-align:center" in style:
|
||||||
|
self.current["align"] = "center"
|
||||||
|
elif "text-align:right" in style:
|
||||||
|
self.current["align"] = "right"
|
||||||
|
elif "text-align:justify" in style:
|
||||||
|
self.current["align"] = "justify"
|
||||||
elif tag_lower in ("h1", "h2", "h3", "h4", "h5", "h6"):
|
elif tag_lower in ("h1", "h2", "h3", "h4", "h5", "h6"):
|
||||||
self.in_paragraph = True
|
self.in_paragraph = True
|
||||||
self.heading_level = int(tag_lower[1])
|
self.heading_level = int(tag_lower[1])
|
||||||
|
|
@ -82,6 +121,14 @@ def html_to_docx_bytes(html_content: str) -> bytes:
|
||||||
else:
|
else:
|
||||||
p = doc.add_paragraph()
|
p = doc.add_paragraph()
|
||||||
|
|
||||||
|
align_val = para_data.get("align")
|
||||||
|
if align_val == "center":
|
||||||
|
p.alignment = WD_ALIGN_PARAGRAPH.CENTER
|
||||||
|
elif align_val == "right":
|
||||||
|
p.alignment = WD_ALIGN_PARAGRAPH.RIGHT
|
||||||
|
elif align_val == "justify":
|
||||||
|
p.alignment = WD_ALIGN_PARAGRAPH.JUSTIFY
|
||||||
|
|
||||||
for run_data in para_data.get("runs", []):
|
for run_data in para_data.get("runs", []):
|
||||||
run = p.add_run(run_data["text"])
|
run = p.add_run(run_data["text"])
|
||||||
if run_data.get("bold"):
|
if run_data.get("bold"):
|
||||||
|
|
|
||||||
Loading…
Reference in New Issue