From beacf1ec624bc2c3658dc0b5b7c32ef829a290f9 Mon Sep 17 00:00:00 2001 From: CocoRoF Date: Mon, 31 Aug 2026 19:56:17 +0900 Subject: [PATCH] =?UTF-8?q?feat(design):=20=EB=94=94=EC=9E=90=EC=9D=B8=20?= =?UTF-8?q?=EC=B6=A9=EC=8B=A4=EB=8F=84=20=E2=80=94=20=EC=85=80=20=EB=B3=80?= =?UTF-8?q?=EB=B3=84=20=ED=85=8C=EB=91=90=EB=A6=AC/=EC=88=98=EC=A7=81?= =?UTF-8?q?=EC=A0=95=EB=A0=AC/=EC=97=AC=EB=B0=B1/=EC=A4=84=EA=B0=84?= =?UTF-8?q?=EA=B2=A9=20(0.22.0)?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit hwp 변환 (표 번호 = 한컴 스펙): - BORDER_FILL 전체 해석(표 18/20/21): 좌/우/상/하 **변별 stroke(실선/대시/ 점선/이중/없음)·굵기(mm 표)·색** → w:tcBorders (sz=1/8pt). '테두리 없음'(stroke 0)은 nil — 한글 표의 무테두리 셀이 그대로 나온다 - 셀 수직 정렬: 표 60 listflags bits5-6 → w:vAlign (한글 기본 = 가운데) - 셀 안쪽 여백: 표 75 padding → w:tcMar - 문단 모양(표 38): 줄간격(RATIO %) → line_spacing, 문단 앞/뒤 간격(doubled margin) → space_before/after docx_pages 렌더: - w:tcBorders 변별 렌더 — 변마다 (dash 패턴, 이중선 2선 근사, nil 은 그리지 않음). tcBorders 없는 문서는 기존 회색 격자 유지 - w:tcMar per-cell 안쪽 여백, w:spacing 줄간격 배수(_Line.spacing_mult) + 문단 앞/뒤 간격 — 본문/셀 모두 검증: 픽스처에 테두리 3종(실선/대시/없음)·valign·여백·줄간격 200% 추가 (신규 5건) + 렌더 단위 2건. pyhwp 실파일 36개 스윕 34/2/0 유지 — 실파일 표가 변별 테두리 선()으로 렌더됨 확인. 921 passed. --- pyproject.toml | 2 +- src/xgen_edit2docs/documents/docx_pages.py | 173 +++++++++++++++-- .../documents/legacy/hwp_convert.py | 180 +++++++++++++++--- tests/unit/test_docx_pages.py | 52 +++++ tests/unit/test_legacy_formats.py | 51 ++++- 5 files changed, 406 insertions(+), 52 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 114f4ebc..fc152042 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "xgen-edit2docs" -version = "0.21.1" +version = "0.22.0" description = "AI-agent-native document engine: generate and chat-edit DOCX, XLSX and PPTX as a Python library, agent tool set, MCP server or hosted service. English-first with first-class Korean support. Sister project of edit2ppt." readme = "README.md" requires-python = ">=3.12" diff --git a/src/xgen_edit2docs/documents/docx_pages.py b/src/xgen_edit2docs/documents/docx_pages.py index c6ece635..16d399ec 100644 --- a/src/xgen_edit2docs/documents/docx_pages.py +++ b/src/xgen_edit2docs/documents/docx_pages.py @@ -13,8 +13,10 @@ numbered lists, hard + automatic page breaks, tables (tblGrid widths, gridSpan/vMerge merges drawn as one spanning rect, per-run cell styles + per-paragraph cell alignment, w:vAlign, trHeight minimums, cell -shading, row-boundary page splitting, nested tables flattened to -text), inline images (extent-scaled, base64), single-section page +shading, **per-side w:tcBorders (single/dashed/dotted/double/nil) with +width+color, w:tcMar cell padding, w:spacing line-spacing multiples and +before/after gaps**, row-boundary page splitting, nested tables +flattened to text), inline images (extent-scaled, base64), single-section page size/margins, first-section header/footer text with PAGE field support. Floating shapes, multi-column sections and footnote blocks are out of scope — the HTML preview covers reading those. @@ -126,10 +128,14 @@ def deco(self) -> str: @dataclass class _Line: segs: list[_Seg] = field(default_factory=list) + #: 문단 줄간격 배수 (w:spacing lineRule="auto" line/240) — 1.0 = 기본. + spacing_mult: float = 1.0 @property def height(self) -> float: - return max((s.size_px for s in self.segs), default=_DEFAULT_FONT_PT * 96 / 72) * _LINE_SPACING + base = max((s.size_px for s in self.segs), + default=_DEFAULT_FONT_PT * 96 / 72) + return base * _LINE_SPACING * self.spacing_mult @property def ascent(self) -> float: @@ -313,6 +319,32 @@ def _num_pr(paragraph): return paragraph._p.find(f"{_w('pPr')}/{_w('numPr')}") +def _para_spacing(p_el) -> tuple[float, float, float]: + """w:pPr/w:spacing → (앞 px, 뒤 px, 줄간격 배수). + + before/after 는 twips, line 은 lineRule="auto" 일 때 240 = 1배. + (HWP 변환물의 160%/200% 줄간격이 여기로 온다.)""" + sp = p_el.find(f"{_w('pPr')}/{_w('spacing')}") + if sp is None: + return 0.0, 0.0, 1.0 + + def twips(attr: str) -> float: + try: + return float(sp.get(_w(attr))) / _TWIPS_PER_PX + except (TypeError, ValueError): + return 0.0 + + mult = 1.0 + rule = (sp.get(_w("lineRule")) or "auto").lower() + try: + line = float(sp.get(_w("line"))) + if rule == "auto" and 60 <= line <= 1200: + mult = line / 240.0 + except (TypeError, ValueError): + pass + return twips("before"), twips("after"), mult + + def _para_align(p_el) -> str: """w:pPr/w:jc → 'left' | 'center' | 'right'. justify(both)/distribute 는 좌측 흘림으로 근사한다 (이 엔진은 자간 조정을 하지 않는다).""" @@ -488,11 +520,63 @@ class _CellBox: lines: list = field(default_factory=list) fill: Optional[str] = None valign: str = "top" # w:tcPr/w:vAlign — top | center | bottom + #: 변별 테두리 {side: (val, width_px, color)} — None = 기본 회색 격자. + borders: Optional[dict] = None + #: 안쪽 여백 (left, right, top, bottom) px. + pad: tuple = (4.0, 4.0, 4.0, 4.0) def content_h(self) -> float: return sum(ln.height for ln, _a in self.lines) +#: 테두리 val → SVG dasharray (None = 실선, "skip" = 안 그림) +_BORDER_DASH = { + "nil": "skip", "none": "skip", + "dashed": "6 3", "dotted": "2 2", "dotdash": "6 3 2 3", + "dotdotdash": "6 3 2 3 2 3", +} + + +def _cell_borders(tc_el) -> Optional[dict]: + """w:tcPr/w:tcBorders → {side: (val, width_px, color)}. + + 없으면 None — 표 전체 기본 격자(회색 0.8)로 그린다. sz 는 1/8pt + (px = sz/8 × 96/72), 색은 RRGGBB 또는 auto.""" + tcb = tc_el.find(f"{_w('tcPr')}/{_w('tcBorders')}") + if tcb is None: + return None + out: dict = {} + for side in ("left", "right", "top", "bottom"): + el = tcb.find(_w(side)) + if el is None: + continue + val = (el.get(_w("val")) or "single").lower() + try: + width_px = max(0.5, float(el.get(_w("sz")) or 4) / 8 * 96 / 72) + except ValueError: + width_px = 0.8 + color = el.get(_w("color")) or "444444" + if color.lower() == "auto": + color = "444444" + out[side] = (val, min(width_px, 6.0), f"#{color}") + return out or None + + +def _cell_pad(tc_el, default: float) -> tuple: + """w:tcPr/w:tcMar → (l, r, t, b) px (dxa=twips).""" + mar = tc_el.find(f"{_w('tcPr')}/{_w('tcMar')}") + pads = [default] * 4 + if mar is not None: + for i, side in enumerate(("left", "right", "top", "bottom")): + el = mar.find(_w(side)) + if el is not None: + try: + pads[i] = min(40.0, max(0.0, float(el.get(_w("w"))) / _TWIPS_PER_PX)) + except (TypeError, ValueError): + pass + return tuple(pads) + + def _cell_valign(tc_el) -> str: va = tc_el.find(f"{_w('tcPr')}/{_w('vAlign')}") val = (va.get(_w("val")) or "").lower() if va is not None else "" @@ -516,11 +600,13 @@ def _cell_lines(tc_el, document, avail_w: float) -> list: if child.tag == _w("p"): paragraph = Paragraph(child, document) align = _para_align(child) + _bf, _af, mult = _para_spacing(child) for segs in _paragraph_segments(paragraph, 0): if not any(s.text for s in segs): continue for ln in _wrap_segments(segs, avail_w): if ln.segs: + ln.spacing_mult = mult out.append((ln, align)) elif child.tag == _w("tbl"): for tr in child.findall(_w("tr")): @@ -576,10 +662,13 @@ def _table_model(tbl_el, document, widths: list[float], pad: float): col_cursor += span continue width = sum(widths[col_cursor:col_cursor + span]) or widths[-1] + cpad = _cell_pad(tc, pad) box = _CellBox( row=r_i, col=col_cursor, colspan=span, - lines=_cell_lines(tc, document, max(width - pad * 2, 10.0)), + lines=_cell_lines(tc, document, + max(width - cpad[0] - cpad[1], 10.0)), fill=_cell_fill(tc), valign=_cell_valign(tc), + borders=_cell_borders(tc), pad=cpad, ) boxes.append(box) if vmerge == "restart": @@ -595,12 +684,12 @@ def _table_model(tbl_el, document, widths: list[float], pad: float): if box.rowspan == 1: r = box.row if r < len(row_h): - row_h[r] = max(row_h[r], box.content_h() + pad * 2) + row_h[r] = max(row_h[r], box.content_h() + box.pad[2] + box.pad[3]) for box in boxes: if box.rowspan > 1: end = min(box.row + box.rowspan, len(row_h)) have = sum(row_h[box.row:end]) - need = box.content_h() + pad * 2 + need = box.content_h() + box.pad[2] + box.pad[3] if need > have and end - 1 >= box.row: row_h[end - 1] += need - have return boxes, row_h @@ -608,26 +697,73 @@ def _table_model(tbl_el, document, widths: list[float], pad: float): def _draw_cell_box(writer: _PageWriter, table_idx: int, box: _CellBox, x: float, y: float, w: float, h: float, pad: float) -> None: - attrs = f' fill="{box.fill}"' if box.fill else ' fill="none"' writer.raw( f'' - f'' ) + fill_attr = f' fill="{box.fill}"' if box.fill else ' fill="none"' + if box.borders is None: + # 변별 테두리 없음 — 기본 격자 (기존과 동일) + writer.raw( + f'' + ) + else: + if box.fill: + writer.raw( + f'' + ) + coords = { + "left": (x, y, x, y + h), "right": (x + w, y, x + w, y + h), + "top": (x, y, x + w, y), "bottom": (x, y + h, x + w, y + h), + } + for side, (x1, y1, x2, y2) in coords.items(): + spec = box.borders.get(side) + if spec is None: + # 선언 안 된 변 — 기본 격자선으로 이음새를 메운다 + writer.raw( + f'' + ) + continue + val, width_px, color = spec + dash = _BORDER_DASH.get(val) + if dash == "skip": + continue # 선 없음 (한글 표의 '테두리 없음' 셀) + dash_attr = f' stroke-dasharray="{dash}"' if dash else "" + if val in ("double", "triple"): + # 이중선 근사 — 가는 선 2개 + thin = max(0.6, width_px / 3) + off = max(1.2, width_px) + dx, dy = ((off, 0) if side in ("left", "right") else (0, off)) + writer.raw( + f'' + f'' + ) + else: + writer.raw( + f'' + ) + pl, pr, pt, pb = box.pad content_h = box.content_h() if box.valign == "center": - ty = y + max(pad, (h - content_h) / 2) + ty = y + max(pt, (h - content_h) / 2) elif box.valign == "bottom": - ty = y + max(pad, h - content_h - pad) + ty = y + max(pt, h - content_h - pb) else: - ty = y + pad - inner_w = max(w - pad * 2, 1.0) + ty = y + pt + inner_w = max(w - pl - pr, 1.0) for ln, align in box.lines: if ty + ln.height > y + h + 0.5: # 넘치는 줄은 셀 경계에서 끊는다 break baseline = ty + ln.ascent line_w = sum(s.width() for s in ln.segs) - tx = x + pad + tx = x + pl if align == "center": tx += max(0.0, (inner_w - line_w) / 2) elif align == "right": @@ -697,7 +833,8 @@ def _layout_table(writer: _PageWriter, table, table_idx: int) -> None: w = sum(widths[box.col:box.col + box.colspan]) or widths[-1] h = sum(row_h[top_r:bot_r]) draw = box if top_r == box.row else _CellBox( - row=box.row, col=box.col, fill=box.fill) # 이월 조각은 빈 칸 + row=box.row, col=box.col, fill=box.fill, + borders=box.borders, pad=box.pad) # 이월 조각은 빈 칸 _draw_cell_box(writer, table_idx, draw, x, y_of[top_r], w, h, pad) writer.y += acc r = chunk_end @@ -835,6 +972,9 @@ def _hf_segments(container) -> list[_Seg]: if heading: writer.y += _PARA_GAP_PX # breathing room above headings align = _para_align(child) + before_px, after_px, mult = _para_spacing(child) + if before_px > 0: + writer.y += before_px first = True for segs in logical_lines: if bullet and first and segs: @@ -842,9 +982,10 @@ def _hf_segments(container) -> list[_Seg]: bold=segs[0].bold, color="#222222")] + segs for ln in _wrap_segments(segs, writer.content_w - indent): if ln.segs: + ln.spacing_mult = mult writer.emit_line(ln, indent=indent, align=align) first = False - writer.y += _PARA_GAP_PX + writer.y += _PARA_GAP_PX + after_px elif not drew_image: # empty paragraph — vertical rhythm (Word keeps them) writer.y += _DEFAULT_FONT_PT * 96 / 72 * 0.9 diff --git a/src/xgen_edit2docs/documents/legacy/hwp_convert.py b/src/xgen_edit2docs/documents/legacy/hwp_convert.py index 81f0b5ee..f4842da6 100644 --- a/src/xgen_edit2docs/documents/legacy/hwp_convert.py +++ b/src/xgen_edit2docs/documents/legacy/hwp_convert.py @@ -24,8 +24,10 @@ - 문단: 정렬(PARA_SHAPE align), 런 스타일(글꼴/크기/굵게/기울임/밑줄/ 취소선/색 — CHAR_SHAPE 표 28/30) - 표(표 70~75): rows×cols 격자, **병합(colspan/rowspan → gridSpan/vMerge)**, - 열 너비/행 높이, 셀 배경(BORDER_FILL 표 18/23), 셀 안 문단 전체 스타일, - 중첩 표(셀 안 표 — 재귀), 캡션 + 열 너비/행 높이, 셀 배경·**변별 테두리(표 18/20/21 — 실선/대시/없음, + 굵기, 색)**, 셀 수직 정렬(표 60 listflags), 셀 안쪽 여백(표 75 → tcMar), + 셀 안 문단 전체 스타일, 중첩 표(셀 안 표 — 재귀), 캡션 +- 문단 간격: 줄간격(표 38 RATIO)·문단 앞/뒤 간격 → w:spacing - 그림(표 102): BinData 임베딩 → docx 인라인 이미지 (개체 요소 크기 반영) - 글상자: 텍스트를 본문 문단으로 (위치는 범위 밖 — 내용 유실 없음) - 머리말/꼬리말: 첫 정의를 docx 섹션 header/footer 텍스트로 @@ -98,13 +100,35 @@ class _CharShape: face_id: Optional[int] = None # 한글(ko) FaceName 참조 +@dataclass +class _BorderSide: + """표 20/21 테두리선 — stroke 0 = 선 없음.""" + stroke: int = 1 + width_mm: float = 0.12 + color: str = "000000" + + +@dataclass +class _BorderFill: + """표 18 테두리/배경 — 좌/우/상/하 선 + 배경색.""" + bg: Optional[str] = None + sides: Optional[List[_BorderSide]] = None # [left, right, top, bottom] + + +@dataclass +class _ParaProps: + """표 38 문단 모양 — 렌더에 실리는 부분집합.""" + align: str = "left" + line_spacing: Optional[float] = None # 배수 (RATIO 형만) + space_before_pt: float = 0.0 + space_after_pt: float = 0.0 + + @dataclass class _DocInfo: char_shapes: List[_CharShape] = field(default_factory=list) - #: PARA_SHAPE align — 'left' | 'center' | 'right' | 'justify' - para_aligns: List[str] = field(default_factory=list) - #: BORDER_FILL → 배경색 RRGGBB (채우기 없음/흰색은 None) - border_fill_bg: List[Optional[str]] = field(default_factory=list) + para_shapes: List[_ParaProps] = field(default_factory=list) + border_fills: List[_BorderFill] = field(default_factory=list) #: 한글(ko) 글꼴 이름 목록 — CHAR_SHAPE face_id 가 가리킨다. ko_faces: List[str] = field(default_factory=list) #: BIN_DATA 레코드 순서(1-based id) → (storage_id, ext). 링크형은 None. @@ -153,24 +177,56 @@ def _parse_char_shape(payload: bytes) -> _CharShape: 4: "justify", 5: "justify"} -def _parse_para_shape(payload: bytes) -> str: +def _parse_para_shape(payload: bytes) -> _ParaProps: + """표 38: flags(4) margins×4(doubled, 1/7200in ×2) @4..20, + linespacing @24 (flags bits0-1: 0=RATIO %, 1=FIXED …).""" + pp = _ParaProps() if len(payload) >= 4: (flags,) = struct.unpack_from("> 2) & 0x7, "left") - return "left" - - -def _parse_border_fill(payload: bytes) -> Optional[str]: - """표 18: borderflags(2) + Border(6)×5 = 32, fillflags UINT32 @32, - colorpattern 이면 background COLORREF @36.""" + pp.align = _ALIGN_MAP.get((flags >> 2) & 0x7, "left") + if len(payload) >= 28: + top2, bottom2, ls = struct.unpack_from("<3i", payload, 16) + # doubled margin: 1/7200 inch × 2 → pt = v/2 × 72/7200 = v/200 + if 0 < top2 <= 7200 * 8: + pp.space_before_pt = top2 / 200.0 + if 0 < bottom2 <= 7200 * 8: + pp.space_after_pt = bottom2 / 200.0 + if (flags & 0x3) == 0 and 50 <= ls <= 500: # RATIO(%) + pp.line_spacing = ls / 100.0 + return pp + + +#: 표 21 테두리선 굵기 인덱스 → mm (pyhwp Border.widths) +_BORDER_WIDTH_MM = (0.1, 0.12, 0.15, 0.2, 0.25, 0.3, 0.4, 0.5, + 0.6, 0.7, 1.0, 1.5, 2.0, 3.0, 4.0, 5.0) + + +def _parse_border_fill(payload: bytes) -> _BorderFill: + """표 18: borderflags(2) + Border(stroke 1B, width 1B, COLORREF 4B)×5 + (좌/우/상/하/대각) = 32, fillflags UINT32 @32, colorpattern 이면 + background COLORREF @36.""" + bf = _BorderFill() + if len(payload) >= 32: + sides: List[_BorderSide] = [] + for k in range(4): # left, right, top, bottom (대각선은 범위 밖) + off = 2 + k * 6 + stroke = payload[off] & 0x1F + width_idx = payload[off + 1] & 0x0F + (colorref,) = struct.unpack_from("> 8) & 0xFF, (colorref >> 16) & 0xFF + sides.append(_BorderSide( + stroke=stroke, + width_mm=_BORDER_WIDTH_MM[width_idx], + color=f"{r:02X}{g:02X}{b:02X}")) + bf.sides = sides if len(payload) >= 40: (fillflags,) = struct.unpack_from("> 8) & 0xFF, (colorref >> 16) & 0xFF if (r, g, b) != (255, 255, 255): - return f"{r:02X}{g:02X}{b:02X}" - return None + bf.bg = f"{r:02X}{g:02X}{b:02X}" + return bf def _parse_bin_data(payload: bytes) -> Optional[Tuple[int, str]]: @@ -204,11 +260,11 @@ def _parse_doc_info(data: bytes) -> _DocInfo: name, _ = _read_bstr(payload, 1) face_names.append(name) elif tagid == TAG_BORDER_FILL: - info.border_fill_bg.append(_parse_border_fill(payload)) + info.border_fills.append(_parse_border_fill(payload)) elif tagid == TAG_CHAR_SHAPE: info.char_shapes.append(_parse_char_shape(payload)) elif tagid == TAG_PARA_SHAPE: - info.para_aligns.append(_parse_para_shape(payload)) + info.para_shapes.append(_parse_para_shape(payload)) elif tagid == TAG_BIN_DATA: info.bin_data.append(_parse_bin_data(payload)) # FACE_NAME 레코드는 언어 그룹 순서(ko 먼저)로 나온다 — ko 수만큼이 @@ -245,6 +301,9 @@ class _Cell: width_hu: int = 0 height_hu: int = 0 borderfill_id: int = 0 + valign: str = "center" # 표 60 listflags bits5-6 — 한글 기본은 가운데 + #: 표 75 안쪽 여백 (left,right,top,bottom) HWPUNIT16 + padding_hu: Tuple[int, int, int, int] = (0, 0, 0, 0) paras: List[_Para] = field(default_factory=list) @@ -384,6 +443,11 @@ def _parse_cell_props(payload: bytes) -> _Cell: 행 우선 순서로 재배정한다. """ c = _Cell() + if len(payload) >= 8: + # 표 60 리스트 헤더 flags @4 — bits5-6 VAlign(0 top/1 middle/2 bottom) + (listflags,) = struct.unpack_from("> 5) & 0x3, "center") if len(payload) >= 16: c.col, c.row, c.colspan, c.rowspan = struct.unpack_from("<4H", payload, 8) c.colspan = max(1, c.colspan) @@ -392,6 +456,8 @@ def _parse_cell_props(payload: bytes) -> _Cell: c.col = c.row = -1 if len(payload) >= 24: c.width_hu, c.height_hu = struct.unpack_from("<2i", payload, 16) + if len(payload) >= 32: + c.padding_hu = struct.unpack_from("<4H", payload, 24) if len(payload) >= 34: (c.borderfill_id,) = struct.unpack_from(" Optional[str]: def apply_align(para_obj, para: _Para) -> None: pid = para.parashape_id - if pid is None or not (0 <= pid < len(info.para_aligns)): + if pid is None or not (0 <= pid < len(info.para_shapes)): return - align = info.para_aligns[pid] - if align != "left": - para_obj.alignment = _WD_ALIGN[align] + pp = info.para_shapes[pid] + if pp.align != "left": + para_obj.alignment = _WD_ALIGN[pp.align] + pf = para_obj.paragraph_format + if pp.line_spacing is not None and abs(pp.line_spacing - 1.0) > 0.01: + pf.line_spacing = pp.line_spacing + if pp.space_before_pt > 0.05: + pf.space_before = Pt(pp.space_before_pt) + if pp.space_after_pt > 0.05: + pf.space_after = Pt(pp.space_after_pt) def emit_runs(para_obj, para: _Para) -> None: text = para.text @@ -719,6 +792,54 @@ def set_cell_bg(cell_obj, rgb: str) -> None: shd.set(qn("w:val"), "clear") shd.set(qn("w:fill"), rgb) + #: 표 20 stroke → docx 테두리 val (0 = 선 없음 → nil) + _STROKE_VAL = {0: "nil", 1: "single", 2: "dashed", 3: "dotted", + 4: "dotDash", 5: "dotDotDash", 6: "dashed", + 7: "dotted", 8: "double", 9: "double", 10: "double", + 11: "triple", 12: "wave", 13: "doubleWave"} + + def set_cell_borders(cell_obj, sides: List[_BorderSide]) -> None: + """표 18 좌/우/상/하 → w:tcBorders (sz = 1/8pt).""" + tc_pr = cell_obj._tc.get_or_add_tcPr() + borders = tc_pr.find(qn("w:tcBorders")) + if borders is None: + borders = OxmlElement("w:tcBorders") + tc_pr.append(borders) + for name, side in zip(("left", "right", "top", "bottom"), sides): + el = borders.find(qn(f"w:{name}")) + if el is None: + el = OxmlElement(f"w:{name}") + borders.append(el) + el.set(qn("w:val"), _STROKE_VAL.get(side.stroke, "single")) + # mm → pt(×72/25.4) → 1/8pt + el.set(qn("w:sz"), str(max(2, int(round(side.width_mm * 72 / 25.4 * 8))))) + el.set(qn("w:color"), side.color) + + def set_cell_margins(cell_obj, padding_hu: Tuple[int, int, int, int]) -> None: + """표 75 안쪽 여백 → w:tcMar (twips = hwpunit/5).""" + if not any(padding_hu): + return + tc_pr = cell_obj._tc.get_or_add_tcPr() + mar = tc_pr.find(qn("w:tcMar")) + if mar is None: + mar = OxmlElement("w:tcMar") + tc_pr.append(mar) + for name, hu in zip(("left", "right", "top", "bottom"), padding_hu): + el = OxmlElement(f"w:{name}") + el.set(qn("w:w"), str(int(hu / 5))) + el.set(qn("w:type"), "dxa") + mar.append(el) + + def set_cell_valign(cell_obj, valign: str) -> None: + if valign == "top": + return # docx 기본 + tc_pr = cell_obj._tc.get_or_add_tcPr() + va = tc_pr.find(qn("w:vAlign")) + if va is None: + va = OxmlElement("w:vAlign") + tc_pr.append(va) + va.set(qn("w:val"), valign) + def fill_cell_paras(cell_obj, paras: List[_Para]) -> None: first = True for para in paras: @@ -785,11 +906,16 @@ def emit_table(block: _Table, container_cell=None) -> None: cell_obj = tbl.cell(cell.row, cell.col) except Exception: # noqa: BLE001 continue - bg = None - if 0 <= cell.borderfill_id - 1 < len(info.border_fill_bg): - bg = info.border_fill_bg[cell.borderfill_id - 1] - if bg: - set_cell_bg(cell_obj, bg) + bf = None + if 0 <= cell.borderfill_id - 1 < len(info.border_fills): + bf = info.border_fills[cell.borderfill_id - 1] + if bf is not None: + if bf.bg: + set_cell_bg(cell_obj, bf.bg) + if bf.sides is not None: + set_cell_borders(cell_obj, bf.sides) + set_cell_valign(cell_obj, cell.valign) + set_cell_margins(cell_obj, cell.padding_hu) fill_cell_paras(cell_obj, cell.paras) for cap in block.caption: p = doc.add_paragraph() if container_cell is None \ diff --git a/tests/unit/test_docx_pages.py b/tests/unit/test_docx_pages.py index 79c9e87f..ee2a30eb 100644 --- a/tests/unit/test_docx_pages.py +++ b/tests/unit/test_docx_pages.py @@ -139,6 +139,58 @@ def test_trheight_minimum_is_honored(self): heights = [float(h) for h in re.findall(r']*height="([\d.]+)"', joined)] assert any(abs(h - 200.0) < 1.0 for h in heights) + def test_tc_borders_nil_and_dashed(self): + from docx import Document + from docx.oxml import OxmlElement + from docx.oxml.ns import qn + + doc = Document() + t = doc.add_table(rows=1, cols=2) + t.cell(0, 0).text = "테두리없음" + t.cell(0, 1).text = "대시" + + def borders(cell, val): + tc_pr = cell._tc.get_or_add_tcPr() + tcb = OxmlElement("w:tcBorders") + tc_pr.append(tcb) + for side in ("left", "right", "top", "bottom"): + el = OxmlElement(f"w:{side}") + el.set(qn("w:val"), val) + el.set(qn("w:sz"), "8") + el.set(qn("w:color"), "FF0000") + tcb.append(el) + + borders(t.cell(0, 0), "nil") + borders(t.cell(0, 1), "dashed") + joined = "".join(docx_to_page_svgs(_docx_bytes(doc))) + assert 'stroke-dasharray="6 3"' in joined # 대시 변 + assert joined.count('stroke="#FF0000"') == 4 # 대시 셀 4변만 (nil 은 0변) + + def test_paragraph_spacing_expands_flow(self): + import re + + from docx import Document + + doc = Document() + a = doc.add_paragraph("첫문장") + b = doc.add_paragraph("둘문장") + base = "".join(docx_to_page_svgs(_docx_bytes(doc))) + + doc2 = Document() + a2 = doc2.add_paragraph("첫문장") + a2.paragraph_format.line_spacing = 2.0 + doc2.add_paragraph("둘문장") + spaced = "".join(docx_to_page_svgs(_docx_bytes(doc2))) + + def y_of(svg, text): + m = re.search(rf']*y="([\d.]+)"[^>]*>{text}', svg) + assert m, text + return float(m.group(1)) + + gap_base = y_of(base, "둘문장") - y_of(base, "첫문장") + gap_spaced = y_of(spaced, "둘문장") - y_of(spaced, "첫문장") + assert gap_spaced > gap_base * 1.5 # 줄간격 200% 가 흐름에 반영 + def test_nested_table_content_not_lost(self): from docx import Document diff --git a/tests/unit/test_legacy_formats.py b/tests/unit/test_legacy_formats.py index 60737ffd..915f02de 100644 --- a/tests/unit/test_legacy_formats.py +++ b/tests/unit/test_legacy_formats.py @@ -274,8 +274,16 @@ def bstr(s: str) -> bytes: bin_data = struct.pack(" bytes: + def border_fill(bg_colorref: int | None, *, stroke: int = 1, + width_idx: int = 1, color: int = 0x000000, + left_stroke: int | None = None) -> bytes: p = bytearray(44) + for k in range(4): # left, right, top, bottom (표 18 순서) + off = 2 + k * 6 + st = left_stroke if (k == 0 and left_stroke is not None) else stroke + p[off] = st + p[off + 1] = width_idx + struct.pack_into(" bytes: struct.pack_into(" bytes: - return struct.pack(" bytes: + p = bytearray(44) + struct.pack_into(" bytes: p = bytearray(40) struct.pack_into(" bytes: _rec(0x4D, 2, table_rec), cell(2, cell_props(0, 0, 2, 1, 16000, 2000, 2), "병합 머리"), cell(2, cell_props(2, 0, 1, 2, 8000, 2000, 1), "세로 병합"), - cell(2, cell_props(0, 1, 1, 1, 8000, 2000, 1), "좌"), + cell(2, cell_props(0, 1, 1, 1, 8000, 2000, 3), "좌"), cell(2, cell_props(1, 1, 1, 1, 8000, 2000, 1), "우"), # 그림 개체 (gso) — BinData/BIN0001.png _rec(0x47, 1, b" osg" + b"\x00" * 8), @@ -824,6 +839,24 @@ def test_column_widths_reach_grid(self, rich_docx): xml = rich_docx.tables[0]._tbl.xml assert "gridCol" in xml + def test_cell_borders_from_borderfill(self, rich_docx): + """표 18 변별 테두리 — 실선/대시/없음이 tcBorders 로 간다.""" + xml = rich_docx.tables[0]._tbl.xml + assert 'w:tcBorders' in xml + assert 'w:val="single"' in xml + assert 'w:val="dashed"' in xml # borderfill 2 좌변 + assert 'w:val="nil"' in xml # borderfill 3 (테두리 없음, '좌' 셀) + + def test_cell_valign_and_margins(self, rich_docx): + xml = rich_docx.tables[0]._tbl.xml + assert 'w:vAlign' in xml and 'w:val="center"' in xml # listflags 세로 가운데 + assert 'w:tcMar' in xml and 'w:w="170"' in xml # 850 HWPUNIT → 170 dxa + + def test_line_spacing_from_parashape(self, rich_docx): + first = rich_docx.paragraphs[0] + assert first.text == "가운데 취소선" + assert abs(first.paragraph_format.line_spacing - 2.0) < 0.01 + def test_e2e_svg_render(self, tmp_path): svg = _render_svg_pages(tmp_path, "rich.hwp", make_hwp_rich()) assert "병합 머리" in svg and svg.count("병합 머리") == 1 @@ -832,6 +865,8 @@ def test_e2e_svg_render(self, tmp_path): assert "line-through" in svg # 취소선 assert "