diff --git a/CHANGELOG.md b/CHANGELOG.md index c1f6e834..cf0fd473 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -49,6 +49,8 @@ - 개체 앞뒤의 줄 바꿈도 새 줄을 연다. 개체 앞의 빈 줄은 글 한 줄만큼 내려간다. - 이제 그런 문단을 FormFit 줄 나눔으로 나누고, 줄마다 높이를 센다(본문과 표 칸). 넘치지 않으면 전처럼 개체 높이의 한 줄이다. + - 유효한 줄 캐시가 있으면 다른 문단처럼 캐시의 줄(한/글이 파일을 열 때 쓰는 줄)을 그대로 쓰고, 캐시가 없을 + 때만 FormFit으로 나눈다. - FormFit `hancom_line_starts`가 개체 자리(`objects`)를 받아, 개체 바로 뒤 둘째 공백도 여백 밖에 매단다. - 쪽 수 추정(실험, `estimate_pages`)이 줄 캐시 없는 표 칸에서 글자처럼 둔 표가 글 사이에 있거나 여럿인 문단을 지원 밖(`a nested table`)으로 두던 것을 고친다. diff --git a/src/hwpx/layout/pages.py b/src/hwpx/layout/pages.py index b478abe9..e92d9fad 100644 --- a/src/hwpx/layout/pages.py +++ b/src/hwpx/layout/pages.py @@ -572,7 +572,8 @@ def stack(self, paragraphs: list[Any], width: int, caches: bool) -> tuple[int, i pitch = _pitch(shape.kind, shape.value, size) table = _table_alone(runs) alone = None if table is not None or caches and _cached_metrics(paragraph) else _object_alone(runs) - spread = self.spread_lines(paragraph, runs, width) if table is not None or alone is not None else () + spread = self.spread_lines(paragraph, runs, width, caches=caches) \ + if table is not None or alone is not None else () if spread: # spaces besides it, some of which go on to the next line table = alone = None if table is not None: # one line as tall as the table, spaced like the text @@ -636,7 +637,7 @@ def stack_lines(self, paragraphs: list[Any], width: int, caches: bool) -> tuple[ metrics[-1] = (last_height, last_advance + shape.prev) runs = paragraph.findall(f"{HP}run") cached = () if _table_alone(runs) is not None or not caches else _cached_metrics(paragraph) - cached = cached or self.spread_lines(paragraph, runs, width) + cached = cached or self.spread_lines(paragraph, runs, width, caches=caches) if not cached and _table_alone(runs) is None: # each line as tall as stack makes it cached = self.pushed_lines(paragraph, runs, width) or self.marked_lines(paragraph, runs, width) \ or self.mixed_lines(paragraph, runs, shape, width) @@ -648,19 +649,22 @@ def stack_lines(self, paragraphs: list[Any], width: int, caches: bool) -> tuple[ metrics += [(size, pitch)] * (count - 1) + [(last, (last if last > size else pitch) + shape.next)] return tuple(metrics) - def spread_lines(self, paragraph: Any, runs: list[Any], width: int, end: int = 0, - head: int = 0) -> tuple[tuple[int, int], ...]: + def spread_lines(self, paragraph: Any, runs: list[Any], width: int, end: int = 0, head: int = 0, + caches: bool = False) -> tuple[tuple[int, int], ...]: """(height, advance) of each line of a paragraph holding one object set as a character and, besides it, only spaces or line breaks, when Hancom lays it out on several lines: a space before the object is text (the object goes on to the next line when it does not fit after it), the two spaces right after it hang past the margin and a further one starting there begins the next line, and a line break begins one (an empty line before the object goes down as a line of text); empty for any other paragraph, or one of a - line.""" + line. With *caches* a paragraph with a valid layout cache keeps Hancom's lines.""" objects = _placed_objects(runs) text = _run_text(runs) if len(objects) != 1 or not text or text.strip() or objects[0].find(f"{HP}pos").get("treatAsChar") != "1": return () + own = _cached_metrics(paragraph) if caches else () + if own: + return own if len(own) > 1 else () shape = self.shape(paragraph.get("paraPrIDRef")) if shape.kind not in ("PERCENT", "FIXED"): return () @@ -1853,7 +1857,7 @@ def _paragraph(measure: _Measure, page: _Page, paragraph: Any, wrap: _Wrap | Non looks = _char_styles(measure, paragraph, runs) lead = 0 # how far the paragraph's first line goes down below a square-wrapped object's band if alone and wrap is None: # spaces besides it, those that do not fit going on to the next line - spread = measure.spread_lines(paragraph, runs, page.column_width, end, head) + spread = measure.spread_lines(paragraph, runs, page.column_width, end, head, caches=True) alone, cached = (False, spread) if spread else (alone, cached) if among or beside: inline_text, inline_sizes, inline_looks, placed, marked = _inline_content(measure, paragraph, runs, anchored) diff --git a/tests/test_layout_page_estimate.py b/tests/test_layout_page_estimate.py index e4790a4c..195c02da 100644 --- a/tests/test_layout_page_estimate.py +++ b/tests/test_layout_page_estimate.py @@ -705,6 +705,42 @@ def test_a_typed_ideographic_space_is_not_a_fixed_width_one() -> None: assert (len(cells), typed) == (30, 22) +def _with_wider_tables_set_as_characters(data: bytes, wider: int) -> bytes: + """*data* with every table set as a character *wider* wider (narrower when negative).""" + + out = io.BytesIO() + with zipfile.ZipFile(io.BytesIO(data)) as source, zipfile.ZipFile(out, "w", zipfile.ZIP_DEFLATED) as target: + for info in source.infolist(): + payload = source.read(info.filename) + if info.filename.startswith("Contents/section"): + root = etree.fromstring(payload) + for table in root.iter(f"{HP}tbl"): + if table.find(f"{HP}pos").get("treatAsChar") == "1": + size = table.find(f"{HP}sz") + size.set("width", str(int(size.get("width")) + wider)) + payload = etree.tostring(root, xml_declaration=True, encoding="UTF-8", standalone=True) + target.writestr(info, payload) + return out.getvalue() + + +@pytest.mark.parametrize( + ("name", "wider"), + [ + ("pages_table_as_character_2000_short_then_three_spaces", 1800), # one line: 200 short, the third space + # would go on to a line of its own + ("pages_table_cell_table_as_character_300_short_then_three_spaces", -2000), # two lines in a cell: 2300 + # short, all would fit + ], +) +def test_a_paragraph_of_an_object_and_spaces_keeps_its_cached_lines(name: str, wider: int) -> None: + # A table set as a character and three spaces after it in its paragraph, laid out by Hancom, then the + # table made wider or narrower: with the caches the paragraph keeps Hancom's lines, as Hancom does when + # it opens the file, whatever FormFit would make of the new width. + data = (FIXTURES / f"{name}.hwpx").read_bytes() + + _assert_like_hancom(estimate_pages(_with_wider_tables_set_as_characters(data, wider)), data, 1) + + def test_a_header_wrapped_square_with_room_beside_it_is_not_followed() -> None: # The exam header 3000 narrower: a line fits beside it, and the lines reaching it are not followed. out = io.BytesIO()