Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,8 @@
- 개체 앞뒤의 줄 바꿈도 새 줄을 연다. 개체 앞의 빈 줄은 글 한 줄만큼 내려간다.
- 이제 그런 문단을 FormFit 줄 나눔으로 나누고, 줄마다 높이를 센다(본문과 표 칸). 넘치지 않으면 전처럼 개체
높이의 한 줄이다.
- 유효한 줄 캐시가 있으면 다른 문단처럼 캐시의 줄(한/글이 파일을 열 때 쓰는 줄)을 그대로 쓰고, 캐시가 없을
때만 FormFit으로 나눈다.
- FormFit `hancom_line_starts`가 개체 자리(`objects`)를 받아, 개체 바로 뒤 둘째 공백도 여백 밖에 매단다.
- 쪽 수 추정(실험, `estimate_pages`)이 줄 캐시 없는 표 칸에서 글자처럼 둔 표가 글 사이에 있거나 여럿인 문단을
지원 밖(`a nested table`)으로 두던 것을 고친다.
Expand Down
16 changes: 10 additions & 6 deletions src/hwpx/layout/pages.py
Original file line number Diff line number Diff line change
Expand Up @@ -572,7 +572,8 @@ def stack(self, paragraphs: list[Any], width: int, caches: bool) -> tuple[int, i
pitch = _pitch(shape.kind, shape.value, size)
table = _table_alone(runs)
alone = None if table is not None or caches and _cached_metrics(paragraph) else _object_alone(runs)
spread = self.spread_lines(paragraph, runs, width) if table is not None or alone is not None else ()
spread = self.spread_lines(paragraph, runs, width, caches=caches) \
if table is not None or alone is not None else ()
if spread: # spaces besides it, some of which go on to the next line
table = alone = None
if table is not None: # one line as tall as the table, spaced like the text
Expand Down Expand Up @@ -636,7 +637,7 @@ def stack_lines(self, paragraphs: list[Any], width: int, caches: bool) -> tuple[
metrics[-1] = (last_height, last_advance + shape.prev)
runs = paragraph.findall(f"{HP}run")
cached = () if _table_alone(runs) is not None or not caches else _cached_metrics(paragraph)
cached = cached or self.spread_lines(paragraph, runs, width)
cached = cached or self.spread_lines(paragraph, runs, width, caches=caches)
if not cached and _table_alone(runs) is None: # each line as tall as stack makes it
cached = self.pushed_lines(paragraph, runs, width) or self.marked_lines(paragraph, runs, width) \
or self.mixed_lines(paragraph, runs, shape, width)
Expand All @@ -648,19 +649,22 @@ def stack_lines(self, paragraphs: list[Any], width: int, caches: bool) -> tuple[
metrics += [(size, pitch)] * (count - 1) + [(last, (last if last > size else pitch) + shape.next)]
return tuple(metrics)

def spread_lines(self, paragraph: Any, runs: list[Any], width: int, end: int = 0,
head: int = 0) -> tuple[tuple[int, int], ...]:
def spread_lines(self, paragraph: Any, runs: list[Any], width: int, end: int = 0, head: int = 0,
caches: bool = False) -> tuple[tuple[int, int], ...]:
"""(height, advance) of each line of a paragraph holding one object set as a character and, besides it,
only spaces or line breaks, when Hancom lays it out on several lines: a space before the object is text
(the object goes on to the next line when it does not fit after it), the two spaces right after it hang
past the margin and a further one starting there begins the next line, and a line break begins one (an
empty line before the object goes down as a line of text); empty for any other paragraph, or one of a
line."""
line. With *caches* a paragraph with a valid layout cache keeps Hancom's lines."""

objects = _placed_objects(runs)
text = _run_text(runs)
if len(objects) != 1 or not text or text.strip() or objects[0].find(f"{HP}pos").get("treatAsChar") != "1":
return ()
own = _cached_metrics(paragraph) if caches else ()
if own:
return own if len(own) > 1 else ()
shape = self.shape(paragraph.get("paraPrIDRef"))
if shape.kind not in ("PERCENT", "FIXED"):
return ()
Expand Down Expand Up @@ -1853,7 +1857,7 @@ def _paragraph(measure: _Measure, page: _Page, paragraph: Any, wrap: _Wrap | Non
looks = _char_styles(measure, paragraph, runs)
lead = 0 # how far the paragraph's first line goes down below a square-wrapped object's band
if alone and wrap is None: # spaces besides it, those that do not fit going on to the next line
spread = measure.spread_lines(paragraph, runs, page.column_width, end, head)
spread = measure.spread_lines(paragraph, runs, page.column_width, end, head, caches=True)
alone, cached = (False, spread) if spread else (alone, cached)
if among or beside:
inline_text, inline_sizes, inline_looks, placed, marked = _inline_content(measure, paragraph, runs, anchored)
Expand Down
36 changes: 36 additions & 0 deletions tests/test_layout_page_estimate.py
Original file line number Diff line number Diff line change
Expand Up @@ -705,6 +705,42 @@ def test_a_typed_ideographic_space_is_not_a_fixed_width_one() -> None:
assert (len(cells), typed) == (30, 22)


def _with_wider_tables_set_as_characters(data: bytes, wider: int) -> bytes:
"""*data* with every table set as a character *wider* wider (narrower when negative)."""

out = io.BytesIO()
with zipfile.ZipFile(io.BytesIO(data)) as source, zipfile.ZipFile(out, "w", zipfile.ZIP_DEFLATED) as target:
for info in source.infolist():
payload = source.read(info.filename)
if info.filename.startswith("Contents/section"):
root = etree.fromstring(payload)
for table in root.iter(f"{HP}tbl"):
if table.find(f"{HP}pos").get("treatAsChar") == "1":
size = table.find(f"{HP}sz")
size.set("width", str(int(size.get("width")) + wider))
payload = etree.tostring(root, xml_declaration=True, encoding="UTF-8", standalone=True)
target.writestr(info, payload)
return out.getvalue()


@pytest.mark.parametrize(
("name", "wider"),
[
("pages_table_as_character_2000_short_then_three_spaces", 1800), # one line: 200 short, the third space
# would go on to a line of its own
("pages_table_cell_table_as_character_300_short_then_three_spaces", -2000), # two lines in a cell: 2300
# short, all would fit
],
)
def test_a_paragraph_of_an_object_and_spaces_keeps_its_cached_lines(name: str, wider: int) -> None:
# A table set as a character and three spaces after it in its paragraph, laid out by Hancom, then the
# table made wider or narrower: with the caches the paragraph keeps Hancom's lines, as Hancom does when
# it opens the file, whatever FormFit would make of the new width.
data = (FIXTURES / f"{name}.hwpx").read_bytes()

_assert_like_hancom(estimate_pages(_with_wider_tables_set_as_characters(data, wider)), data, 1)


def test_a_header_wrapped_square_with_room_beside_it_is_not_followed() -> None:
# The exam header 3000 narrower: a line fits beside it, and the lines reaching it are not followed.
out = io.BytesIO()
Expand Down
Loading