Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
10 changes: 10 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -13,6 +13,16 @@
않는다. `hwpx.TextExtractor`·`hwpx.doc_diff`·`hwpx.validate_package`처럼 `hwpx.tools`에 사는
이름은 처음 쓸 때 경고 없이 읽고, 같은 객체다. 위 `hwpx.hwp5`와 합쳐 `import hwpx`가 읽는
hwpx 모듈은 117개에서 97개로 준다. 저장은 여전히 처음 저장할 때 패키지 검증 도구를 읽는다.
- FormFit이 글꼴 12종(바탕·바탕체·돋움·돋움체·굴림·굴림체·궁서·궁서체·한컴 고딕·함초롬바탕·함초롬돋움·맑은 고딕)에서
KS X 1001 특수 기호와 전각 문자, 라틴-1·라틴 확장-A·일반 문장 부호·통화 기호·수학 연산자·도형·기타 기호·딩뱃·
반각/전각 형태 블록의 글자를 글꼴의 설계 폭으로 잰다. 전에는 흔한 문장 부호와 기호 몇십 자 말고는 글자 종류의 평균 폭을
썼다. 글꼴에 없는 글자는 한/글이 대신 그리는 글꼴(바탕·궁서와 -체 글꼴은 함초롬바탕, 굴림·돋움·맑은 고딕·한컴
고딕은 함초롬돋움, 함초롬 글꼴은 Segoe UI Symbol)의 폭을 그 글꼴 단위로 옮겨 쓴다. 이런 글자가 든 칸은 채우기
결과(글자 크기, 줄 수)가 달라질 수 있다.
- 기호 글꼴 Wingdings·Symbol도 문서가 글자를 두는 사용자 영역(U+F020–F0FF)의 설계 폭으로 잰다. 실제 너비를 쓰는
Wingdings 글머리표(U+F09F)는 10 pt에서 456을 차지한다.
- HY헤드라인M도 글꼴의 설계 폭으로 잰다. 글꼴에 없는 글자와 `` ` ``는 함초롬바탕의 폭을 쓴다.
- 휴먼명조도 글꼴의 설계 폭으로 잰다. 글꼴에 없는 글자는 함초롬바탕의 폭을 쓴다.

### 고침

Expand Down
6 changes: 6 additions & 0 deletions docs/architecture/module-ownership.json
Original file line number Diff line number Diff line change
Expand Up @@ -716,6 +716,12 @@
"disposition": "core",
"approvedBy": "shape-curves-and-connectors",
"rationale": "HwpxOxmlParagraph.add_curve and add_connector: builds hp:curve (CURVE segments through the given points, the box of the Catmull-Rom curve through them, the points relative to that box) and hp:connectLine (STRAIGHT or STROKE, NOARROW) whose ends sit at the middle of a side of two floating objects placed in the same frame, with the connector's start/end object references. A new module rather than more objects.py lines (the 1,600-line owner-file cap); the two builders are attached to HwpxOxmlParagraph as plain class attributes, the same escape valve as shape_position.py and drop_cap.py. Format-level only: no renderer, application workflow or agent policy."
},
{
"path": "src/hwpx/form_fit/_glyph_table.py",
"disposition": "core",
"approvedBy": "formfit-glyph-widths",
"rationale": "FormFit glyph widths beyond measure._GLYPHS: per face, the design advance of each glyph of the KS X 1001 symbols, the full-width forms and eight common Unicode blocks, read from the font files, and the face Hancom lays a glyph out from when a face lacks it. Data only; imports nothing."
}
],
"removed": [
Expand Down
1,444 changes: 1,444 additions & 0 deletions src/hwpx/form_fit/_glyph_table.py

Large diffs are not rendered by default.

87 changes: 72 additions & 15 deletions src/hwpx/form_fit/measure.py
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,7 @@
from typing import Any, Literal

from ..oxml.table_sizes import cell_margins_of, grid_widths_of
from ._glyph_table import ADVANCES, FALLBACK, UNITS_PER_EM

# Advance width as a fraction of the em (font height in HWPUNIT). Hangul/wide are
# exact (full-width cells); the Latin/digit/punct values are conservative class
Expand Down Expand Up @@ -132,8 +133,10 @@
# at that em, rounded half up at 100% 장평 and down at any other 장평, and 자간
# adds its share of that advance, rounded half away from zero. The half-em space
# is half the em, rounded down, at the 장평, rounded half up. Bold text keeps the
# regular advances. A face or glyph not listed below falls back to the class
# averages, unrounded.
# regular advances. A glyph a face lacks takes the advance of the face Hancom
# lays it out from (``_glyph_table.FALLBACK``), moved into the face's own units
# and rounded. A face or glyph listed nowhere falls back to the class averages,
# unrounded.
_LAYOUT_UNIT = 4

_HP = "{http://www.hancom.co.kr/hwpml/2011/paragraph}"
Expand All @@ -145,7 +148,8 @@
_GLYPHS = '!"#$%&\'()*+,-./0123456789:;<=>?@ABCDEFGHIJKLMNOPQRSTUVWXYZ[\\]^_`abcdefghijklmnopqrstuvwxyz{|}~·…“”‘’「」『』〈〉《》※○●□■△▲◇◆☆★→←↑↓ㆍ×÷±°℃‰—–ⅠⅡⅢⅣⅤⅥⅦⅧⅨⅩⅪⅫⅰⅱⅲⅳⅴⅵⅶⅷⅸⅹⅺⅻ①②③④⑤⑥⑦⑧⑨⑩⑪⑫⑬⑭⑮⑯⑰⑱⑲⑳⑴⑵⑶⑷⑸⑹⑺⑻⑼⑽⑾⑿⒀⒁⒂⒃⒄⒅⒆⒇'

#: Design advances per face in font units: units per em, a Hangul syllable, the
#: space, and each glyph of ``_GLYPHS`` (0: not in the face).
#: space, and each glyph of ``_GLYPHS`` (0: not in the face, or one Hancom lays
#: out from the fallback face all the same).
_DESIGN: dict[str, tuple[int, int, int, tuple[int, ...]]] = {
"함초롬바탕": (1000, 970, 300, (
320, 320, 610, 610, 830, 724, 320, 320, 320, 550, 550, 320, 550, 320, 550, 550,
Expand Down Expand Up @@ -327,6 +331,36 @@
1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 0,
0, 0, 0, 0,
)),
"HY헤드라인M": (1024, 1024, 341, (
256, 384, 853, 597, 810, 640, 213, 298, 298, 341, 597, 213, 597, 213, 384, 597,
597, 597, 597, 597, 597, 597, 597, 597, 597, 256, 256, 469, 597, 469, 597, 853,
597, 640, 640, 640, 469, 469, 640, 640, 256, 341, 597, 426, 810, 597, 597, 554,
597, 597, 597, 512, 640, 640, 938, 597, 554, 469, 341, 384, 341, 554, 512, 0,
597, 597, 597, 597, 640, 384, 597, 597, 256, 298, 554, 256, 853, 597, 640, 597,
597, 426, 554, 384, 597, 554, 810, 512, 512, 426, 426, 298, 426, 725, 1024, 1024,
1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024,
1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024,
1024, 1024, 0, 0, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 0, 0,
1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 0, 0, 1024, 1024, 1024, 1024,
1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 0, 0, 0, 0, 0,
1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 1024, 0,
0, 0, 0, 0,
)),
"휴먼명조": (512, 512, 256, (
104, 256, 336, 258, 384, 394, 148, 160, 160, 256, 256, 138, 256, 138, 160, 256,
256, 256, 256, 256, 256, 256, 256, 256, 256, 138, 138, 256, 256, 256, 220, 398,
338, 308, 328, 348, 320, 298, 360, 348, 132, 202, 330, 296, 400, 354, 366, 294,
352, 334, 254, 320, 346, 340, 478, 364, 344, 306, 164, 160, 164, 186, 256, 166,
260, 278, 248, 260, 268, 200, 264, 284, 128, 136, 268, 128, 408, 282, 262, 276,
276, 206, 210, 174, 276, 276, 370, 250, 262, 232, 150, 104, 150, 256, 512, 512,
512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512,
512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512,
512, 512, 0, 0, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 0, 0,
512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 0, 0, 512, 512, 512, 512,
512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 0, 0, 0, 0, 0,
512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 512, 0,
0, 0, 0, 0,
)),
}


Expand All @@ -337,22 +371,45 @@ def glyph_advance_em(face: str, ch: str) -> float | None:
return design[0] / design[1] if design is not None else None


@lru_cache(maxsize=8192)
def _design_units(face: str, ch: str | None) -> tuple[int, int] | None:
"""``(advance, units per em)`` of *ch* in *face*, a Hangul syllable when *ch*
is ``None``, or ``None`` when not listed."""
is ``None``, or ``None`` when not listed. A glyph the face lacks takes the
advance of its fallback face, in the face's own units."""

entry = _DESIGN.get(face)
if entry is None:
return None
upem, hangul, space, glyphs = entry
if ch is None:
units = hangul
elif ch == " ":
units = space
else:
index = _GLYPHS.find(ch) if len(ch) == 1 else -1
units = glyphs[index] if index >= 0 else 0
return (units, upem) if units else None
if entry is None: # a symbol face only the glyph table lists (no Hangul, no space of its own)
units = _face_glyphs(face).get(ch, 0) if ch is not None and ch != " " else 0
return (units, UNITS_PER_EM[face]) if units else None
upem, hangul, space, _ = entry
if ch is None or ch == " ":
return (hangul if ch is None else space), upem
source = face
while source:
units = _own_units(source, ch)
if units:
if source != face:
units = math.floor(Fraction(units * upem, UNITS_PER_EM[source]) + Fraction(1, 2))
return units, upem
source = FALLBACK.get(source, "")
return None


def _own_units(face: str, ch: str) -> int:
"""Design advance of *ch* in *face* itself; 0 when the face does not have it."""

entry = _DESIGN.get(face)
index = _GLYPHS.find(ch) if len(ch) == 1 else -1
if entry is not None and index >= 0:
return entry[3][index]
return _face_glyphs(face).get(ch, 0)


@lru_cache(maxsize=None)
def _face_glyphs(face: str) -> dict[str, int]:
"""Design advance of every glyph ``_glyph_table`` lists for *face*."""

return {ch: units for units, chars in ADVANCES.get(face, {}).items() for ch in chars}


@lru_cache(maxsize=4096)
Expand Down
Binary file not shown.
Binary file not shown.
Binary file not shown.
77 changes: 75 additions & 2 deletions tests/test_form_fit_hancom_rules.py
Original file line number Diff line number Diff line change
Expand Up @@ -62,6 +62,8 @@
for name in ("margin_word", "space_without_spacing", "text_reaching_margin")
]
LABELS = Path(__file__).parent / "fixtures" / "hancom_saved" / "pages_bullet_and_number_labels.hwpx"
WINGDINGS_LABEL = Path(__file__).parent / "fixtures" / "hancom_saved" / "formfit_wingdings_label_rows.hwpx"
HY_HEADLINE_CELLS = Path(__file__).parent / "fixtures" / "hancom_saved" / "formfit_hy_headline_cells.hwpx"
NO_BREAK_SPACES = Path(__file__).parent / "fixtures" / "hancom_saved" / "pages_no_break_spaces.hwpx"
FIXED_WIDTH_SPACES = Path(__file__).parent / "fixtures" / "hancom_saved" / "pages_fixed_width_spaces.hwpx"
SCRIPT_ROWS = [
Expand Down Expand Up @@ -93,6 +95,61 @@ def test_each_character_can_take_its_own_size() -> None:
assert hancom_line_starts("가나다라마", [5000], 10, style, sizes=[10, 10, 20, 10, 10]) == [0, 4]


def test_cells_of_symbols_break_where_hancom_breaks_them() -> None:
# Hancom laid this document out. Each cell holds a symbol and five Hangul syllables at one of the two
# widths where its line breaks change: symbols a face lacks (drawn from its fallback face, in faces of
# 1000, 1024 and 2048 units per em), the Greek of 함초롬돋움, 한컴 고딕's я, and design advances.
doc = HwpxDocument.open(Path(__file__).parent / "fixtures" / "hancom_saved" / "formfit_glyph_cells.hwpx")
cells = list(doc.oxml.sections[0].element.iter(f"{HP}tc"))
for cell in cells:
paragraph = cell.find(f"{HP}subList/{HP}p")
text = "".join(t.text or "" for t in paragraph.iter(f"{HP}t"))
ref = paragraph.find(f"{HP}run").get("charPrIDRef")
style = text_style_from_refs(doc.oxml, paragraph.get("paraPrIDRef"), [ref])
margins = cell.find(f"{HP}cellMargin")
width = int(cell.find(f"{HP}cellSz").get("width")) - int(margins.get("left")) - int(margins.get("right"))
hancom = len(paragraph.findall(f"{HP}linesegarray/{HP}lineseg"))
assert estimate_lines(text, width, 10, style) == hancom, (style.glyph_face, f"U+{ord(text[0]):04X}", width)
assert len(cells) == 24


def test_cells_in_hy_headline_break_where_hancom_breaks_them() -> None:
# Hancom laid this document out: each cell holds a glyph and five Hangul syllables in HY헤드라인M at one of the
# two widths where its line count changes, at 30 pt and at 10 pt. The glyphs take the face's design advance;
# those it lacks (₩ €) and the grave accent take 함초롬바탕's, moved into the face's units; я and ✀ take the
# advance Hancom gives them.
doc = HwpxDocument.open(HY_HEADLINE_CELLS)
cells = list(doc.oxml.sections[0].element.iter(f"{HP}tc"))
for cell in cells:
paragraph = cell.find(f"{HP}subList/{HP}p")
text = "".join(t.text or "" for t in paragraph.iter(f"{HP}t"))
ref = paragraph.find(f"{HP}run").get("charPrIDRef")
style = text_style_from_refs(doc.oxml, paragraph.get("paraPrIDRef"), [ref])
points = int(doc.oxml.char_property(ref).attributes["height"]) / 100
margins = cell.find(f"{HP}cellMargin")
width = int(cell.find(f"{HP}cellSz").get("width")) - int(margins.get("left")) - int(margins.get("right"))
hancom = len(paragraph.findall(f"{HP}linesegarray/{HP}lineseg"))
assert style.glyph_face == "HY헤드라인M"
assert estimate_lines(text, width, points, style) == hancom, (f"U+{ord(text[0]):04X}", points, width)
assert len(cells) == 22


def test_human_myeongjo_takes_the_design_advances_of_its_font() -> None:
# 휴먼명조's font has 512 units per em: a Hangul syllable takes the full em, a parenthesis 160 units, the comma
# 138 and A 338.
style = TextStyle(hangul_face="휴먼명조", glyph_face="휴먼명조")
assert glyph_advance_em("휴먼명조", "(") == 160 / 512
assert [char_advance(ch, 10, style) for ch in "가(,A"] == [1000, 312, 268, 660]


def test_every_fallback_face_has_a_table() -> None:
from hwpx.form_fit import _glyph_table as table

for face, fallback in table.FALLBACK.items():
assert face in table.ADVANCES and fallback in table.ADVANCES
assert face in table.UNITS_PER_EM and fallback in table.UNITS_PER_EM


def test_each_character_can_take_its_own_style() -> None:
wide = TextStyle(break_non_latin_word="KEEP_WORD")
narrow = TextStyle(break_non_latin_word="KEEP_WORD", ratio=50)
Expand Down Expand Up @@ -456,8 +513,9 @@ def test_a_numeral_a_face_does_not_list_is_full_width() -> None:


def test_a_symbol_in_a_face_the_table_does_not_list_breaks_where_hancom_breaks_it() -> None:
# Five ○ in 휴먼명조 10 pt, which the glyph table does not list, in cells 4400 and 5400 wide inside:
# Hancom laid them out in two lines and in one.
# Five ○ at 10 pt in cells 4400 and 5400 wide inside: Hancom laid them out in two lines and in one. The
# cells are in 휴먼명조, which the glyph table lists, so they are measured with the face left unnamed as well:
# a face the table does not list draws ○ full width.
doc = HwpxDocument.open(UNLISTED_SYMBOLS.read_bytes())
tables = [table for paragraph in doc.paragraphs for table in paragraph.tables]

Expand All @@ -466,6 +524,9 @@ def test_a_symbol_in_a_face_the_table_does_not_list_breaks_where_hancom_breaks_i
cell = table.cell(0, 0)
segs = cell.paragraphs[0].element.findall(f"{HP}linesegarray/{HP}lineseg")
slot = resolve_slot_metrics(cell, doc, max_lines=10, safety=1.0)
assert slot.text_style.hangul_face == "휴먼명조"
assert measure(cell.text, slot).lines == len(segs), (cell.width, len(segs))
slot.text_style = replace(slot.text_style, hangul_face="", glyph_face="")
assert glyph_advance_em(slot.text_style.hangul_face, "○") is None
assert measure(cell.text, slot).lines == len(segs), (cell.width, len(segs))
lines.append(len(segs))
Expand Down Expand Up @@ -725,6 +786,18 @@ def test_the_rows_under_bullets_and_numbers_break_where_hancom_breaks() -> None:

assert len(rows) == 168
assert [starts for starts, _ in rows[:144]] == [hancom for _, hancom in rows[:144]]


def test_a_wingdings_bullet_at_its_own_width_takes_the_glyph_design_advance() -> None:
# 25 rows of 120 syllables, 0.15 mm narrower one after another, under the bullet U+F09F in its own Wingdings
# shape with its own width on (useInstWidth): Hancom takes the glyph's design advance, 456 at 10 pt, and the
# half-em gap off every line, laid out and saved by Hancom.
rows = _row_line_starts(WINGDINGS_LABEL, labelled=True)

assert len(rows) == 25
assert [starts for starts, _ in rows] == [hancom for _, hancom in rows]


def test_a_no_break_space_is_half_an_em_and_keeps_the_words_together() -> None:
assert char_advance("\u00a0", 10, TextStyle()) == char_advance(" ", 10, TextStyle()) == 500
# "다라" and "마바" stay together: the line breaks at the space before them, not after the no-break space.
Expand Down
3 changes: 2 additions & 1 deletion tests/test_product_boundary.py
Original file line number Diff line number Diff line change
Expand Up @@ -318,7 +318,8 @@ def test_real_tree_gate_runs_from_a_gitless_source_copy(tmp_path: Path) -> None:
# +1: the experimental page estimate (layout/pages.py), re-exported from hwpx.experimental.
# +1: curves and connectors (oxml/curves.py) -- objects.py's overflow module under the
# 1600-line cap.
assert report["classifiedFiles"] == 179
# +1: FormFit's glyph width table (form_fit/_glyph_table.py), read from the font files.
assert report["classifiedFiles"] == 180


def test_gitless_cli_reproduces_literal_dynamic_import_failure_without_mutating_source(
Expand Down
Loading