From 65bee91bfe6c8a7f51db78b4a302909b087b2b03 Mon Sep 17 00:00:00 2001 From: airmang <38392618+airmang@users.noreply.github.com> Date: Thu, 1 Oct 2026 18:53:20 +0900 Subject: [PATCH] fix(equation): quote rm, it and bold as identifiers, and read bare braces back latex_to_eqedit wrote identifiers spelled rm, it or bold as they were (e^{it} as e ^{it}). Hancom reads those words as font switches, so alone in a script they draw nothing; they are now quoted like the other reserved words (e ^{"it"}), and eqedit_to_latex reads them back as letters. \lbrace and \rbrace convert like \{ and \} (also after \left and \right), and eqedit_to_latex reads a bare LBRACE or RBRACE outside LEFT/RIGHT as \{ or \}. The equation size notes say that Hancom on Windows keeps the stored box and baseLine when it saves and lays the page out with them. Co-Authored-By: Claude Opus 5.5 --- CHANGELOG.md | 8 ++++++++ src/hwpx/equation/authoring.py | 9 ++++++--- src/hwpx/equation/eqedit.py | 9 +++++++-- src/hwpx/equation/measure.py | 3 ++- tests/test_equation_authoring.py | 5 +++++ tests/test_equation_converter.py | 13 +++++++++++++ 6 files changed, 41 insertions(+), 6 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d48a600a..9777cf51 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,14 @@ ### 고침 +- `hwpx.experimental.latex_to_eqedit`(실험)가 `e^{it}`, `x_{rm}`처럼 글자가 `rm`·`it`·`bold`가 되는 식별자를 그대로 + 쓰던 것을 고친다. 한/글은 이 낱말을 글꼴 전환으로 읽어서, 첨자에 홀로 오면 아무것도 그리지 않는다. 이제 다른 + 예약어처럼 따옴표로 감싸 쓰고(`e ^{"it"}`), `eqedit_to_latex`는 그것을 다시 글자로 읽는다. +- `latex_to_eqedit`가 `\lbrace`·`\rbrace`를 `\{`·`\}`처럼 `LBRACE`·`RBRACE`로 옮긴다(`\left`·`\right` 뒤에서도). 전에는 + 지원하지 않는 명령이었다. `eqedit_to_latex`는 `LEFT`·`RIGHT` 밖의 맨 `LBRACE`·`RBRACE`를 `\{`·`\}`로 읽는다. 전에는 + 낱말 그대로 남았다. +- 수식 상자 설명에, Windows의 한/글은 저장할 때 상자(`hp:sz`)와 `baseLine`을 그대로 두고 그 상자로 쪽을 배치한다는 + 점을 더한다. - 쪽 수 추정(실험, `estimate_pages`)이 곁에 줄 하나 들어갈 자리를 남기는 종이·쪽 기준 어울림(SQUARE) 개체를 받는다(한 단일 때). - 그 쪽의 줄·표·개체가 개체의 띠에 닿지 않으면 개체가 없을 때와 같다. 띠 위 끝에 바닥이 맞닿은 줄도 그렇다. diff --git a/src/hwpx/equation/authoring.py b/src/hwpx/equation/authoring.py index a72a7ebb..7e528eed 100644 --- a/src/hwpx/equation/authoring.py +++ b/src/hwpx/equation/authoring.py @@ -270,6 +270,8 @@ def _atom(self, depth: int) -> str: radicand = self._group_or_atom(depth) return f"root {{{' '.join(index_parts)}}} of {{{radicand}}}" return f"sqrt {{{self._group_or_atom(depth)}}}" + if token in ("\\lbrace", "\\rbrace"): # the command spellings of \{ and \} + return "LBRACE" if token == "\\lbrace" else "RBRACE" if token in _TEXT_COMMANDS: return self._text_literal() if token == "\\mathbf": @@ -407,8 +409,8 @@ def _delimiter(self, token: str | None) -> str: raise UnsupportedLatexError("missing \\left/\\right delimiter") if token in _LATEX_DELIMITERS: return _LATEX_DELIMITERS[token] - if token in ("\\{", "\\}"): - return "LBRACE" if token == "\\{" else "RBRACE" + if token in ("\\{", "\\}", "\\lbrace", "\\rbrace"): + return "LBRACE" if token in ("\\{", "\\lbrace") else "RBRACE" if token in DELIMITERS and len(token) == 1: return token raise UnsupportedLatexError(f"unsupported \\left/\\right delimiter: {token}") @@ -450,7 +452,8 @@ def estimate_equation_size(script: str, *, base_unit: int = 1100) -> tuple[int, A reader that lays the page out from the file uses the stored box, so the box has to fit the script; Hancom itself was observed (on macOS) to lay the - equation out again and rewrite the box when it saves the document (see + equation out again and rewrite the box when it saves the document, while on + Windows it keeps the stored box and lays the page out with it (see :func:`hwpx.equation.measure.measure_equation`). """ diff --git a/src/hwpx/equation/eqedit.py b/src/hwpx/equation/eqedit.py index 1479087e..01ac017c 100644 --- a/src/hwpx/equation/eqedit.py +++ b/src/hwpx/equation/eqedit.py @@ -42,8 +42,10 @@ MAX_SOURCE_LENGTH = 10_000 MAX_GROUP_DEPTH = 64 -# Words EqEdit reads as symbols or structure. The writer quotes a bare identifier that collides with one -# (``T_{int}`` → ``T _{"int"}``); the reader turns such a quoted word back into plain letters. +# Words EqEdit reads as symbols, structure or font switches. The writer quotes a bare identifier that collides +# with one (``T_{int}`` → ``T _{"int"}``, ``e^{it}`` → ``e ^{"it"}``: a lone ``it`` in a superscript switches to +# italic and draws nothing); the reader turns such a quoted word back into plain letters. +_FONT_SWITCHES = frozenset({"rm", "it", "bold", "RM", "IT", "BOLD"}) _RESERVED_WORDS = ( frozenset(GREEK) | frozenset(OPERATORS) @@ -52,6 +54,7 @@ | frozenset(ACCENTS) | frozenset(MATRIX_ENVIRONMENTS) | STRUCTURAL + | _FONT_SWITCHES ) # Characters that always terminate a token even when not whitespace separated. @@ -338,6 +341,8 @@ def _map_token(self, token: str) -> str: return OPERATORS[token] if token in FUNCTIONS: return FUNCTIONS[token] + if token in ("LBRACE", "RBRACE"): # a literal brace outside LEFT/RIGHT + return r"\{" if token == "LBRACE" else r"\}" if token.startswith('"') and token.endswith('"') and len(token) >= 2: literal = token[1:-1] if literal.isalpha() and literal in _RESERVED_WORDS: diff --git a/src/hwpx/equation/measure.py b/src/hwpx/equation/measure.py index 4ce5a38e..81514fbb 100644 --- a/src/hwpx/equation/measure.py +++ b/src/hwpx/equation/measure.py @@ -8,7 +8,8 @@ lets the next characters overlap it. Hancom itself was observed (on macOS) to lay the equation out again and rewrite ```` when it saves the document, whatever size was stored; the stored box still matters until then -and for every other reader. +and for every other reader. On Windows Hancom keeps the stored box and +``baseLine`` when it saves, and lays the page out with them. This module lays the script out as boxes (width, ascent, descent) by its structure -- characters by kind, ``over`` fractions (and ``atop``, ``choose``/ diff --git a/tests/test_equation_authoring.py b/tests/test_equation_authoring.py index 38e0ee6d..0d428e0a 100644 --- a/tests/test_equation_authoring.py +++ b/tests/test_equation_authoring.py @@ -218,6 +218,11 @@ class TestLatexToEqedit: (r"\bar{x} + \vec{v}", "bar {x} + vec {v}"), (r"\text{판별식} = 0", '{rm "판별식" it} = 0'), (r"T_{int}", 'T _{"int"}'), + # rm, it and bold switch the font: alone in a script they would draw nothing + (r"e^{it}", 'e ^{"it"}'), + (r"x_{rm} + a_{bold}", 'x _{"rm"} + a _{"bold"}'), + (r"\lbrace x \rbrace", "LBRACE x RBRACE"), + (r"\left\lbrace x \right\rbrace", "LEFT LBRACE x RIGHT RBRACE"), (r"$$\frac{1}{2}$$", "{1} over {2}"), (r"$x + 1$", "x + 1"), (r"\le \ge \ne", "leq geq neq"), diff --git a/tests/test_equation_converter.py b/tests/test_equation_converter.py index c09823b2..b1578bf5 100644 --- a/tests/test_equation_converter.py +++ b/tests/test_equation_converter.py @@ -140,6 +140,8 @@ def test_thin_space_two_headed_arrow_and_conclusion_signs(script: str, expected: ('{rm "d"} x', r"{rm \text{d}} x"), # not the authored shape: as before # a quoted reserved word is the writer's identifier protection: plain letters again ('T _{"int"}', r"T_{int}"), + ('e ^{"it"}', r"e^{it}"), # so with the font switches + ('x _{"rm"} + a _{"bold"}', r"x_{rm} + a_{bold}"), ('"therefore" x', r"therefore x"), ('"sin x"', r"\text{sin x}"), # not a single word: text, as before ], @@ -148,6 +150,17 @@ def test_upright_text_group_reads_as_text(script: str, expected: str) -> None: assert eqedit_to_latex(script) == expected +@pytest.mark.parametrize( + "script,expected", + [ + ("LBRACE x RBRACE", r"\{ x \}"), # literal braces, as latex_to_eqedit writes \{ and \} + ("LEFT LBRACE x RIGHT RBRACE", r"\left\{ x \right\}"), # as before + ], +) +def test_brace_words_read_as_braces(script: str, expected: str) -> None: + assert eqedit_to_latex(script) == expected + + @pytest.mark.parametrize( "script,expected", [