From 568bfc4beae2ac20b85b973d16991042c5c38896 Mon Sep 17 00:00:00 2001 From: seonghobae <8172694+seonghobae@users.noreply.github.com> Date: Sun, 6 Sep 2026 21:36:38 +0000 Subject: [PATCH 1/5] =?UTF-8?q?=E2=9A=A1=20Bolt:=20=EB=94=95=EC=85=94?= =?UTF-8?q?=EB=84=88=EB=A6=AC=20=EC=88=9C=ED=9A=8C=20=EC=B5=9C=EC=A0=81?= =?UTF-8?q?=ED=99=94=EB=A1=9C=20DOM=20=EC=83=9D=EC=84=B1=20=EC=84=B1?= =?UTF-8?q?=EB=8A=A5=20=EA=B0=9C=EC=84=A0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .jules/bolt.md | 3 +++ src/newsdom_api/dom_builder.py | 9 +++++---- 2 files changed, 8 insertions(+), 4 deletions(-) diff --git a/.jules/bolt.md b/.jules/bolt.md index 1d2f017a..a175baa1 100644 --- a/.jules/bolt.md +++ b/.jules/bolt.md @@ -63,3 +63,6 @@ ## 2024-07-30 - Avoid chained string replace when checking character sets **Learning:** Using chained `.replace(a, "").replace(b, "")` to check if a string consists entirely of specific characters requires intermediate string allocations for every call. In benchmarks, using `.strip("ab")` is ~30% faster and avoids multiple allocations in the hot path. **Action:** When checking if a string is solely composed of specific characters, use `.strip(chars)` instead of chained `.replace()` calls to improve performance. +## 2024-09-06 - 딕셔너리 순회 성능 최적화 +**Learning:** Python에서 성능이 중요한 상황에 딕셔너리의 키와 값을 모두 접근할 때, `sorted(dict.items())`를 사용하여 딕셔너리를 순회하면 내부 구조상 튜플 비교 시 첫 번째 원소(키)에서 이미 유일성이 보장되어 숏서킷(short-circuit)이 발생하므로 성능 상의 불이익 없이 중복된 키 조회(lookup) 연산을 제거할 수 있음을 확인했습니다. +**Action:** 앞으로 딕셔너리를 순회하며 값에 접근해야 하는 핫 패스 루프(hot path loop)에서는 단순히 키를 가져와 다시 조회하는 대신 `.items()`를 적극 활용할 것입니다. diff --git a/src/newsdom_api/dom_builder.py b/src/newsdom_api/dom_builder.py index 74778b2a..150ef05f 100644 --- a/src/newsdom_api/dom_builder.py +++ b/src/newsdom_api/dom_builder.py @@ -386,8 +386,8 @@ def _build_pages_without_page_idx( "Some blocks are missing page_idx; content was assigned to page_idx 0 while preserving model-declared page count." ) pages = [] - for page_idx in sorted(page_info_by_idx): - page_info = page_info_by_idx.get(page_idx, {}) + # ⚡ Bolt: Iterate over items() to avoid redundant dictionary lookups during DOM generation + for page_idx, page_info in sorted(page_info_by_idx.items()): pages.append( _build_page_dom( content_list if page_idx == 0 else [], @@ -425,11 +425,12 @@ def _build_pages_with_page_idx( pages = [] article_seq = count(1) - for page_idx in sorted(blocks_by_page_idx): + # ⚡ Bolt: Iterate over items() to avoid redundant dictionary lookups during DOM generation + for page_idx, blocks in sorted(blocks_by_page_idx.items()): page_info = page_info_by_idx.get(page_idx, {}) pages.append( _build_page_dom( - blocks_by_page_idx[page_idx], + blocks, page_number=_page_number_from_info(page_info, page_idx + 1), article_seq=article_seq, width=page_info.get("width"), From 86795cd0b8dd8322614d88416f8b1fac2816edf0 Mon Sep 17 00:00:00 2001 From: seonghobae <8172694+seonghobae@users.noreply.github.com> Date: Sun, 6 Sep 2026 22:19:52 +0000 Subject: [PATCH 2/5] =?UTF-8?q?=E2=9A=A1=20Bolt:=20=EB=94=95=EC=85=94?= =?UTF-8?q?=EB=84=88=EB=A6=AC=20=EC=88=9C=ED=9A=8C=20=EC=B5=9C=EC=A0=81?= =?UTF-8?q?=ED=99=94=EB=A1=9C=20DOM=20=EC=83=9D=EC=84=B1=20=EC=84=B1?= =?UTF-8?q?=EB=8A=A5=20=EA=B0=9C=EC=84=A0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit From 9e812b150f29de71dce0eafe4dddfcd8fff867e5 Mon Sep 17 00:00:00 2001 From: seonghobae <8172694+seonghobae@users.noreply.github.com> Date: Sun, 6 Sep 2026 23:11:59 +0000 Subject: [PATCH 3/5] =?UTF-8?q?=E2=9A=A1=20Bolt:=20=EB=94=95=EC=85=94?= =?UTF-8?q?=EB=84=88=EB=A6=AC=20=EC=88=9C=ED=9A=8C=20=EC=B5=9C=EC=A0=81?= =?UTF-8?q?=ED=99=94=EB=A1=9C=20DOM=20=EC=83=9D=EC=84=B1=20=EC=84=B1?= =?UTF-8?q?=EB=8A=A5=20=EA=B0=9C=EC=84=A0?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .jules/bolt.md | 3 --- 1 file changed, 3 deletions(-) diff --git a/.jules/bolt.md b/.jules/bolt.md index a175baa1..1d2f017a 100644 --- a/.jules/bolt.md +++ b/.jules/bolt.md @@ -63,6 +63,3 @@ ## 2024-07-30 - Avoid chained string replace when checking character sets **Learning:** Using chained `.replace(a, "").replace(b, "")` to check if a string consists entirely of specific characters requires intermediate string allocations for every call. In benchmarks, using `.strip("ab")` is ~30% faster and avoids multiple allocations in the hot path. **Action:** When checking if a string is solely composed of specific characters, use `.strip(chars)` instead of chained `.replace()` calls to improve performance. -## 2024-09-06 - 딕셔너리 순회 성능 최적화 -**Learning:** Python에서 성능이 중요한 상황에 딕셔너리의 키와 값을 모두 접근할 때, `sorted(dict.items())`를 사용하여 딕셔너리를 순회하면 내부 구조상 튜플 비교 시 첫 번째 원소(키)에서 이미 유일성이 보장되어 숏서킷(short-circuit)이 발생하므로 성능 상의 불이익 없이 중복된 키 조회(lookup) 연산을 제거할 수 있음을 확인했습니다. -**Action:** 앞으로 딕셔너리를 순회하며 값에 접근해야 하는 핫 패스 루프(hot path loop)에서는 단순히 키를 가져와 다시 조회하는 대신 `.items()`를 적극 활용할 것입니다. From d7f1c307385148f75e4f108dab77351de0b91577 Mon Sep 17 00:00:00 2001 From: seonghobae <8172694+seonghobae@users.noreply.github.com> Date: Mon, 7 Sep 2026 00:20:40 +0000 Subject: [PATCH 4/5] =?UTF-8?q?=E2=9A=A1=20Bolt:=20=EB=94=95=EC=85=94?= =?UTF-8?q?=EB=84=88=EB=A6=AC=20=EC=88=9C=ED=9A=8C=20=EC=B5=9C=EC=A0=81?= =?UTF-8?q?=ED=99=94=20=EB=B0=8F=20=ED=9A=8C=EA=B7=80=20=ED=85=8C=EC=8A=A4?= =?UTF-8?q?=ED=8A=B8=20=EC=B6=94=EA=B0=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- src/newsdom_api/dom_builder.py | 4 +-- tests/test_dom_builder.py | 56 ++++++++++++++++++++++++++++++++++ 2 files changed, 58 insertions(+), 2 deletions(-) diff --git a/src/newsdom_api/dom_builder.py b/src/newsdom_api/dom_builder.py index 150ef05f..4ef89992 100644 --- a/src/newsdom_api/dom_builder.py +++ b/src/newsdom_api/dom_builder.py @@ -386,7 +386,7 @@ def _build_pages_without_page_idx( "Some blocks are missing page_idx; content was assigned to page_idx 0 while preserving model-declared page count." ) pages = [] - # ⚡ Bolt: Iterate over items() to avoid redundant dictionary lookups during DOM generation + for page_idx, page_info in sorted(page_info_by_idx.items()): pages.append( _build_page_dom( @@ -425,7 +425,7 @@ def _build_pages_with_page_idx( pages = [] article_seq = count(1) - # ⚡ Bolt: Iterate over items() to avoid redundant dictionary lookups during DOM generation + for page_idx, blocks in sorted(blocks_by_page_idx.items()): page_info = page_info_by_idx.get(page_idx, {}) pages.append( diff --git a/tests/test_dom_builder.py b/tests/test_dom_builder.py index 3499abef..48370430 100644 --- a/tests/test_dom_builder.py +++ b/tests/test_dom_builder.py @@ -552,3 +552,59 @@ def test_bbox_helper_returns_none_for_invalid_y0_x1_y1(): assert _bbox_from_values([0, "bad", 1, 1]) is None assert _bbox_from_values([0, 0, "bad", 1]) is None assert _bbox_from_values([0, 0, 1, "bad"]) is None + + +def test_missing_page_idx_multi_metadata(): + content = [{"type": "text", "text": "hello"}] + model = [{"page_info": {"width": 100}}, {"page_info": {"width": 200}}] + res = build_dom(content, "doc1", model) + assert len(res.pages) == 2 + assert len(res.pages[0].articles) == 1 + assert len(res.pages[1].articles) == 0 + + +def test_sparse_page_idx(): + content = [{"type": "text", "text": "hello", "page_idx": 2}] + model = [ + {"page_info": {"width": 100}}, + {"page_info": {"width": 200}}, + {"page_info": {"width": 300}}, + ] + res = build_dom(content, "doc1", model) + assert len(res.pages) == 1 + assert res.pages[0].page_number == 3 + assert len(res.pages[0].articles) == 1 + + +def test_mixed_missing_page_idx(): + content = [ + {"type": "text", "text": "hello", "page_idx": 1}, + {"type": "text", "text": "world"}, + ] + model = [{"page_info": {"width": 100}}, {"page_info": {"width": 200}}] + res = build_dom(content, "doc1", model) + assert len(res.pages) == 2 + assert len(res.pages[0].articles) == 1 + assert res.pages[0].articles[0].body_blocks[0] == "world" + assert len(res.pages[1].articles) == 1 + assert res.pages[1].articles[0].body_blocks[0] == "hello" + + +def test_metadata_fallback(): + content = [{"type": "text", "text": "hello"}] + res = build_dom(content, "doc1", [{"page_info": {}}]) + assert len(res.pages) == 1 + assert res.pages[0].page_number == 1 + + +def test_deterministic_ordering(): + content = [ + {"type": "text", "text": "page1", "page_idx": 0}, + {"type": "text", "text": "page3", "page_idx": 2}, + {"type": "text", "text": "page2", "page_idx": 1}, + ] + res = build_dom(content, "doc1") + assert len(res.pages) == 3 + assert res.pages[0].articles[0].body_blocks[0] == "page1" + assert res.pages[1].articles[0].body_blocks[0] == "page2" + assert res.pages[2].articles[0].body_blocks[0] == "page3" From 95c0d93e16f28af24eef5fc404ae073c200861ca Mon Sep 17 00:00:00 2001 From: seonghobae <8172694+seonghobae@users.noreply.github.com> Date: Mon, 7 Sep 2026 01:46:31 +0000 Subject: [PATCH 5/5] =?UTF-8?q?=E2=9A=A1=20Bolt:=20=EB=94=95=EC=85=94?= =?UTF-8?q?=EB=84=88=EB=A6=AC=20=EC=88=9C=ED=9A=8C=20=EC=B5=9C=EC=A0=81?= =?UTF-8?q?=ED=99=94=20=EB=B0=8F=20=ED=9A=8C=EA=B7=80=20=ED=85=8C=EC=8A=A4?= =?UTF-8?q?=ED=8A=B8=20=EC=B6=94=EA=B0=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit