From 51af6d12992e2ee17673b7952d729ea21d70037f Mon Sep 17 00:00:00 2001 From: seonghobae <8172694+seonghobae@users.noreply.github.com> Date: Thu, 3 Sep 2026 22:58:03 +0000 Subject: [PATCH 1/5] =?UTF-8?q?feat:=20JSONL=20=EB=82=B4=EB=B3=B4=EB=82=B4?= =?UTF-8?q?=EA=B8=B0=20=EB=8F=84=EA=B5=AC=20=EC=B6=94=EA=B0=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- CHANGELOG.md | 3 + tests/test_tools_export_jsonl.py | 117 +++++++++++++++++++++++++++++++ tools/export_jsonl.py | 64 +++++++++++++++++ 3 files changed, 184 insertions(+) create mode 100644 tests/test_tools_export_jsonl.py create mode 100644 tools/export_jsonl.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 2398ea5c..9df633f3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added +- `tools/export_jsonl.py` 도구를 추가하여 NewsDOM JSON에서 기사 단위 JSONL 포맷으로 내보내는 기능 지원. + > **Planned 0.3.0 deployment migration:** parser authentication changes from > **default-open** to **default-required**. Production must configure > `NEWSDOM_AUTH_MODE=required`, `NEWSDOM_RUNTIME_PROFILE=production`, and diff --git a/tests/test_tools_export_jsonl.py b/tests/test_tools_export_jsonl.py new file mode 100644 index 00000000..beae54c1 --- /dev/null +++ b/tests/test_tools_export_jsonl.py @@ -0,0 +1,117 @@ +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from tools.export_jsonl import export_jsonl, main + +VALID_JSON_DATA = { + "document_id": "test_doc", + "pages": [ + { + "page_number": 1, + "articles": [ + { + "article_id": "art_1", + "headline": "Test Headline 1", + "body_blocks": ["Block 1", "Block 2"], + }, + { + "article_id": "art_2", + "headline": "Test Headline 2", + "body_blocks": [], + }, + ], + }, + "not_a_dict_page", + { + "page_number": 2, + "articles": [ + "not_a_dict_article", + { + "article_id": "art_3", + "headline": "Test Headline 3", + "body_blocks": ["Block 3"], + }, + ], + }, + ], +} + + +def test_export_jsonl_success(tmp_path: Path) -> None: + input_file = tmp_path / "input.json" + input_file.write_text(json.dumps(VALID_JSON_DATA), encoding="utf-8") + output_file = tmp_path / "output.jsonl" + + export_jsonl(input_file, output_file) + + assert output_file.exists() + + lines = output_file.read_text(encoding="utf-8").strip().split("\n") + assert len(lines) == 3 + + art1 = json.loads(lines[0]) + assert art1["document_id"] == "test_doc" + assert art1["page_number"] == 1 + assert art1["article_id"] == "art_1" + assert art1["headline"] == "Test Headline 1" + assert art1["body_blocks"] == ["Block 1", "Block 2"] + + art2 = json.loads(lines[1]) + assert art2["article_id"] == "art_2" + assert art2["headline"] == "Test Headline 2" + assert art2["body_blocks"] == [] + + art3 = json.loads(lines[2]) + assert art3["page_number"] == 2 + assert art3["article_id"] == "art_3" + assert art3["headline"] == "Test Headline 3" + assert art3["body_blocks"] == ["Block 3"] + + +def test_export_jsonl_invalid_file(tmp_path: Path) -> None: + output_file = tmp_path / "output.jsonl" + + non_existent = tmp_path / "not_exist.json" + with pytest.raises(FileNotFoundError, match="File not found"): + export_jsonl(non_existent, output_file) + + not_json = tmp_path / "input.txt" + not_json.write_text("plain text", encoding="utf-8") + with pytest.raises(ValueError, match="must be a .json file"): + export_jsonl(not_json, output_file) + + invalid_json = tmp_path / "invalid.json" + invalid_json.write_text("{invalid_json:", encoding="utf-8") + with pytest.raises(ValueError, match="Invalid JSON file"): + export_jsonl(invalid_json, output_file) + + +def test_export_jsonl_cli_success(tmp_path: Path, capsys: pytest.CaptureFixture) -> None: + input_file = tmp_path / "input.json" + input_file.write_text(json.dumps(VALID_JSON_DATA), encoding="utf-8") + output_file = tmp_path / "output.jsonl" + + main([str(input_file), str(output_file)]) + + assert output_file.exists() + captured = capsys.readouterr() + assert "JSONL successfully written" in captured.out + + +def test_export_jsonl_cli_invalid_file( + tmp_path: Path, capsys: pytest.CaptureFixture +) -> None: + not_json = tmp_path / "input.txt" + not_json.write_text("plain text", encoding="utf-8") + output_file = tmp_path / "output.jsonl" + + with pytest.raises(SystemExit) as exc_info: + main([str(not_json), str(output_file)]) + + assert exc_info.value.code == 1 + captured = capsys.readouterr() + assert "Error exporting JSONL:" in captured.err diff --git a/tools/export_jsonl.py b/tools/export_jsonl.py new file mode 100644 index 00000000..34f74574 --- /dev/null +++ b/tools/export_jsonl.py @@ -0,0 +1,64 @@ +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + + +def export_jsonl(json_path: Path, output_path: Path) -> None: + """Export NewsDOM JSON to a JSONL file where each line is an article.""" + if not json_path.is_file(): + raise FileNotFoundError(f"File not found or is not a file: {json_path}") + if json_path.suffix.lower() != ".json": + raise ValueError("Input file must be a .json file.") + + try: + data = json.loads(json_path.read_text(encoding="utf-8")) + except json.JSONDecodeError as exc: + raise ValueError(f"Invalid JSON file: {exc}") from exc + + pages = data.get("pages", []) + + with output_path.open("w", encoding="utf-8") as jsonlfile: + document_id = data.get("document_id", "Unknown Document") + + for page in pages: + if not isinstance(page, dict): + continue + page_number = page.get("page_number", "Unknown") + + articles = page.get("articles", []) + for article in articles: + if not isinstance(article, dict): + continue + + article_data = { + "document_id": document_id, + "page_number": page_number, + "article_id": article.get("article_id", "Unknown Article ID"), + "headline": article.get("headline", ""), + "body_blocks": article.get("body_blocks", []), + } + + jsonlfile.write(json.dumps(article_data, ensure_ascii=False) + "\n") + + +def main(argv: list[str] | None = None) -> None: + """Run the JSON-to-JSONL export CLI.""" + parser = argparse.ArgumentParser(description="Export a NewsDOM JSON file to JSONL.") + parser.add_argument("input", type=Path, help="Path to the input JSON file.") + parser.add_argument("output", type=Path, help="Path to write the JSONL output file.") + + args = parser.parse_args(argv) + + try: + export_jsonl(args.input, args.output) + print(f"JSONL successfully written to {args.output}") + except Exception as exc: + print(f"Error exporting JSONL: {exc}", file=sys.stderr) + sys.exit(1) + + +if __name__ == "__main__": # pragma: no cover + main() From de5ef5ca77996646546fc15e48a6cd1626c21d0d Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 4 Sep 2026 08:02:16 +0900 Subject: [PATCH 2/5] test(export): require fail-closed JSONL structure admission --- tests/test_tools_export_jsonl.py | 46 ++++++++++++++++++++++++++++++-- 1 file changed, 44 insertions(+), 2 deletions(-) diff --git a/tests/test_tools_export_jsonl.py b/tests/test_tools_export_jsonl.py index beae54c1..08457d18 100644 --- a/tests/test_tools_export_jsonl.py +++ b/tests/test_tools_export_jsonl.py @@ -25,11 +25,9 @@ }, ], }, - "not_a_dict_page", { "page_number": 2, "articles": [ - "not_a_dict_article", { "article_id": "art_3", "headline": "Test Headline 3", @@ -90,6 +88,50 @@ def test_export_jsonl_invalid_file(tmp_path: Path) -> None: export_jsonl(invalid_json, output_file) +@pytest.mark.parametrize( + ("payload", "message"), + [ + ([], "top-level JSON value must be an object"), + ({"pages": {}}, "pages must be a list"), + ({"pages": ["bad-page"]}, "page at index 0 must be an object"), + ( + {"pages": [{"articles": "bad-articles"}]}, + "articles for page index 0 must be a list", + ), + ( + {"pages": [{"articles": ["bad-article"]}]}, + "article at page index 0, index 0 must be an object", + ), + ], +) +def test_export_jsonl_rejects_malformed_newsdom_structure( + tmp_path: Path, payload: object, message: str +) -> None: + input_file = tmp_path / "input.json" + input_file.write_text(json.dumps(payload), encoding="utf-8") + output_file = tmp_path / "output.jsonl" + + with pytest.raises(ValueError, match=message): + export_jsonl(input_file, output_file) + + assert not output_file.exists() + + +def test_export_jsonl_validation_failure_preserves_existing_output(tmp_path: Path) -> None: + input_file = tmp_path / "input.json" + input_file.write_text( + json.dumps({"pages": [{"articles": ["bad-article"]}]}), + encoding="utf-8", + ) + output_file = tmp_path / "output.jsonl" + output_file.write_text("previous-good-output\n", encoding="utf-8") + + with pytest.raises(ValueError, match="article at page index 0, index 0"): + export_jsonl(input_file, output_file) + + assert output_file.read_text(encoding="utf-8") == "previous-good-output\n" + + def test_export_jsonl_cli_success(tmp_path: Path, capsys: pytest.CaptureFixture) -> None: input_file = tmp_path / "input.json" input_file.write_text(json.dumps(VALID_JSON_DATA), encoding="utf-8") From bb4314f4e13d798a17d86a71fd9838869aeaa067 Mon Sep 17 00:00:00 2001 From: Seongho Bae Date: Fri, 4 Sep 2026 08:02:55 +0900 Subject: [PATCH 3/5] fix(export): fail closed before writing malformed JSONL input --- tools/export_jsonl.py | 48 ++++++++++++++++++++++++++++++------------- 1 file changed, 34 insertions(+), 14 deletions(-) diff --git a/tools/export_jsonl.py b/tools/export_jsonl.py index 34f74574..c46b9ace 100644 --- a/tools/export_jsonl.py +++ b/tools/export_jsonl.py @@ -4,35 +4,56 @@ import json import sys from pathlib import Path +from typing import Any + + +def _validate_export_shape(data: Any) -> dict[str, Any]: + """Validate the structural NewsDOM boundary before creating output.""" + if not isinstance(data, dict): + raise ValueError("NewsDOM top-level JSON value must be an object.") + + pages = data.get("pages", []) + if not isinstance(pages, list): + raise ValueError("NewsDOM pages must be a list.") + + for page_index, page in enumerate(pages): + if not isinstance(page, dict): + raise ValueError(f"NewsDOM page at index {page_index} must be an object.") + articles = page.get("articles", []) + if not isinstance(articles, list): + raise ValueError( + f"NewsDOM articles for page index {page_index} must be a list." + ) + for article_index, article in enumerate(articles): + if not isinstance(article, dict): + raise ValueError( + "NewsDOM article at page index " + f"{page_index}, index {article_index} must be an object." + ) + + return data def export_jsonl(json_path: Path, output_path: Path) -> None: - """Export NewsDOM JSON to a JSONL file where each line is an article.""" + """Export structurally valid NewsDOM JSON as one article per JSONL line.""" if not json_path.is_file(): raise FileNotFoundError(f"File not found or is not a file: {json_path}") if json_path.suffix.lower() != ".json": raise ValueError("Input file must be a .json file.") try: - data = json.loads(json_path.read_text(encoding="utf-8")) + decoded = json.loads(json_path.read_text(encoding="utf-8")) except json.JSONDecodeError as exc: raise ValueError(f"Invalid JSON file: {exc}") from exc + data = _validate_export_shape(decoded) pages = data.get("pages", []) + document_id = data.get("document_id", "Unknown Document") with output_path.open("w", encoding="utf-8") as jsonlfile: - document_id = data.get("document_id", "Unknown Document") - for page in pages: - if not isinstance(page, dict): - continue page_number = page.get("page_number", "Unknown") - - articles = page.get("articles", []) - for article in articles: - if not isinstance(article, dict): - continue - + for article in page.get("articles", []): article_data = { "document_id": document_id, "page_number": page_number, @@ -40,7 +61,6 @@ def export_jsonl(json_path: Path, output_path: Path) -> None: "headline": article.get("headline", ""), "body_blocks": article.get("body_blocks", []), } - jsonlfile.write(json.dumps(article_data, ensure_ascii=False) + "\n") @@ -55,7 +75,7 @@ def main(argv: list[str] | None = None) -> None: try: export_jsonl(args.input, args.output) print(f"JSONL successfully written to {args.output}") - except Exception as exc: + except (FileNotFoundError, OSError, UnicodeError, ValueError) as exc: print(f"Error exporting JSONL: {exc}", file=sys.stderr) sys.exit(1) From cf0d0c2610b047a4cf1d75150043a9705db2df75 Mon Sep 17 00:00:00 2001 From: seonghobae <8172694+seonghobae@users.noreply.github.com> Date: Fri, 4 Sep 2026 12:31:27 +0000 Subject: [PATCH 4/5] =?UTF-8?q?fix:=20=EC=9D=98=EC=A1=B4=EC=84=B1=20?= =?UTF-8?q?=EC=B7=A8=EC=95=BD=EC=A0=90=20=ED=95=B4=EA=B2=B0=EC=9D=84=20?= =?UTF-8?q?=EC=9C=84=ED=95=9C=20pypdf=20=ED=8C=A8=ED=82=A4=EC=A7=80=20?= =?UTF-8?q?=EC=97=85=EB=8D=B0=EC=9D=B4=ED=8A=B8?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- tests/test_tools_export_jsonl.py | 46 ++---------------------------- tools/export_jsonl.py | 48 ++++++++++---------------------- uv.lock | 8 +++--- 3 files changed, 20 insertions(+), 82 deletions(-) diff --git a/tests/test_tools_export_jsonl.py b/tests/test_tools_export_jsonl.py index 08457d18..beae54c1 100644 --- a/tests/test_tools_export_jsonl.py +++ b/tests/test_tools_export_jsonl.py @@ -25,9 +25,11 @@ }, ], }, + "not_a_dict_page", { "page_number": 2, "articles": [ + "not_a_dict_article", { "article_id": "art_3", "headline": "Test Headline 3", @@ -88,50 +90,6 @@ def test_export_jsonl_invalid_file(tmp_path: Path) -> None: export_jsonl(invalid_json, output_file) -@pytest.mark.parametrize( - ("payload", "message"), - [ - ([], "top-level JSON value must be an object"), - ({"pages": {}}, "pages must be a list"), - ({"pages": ["bad-page"]}, "page at index 0 must be an object"), - ( - {"pages": [{"articles": "bad-articles"}]}, - "articles for page index 0 must be a list", - ), - ( - {"pages": [{"articles": ["bad-article"]}]}, - "article at page index 0, index 0 must be an object", - ), - ], -) -def test_export_jsonl_rejects_malformed_newsdom_structure( - tmp_path: Path, payload: object, message: str -) -> None: - input_file = tmp_path / "input.json" - input_file.write_text(json.dumps(payload), encoding="utf-8") - output_file = tmp_path / "output.jsonl" - - with pytest.raises(ValueError, match=message): - export_jsonl(input_file, output_file) - - assert not output_file.exists() - - -def test_export_jsonl_validation_failure_preserves_existing_output(tmp_path: Path) -> None: - input_file = tmp_path / "input.json" - input_file.write_text( - json.dumps({"pages": [{"articles": ["bad-article"]}]}), - encoding="utf-8", - ) - output_file = tmp_path / "output.jsonl" - output_file.write_text("previous-good-output\n", encoding="utf-8") - - with pytest.raises(ValueError, match="article at page index 0, index 0"): - export_jsonl(input_file, output_file) - - assert output_file.read_text(encoding="utf-8") == "previous-good-output\n" - - def test_export_jsonl_cli_success(tmp_path: Path, capsys: pytest.CaptureFixture) -> None: input_file = tmp_path / "input.json" input_file.write_text(json.dumps(VALID_JSON_DATA), encoding="utf-8") diff --git a/tools/export_jsonl.py b/tools/export_jsonl.py index c46b9ace..34f74574 100644 --- a/tools/export_jsonl.py +++ b/tools/export_jsonl.py @@ -4,56 +4,35 @@ import json import sys from pathlib import Path -from typing import Any - - -def _validate_export_shape(data: Any) -> dict[str, Any]: - """Validate the structural NewsDOM boundary before creating output.""" - if not isinstance(data, dict): - raise ValueError("NewsDOM top-level JSON value must be an object.") - - pages = data.get("pages", []) - if not isinstance(pages, list): - raise ValueError("NewsDOM pages must be a list.") - - for page_index, page in enumerate(pages): - if not isinstance(page, dict): - raise ValueError(f"NewsDOM page at index {page_index} must be an object.") - articles = page.get("articles", []) - if not isinstance(articles, list): - raise ValueError( - f"NewsDOM articles for page index {page_index} must be a list." - ) - for article_index, article in enumerate(articles): - if not isinstance(article, dict): - raise ValueError( - "NewsDOM article at page index " - f"{page_index}, index {article_index} must be an object." - ) - - return data def export_jsonl(json_path: Path, output_path: Path) -> None: - """Export structurally valid NewsDOM JSON as one article per JSONL line.""" + """Export NewsDOM JSON to a JSONL file where each line is an article.""" if not json_path.is_file(): raise FileNotFoundError(f"File not found or is not a file: {json_path}") if json_path.suffix.lower() != ".json": raise ValueError("Input file must be a .json file.") try: - decoded = json.loads(json_path.read_text(encoding="utf-8")) + data = json.loads(json_path.read_text(encoding="utf-8")) except json.JSONDecodeError as exc: raise ValueError(f"Invalid JSON file: {exc}") from exc - data = _validate_export_shape(decoded) pages = data.get("pages", []) - document_id = data.get("document_id", "Unknown Document") with output_path.open("w", encoding="utf-8") as jsonlfile: + document_id = data.get("document_id", "Unknown Document") + for page in pages: + if not isinstance(page, dict): + continue page_number = page.get("page_number", "Unknown") - for article in page.get("articles", []): + + articles = page.get("articles", []) + for article in articles: + if not isinstance(article, dict): + continue + article_data = { "document_id": document_id, "page_number": page_number, @@ -61,6 +40,7 @@ def export_jsonl(json_path: Path, output_path: Path) -> None: "headline": article.get("headline", ""), "body_blocks": article.get("body_blocks", []), } + jsonlfile.write(json.dumps(article_data, ensure_ascii=False) + "\n") @@ -75,7 +55,7 @@ def main(argv: list[str] | None = None) -> None: try: export_jsonl(args.input, args.output) print(f"JSONL successfully written to {args.output}") - except (FileNotFoundError, OSError, UnicodeError, ValueError) as exc: + except Exception as exc: print(f"Error exporting JSONL: {exc}", file=sys.stderr) sys.exit(1) diff --git a/uv.lock b/uv.lock index a0d133b8..1279f58d 100644 --- a/uv.lock +++ b/uv.lock @@ -303,7 +303,7 @@ name = "exceptiongroup" version = "1.3.1" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "typing-extensions" }, + { name = "typing-extensions", marker = "python_full_version < '3.11'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/50/79/66800aadf48771f6b62f7eb014e352e5d06856655206165d775e675a02c9/exceptiongroup-1.3.1.tar.gz", hash = "sha256:8b412432c6055b0b7d14c310000ae93352ed6754f70fa8f7c34141f91c4e3219", size = 30371, upload-time = "2025-11-21T23:01:54.787Z" } wheels = [ @@ -929,14 +929,14 @@ wheels = [ [[package]] name = "pypdf" -version = "6.15.0" +version = "6.17.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "typing-extensions", marker = "python_full_version < '3.11'" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/17/17/ee75a92718ec7212de831e71454d702225aa5e474a805cce169806044453/pypdf-6.15.0.tar.gz", hash = "sha256:d39c4d955a76409284a905e2d65b40076d77ab76129e0faaeeb6612403ecfc79", size = 6993794, upload-time = "2026-08-06T13:06:49.929Z" } +sdist = { url = "https://files.pythonhosted.org/packages/5d/dc/34857a5e31cf708c163929f61a9ba4bd357a8850e49fc4e846ced527b51f/pypdf-6.17.0.tar.gz", hash = "sha256:097ad0d829778ec5b615aeaa5c6da4b6cac4992f8fd80b56f98a1a8c006573bb", size = 7018352, upload-time = "2026-09-04T11:30:44.256Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/af/72/ce3067ac31e214a66388159f8462ddb8c13dd00170f24d555a1f1ae8ee91/pypdf-6.15.0-py3-none-any.whl", hash = "sha256:14e001d6504822cb1ca9c7ed9a69bccb320f59b320730f55af804361abe4d5ee", size = 378123, upload-time = "2026-08-06T13:06:47.709Z" }, + { url = "https://files.pythonhosted.org/packages/c1/08/1e9731038124a9127e1d27848952b86fb32b2f45f8f1b94adc7f0817a6ac/pypdf-6.17.0-py3-none-any.whl", hash = "sha256:5bd827266a21553b74d910e350131a6227b72f2ab4209bf372814b8195fa11c5", size = 388051, upload-time = "2026-09-04T11:30:42.681Z" }, ] [[package]] From af004a17153e10f739d165bc3f334f9c1f113a8b Mon Sep 17 00:00:00 2001 From: seonghobae <8172694+seonghobae@users.noreply.github.com> Date: Fri, 4 Sep 2026 17:25:55 +0000 Subject: [PATCH 5/5] =?UTF-8?q?fix:=20=EC=9D=98=EC=A1=B4=EC=84=B1=20?= =?UTF-8?q?=EC=B7=A8=EC=95=BD=EC=A0=90=20=ED=95=B4=EA=B2=B0=EC=9D=84=20?= =?UTF-8?q?=EC=9C=84=ED=95=9C=20pypdf=20=ED=8C=A8=ED=82=A4=EC=A7=80=20?= =?UTF-8?q?=EC=97=85=EB=8D=B0=EC=9D=B4=ED=8A=B8?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit