diff --git a/CHANGELOG.md b/CHANGELOG.md index 2398ea5c..9df633f3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,9 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added +- `tools/export_jsonl.py` 도구를 추가하여 NewsDOM JSON에서 기사 단위 JSONL 포맷으로 내보내는 기능 지원. + > **Planned 0.3.0 deployment migration:** parser authentication changes from > **default-open** to **default-required**. Production must configure > `NEWSDOM_AUTH_MODE=required`, `NEWSDOM_RUNTIME_PROFILE=production`, and diff --git a/tests/test_tools_export_jsonl.py b/tests/test_tools_export_jsonl.py new file mode 100644 index 00000000..beae54c1 --- /dev/null +++ b/tests/test_tools_export_jsonl.py @@ -0,0 +1,117 @@ +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +from tools.export_jsonl import export_jsonl, main + +VALID_JSON_DATA = { + "document_id": "test_doc", + "pages": [ + { + "page_number": 1, + "articles": [ + { + "article_id": "art_1", + "headline": "Test Headline 1", + "body_blocks": ["Block 1", "Block 2"], + }, + { + "article_id": "art_2", + "headline": "Test Headline 2", + "body_blocks": [], + }, + ], + }, + "not_a_dict_page", + { + "page_number": 2, + "articles": [ + "not_a_dict_article", + { + "article_id": "art_3", + "headline": "Test Headline 3", + "body_blocks": ["Block 3"], + }, + ], + }, + ], +} + + +def test_export_jsonl_success(tmp_path: Path) -> None: + input_file = tmp_path / "input.json" + input_file.write_text(json.dumps(VALID_JSON_DATA), encoding="utf-8") + output_file = tmp_path / "output.jsonl" + + export_jsonl(input_file, output_file) + + assert output_file.exists() + + lines = output_file.read_text(encoding="utf-8").strip().split("\n") + assert len(lines) == 3 + + art1 = json.loads(lines[0]) + assert art1["document_id"] == "test_doc" + assert art1["page_number"] == 1 + assert art1["article_id"] == "art_1" + assert art1["headline"] == "Test Headline 1" + assert art1["body_blocks"] == ["Block 1", "Block 2"] + + art2 = json.loads(lines[1]) + assert art2["article_id"] == "art_2" + assert art2["headline"] == "Test Headline 2" + assert art2["body_blocks"] == [] + + art3 = json.loads(lines[2]) + assert art3["page_number"] == 2 + assert art3["article_id"] == "art_3" + assert art3["headline"] == "Test Headline 3" + assert art3["body_blocks"] == ["Block 3"] + + +def test_export_jsonl_invalid_file(tmp_path: Path) -> None: + output_file = tmp_path / "output.jsonl" + + non_existent = tmp_path / "not_exist.json" + with pytest.raises(FileNotFoundError, match="File not found"): + export_jsonl(non_existent, output_file) + + not_json = tmp_path / "input.txt" + not_json.write_text("plain text", encoding="utf-8") + with pytest.raises(ValueError, match="must be a .json file"): + export_jsonl(not_json, output_file) + + invalid_json = tmp_path / "invalid.json" + invalid_json.write_text("{invalid_json:", encoding="utf-8") + with pytest.raises(ValueError, match="Invalid JSON file"): + export_jsonl(invalid_json, output_file) + + +def test_export_jsonl_cli_success(tmp_path: Path, capsys: pytest.CaptureFixture) -> None: + input_file = tmp_path / "input.json" + input_file.write_text(json.dumps(VALID_JSON_DATA), encoding="utf-8") + output_file = tmp_path / "output.jsonl" + + main([str(input_file), str(output_file)]) + + assert output_file.exists() + captured = capsys.readouterr() + assert "JSONL successfully written" in captured.out + + +def test_export_jsonl_cli_invalid_file( + tmp_path: Path, capsys: pytest.CaptureFixture +) -> None: + not_json = tmp_path / "input.txt" + not_json.write_text("plain text", encoding="utf-8") + output_file = tmp_path / "output.jsonl" + + with pytest.raises(SystemExit) as exc_info: + main([str(not_json), str(output_file)]) + + assert exc_info.value.code == 1 + captured = capsys.readouterr() + assert "Error exporting JSONL:" in captured.err diff --git a/tools/export_jsonl.py b/tools/export_jsonl.py new file mode 100644 index 00000000..34f74574 --- /dev/null +++ b/tools/export_jsonl.py @@ -0,0 +1,64 @@ +from __future__ import annotations + +import argparse +import json +import sys +from pathlib import Path + + +def export_jsonl(json_path: Path, output_path: Path) -> None: + """Export NewsDOM JSON to a JSONL file where each line is an article.""" + if not json_path.is_file(): + raise FileNotFoundError(f"File not found or is not a file: {json_path}") + if json_path.suffix.lower() != ".json": + raise ValueError("Input file must be a .json file.") + + try: + data = json.loads(json_path.read_text(encoding="utf-8")) + except json.JSONDecodeError as exc: + raise ValueError(f"Invalid JSON file: {exc}") from exc + + pages = data.get("pages", []) + + with output_path.open("w", encoding="utf-8") as jsonlfile: + document_id = data.get("document_id", "Unknown Document") + + for page in pages: + if not isinstance(page, dict): + continue + page_number = page.get("page_number", "Unknown") + + articles = page.get("articles", []) + for article in articles: + if not isinstance(article, dict): + continue + + article_data = { + "document_id": document_id, + "page_number": page_number, + "article_id": article.get("article_id", "Unknown Article ID"), + "headline": article.get("headline", ""), + "body_blocks": article.get("body_blocks", []), + } + + jsonlfile.write(json.dumps(article_data, ensure_ascii=False) + "\n") + + +def main(argv: list[str] | None = None) -> None: + """Run the JSON-to-JSONL export CLI.""" + parser = argparse.ArgumentParser(description="Export a NewsDOM JSON file to JSONL.") + parser.add_argument("input", type=Path, help="Path to the input JSON file.") + parser.add_argument("output", type=Path, help="Path to write the JSONL output file.") + + args = parser.parse_args(argv) + + try: + export_jsonl(args.input, args.output) + print(f"JSONL successfully written to {args.output}") + except Exception as exc: + print(f"Error exporting JSONL: {exc}", file=sys.stderr) + sys.exit(1) + + +if __name__ == "__main__": # pragma: no cover + main() diff --git a/uv.lock b/uv.lock index a0d133b8..1279f58d 100644 --- a/uv.lock +++ b/uv.lock @@ -303,7 +303,7 @@ name = "exceptiongroup" version = "1.3.1" source = { registry = "https://pypi.org/simple" } dependencies = [ - { name = "typing-extensions" }, + { name = "typing-extensions", marker = "python_full_version < '3.11'" }, ] sdist = { url = "https://files.pythonhosted.org/packages/50/79/66800aadf48771f6b62f7eb014e352e5d06856655206165d775e675a02c9/exceptiongroup-1.3.1.tar.gz", hash = "sha256:8b412432c6055b0b7d14c310000ae93352ed6754f70fa8f7c34141f91c4e3219", size = 30371, upload-time = "2025-11-21T23:01:54.787Z" } wheels = [ @@ -929,14 +929,14 @@ wheels = [ [[package]] name = "pypdf" -version = "6.15.0" +version = "6.17.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "typing-extensions", marker = "python_full_version < '3.11'" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/17/17/ee75a92718ec7212de831e71454d702225aa5e474a805cce169806044453/pypdf-6.15.0.tar.gz", hash = "sha256:d39c4d955a76409284a905e2d65b40076d77ab76129e0faaeeb6612403ecfc79", size = 6993794, upload-time = "2026-08-06T13:06:49.929Z" } +sdist = { url = "https://files.pythonhosted.org/packages/5d/dc/34857a5e31cf708c163929f61a9ba4bd357a8850e49fc4e846ced527b51f/pypdf-6.17.0.tar.gz", hash = "sha256:097ad0d829778ec5b615aeaa5c6da4b6cac4992f8fd80b56f98a1a8c006573bb", size = 7018352, upload-time = "2026-09-04T11:30:44.256Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/af/72/ce3067ac31e214a66388159f8462ddb8c13dd00170f24d555a1f1ae8ee91/pypdf-6.15.0-py3-none-any.whl", hash = "sha256:14e001d6504822cb1ca9c7ed9a69bccb320f59b320730f55af804361abe4d5ee", size = 378123, upload-time = "2026-08-06T13:06:47.709Z" }, + { url = "https://files.pythonhosted.org/packages/c1/08/1e9731038124a9127e1d27848952b86fb32b2f45f8f1b94adc7f0817a6ac/pypdf-6.17.0-py3-none-any.whl", hash = "sha256:5bd827266a21553b74d910e350131a6227b72f2ab4209bf372814b8195fa11c5", size = 388051, upload-time = "2026-09-04T11:30:42.681Z" }, ] [[package]]