This repository was archived by the owner on Jan 18, 2025. It is now read-only.
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathapp.py
More file actions
89 lines (69 loc) · 2.64 KB
/
Copy pathapp.py
File metadata and controls
89 lines (69 loc) · 2.64 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
#!/usr/bin/env python3
# Copyright 2024 Code Inc. <https://www.codeinc.co>
#
# Use of this source code is governed by an MIT-style
# license that can be found in the LICENSE file or at
# https://opensource.org/licenses/MIT.
from datetime import datetime
from os import getenv
import spacy
from fastapi import FastAPI
from fastapi import HTTPException
from models import SentencesRequest, WordsRequest, HealthResponse, TokenizerResponse, Language, ParagraphsRequest
# Load SpaCy language model
nlps = {
Language.en: spacy.load("en_core_web_md"),
Language.fr: spacy.load("fr_core_news_md"),
}
app = FastAPI()
start_time = datetime.now()
@app.get("/health", response_model=HealthResponse)
async def health():
"""Health check endpoint."""
return {
"status": "ok",
"uptime": str(datetime.now() - start_time),
"version": getenv("VERSION", "0.0.0"),
"build_id": getenv("BUILD_ID", "0")
}
@app.post("/tokenize/sentences", response_model=TokenizerResponse)
async def tokenize_sentences(request: SentencesRequest):
"""Tokenize text into sentences."""
try:
doc = nlps[request.lang](request.text)
return {"tokens": [sent.text for sent in doc.sents]}
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@app.post("/tokenize/words", response_model=TokenizerResponse)
async def tokenize_words(request: WordsRequest):
"""Tokenize text into words."""
try:
doc = nlps[request.lang](request.text)
# Extract tokens excluding punctuations
if request.exclude_punct:
tokens = [token.text for token in doc if not token.is_punct]
# Extract all tokens, including punctuations
else:
tokens = [token.text for token in doc]
# Lowercase tokens if requested
if request.lowercase:
tokens = [token.lower() for token in tokens]
return {"tokens": tokens}
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))
@app.post("/tokenize/paragraphs", response_model=TokenizerResponse)
async def tokenize_words(request: ParagraphsRequest):
"""Tokenize text into paragraphs."""
try:
doc = nlps[request.lang](request.text)
# Extract paragraphs
start = 0
paragraphs = []
for token in doc:
if token.is_space and token.text.count("\n") > 1:
paragraphs.append(doc[start:token.i].text.strip())
start = token.i
paragraphs.append(doc[start:].text.strip())
return {"tokens": paragraphs}
except Exception as e:
raise HTTPException(status_code=500, detail=str(e))