From 161627de58b863a0edf4e0a3f329ed53239fb16e Mon Sep 17 00:00:00 2001 From: torrid-fish Date: Mon, 14 Sep 2026 15:14:50 +0000 Subject: [PATCH] feat(api)!: remove DictQuery / SentenceQuery / UsageQuery MarkAccent is the service's only purpose. The other three endpoints were HTML scrapes of external sites, all saw little use, and DictQuery had been returning 404 for every input since EDRDG moved JMdictDB behind a monthly-rotating password gate. Removing them makes the whole service request-time offline, matching what the README already claimed for the accent pipeline: - delete api/dict_query.py, api/sentence_query.py, api/usage_query.py - delete api/dependencies.py (get_http_client had no other consumer) and the app.state.http_client lifespan setup/teardown in main.py - drop beautifulsoup4 from runtime deps; move httpx to the dev group (only tests/test_request_limits.py still needs it) - README: trim the endpoint table, record why the three went, and drop the shared-httpx.AsyncClient guide that documented api.dependencies BREAKING CHANGE: remove `POST /api/DictQuery/`, `/api/SentenceQuery/`, `/api/UsageQuery/HeadWords/`, `/api/UsageQuery/URL/` and `/api/UsageQuery/IdDetails/`; they now return 404. Only `/api/MarkAccent/` and `/api/MarkAccent/stream/` remain. Refs #54. Refs #62. Co-Authored-By: Claude Opus 5 (1M context) --- README.md | 36 +--- api/dependencies.py | 20 -- api/dict_query.py | 151 -------------- api/sentence_query.py | 109 ---------- api/usage_query.py | 454 ------------------------------------------ main.py | 21 +- pyproject.toml | 3 +- uv.lock | 28 +-- 8 files changed, 16 insertions(+), 806 deletions(-) delete mode 100644 api/dependencies.py delete mode 100644 api/dict_query.py delete mode 100644 api/sentence_query.py delete mode 100644 api/usage_query.py diff --git a/README.md b/README.md index 6d90081..d7649f2 100644 --- a/README.md +++ b/README.md @@ -16,10 +16,14 @@ network calls, external API keys, or `.env` setup required. |--|--| | `POST /api/MarkAccent/` | Mark pitch accent + furigana on the whole input, returns one `AccentResponse`. | | `POST /api/MarkAccent/stream/` | Same pipeline, streams one NDJSON object per `\n`-split sentence in input order. | -| `POST /api/UsageQuery/HeadWords/` | Look up Yahoo Realtime/News headwords for a query (delegates to an external HTTP endpoint). | -| `POST /api/UsageQuery/URL/` | Resolve headword references to URLs. | -| `POST /api/DictQuery/` | JMdict dictionary lookup. | -| `POST /api/SentenceQuery/` | Example-sentence search. | + +`MarkAccent` is the whole service. The `DictQuery`, +`SentenceQuery` and `UsageQuery` endpoints that lived here previously +were removed in #54: each was an HTML scrape of an external site +(EDRDG JMdictDB, EDRDG WWWJDIC, Yahoo), all three saw little use, and +JMdictDB moved behind a password gate that broke `DictQuery` +outright. Nothing in the service makes a network call at request time +any more. The MarkFurigana endpoint that lived in earlier versions of the service was removed during the Yahoo MA → local fugashi migration; @@ -189,30 +193,6 @@ curl -s -X POST http://127.0.0.1:8000/api/MarkAccent/ \ -d '{"text":"三月五日(土)"}' | python -m json.tool ``` -## How to use a shared `httpx.AsyncClient` - -If your router needs to send HTTP requests, follow this pattern to -reuse the connection pool managed by `api.dependencies`: - -```python -import httpx -from fastapi import APIRouter, Depends -from api.dependencies import get_http_client - -router = APIRouter() - -@router.post("/Foo/", tags=["Foo"], response_model=FooResponse) -async def foo( - request: FooRequest, client: httpx.AsyncClient = Depends(get_http_client) -): - try: - response = await client.post(url) - except httpx.TimeoutException: - ... - except httpx.HTTPError as e: - ... -``` - ## Known limitations - **UniDic-vs-OpenJTalk reading mismatches.** A handful of kanji come diff --git a/api/dependencies.py b/api/dependencies.py deleted file mode 100644 index 93dcb41..0000000 --- a/api/dependencies.py +++ /dev/null @@ -1,20 +0,0 @@ -import httpx -from fastapi import Request - - -async def get_http_client(request: Request) -> httpx.AsyncClient: - """ - Dependency to provide an HTTP client for making asynchronous requests. - - This function retrieves the HTTP client instance stored in the FastAPI - application's state. It can be used as a dependency in route handlers - to perform HTTP requests. - - Args: - request (Request): The FastAPI request object. - - Returns: - httpx.AsyncClient: The HTTP client instance. - """ - ret: httpx.AsyncClient = request.app.state.http_client - return ret diff --git a/api/dict_query.py b/api/dict_query.py deleted file mode 100644 index e3a12b4..0000000 --- a/api/dict_query.py +++ /dev/null @@ -1,151 +0,0 @@ -import asyncio -from urllib.parse import parse_qs, urlparse - -import httpx -from bs4 import BeautifulSoup -from fastapi import APIRouter, Depends -from pydantic import BaseModel, Field - -from api.dependencies import get_http_client - - -class Request(BaseModel): - word: str = Field(description="The word to query") - - -class Definition(BaseModel): - pos: list[str] = Field(description="pos list") - meanings: list[str] = Field(description="Meanings of the word") - - -class WordResult(BaseModel): - kanji: list[str] = Field(description="Kanji") - furigana: list[str] = Field(description="Furigana") - definitions: list[Definition] = Field(description="Definitions of the word") - id: int = Field(description="ID") - - -class ErrorInfo(BaseModel): - code: int = Field(description="Error code (similar to HTTP status)") - message: str = Field(description="Details of the error") - - -class Response(BaseModel): - status: int = Field(default=200, description="Status code") - result: list[WordResult] | None = Field(default=None, description="List of results") - error: ErrorInfo | None = Field(default=None, description="Error details") - - -router = APIRouter() - - -# 取得所有符合資料的 url -async def get_all_url(search_word: str, client: httpx.AsyncClient) -> list[str]: - url = f"https://www.edrdg.org/jmwsgi/srchres.py?s1=1&y1=1&t1={search_word}&src=1&search=Search&svc=jmdict" - - try: - response = await client.get(url, follow_redirects=True) - response.encoding = response.charset_encoding or "utf-8" - except httpx.RequestError as e: - raise RuntimeError(f"Network error: {str(e)}") - - # 判斷是否因為只有一個結果而直接跳轉 - if "entr.py" in str(response.url): - entry_id = parse_qs(urlparse(str(response.url)).query).get("e", [None])[0] - return ( - [f"https://www.edrdg.org/jmwsgi/entr.py?svc=jmdict&e={entry_id}"] - if entry_id - else [] - ) - - soup = BeautifulSoup(response.text, "html.parser") - rows = soup.find_all("tr", class_="resrow") - - url_list = [] - for row in rows: - inp = row.find("input", {"name": "e"}) - if inp and inp.has_attr("value"): - url_list.append( - f"https://www.edrdg.org/jmwsgi/entr.py?svc=jmdict&e={inp['value']}" - ) - - return url_list - - -# 根據 url 清單回傳查詢結果 -async def get_dict(url_list: list[str], client: httpx.AsyncClient) -> list[WordResult]: - results: list[WordResult] = [] - for url in url_list: - try: - response = await client.get(url) - response.encoding = response.charset_encoding or "utf-8" - except httpx.RequestError as e: - raise RuntimeError(f"Network error: {str(e)}") - - soup = BeautifulSoup(response.text, "html.parser") - - try: - kanji = [k.get_text(strip=True) for k in soup.select("span.kanj")] - furigana = [k.get_text(strip=True) for k in soup.select("span.rdng")] - - definitions: list[Definition] = [] - for sense in soup.select("tr.sense"): - pos = [ - k.get_text(" ", strip=True) - for k in sense.select("span.pos span.abbr") - ] - meanings = [ - k.get_text(" ", strip=True).replace("▶", "").strip() - for k in sense.select("span.glossx") - ] - definitions.append(Definition(pos=pos, meanings=meanings)) - - jmdict_id = soup.select_one('a[href^="srchres.py"]') - id = int(jmdict_id.get_text(strip=True)) if jmdict_id else 0 - - results.append( - WordResult( - kanji=kanji, furigana=furigana, definitions=definitions, id=id - ) - ) - except Exception as e: - raise RuntimeError(f"Parse error: {str(e)}") - - return results - - -@router.post("/DictQuery/", tags=["DictionaryQuery"], response_model=Response) -async def dict_query( - request: Request, client: httpx.AsyncClient = Depends(get_http_client) -) -> Response: - """Query JMdict dictionary for the given word.""" - - try: - url_list = await get_all_url(request.word, client) - if not url_list: - return Response( - status=404, - result=None, - error=ErrorInfo(code=404, message="No results found"), - ) - - results = await get_dict(url_list, client) - return Response(status=200, result=results, error=None) - - except RuntimeError as e: - return Response( - status=500, result=None, error=ErrorInfo(code=500, message=str(e)) - ) - - -# Test codes -if __name__ == "__main__": - - async def test() -> None: - async with httpx.AsyncClient() as client: - print(await dict_query(Request(word="先生"), client)) - print(await dict_query(Request(word="生所"), client)) - print(await dict_query(Request(word="みたび"), client)) - print(await dict_query(Request(word="つらら"), client)) - - asyncio.run(test()) diff --git a/api/sentence_query.py b/api/sentence_query.py deleted file mode 100644 index c4f3123..0000000 --- a/api/sentence_query.py +++ /dev/null @@ -1,109 +0,0 @@ -import asyncio -import re - -import httpx -from bs4 import BeautifulSoup, Comment -from fastapi import APIRouter, Depends -from pydantic import BaseModel, Field - -from api.dependencies import get_http_client - - -class Request(BaseModel): - """Class representing a request object""" - - word: str = Field(description="The word to query") - id: int = Field( - description="The unique ID of a word in JMdict.\n" - "Must be obtained through DictionaryQuery to fetch example sentences." - ) - - -class WordSentence(BaseModel): - jp: str = Field(description="jp sentence") - en: str = Field(description="en sentence") - - -class WordResult(BaseModel): - word: str = Field(description="Word") - id: int = Field(description="ID") - sentence: list[WordSentence] = Field(description="A list of sentence") - - -class Response(BaseModel): - status: int = Field(default=200, description="Status code") - result: WordResult | None = Field(description="Results") - error: str | None = Field(default=None, description="Error message if any") - - -router = APIRouter() - - -# 根據接收到的漢字及ID回傳可能的例句 -@router.post("/SentenceQuery/", tags=["SentenceQuery"], response_model=Response) -async def sentence_query( - request: Request, - client: httpx.AsyncClient = Depends(get_http_client), -) -> Response: - """ - Example sentences from JMdict. - - - Uses the provided `word`(kanji or furigana) and `id` from DictionaryQuery. - - Returns a list of sentences containing - both Japanese text and their English translations. - """ - url = "https://www.edrdg.org/cgi-bin/wwwjdic/wwwjdic?1E" - payload = {"dsrchkey": request.word, "dicsel": "1"} - - try: - response = await client.post(url, data=payload) - response.encoding = response.charset_encoding or "utf-8" - except httpx.RequestError as e: - return Response(status=500, result=None, error=f"Network error: {str(e)}") - - try: - soup = BeautifulSoup(response.text, "html.parser") - - sentences = [] - found_block = False - for block in soup.select("div[style*=clear]"): - comments = block.find_all(string=lambda text: isinstance(text, Comment)) - if not any(f"ent_seq={request.id}" in c for c in comments): - continue - found_block = True - for br in block.find_all("br"): - nxt = br.find_next_sibling("font") - - if nxt and nxt.get("size") == "-1": - s = nxt.get_text(" ", strip=True) - - s = re.sub(r"^\(\d+\)\s*", "", s) - jp, en = re.split(r"\t+|\s{2,}", s, maxsplit=1) - jp = jp.replace(" ", "") - - sentences.append(WordSentence(jp=jp.strip(), en=en.strip())) - - if not found_block or not sentences: - return Response(status=404, result=None, error="No results found") - - return Response( - status=200, - result=WordResult(word=request.word, id=request.id, sentence=sentences), - error=None, - ) - - except Exception as e: - return Response(status=500, result=None, error=f"Parse error: {str(e)}") - - -# Test codes -if __name__ == "__main__": - - async def test() -> None: - async with httpx.AsyncClient() as client: - print(await sentence_query(Request(word="先生", id=1387990), client)) - print(await sentence_query(Request(word="せんせい", id=1387990), client)) - print(await sentence_query(Request(word="少女", id=1580290), client)) - print(await sentence_query(Request(word="嗨嗨", id=1580290), client)) - - asyncio.run(test()) diff --git a/api/usage_query.py b/api/usage_query.py deleted file mode 100644 index 602692f..0000000 --- a/api/usage_query.py +++ /dev/null @@ -1,454 +0,0 @@ -import json -import re -from typing import Any, Literal - -import httpx -import jaconv -from fastapi import APIRouter, Depends -from pydantic import BaseModel, Field - -from api.dependencies import get_http_client - -SITE = { - "NLB": "https://nlb.ninjal.ac.jp", - "NLT": "https://tsukubawebcorpus.jp", -} - - -class HeadWordRequest(BaseModel): - """Class representing a request object""" - - word: str = Field(description="The word to query") - site: Literal["NLB", "NLT"] = Field( - default="NLB", - description="The site to query, either 'NLB' or 'NLT'. Default is 'NLB'.", - ) - - -class IdRequest(BaseModel): - """Class representing a request object for headword ID""" - - headword_id: str = Field(description="The headword ID of the word") - site: Literal["NLB", "NLT"] = Field( - default="NLB", - description="The site to query, either 'NLB' or 'NLT'. Default is 'NLB'.", - ) - - -class ErrorInfo(BaseModel): - """Class representing an error information""" - - code: int = Field(description="The error code that follows JSON-RPC 2.0") - message: str = Field( - description="The error message that describe the details of an error" - ) - - -class HeadWord(BaseModel): - """Class representing a headword object""" - - id: int = Field(description="The id of the word") - headword_id: str = Field(description="The headword id of the word") - headword: str = Field(description="The headword of the word in kanji") - yomi_display: str = Field(description="The yomi display of the word in katakana") - romaji_display: str = Field(description="The romaji display of the word") - freq: int = Field(description="The frequency of the word in the corpus") - - -class URL(BaseModel): - """Class representing a URL object""" - - word: str = Field(description="The headword of the word") - url: str = Field(description="The URL of the headword") - - -class IdDetails(BaseModel): - """Class representing details of a word""" - - base: dict[str, Any] = Field(description="The base form of the word") - subcorpus: list[dict[str, Any]] = Field(description="The subcorpus of the word") - shojikei: list[dict[str, Any]] = Field(description="The shojikei of the word") - subcorpus_shojikei: list[dict[str, Any]] = Field( - description="The distribution of shojikei by subcorpus of the word" - ) - katuyokei: list[dict[str, Any]] = Field(description="The katuyokei of the word") - setuzoku: list[dict[str, Any]] = Field( - description="The subsequent auxiliary verbs of the word" - ) - patternfreqorder: list[dict[str, Any]] = Field( - description="The frequency of the word in different patterns" - ) - - -class HeadWordResponse(BaseModel): - """Class representing a response object for headword query""" - - status: int = Field( - default=200, description="Status code of response align with RFC 9110" - ) - result: list[HeadWord] | None = Field( - description="A list contains headword results" - ) - error: ErrorInfo | None = Field( - default=None, - description="An object that describe the details of an error when occur", - ) - - -class URLResponse(BaseModel): - """Class representing a response object for URL query""" - - status: int = Field( - default=200, description="Status code of response align with RFC 9110" - ) - result: list[URL] | None = Field( - description="A list contains URLs for the headwords" - ) - error: ErrorInfo | None = Field( - default=None, - description="An object that describe the details of an error when occur", - ) - - -class IdResponse(BaseModel): - """Class representing a response object for word details""" - - status: int = Field( - default=200, description="Status code of response align with RFC 9110" - ) - result: IdDetails | None = Field( - description="Details of the word with the given headword ID" - ) - error: ErrorInfo | None = Field( - default=None, - description="An object that describe the details of an error when occur", - ) - - -router = APIRouter() - - -def text_type(text: str) -> str | None: - """Determine the type of text: yomi, romaji, or headword.""" - - if not text: - return None - - if bool(re.fullmatch(r"[\u3040-\u309F\u30A0-\u30FF]+", text)): - # The string consists only of hiragana or katakana characters - return "yomi" - elif bool(re.fullmatch(r"[A-Za-z]+", text)): - # The string consists only of romaji - return "romaji" - else: - return "headword" - - -@router.post( - "/UsageQuery/HeadWords/", tags=["UsageQuery"], response_model=HeadWordResponse -) -async def get_headwords( - request: HeadWordRequest, client: httpx.AsyncClient = Depends(get_http_client) -) -> HeadWordResponse: - """Get the lists of information of headwords with the given word.""" - - match text_type(request.word): - case "yomi": - rules = [ - {"field": "yomi1", "op": "eq", "data": jaconv.hira2kata(request.word)}, - {"field": "yomi2", "op": "ew", "data": jaconv.hira2kata(request.word)}, - {"field": "yomi3", "op": "ew", "data": jaconv.hira2kata(request.word)}, - ] - case "romaji": - rules = [ - {"field": "romaji1", "op": "eq", "data": request.word}, - {"field": "romaji2", "op": "ew", "data": request.word}, - {"field": "romaji3", "op": "ew", "data": request.word}, - ] - case "headword": - rules = [{"field": "headword", "op": "eq", "data": request.word}] - - filter = {"groupOp": "OR", "rules": rules} - - payload = { - "_search": "true", - "filters": json.dumps(filter), - } - - url = f"{SITE[request.site]}/headwordlist_all/" - headers = { - "Content-Type": "application/x-www-form-urlencoded; charset=UTF-8", - "Referer": f"{SITE[request.site]}/search/", - "User-Agent": "Mozilla/5.0", - } - - try: - response = await client.post(url, data=payload, headers=headers) - except httpx.TimeoutException: - return HeadWordResponse( - status=504, - result=None, - error=ErrorInfo(code=504, message="Connection timed out"), - ) - except httpx.HTTPError as e: - return HeadWordResponse( - status=500, - result=None, - error=ErrorInfo(code=500, message=f"Request error: {e}"), - ) - - if response.status_code == 200: - try: - data = response.json() - result = [] - if data.get("rows") is not None and len(data["rows"]) > 0: - for row in data["rows"]: - id = row.get("id") - headword_id = row.get("headword_id") - headword = row.get("headword") - yomi_display = row.get("yomi_display") - romaji_display = row.get("romaji_display") - freq = row.get("freq") - if ( - id - and headword_id - and headword - and yomi_display - and romaji_display - and freq - ): - search_result = HeadWord( - id=id, - headword_id=headword_id, - headword=headword, - yomi_display=yomi_display, - romaji_display=romaji_display, - freq=freq, - ) - # print(f"Found result: {search_result}") - result.append(search_result) - return HeadWordResponse(status=200, result=result) - else: - # print("No results found") - return HeadWordResponse( - status=404, - result=None, - error=ErrorInfo( - code=404, message="No results found for the given word" - ), - ) - except Exception as e: - # print(f"Error parsing JSON: {e}") - return HeadWordResponse( - status=500, - result=None, - error=ErrorInfo(code=500, message=f"Error parsing JSON: {e}"), - ) - else: - # print(f"HTTP error {response.status_code}") - return HeadWordResponse( - status=response.status_code, - result=None, - error=ErrorInfo( - code=response.status_code, message=f"HTTP error {response.status_code}" - ), - ) - - -@router.post("/UsageQuery/URL/", tags=["UsageQuery"], response_model=URLResponse) -async def get_urls( - request: HeadWordRequest, client: httpx.AsyncClient = Depends(get_http_client) -) -> URLResponse: - """Get the URLs of words with the given word.""" - response = await get_headwords(request, client) - - if response.status != 200: - return URLResponse( - status=response.status, - result=None, - error=response.error, - ) - - result = [] - for res in response.result or []: - result.append( - URL( - word=res.headword, - url=f"{SITE[request.site]}/headword/{res.headword_id}/", - ) - ) - # print(f"Generated URLs: {result}") - - return URLResponse(status=200, result=result) - - -@router.post("/UsageQuery/IdDetails/", tags=["UsageQuery"], response_model=IdResponse) -async def get_id_details( - request: IdRequest, client: httpx.AsyncClient = Depends(get_http_client) -) -> IdResponse: - """Get the details of the given headword ID.""" - - async def fetch_data( - mode: Literal["get", "post"], endpoint: str, target: str = "" - ) -> IdResponse | Any: - """Helper function to fetch and parse JSON data""" - match mode: - case "get": - headers = { - "Referer": f"{SITE[request.site]}/headword/{request.headword_id}/", - "User-Agent": "Mozilla/5.0", - } - try: - response = await client.get( - f"{SITE[request.site]}/{endpoint}/{request.headword_id}/", - headers=headers, - ) - except httpx.TimeoutException: - return IdResponse( - status=504, - result=None, - error=ErrorInfo(code=504, message="Connection timed out"), - ) - except httpx.HTTPError as e: - return IdResponse( - status=500, - result=None, - error=ErrorInfo(code=500, message=f"Request error: {e}"), - ) - case "post": - headers = { - "Content-Type": "application/x-www-form-urlencoded; charset=UTF-8", - "Referer": f"{SITE[request.site]}/headword/{request.headword_id}/", - "User-Agent": "Mozilla/5.0", - } - try: - response = await client.post( - f"{SITE[request.site]}/{endpoint}/{request.headword_id}/", - headers=headers, - ) - except httpx.TimeoutException: - return IdResponse( - status=504, - result=None, - error=ErrorInfo(code=504, message="Connection timed out"), - ) - except httpx.HTTPError as e: - return IdResponse( - status=500, - result=None, - error=ErrorInfo(code=500, message=f"Request error: {e}"), - ) - - if response.status_code == 200: - try: - data = response.json() - ret = data[target] if target and data.get(target) is not None else data - # print(f"Fetched {target}: {ret}") - return ret - except Exception as e: - # print(f"Error parsing JSON while fetching {endpoint}: {e}") - return IdResponse( - status=500, - result=None, - error=ErrorInfo( - code=500, - message=f"Error parsing JSON while fetching {endpoint}: {e}", - ), - ) - else: - # print(f"HTTP error {response.status_code} while fetching {endpoint}") - return IdResponse( - status=response.status_code, - result=None, - error=ErrorInfo( - code=response.status_code, - message=( - f"HTTP error {response.status_code} while fetching {endpoint}" - ), - ), - ) - - # IdResponse.base - base = await fetch_data("get", "basicinfob") - if isinstance(base, IdResponse): - return base - # IdResponse.subcorpus - subcorpus = ( - [] - if request.site == "NLT" - else await fetch_data("get", "basicinfosc", "subcorpus") - ) - if isinstance(subcorpus, IdResponse): - return subcorpus - # IdResponse.shojikei - shojikei = await fetch_data("get", "basicinfosj", "shojikei") - if isinstance(shojikei, IdResponse): - return shojikei - # IdResponse.subcorpus_shojikei - subcorpus_shojikei = ( - [] - if request.site == "NLT" - else await fetch_data("post", "basicinfoss", "subcorpus") - ) - if isinstance(subcorpus_shojikei, IdResponse): - return subcorpus_shojikei - # IdResponse.katuyokei - katuyokei = ( - [] - if request.site == "NLT" - else await fetch_data("get", "basicinfoky", "katuyokei") - ) - if isinstance(katuyokei, IdResponse): - return katuyokei - # IdResponse.setuzoku - setuzoku = await fetch_data("get", "basicinfojs", "setuzoku") - if isinstance(setuzoku, IdResponse): - return setuzoku - # IdResponse.patternfreqorder - patternfreqorder = await fetch_data("post", "patternfreqorder", "rows") - if isinstance(patternfreqorder, IdResponse): - return patternfreqorder - - # print(patternfreqorder) - - return IdResponse( - status=200, - result=IdDetails( - base=base, - subcorpus=subcorpus, - shojikei=shojikei, - subcorpus_shojikei=subcorpus_shojikei, - katuyokei=katuyokei, - setuzoku=setuzoku, - patternfreqorder=patternfreqorder, - ), - ) - - -# Test code -if __name__ == "__main__": - print("==== Testing NLB ====") - print(get_urls(HeadWordRequest(word="走る", site="NLB"))) - print(get_urls(HeadWordRequest(word="はしる", site="NLB"))) - print(get_urls(HeadWordRequest(word="hashiru", site="NLB"))) - print("\n==== Testing NLT ====") - print(get_urls(HeadWordRequest(word="走る", site="NLT"))) - print(get_urls(HeadWordRequest(word="はしる", site="NLT"))) - print(get_urls(HeadWordRequest(word="hashiru", site="NLT"))) - # print("==== Testing NLB ====") - # get_headwords(HeadWordRequest(word="走る", site="NLB")) - # print("=" * 10) - # get_headwords(HeadWordRequest(word="はしる", site="NLB")) - # print("=" * 10) - # get_headwords(HeadWordRequest(word="hashiru", site="NLB")) - # print("\n==== Testing NLT ====") - # get_headwords(HeadWordRequest(word="走る", site="NLT")) - # print("=" * 10) - # get_headwords(HeadWordRequest(word="はしる", site="NLT")) - # print("=" * 10) - # get_headwords(HeadWordRequest(word="hashiru", site="NLT")) - # print("=" * 100) - # print("\n==== Testing NLB ====") - # get_id_details(IdRequest(id="V.00093", site="NLB")) - # print("\n==== Testing NLT ====") - # get_id_details(IdRequest(id="V.00128", site="NLT")) diff --git a/main.py b/main.py index 91323eb..bda8fe4 100644 --- a/main.py +++ b/main.py @@ -1,9 +1,6 @@ """ -An API interface that provide the following functionalities -(1) Accent Marker (/api/MarkAccent/ + /api/MarkAccent/stream/) -(2) Usage Query (/api/UsageQuery/) -(3) Dictionary Query (/api/DictQuery/) -(4) Sentence Query (/api/SentenceQuery/) +An API interface that marks Japanese pitch accent and furigana on input text +(/api/MarkAccent/ + /api/MarkAccent/stream/). """ import asyncio @@ -11,10 +8,8 @@ from contextlib import asynccontextmanager from typing import AsyncGenerator -import httpx from fastapi import FastAPI -from api import dict_query, sentence_query, usage_query from api.accent import accent_router from api.accent.openjtalk import warmup as warmup_openjtalk from api.accent.tokenizer import warmup as warmup_tokenizer @@ -35,9 +30,6 @@ async def lifespan(app: FastAPI) -> AsyncGenerator[None, None]: Args: app (FastAPI): The FastAPI application instance. """ - # Set up resources before the application starts - app.state.http_client = httpx.AsyncClient(timeout=10.0) - # Warm the accent engines (fugashi/UniDic tagger + OpenJTalk frontend) so # the first /MarkAccent/ request doesn't pay the one-off dictionary-load # latency. Both are blocking C-extension loads, so run them off the event @@ -50,8 +42,6 @@ async def lifespan(app: FastAPI) -> AsyncGenerator[None, None]: logger.info("Accent engines ready.") yield - # Clean up resources after the application stops - await app.state.http_client.aclose() app = FastAPI(lifespan=lifespan) @@ -60,11 +50,10 @@ async def lifespan(app: FastAPI) -> AsyncGenerator[None, None]: max_body_bytes=MAX_REQUEST_BODY_BYTES, ) -# Include routers from different modules +# The accent pipeline is the only router: it runs fully in-process, so the +# application holds no shared HTTP client (see #54 — the former DictQuery / +# SentenceQuery / UsageQuery scrapes and `api.dependencies` went with it). app.include_router(accent_router, prefix="/api") -app.include_router(usage_query.router, prefix="/api") -app.include_router(dict_query.router, prefix="/api") -app.include_router(sentence_query.router, prefix="/api") logging.basicConfig( level=logging.INFO, format="{asctime} [{levelname:^8s}] {message} ({name}.{module}:{lineno})", diff --git a/pyproject.toml b/pyproject.toml index d739745..d5215dd 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -5,10 +5,8 @@ description = "Add your description here" readme = "README.md" requires-python = "==3.11.*" dependencies = [ - "beautifulsoup4>=4.14.2", "fastapi>=0.120.0", "fugashi>=1.5.2", - "httpx>=0.28.1", "jaconv>=0.4.0", "neologdn>=0.5.4", "pydantic>=2.12.3", @@ -37,6 +35,7 @@ ignore_missing_imports = true [dependency-groups] dev = [ + "httpx>=0.28.1", "mypy>=1.18.2", "pytest>=8.3.0", "ruff>=0.14.2", diff --git a/uv.lock b/uv.lock index 2faa042..428ddec 100644 --- a/uv.lock +++ b/uv.lock @@ -38,10 +38,8 @@ name = "api-tools" version = "0.1.0" source = { virtual = "." } dependencies = [ - { name = "beautifulsoup4" }, { name = "fastapi" }, { name = "fugashi" }, - { name = "httpx" }, { name = "jaconv" }, { name = "neologdn" }, { name = "pydantic" }, @@ -52,6 +50,7 @@ dependencies = [ [package.dev-dependencies] dev = [ + { name = "httpx" }, { name = "mypy" }, { name = "pytest" }, { name = "ruff" }, @@ -59,10 +58,8 @@ dev = [ [package.metadata] requires-dist = [ - { name = "beautifulsoup4", specifier = ">=4.14.2" }, { name = "fastapi", specifier = ">=0.120.0" }, { name = "fugashi", specifier = ">=1.5.2" }, - { name = "httpx", specifier = ">=0.28.1" }, { name = "jaconv", specifier = ">=0.4.0" }, { name = "neologdn", specifier = ">=0.5.4" }, { name = "pydantic", specifier = ">=2.12.3" }, @@ -73,24 +70,12 @@ requires-dist = [ [package.metadata.requires-dev] dev = [ + { name = "httpx", specifier = ">=0.28.1" }, { name = "mypy", specifier = ">=1.18.2" }, { name = "pytest", specifier = ">=8.3.0" }, { name = "ruff", specifier = ">=0.14.2" }, ] -[[package]] -name = "beautifulsoup4" -version = "4.14.3" -source = { registry = "https://pypi.org/simple" } -dependencies = [ - { name = "soupsieve" }, - { name = "typing-extensions" }, -] -sdist = { url = "https://files.pythonhosted.org/packages/c3/b0/1c6a16426d389813b48d95e26898aff79abbde42ad353958ad95cc8c9b21/beautifulsoup4-4.14.3.tar.gz", hash = "sha256:6292b1c5186d356bba669ef9f7f051757099565ad9ada5dd630bd9de5fa7fb86", size = 627737, upload-time = "2025-11-30T15:08:26.084Z" } -wheels = [ - { url = "https://files.pythonhosted.org/packages/1a/39/47f9197bdd44df24d67ac8893641e16f386c984a0619ef2ee4c51fbbc019/beautifulsoup4-4.14.3-py3-none-any.whl", hash = "sha256:0918bfe44902e6ad8d57732ba310582e98da931428d231a5ecb9e7c703a735bb", size = 107721, upload-time = "2025-11-30T15:08:24.087Z" }, -] - [[package]] name = "certifi" version = "2025.11.12" @@ -500,15 +485,6 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/74/31/b0e29d572670dca3674eeee78e418f20bdf97fa8aa9ea71380885e175ca0/ruff-0.14.10-py3-none-win_arm64.whl", hash = "sha256:e51d046cf6dda98a4633b8a8a771451107413b0f07183b2bef03f075599e44e6", size = 13729839, upload-time = "2025-12-18T19:28:48.636Z" }, ] -[[package]] -name = "soupsieve" -version = "2.8.1" -source = { registry = "https://pypi.org/simple" } -sdist = { url = "https://files.pythonhosted.org/packages/89/23/adf3796d740536d63a6fbda113d07e60c734b6ed5d3058d1e47fc0495e47/soupsieve-2.8.1.tar.gz", hash = "sha256:4cf733bc50fa805f5df4b8ef4740fc0e0fa6218cf3006269afd3f9d6d80fd350", size = 117856, upload-time = "2025-12-18T13:50:34.655Z" } -wheels = [ - { url = "https://files.pythonhosted.org/packages/48/f3/b67d6ea49ca9154453b6d70b34ea22f3996b9fa55da105a79d8732227adc/soupsieve-2.8.1-py3-none-any.whl", hash = "sha256:a11fe2a6f3d76ab3cf2de04eb339c1be5b506a8a47f2ceb6d139803177f85434", size = 36710, upload-time = "2025-12-18T13:50:33.267Z" }, -] - [[package]] name = "starlette" version = "0.50.0"