여러 줄 주석이 설명보다 경위(예전·실측·지적)를 적고 있어 읽는 사람이 결론을 찾기 어려웠다. - ts·tsx·js·mjs·css·py 478개: 여러 줄 주석은 첫 문장 한 줄로, 과거형·날짜 문장은 삭제 - 주석 위치는 TypeScript 파서·파이썬 tokenize/ast 로 찾는다 — 문자열 안의 # · /* 는 건드리지 않는다 - eslint·ts·noqa·type: ignore 같은 지시 주석은 그대로 둔다 파이썬 275개 정리 전후 AST 동일, TS 298개 주석 뺀 토큰 동일(빈 JSX 주석 10곳만 차이). site·frontend·admin tsc, site vitest 105 passed Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
116 lines
4.3 KiB
Python
116 lines
4.3 KiB
Python
"""상호 → 네이버 플레이스 id 해석."""
|
|
import pytest
|
|
|
|
from services.external import naver_place_lookup as lookup
|
|
|
|
|
|
# 항목 사이를 채우는 잡음.
|
|
_FILLER = "<div class=\"noise\">방문자 리뷰 블로그 리뷰 사진 더보기</div>" * 40
|
|
|
|
|
|
def _html(*entries: tuple[str, str]) -> str:
|
|
"""검색 결과 원문 흉내."""
|
|
parts = []
|
|
for name, place_id in entries:
|
|
parts.append(
|
|
_FILLER
|
|
+ f'<li><a href="/place/{place_id}">'
|
|
f'<span class="name">{name}</span>'
|
|
f'<span class="tag">주차하기 편해요</span></a></li>'
|
|
+ _FILLER
|
|
)
|
|
return "<html><body><ul>" + "".join(parts) + "</ul></body></html>"
|
|
|
|
|
|
def test_ampersand_in_name_is_matched_through_the_entity():
|
|
"""검증: 상호에 `&` 가 있고 원문에는 `&` 로 인코딩돼 있다."""
|
|
html = _html(("누에베 풀빌라&리조트", "1064005604"))
|
|
assert lookup._match_in_html(html, "누에베 풀빌라&리조트") == "1064005604"
|
|
|
|
|
|
def test_plain_name_still_matches():
|
|
"""검증: `&` 도 엔티티도 없는 평범한 상호."""
|
|
html = _html(("보사노바 커피로스터스 강릉점", "37093035"))
|
|
assert lookup._match_in_html(html, "보사노바 커피로스터스 강릉점") == "37093035"
|
|
|
|
|
|
def test_punctuation_differences_are_ignored():
|
|
"""검증: 네이버는 '스테이,머뭄' 처럼 구두점을 넣어 표기한다."""
|
|
html = _html(("스테이, 머뭄", "1133638931"))
|
|
assert lookup._match_in_html(html, "스테이머뭄") == "1133638931"
|
|
assert lookup._match_in_html(html, "스테이,머뭄") == "1133638931"
|
|
|
|
|
|
def test_unrelated_name_is_not_matched():
|
|
"""검증: 원문에 id 는 있지만 그 상호는 없다."""
|
|
html = _html(("통나무파크", "13149475"), ("도치돌알파카목장", "11111111"))
|
|
assert lookup._match_in_html(html, "제주양떼목장") is None
|
|
|
|
|
|
def test_correct_id_is_picked_among_several():
|
|
"""검증: 후보가 여럿 섞인 원문에서 특정 상호를 찾는다."""
|
|
html = _html(
|
|
("통나무파크입구", "99999999"),
|
|
("통나무파크 전기차충전소", "88888888"),
|
|
("누에베 풀빌라&리조트", "1064005604"),
|
|
)
|
|
assert lookup._match_in_html(html, "누에베 풀빌라&리조트") == "1064005604"
|
|
|
|
|
|
def test_empty_name_never_matches():
|
|
"""검증: 상호가 빈 문자열이다."""
|
|
html = _html(("통나무파크", "13149475"))
|
|
assert lookup._match_in_html(html, "") is None
|
|
assert lookup._match_in_html(html, " ") is None
|
|
|
|
|
|
async def test_find_place_ids_returns_only_the_ones_it_found(monkeypatch):
|
|
"""검증: 후보 3건 중 2건만 원문에 있다."""
|
|
calls: list[str] = []
|
|
|
|
async def _fake_fetch(query: str):
|
|
calls.append(query)
|
|
return _html(("누에베 풀빌라&리조트", "1064005604"), ("통나무파크", "13149475"))
|
|
|
|
monkeypatch.setattr(lookup, "_fetch_search_html", _fake_fetch)
|
|
|
|
found = await lookup.find_place_ids(
|
|
"누에베 풀빌라 제주 애월읍",
|
|
["누에베 풀빌라&리조트", "통나무파크", "여기없는가게"],
|
|
)
|
|
|
|
assert found == {"누에베 풀빌라&리조트": "1064005604", "통나무파크": "13149475"}
|
|
assert calls == ["누에베 풀빌라 제주 애월읍"], "후보 수만큼 부르면 네이버가 429 로 막는다"
|
|
|
|
|
|
async def test_find_place_ids_is_empty_when_search_fails(monkeypatch):
|
|
|
|
async def _fake_fetch(query: str):
|
|
return None
|
|
|
|
monkeypatch.setattr(lookup, "_fetch_search_html", _fake_fetch)
|
|
assert await lookup.find_place_ids("아무거나", ["가게"]) == {}
|
|
|
|
|
|
@pytest.mark.parametrize(
|
|
"raw, expected",
|
|
[
|
|
("누에베 풀빌라&리조트", "누에베풀빌라리조트"),
|
|
("스테이, 머뭄", "스테이머뭄"),
|
|
("A-1 Cafe (본점)", "a1cafe본점"),
|
|
(None, ""),
|
|
],
|
|
)
|
|
def test_normalize(raw, expected):
|
|
"""검증: 비교용 정규화 규칙."""
|
|
assert lookup._normalize(raw) == expected
|
|
|
|
|
|
def test_nearby_entries_can_steal_the_id__known_limit():
|
|
"""검증: 두 업소가 조회 창(앞 600자 + 뒤 300자)보다 가깝게 붙어 있다."""
|
|
dense = (
|
|
'<li><a href="/place/11111111">앞가게</a></li>'
|
|
'<li><a href="/place/22222222">뒷가게</a></li>'
|
|
)
|
|
assert lookup._match_in_html(dense, "뒷가게") == "11111111"
|