여러 줄 주석이 설명보다 경위(예전·실측·지적)를 적고 있어 읽는 사람이 결론을 찾기 어려웠다. - ts·tsx·js·mjs·css·py 478개: 여러 줄 주석은 첫 문장 한 줄로, 과거형·날짜 문장은 삭제 - 주석 위치는 TypeScript 파서·파이썬 tokenize/ast 로 찾는다 — 문자열 안의 # · /* 는 건드리지 않는다 - eslint·ts·noqa·type: ignore 같은 지시 주석은 그대로 둔다 파이썬 275개 정리 전후 AST 동일, TS 298개 주석 뺀 토큰 동일(빈 JSX 주석 10곳만 차이). site·frontend·admin tsc, site vitest 105 passed Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
291 lines
12 KiB
Python
291 lines
12 KiB
Python
"""수집 어댑터 계약 — 어디서 긁어오든 결과 모양이 같은지, 그리고 법무 게이트가 코드로 지켜지는지."""
|
|
import pytest
|
|
|
|
from common.category_schema import get_schema
|
|
from common.enums import LinkChannel, PlaceCategory
|
|
from services.collector import (
|
|
AdapterDisabled,
|
|
AdapterNotFound,
|
|
AdapterRegistry,
|
|
CollectedFact,
|
|
CollectedMedia,
|
|
CollectError,
|
|
MockAdapter,
|
|
RawSource,
|
|
adapter_ids,
|
|
get_adapter,
|
|
)
|
|
from services.collector.registry import REGISTRY
|
|
|
|
# 어떤 어댑터도 처리하면 안 되는 URL — 조용히 빈 결과를 주지 않고 AdapterNotFound 로 끊어야 한다.
|
|
_REAL_URLS = (
|
|
"https://www.yanolja.com/pension/1000",
|
|
"https://www.goodchoice.kr/product/detail/1000",
|
|
"https://place.map.kakao.com/26338954",
|
|
"https://www.instagram.com/some_cafe",
|
|
)
|
|
|
|
|
|
# ── 목데이터 ↔ 업종 스키마 정합성 (제일 중요) ────────────────────────────
|
|
@pytest.mark.parametrize("category", list(PlaceCategory))
|
|
async def test_mock_fact_keys_exist_in_category_schema(category):
|
|
"""검증: MockAdapter 가 뱉는 fact key 를 업종 스키마와 대조한다."""
|
|
schema = get_schema(category)
|
|
source = await MockAdapter().fetch(MockAdapter.url_for(category))
|
|
|
|
assert source.ok is True
|
|
assert source.facts, f"{schema.name}: 목데이터 fact 가 비었다"
|
|
for fact in source.facts:
|
|
assert schema.has(fact.key), f"{schema.name} 스키마에 없는 key={fact.key}"
|
|
|
|
|
|
@pytest.mark.parametrize("category", list(PlaceCategory))
|
|
async def test_mock_fact_scope_matches_schema(category):
|
|
"""검증: fact 의 scope 가 스키마 정의와 같은지."""
|
|
schema = get_schema(category)
|
|
source = await MockAdapter().fetch(MockAdapter.url_for(category))
|
|
|
|
for fact in source.facts:
|
|
assert fact.scope == schema.get(fact.key).scope, f"{schema.name}.{fact.key}: scope 불일치"
|
|
if fact.scope == "unit":
|
|
assert fact.unit_name, f"{schema.name}.{fact.key}: unit 스코프인데 단위 이름이 없다"
|
|
|
|
|
|
@pytest.mark.parametrize("category", list(PlaceCategory))
|
|
async def test_mock_does_not_collect_llm_written_fields(category):
|
|
"""검증: 수집물에 소개문 계열(allow_llm=True) 필드가 섞이는지."""
|
|
schema = get_schema(category)
|
|
source = await MockAdapter().fetch(MockAdapter.url_for(category))
|
|
|
|
llm_keys = set(schema.llm_writable_keys())
|
|
collected = {f.key for f in source.facts}
|
|
assert not (collected & llm_keys), f"{schema.name}: 수집물에 LLM 작성 필드가 섞였다 — {collected & llm_keys}"
|
|
|
|
|
|
@pytest.mark.parametrize("category", list(PlaceCategory))
|
|
async def test_mock_returns_three_to_five_photos(category):
|
|
"""검증: 사진 수집 결과."""
|
|
source = await MockAdapter().fetch(MockAdapter.url_for(category))
|
|
assert 3 <= len(source.media) <= 5, f"사진 {len(source.media)}장"
|
|
for item in source.media:
|
|
assert item.origin_url.startswith("http")
|
|
|
|
|
|
async def test_mock_is_deterministic():
|
|
"""검증: 같은 URL 로 두 번 수집한다."""
|
|
adapter = MockAdapter()
|
|
url = MockAdapter.url_for(PlaceCategory.LODGING)
|
|
first, second = await adapter.fetch(url), await adapter.fetch(url)
|
|
|
|
assert [(f.key, f.value, f.unit_name) for f in first.facts] == [(f.key, f.value, f.unit_name) for f in second.facts]
|
|
assert [m.origin_url for m in first.media] == [m.origin_url for m in second.media]
|
|
|
|
|
|
async def test_mock_reads_category_and_channel_from_url():
|
|
"""검증: URL 에서 업종·채널을 읽는지."""
|
|
adapter = MockAdapter()
|
|
lodging = await adapter.fetch("mock://lodging/p1")
|
|
cafe = await adapter.fetch("mock://cafe/c1?channel=naver_place")
|
|
|
|
assert "check_in_time" in lodging.fact_map()
|
|
assert "break_time" in cafe.fact_map()
|
|
assert lodging.channel is LinkChannel.ETC
|
|
assert cafe.channel is LinkChannel.NAVER_PLACE
|
|
|
|
|
|
async def test_mock_unknown_category_fails_softly():
|
|
"""검증: 업종을 못 읽는 mock URL."""
|
|
source = await MockAdapter().fetch("mock://unknown/x1")
|
|
assert source.ok is False
|
|
assert source.error and "업종" in source.error
|
|
assert source.facts == []
|
|
|
|
|
|
# ── can_handle / 레지스트리 ──────────────────────────────────────────────
|
|
def test_can_handle_accepts_mock_urls_only():
|
|
"""검증: MockAdapter 의 처리 범위."""
|
|
adapter = MockAdapter()
|
|
assert adapter.can_handle("mock://lodging/1") is True
|
|
assert adapter.can_handle("https://mock.test/cafe/1") is True
|
|
for url in _REAL_URLS:
|
|
assert adapter.can_handle(url) is False, f"MockAdapter 가 실제 URL 을 처리하려 한다: {url}"
|
|
assert adapter.can_handle("") is False
|
|
|
|
|
|
def test_registry_resolves_mock_url():
|
|
"""검증: 레지스트리로 어댑터를 찾는다."""
|
|
assert get_adapter("mock://lodging/1").id == "mock"
|
|
|
|
|
|
@pytest.mark.parametrize("url", _REAL_URLS)
|
|
def test_registry_raises_for_unhandled_url(url):
|
|
"""검증: 등록된 어댑터가 처리 못 하는 URL 을 조회한다."""
|
|
with pytest.raises(AdapterNotFound):
|
|
get_adapter(url)
|
|
|
|
|
|
def test_registry_rejects_duplicate_adapter_id():
|
|
"""검증: 같은 id 의 어댑터를 두 번 등록한다."""
|
|
registry = AdapterRegistry()
|
|
registry.register(MockAdapter())
|
|
with pytest.raises(ValueError):
|
|
registry.register(MockAdapter())
|
|
|
|
|
|
def test_disabled_adapter_is_blocked():
|
|
"""검증: 등록은 됐지만 ENABLED 목록에 없는 어댑터를 조회한다."""
|
|
registry = AdapterRegistry(enabled=frozenset())
|
|
registry.register(MockAdapter())
|
|
with pytest.raises(AdapterDisabled):
|
|
registry.get_adapter("mock://lodging/1")
|
|
assert registry.can_handle("mock://lodging/1") is False
|
|
|
|
|
|
# ── 법무 게이트 (Phase 1) ────────────────────────────────────────────────
|
|
def test_registers_only_reviewed_adapters():
|
|
"""검증: 기본 레지스트리에 등록된 어댑터 목록."""
|
|
assert adapter_ids() == ["naver_place", "tour_api", "yanolja", "mock", "static_html"], (
|
|
f"예상 밖 어댑터가 등록됐다: {adapter_ids()} — 승인 없이 수집 대상을 늘리지 않는다"
|
|
)
|
|
|
|
|
|
def test_static_html_is_registered_last():
|
|
"""검증: 넓은 어댑터(static_html)가 좁은 어댑터보다 뒤에 있는가."""
|
|
assert adapter_ids()[-1] == "static_html", (
|
|
f"static_html 은 맨 뒤여야 한다: {adapter_ids()}"
|
|
)
|
|
|
|
|
|
def test_static_html_never_touches_blocked_platforms():
|
|
"""검증: 수집 불가 결론이 난 플랫폼 URL 을 static_html 에 직접 물어본다."""
|
|
from services.collector.static_html_adapter import StaticHtmlAdapter
|
|
|
|
adapter = StaticHtmlAdapter()
|
|
for url in (
|
|
"https://www.yanolja.com/pension/1000", # 403 + 민사 10억 선례
|
|
"https://nol.yanolja.com/hotels/1",
|
|
"https://www.goodchoice.kr/product/detail?ano=1",
|
|
"https://m.place.naver.com/restaurant/1/home", # robots Disallow: / · 전용 어댑터 있음
|
|
"https://blog.naver.com/somepension",
|
|
"https://place.map.kakao.com/26338954", # 내부 API 406
|
|
"https://www.instagram.com/some_cafe", # Graph API(OAuth)로만
|
|
"https://app.catchtable.co.kr/ct/shop/x",
|
|
):
|
|
assert adapter.can_handle(url) is False, f"건드리면 안 되는 URL 을 받았다: {url}"
|
|
|
|
|
|
def test_static_html_takes_owner_domains():
|
|
"""검증: 사장님 자체 홈페이지로 보이는 평범한 도메인."""
|
|
for url in ("https://gangmunstay.kr/rooms", "https://offinghouse.com/", "https://www.mulhoe.co.kr/"):
|
|
assert get_adapter(url).id == "static_html", f"static_html 이 받지 않는다: {url}"
|
|
|
|
|
|
def test_adapters_do_not_collect_llm_written_fields():
|
|
"""검증: **등록된 모든 어댑터**가 만드는 fact key 에 allow_llm=True 필드가 섞이는지."""
|
|
from pathlib import Path
|
|
|
|
banned = set()
|
|
for category in PlaceCategory:
|
|
banned |= set(get_schema(category).llm_writable_keys())
|
|
assert banned, "allow_llm 필드가 하나도 없다 — 스키마 로딩이 잘못됐다"
|
|
|
|
package = Path(__file__).resolve().parents[1] / "services" / "collector"
|
|
for path in package.glob("*_adapter.py"):
|
|
body = path.read_text(encoding="utf-8")
|
|
for key in banned:
|
|
for pattern in (f'key="{key}"', f'("{key}",'):
|
|
for line in body.splitlines():
|
|
stripped = line.strip()
|
|
if pattern in stripped and not stripped.startswith("#"):
|
|
raise AssertionError(
|
|
f"{path.name}: LLM 작성 필드 '{key}' 를 수집한다 — {stripped[:80]}"
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize("url", _REAL_URLS)
|
|
def test_unregistered_sites_are_not_reachable(url):
|
|
"""검증: 어댑터가 없는 사이트 URL 로 수집을 시도한다."""
|
|
assert REGISTRY.can_handle(url) is False
|
|
|
|
|
|
def test_naver_place_hosts_are_handled():
|
|
"""검증: 사람이 실제로 공유하는 네이버 플레이스 주소들."""
|
|
for url in (
|
|
"https://map.naver.com/p/entry/place/1133638931",
|
|
"https://m.place.naver.com/accommodation/1133638931/home",
|
|
"https://pcmap.place.naver.com/accommodation/1133638931/home",
|
|
"https://naver.me/xxxxxxx",
|
|
):
|
|
assert REGISTRY.can_handle(url) is True, f"처리하지 못한다: {url}"
|
|
|
|
|
|
def test_no_evasion_code_in_collector_package():
|
|
"""검증: 수집 패키지에 우회 코드가 들어왔는지 소스를 훑는다."""
|
|
from pathlib import Path
|
|
|
|
banned = ("captcha", "solve_captcha", "rotate_ip", "proxy_rotate", "stealth", "undetected")
|
|
package = Path(__file__).resolve().parents[1] / "services" / "collector"
|
|
for path in package.glob("*.py"):
|
|
body = path.read_text(encoding="utf-8").lower()
|
|
for token in banned:
|
|
# 금지 사실을 적어둔 주석은 허용한다 — 실제 식별자로 쓰였는지만 본다.
|
|
for line in body.splitlines():
|
|
stripped = line.strip()
|
|
if token in stripped and not stripped.startswith("#"):
|
|
raise AssertionError(f"{path.name}: 금지된 우회 코드 흔적 '{token}' — {stripped[:80]}")
|
|
|
|
|
|
# ── 출처 강제 ────────────────────────────────────────────────────────────
|
|
async def test_every_fact_and_media_carries_source_url():
|
|
"""검증: 수집 결과의 모든 fact·사진이 출처를 들고 있는지."""
|
|
url = MockAdapter.url_for(PlaceCategory.LODGING, "pension-9")
|
|
source = await MockAdapter().fetch(url)
|
|
|
|
assert source.source_url == url
|
|
for item in (*source.facts, *source.media):
|
|
assert item.source_url == url, f"출처 없이 흘러가는 항목: {item}"
|
|
|
|
|
|
def test_raw_source_stamps_source_url_on_bare_items():
|
|
"""검증: source_url 을 비운 채 항목을 넣어 RawSource 를 만든다."""
|
|
source = RawSource(
|
|
url="mock://lodging/x",
|
|
adapter_id="mock",
|
|
facts=[CollectedFact(key="wifi", value="true")],
|
|
media=[CollectedMedia(origin_url="https://mock.test/img/1.jpg")],
|
|
)
|
|
assert source.facts[0].source_url == "mock://lodging/x"
|
|
assert source.media[0].source_url == "mock://lodging/x"
|
|
|
|
|
|
def test_raw_source_requires_url():
|
|
"""검증: url 없이 RawSource 를 만든다."""
|
|
with pytest.raises(CollectError):
|
|
RawSource(url="", adapter_id="mock")
|
|
|
|
|
|
def test_collected_media_requires_origin_url():
|
|
"""검증: origin_url 없이 사진을 담는다."""
|
|
with pytest.raises(CollectError):
|
|
CollectedMedia(origin_url=" ")
|
|
|
|
|
|
def test_unit_scoped_fact_requires_unit_name():
|
|
"""검증: 단위 이름 없이 unit 스코프 fact 를 만든다."""
|
|
with pytest.raises(CollectError):
|
|
CollectedFact(key="standard_capacity", value="2", scope="unit")
|
|
|
|
|
|
async def test_unit_names_are_listed_in_order():
|
|
"""검증: 수집된 단위 이름 목록."""
|
|
source = await MockAdapter().fetch(MockAdapter.url_for(PlaceCategory.LODGING))
|
|
assert source.unit_names() == ["A동 스탠다드", "B동 복층"]
|
|
|
|
|
|
async def test_failure_result_has_no_facts():
|
|
"""검증: 실패 결과(RawSource.failure)의 모양."""
|
|
source = RawSource.failure("mock://lodging/1", "mock", "타임아웃")
|
|
assert source.ok is False and source.error == "타임아웃"
|
|
assert source.facts == [] and source.media == []
|
|
assert source.source_url == "mock://lodging/1"
|