여러 줄 주석이 설명보다 경위(예전·실측·지적)를 적고 있어 읽는 사람이 결론을 찾기 어려웠다. - ts·tsx·js·mjs·css·py 478개: 여러 줄 주석은 첫 문장 한 줄로, 과거형·날짜 문장은 삭제 - 주석 위치는 TypeScript 파서·파이썬 tokenize/ast 로 찾는다 — 문자열 안의 # · /* 는 건드리지 않는다 - eslint·ts·noqa·type: ignore 같은 지시 주석은 그대로 둔다 파이썬 275개 정리 전후 AST 동일, TS 298개 주석 뺀 토큰 동일(빈 JSX 주석 10곳만 차이). site·frontend·admin tsc, site vitest 105 passed Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
152 lines
5.6 KiB
Python
152 lines
5.6 KiB
Python
"""업종별 fact 스키마 — 업종마다 어떤 key 가 존재하는지의 유일한 소스."""
|
|
import json
|
|
from pathlib import Path
|
|
|
|
from common.enums import PlaceCategory
|
|
|
|
_RESOURCE_DIR = Path(__file__).parent / "resources"
|
|
|
|
_ALLOWED_TYPES = ("text", "number", "bool", "time")
|
|
_ALLOWED_SCOPES = ("place", "unit")
|
|
|
|
_schemas: dict | None = None # category code -> CategorySchema
|
|
|
|
|
|
class CategorySchemaError(RuntimeError):
|
|
"""업종 스키마 로드/검증 실패."""
|
|
|
|
|
|
class FieldSpec:
|
|
"""업종 스키마의 필드 1개."""
|
|
|
|
__slots__ = ("key", "label", "type", "scope", "required", "critical", "allow_llm", "unit")
|
|
|
|
def __init__(self, row: dict, source: str):
|
|
for name in ("key", "label", "type", "scope"):
|
|
if not isinstance(row.get(name), str) or not row[name]:
|
|
raise CategorySchemaError(f"{source}: 필드 '{name}' 이 비었거나 문자열이 아님 — {row}")
|
|
if row["type"] not in _ALLOWED_TYPES:
|
|
raise CategorySchemaError(f"{source}: type 값 오류 key={row['key']} type={row['type']} (허용 {_ALLOWED_TYPES})")
|
|
if row["scope"] not in _ALLOWED_SCOPES:
|
|
raise CategorySchemaError(f"{source}: scope 값 오류 key={row['key']} scope={row['scope']} (허용 {_ALLOWED_SCOPES})")
|
|
for name in ("required", "critical", "allow_llm"):
|
|
if not isinstance(row.get(name), bool):
|
|
raise CategorySchemaError(f"{source}: '{name}' 은 bool 이어야 함 key={row['key']} value={row.get(name)}")
|
|
|
|
self.key = row["key"]
|
|
self.label = row["label"]
|
|
self.type = row["type"]
|
|
self.scope = row["scope"]
|
|
self.required = row["required"]
|
|
self.critical = row["critical"]
|
|
self.allow_llm = row["allow_llm"]
|
|
self.unit = row.get("unit")
|
|
|
|
def to_dict(self) -> dict:
|
|
return {name: getattr(self, name) for name in self.__slots__}
|
|
|
|
|
|
class CategorySchema:
|
|
"""업종 1개의 fact 스키마."""
|
|
|
|
def __init__(self, doc: dict, source: str):
|
|
code = doc.get("code")
|
|
try:
|
|
self.category = PlaceCategory(code)
|
|
except ValueError as ex:
|
|
raise CategorySchemaError(f"{source}: PlaceCategory 에 없는 code={code}") from ex
|
|
self.name = doc.get("category")
|
|
self.label = doc.get("label")
|
|
if not isinstance(self.name, str) or not isinstance(self.label, str):
|
|
raise CategorySchemaError(f"{source}: category/label 이 문자열이 아님")
|
|
|
|
rows = doc.get("fields")
|
|
if not isinstance(rows, list) or not rows:
|
|
raise CategorySchemaError(f"{source}: fields 가 비었음")
|
|
|
|
self.fields: dict[str, FieldSpec] = {}
|
|
for row in rows:
|
|
spec = FieldSpec(row, source)
|
|
if spec.key in self.fields:
|
|
raise CategorySchemaError(f"{source}: key 중복 — {spec.key}")
|
|
self.fields[spec.key] = spec
|
|
|
|
@property
|
|
def code(self) -> int:
|
|
return self.category.value
|
|
|
|
def get(self, key: str) -> FieldSpec | None:
|
|
return self.fields.get(key)
|
|
|
|
def has(self, key: str) -> bool:
|
|
return key in self.fields
|
|
|
|
def keys_by_scope(self, scope: str) -> list[str]:
|
|
return [k for k, f in self.fields.items() if f.scope == scope]
|
|
|
|
def required_keys(self, scope: str | None = None) -> list[str]:
|
|
"""발행 검수 게이트가 존재를 확인하는 필수 key 목록."""
|
|
return [k for k, f in self.fields.items() if f.required and (scope is None or f.scope == scope)]
|
|
|
|
def critical_keys(self) -> list[str]:
|
|
"""미검증 상태로 노출하면 안 되는 key 목록(체크인·취사·반려동물·취소 규정 등)."""
|
|
return [k for k, f in self.fields.items() if f.critical]
|
|
|
|
def llm_writable_keys(self) -> list[str]:
|
|
"""LLM 이 값을 만들어도 되는 key 목록."""
|
|
return [k for k, f in self.fields.items() if f.allow_llm]
|
|
|
|
|
|
def load_schemas() -> None:
|
|
"""리소스 디렉터리 전체 로드 + 검증."""
|
|
global _schemas
|
|
if _schemas is not None:
|
|
return
|
|
|
|
paths = sorted(_RESOURCE_DIR.glob("*.json"))
|
|
if not paths:
|
|
raise CategorySchemaError(f"업종 스키마 파일이 없습니다: {_RESOURCE_DIR}")
|
|
|
|
loaded: dict[int, CategorySchema] = {}
|
|
for path in paths:
|
|
try:
|
|
doc = json.loads(path.read_text(encoding="utf-8"))
|
|
except Exception as ex:
|
|
raise CategorySchemaError(f"업종 스키마 파일 로드 실패: {path}: {ex}") from ex
|
|
schema = CategorySchema(doc, path.name)
|
|
if schema.code in loaded:
|
|
raise CategorySchemaError(f"업종 code 중복: {schema.code} ({path.name})")
|
|
loaded[schema.code] = schema
|
|
|
|
missing = [c.name for c in PlaceCategory if c.value not in loaded]
|
|
if missing:
|
|
raise CategorySchemaError(f"PlaceCategory 에 있으나 스키마 파일이 없는 업종: {missing}")
|
|
|
|
_schemas = loaded
|
|
|
|
|
|
def get_schema(category) -> CategorySchema:
|
|
"""업종 코드(int 또는 PlaceCategory) → 스키마."""
|
|
if _schemas is None:
|
|
load_schemas()
|
|
code = category.value if isinstance(category, PlaceCategory) else category
|
|
schema = _schemas.get(code)
|
|
if schema is None:
|
|
raise CategorySchemaError(f"지원하지 않는 업종 코드: {code}")
|
|
return schema
|
|
|
|
|
|
def all_schemas() -> dict:
|
|
"""전 업종 스키마."""
|
|
if _schemas is None:
|
|
load_schemas()
|
|
return dict(_schemas)
|
|
|
|
|
|
def is_valid_key(category, key: str) -> bool:
|
|
"""해당 업종에 존재하는 fact key 인지."""
|
|
try:
|
|
return get_schema(category).has(key)
|
|
except CategorySchemaError:
|
|
return False
|