diff --git a/.env.example b/.env.example index 30db340..056af5d 100644 --- a/.env.example +++ b/.env.example @@ -47,6 +47,8 @@ SUNO_CALLBACK_URL=https://example.com/api/suno/callback # ★ 워커가 부르는 주소다. compose 로 띄우면 컨테이너 안에서 보는 주소(http://host.docker.internal:3100), # 백엔드를 네이티브로 돌리면 http://127.0.0.1:3100 SITE_ONTOLOGY_URL= +# 새 숙소 키워드를 OpenAI 로 만든다(OPENAI_API_KEY 를 함께 쓴다). mock 이면 생성분이 매칭에 나가지 않는다. +ONTOLOGY_LLM_PROVIDER=mock # 프리렌더가 절대 굽지 않는 슬러그(쉼표 구분). 손으로 만든 목업(/s/stay·stay2·stay3·stay4·stay5) # 이름과 같은 슬러그로 실제 발행이 생기면 그 payload 로 목업을 덮어 구워버린다 — 비우지 않는다. diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index e7ed3e2..d6f898b 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -42,7 +42,7 @@ React 를 렌더해야 하고, 그때부터 디자인 수정에 백엔드 배포 BUILD 잡 (worker) ─ services/build_service.py:99 run_build() ├ build_snapshot → 원본 데이터(JSONB) ├ seo_keywords.fetch() → SiteOntology 추천을 이 가게 자료로 거른 키워드 → snapshot["seo"] - │ (숙박만 · 설정 없거나 실패하면 생략 · 발행은 계속) + │ (숙박만 · 처음 보는 숙소는 OpenAI 키워드 생성을 기다림 · 실패하면 생략 · 발행은 계속) │ → site_versions 행 insert ├ 1차 게이트 (DB 사실 기준) → publish_gate.evaluate() ├ site_payload.emit_payload() → out/payloads/.json ★ 백엔드의 유일한 산출물 diff --git a/docs/DEVLOG.md b/docs/DEVLOG.md index 5f9c360..45b7ae1 100644 --- a/docs/DEVLOG.md +++ b/docs/DEVLOG.md @@ -4,6 +4,25 @@ **나중에 같은 실수를 막아 주는 것**(결정의 이유·밟은 함정·실측값)만 남긴다. 2026-09-29에 요약본으로 다시 썼다. 원문 전체는 git 히스토리(이 파일의 09-29 이전 버전)에 있다. +## 2026-10-01 — 온톨로지: 새 숙소 키워드를 OpenAI 로 쌓는다 · 호텔 업종 · 숙소 검색 + +- 발행이 `generate:false` 로 불러 사전이 고정이었다 — 사전에 없는 숙소는 새 단어를 못 받았다. + 이제 **처음 보는 숙소는 첫 발행에서 생성을 기다린다**(실측 7초, 키워드 15·질문 5). 재빌드·재발행은 + 30일 안이면 부르지 않는다 — 재빌드마다 부르는 창구라 조건이 없으면 저장마다 요금이 나간다. +- 생성분이 매칭에 안 나오던 이유는 `MATCH_SOURCES` 가 `llm` 을 뺀 것. 업종 뿌리(`stay`)로 거르고 + brand 는 그 업체 것만 받게 해서 넣었다. **mock 생성분은 `source='mock'`** — compose 기본이 mock 이라 + 키 없이 띄운 환경마다 가짜 단어가 발행본에 나갈 뻔했다. +- 호텔(`stay.hotel`): 외부 분류(`places.external_category`, 없으면 상호)에 "호텔" 이면. 매칭은 펜션 사전도 + 받되 메타 게이트가 **자기 유형어만** 자료 없이 통과시킨다 — 호텔에 `군산 독채펜션` 이 붙지 않는다. +- `/v1/lodgings/search`: ★ 가까운 키워드를 사전 전체(7천)에서 뽑으면 상위가 아무 숙소에도 안 붙은 + 데이터셋 단어로 차서 `선유도 독채` 가 0건이었다 → **숙소에 연결된 키워드만** 비교한다. + ★ e5-small 은 지역이 틀려도 0.85 가 나온다(`여수 호텔` → 군산 호텔 0.848) — 지역은 임계값이 아니라 + **지역 이름**으로 가른다. +- ★ 로컬에서 임베딩 모델 내려받기가 Node 에서 `ECONNRESET`(terminated)으로 끊겨 3분 걸리고 실패했다. + curl 로는 받힌다. 실패한 로딩 promise 가 프로세스에 남아 **재시작 전까지 모든 임베딩이 500**이다. +- 실키 검증(가짜 숙소 2곳): 메타에 `군산 시내 호텔`·`군산 오션뷰 호텔` 이 실렸고 미확인 `주차`·`애견동반` + 은 안 나왔다. 백엔드 `1041 passed / 47 failed` — 47건은 이 변경 전 main 과 같은 목록. + ## 2026-09-30 — 템플릿 검수 · 코랄 · 미니멀 · 솔숲 추가 - 한국 펜션 사이트(코랄트리 · 바다동화)를 참고해 `coral` · `minimal`, 디자인 스킬 시안에서 `pine`(솔숲). diff --git a/ontology/README.md b/ontology/README.md index ee479e7..18d4a9d 100644 --- a/ontology/README.md +++ b/ontology/README.md @@ -2,7 +2,9 @@ o2o-site-AEO 가 발행한 사이트에 **업체별 SEO/AEO 키워드**를 제공하는 온톨로지 서비스. -- **고정 데이터셋 1회 적재** 정책 — 주기 수집 없음 (`data/gunsan-pension-keywords.json`, 1,000건) +- 사전은 **고정 데이터셋 + 업체별 LLM 생성분**이다. 처음 보는 숙소가 발행되면 OpenAI 로 키워드를 만들어 + 쌓고(`source='llm'`), 이후 매칭 후보가 된다. mock 이 만든 단어는 `source='mock'` 으로 남아 후보가 되지 않는다 +- 업종은 숙박(`stay`) 아래 펜션(`stay.pension`)·호텔(`stay.hotel`). 매칭·검색은 뿌리(`stay`)로 묶는다 - 로컬 임베딩(`multilingual-e5-small`, 384차원)으로 pgvector 에 적재 후 의미 검색 - 업체명 또는 자연어 문장 → 사전에서 잘 맞는 키워드를 골라주는 **매칭 API + 데모 콘솔** - 어휘 단계 중복제거는 자동, 벡터 근접쌍은 자동 병합하지 않고 검토 목록으로만 @@ -245,13 +247,14 @@ npm run dataset:ingest && npm run dataset:ingest-nationwide | 메서드 | 경로 | 용도 | |---|---|---| | `GET` | `/health` | 헬스체크 | -| `POST` | `/v1/merchants/publish` | **사이트 발행 웹훅** — 업체 upsert + 생성 예약 (`sync:true` 면 동기 실행) | +| `POST` | `/v1/merchants/publish` | **사이트 발행 웹훅** — 업체 upsert + 생성. 처음 보는 업체·`REFRESH_INTERVAL_DAYS` 지난 업체만 생성하고, `sync:true` 는 처음 보는 업체에만 동기로 기다린다. 동기 생성이 실패해도 200 (`generation.failed`) | | `POST` | `/v1/merchants/:id/generate?sync=true` | 수동 재생성 | | `GET` | `/v1/merchants` `/v1/merchants/:id` | 조회 | | `GET` | `/v1/sites/:id/seo?limit=20` | **발행 사이트가 렌더링 시 호출** — title/description/keywords/tags | | `GET` | `/v1/sites/:id/aeo?limit=10` | 답변엔진용 topics/FAQ/structuredDataHints | | `POST` | `/v1/match` | **업체명 또는 문장 → 사전에서 잘 맞는 키워드.** `mode=fusion`(기본) / `single`(통짜, 비교용) | | `POST` | `/v1/keywords/search` | 의미 기반 키워드 검색 (어드민) | +| `POST` | `/v1/lodgings/search` | **검색어 → 맞는 숙소와 각 숙소의 SEO 메타.** 숙소에 연결된 키워드와의 유사도(≥0.87, 표기 변형은 aliases) + 지역 이름. 검색어가 지역을 말하면 그 지역 숙소만 | | `GET` | `/demo` | 매칭 콘솔 (로컬 확인용) | | `POST` | `/v1/sites/:id/performance` | Search Console·유입 로그 피드백 → 저성과 키워드 강등 | @@ -401,7 +404,8 @@ curl -s -X POST http://localhost:3100/v1/match \ ## 생성 주기 -- **발행 즉시** — `/v1/merchants/publish` 가 BullMQ 에 적재 (60초 dedupe 창) +- **첫 발행** — 처음 보는 업체는 `/v1/merchants/publish` 가 동기로 생성한다 (o2o-site-AEO 가 `sync:true` 로 부른다) +- **기한 지난 업체의 발행** — BullMQ 에 적재 (60초 dedupe 창). 기한 안이면 생성하지 않는다(`generation: 'fresh'`) - **주기 리프레시** — 매일 03:00 크론이 `REFRESH_INTERVAL_DAYS`(기본 30일) 지난 업체를 적재 - **성과 기반** — 노출 100회 이상 & CTR < 0.2% 인 키워드는 `demoted` 로 강등, 다음 사이클에서 대체 diff --git a/ontology/drizzle/0002_industry_hotel.sql b/ontology/drizzle/0002_industry_hotel.sql new file mode 100644 index 0000000..94d619d --- /dev/null +++ b/ontology/drizzle/0002_industry_hotel.sql @@ -0,0 +1,4 @@ +-- 호텔 업종. o2o-site-AEO 가 외부 분류에 "호텔" 이 있는 숙소를 stay.hotel 로 보낸다. +-- ★ 시드에만 두면 이미 떠 있는 DB 에는 없고, merchant.industry_id 외래키 위반으로 publish 가 500 이 된다. +INSERT INTO industry (id, path, name) VALUES ('stay.hotel', 'stay.hotel', '호텔') +ON CONFLICT (id) DO NOTHING; diff --git a/ontology/src/db/seed.ts b/ontology/src/db/seed.ts index e017f0c..19ed8ba 100644 --- a/ontology/src/db/seed.ts +++ b/ontology/src/db/seed.ts @@ -10,6 +10,7 @@ const industries = [ ['health.dental', 'health.dental', '치과'], ['stay', 'stay', '숙박'], ['stay.pension', 'stay.pension', '펜션'], + ['stay.hotel', 'stay.hotel', '호텔'], ]; const regions = [ diff --git a/ontology/src/generation/generation.service.ts b/ontology/src/generation/generation.service.ts index b866415..33eff1f 100644 --- a/ontology/src/generation/generation.service.ts +++ b/ontology/src/generation/generation.service.ts @@ -103,6 +103,7 @@ export class GenerationService { locale: 'ko-KR', industryId: merchant.industry_id, regionId: merchant.region_id, + source: keywordSource(this.llm.name), }); stats.details.push({ @@ -214,3 +215,8 @@ export class GenerationService { return created; } } + +// mock 이 만든 단어를 'llm' 으로 두면 매칭 후보가 되어 가짜 단어가 발행본에 나간다. +function keywordSource(provider: string): string { + return provider === 'mock' ? 'mock' : 'llm'; +} diff --git a/ontology/src/keywords/dedup.service.ts b/ontology/src/keywords/dedup.service.ts index b2ddf04..01772af 100644 --- a/ontology/src/keywords/dedup.service.ts +++ b/ontology/src/keywords/dedup.service.ts @@ -26,6 +26,8 @@ export interface ResolveInput { locale: string; industryId: string | null; regionId: string | null; + /** 새로 등록될 때 keyword.source — 매칭 후보 여부가 이 값으로 갈린다 */ + source: string; } /** 4단계 계단식 중복제거. */ @@ -98,6 +100,7 @@ export class DedupService { embedding: input.embedding, industryId: input.industryId, regionId: input.regionId, + source: input.source, }); return { action: 'created', keywordId: created.id, canonical: created.canonical }; } diff --git a/ontology/src/keywords/keyword.repository.ts b/ontology/src/keywords/keyword.repository.ts index b87952a..ade6957 100644 --- a/ontology/src/keywords/keyword.repository.ts +++ b/ontology/src/keywords/keyword.repository.ts @@ -81,11 +81,12 @@ export class KeywordRepository { embedding: number[]; industryId: string | null; regionId: string | null; + source: string; }): Promise { const rows = await this.sql` - INSERT INTO keyword (canonical, normalized, locale, intent, embedding, industry_id, region_id, usage_count) + INSERT INTO keyword (canonical, normalized, locale, intent, embedding, industry_id, region_id, usage_count, source) VALUES (${input.canonical}, ${input.normalized}, ${input.locale}, ${input.intent}, - ${toVector(input.embedding)}::vector, ${input.industryId}, ${input.regionId}, 0) + ${toVector(input.embedding)}::vector, ${input.industryId}, ${input.regionId}, 0, ${input.source}) ON CONFLICT (normalized, locale) DO UPDATE SET updated_at = now() RETURNING id, canonical, normalized, aliases, intent, usage_count`; return rows[0]; diff --git a/ontology/src/llm/openai.provider.ts b/ontology/src/llm/openai.provider.ts index 1b90db7..f6fbd56 100644 --- a/ontology/src/llm/openai.provider.ts +++ b/ontology/src/llm/openai.provider.ts @@ -92,6 +92,10 @@ const SYSTEM_PROMPT = `당신은 한국 로컬 비즈니스 SEO/AEO 전문가입 - 지역명 + 업종 + 의도어 조합을 적극 활용합니다. - 과장광고 표현(최고, 1위, 100%, 무조건, 완치 등)은 절대 사용하지 않습니다. - 제공된 "이미 보유한 키워드"와 의미가 겹치는 것은 생성하지 않습니다. +- 시설·전망·객실 유형은 "상세"의 features·nearby 에 있는 것만 씁니다. 없는 것을 지어내지 않습니다. +- "상세"의 unverified 에 있는 항목은 확인되지 않은 것이므로 키워드와 답변 어디에도 쓰지 않습니다. +- 숙박 유형어는 주어진 업종의 것만 씁니다. 업종이 호텔이면 '펜션'을, 펜션이면 '호텔'을 붙이지 않습니다. +- 가격·최저가·할인 같은 가격 주장은 쓰지 않습니다. - relevance 는 해당 업체와의 관련도를 0.0~1.0 으로 매깁니다. - 답변(answer)은 2~3문장, 업체 정보에 근거한 사실만 씁니다.`; diff --git a/ontology/src/merchants/merchants.controller.ts b/ontology/src/merchants/merchants.controller.ts index b0eed6f..a2036c7 100644 --- a/ontology/src/merchants/merchants.controller.ts +++ b/ontology/src/merchants/merchants.controller.ts @@ -1,10 +1,13 @@ -import { Body, Controller, Get, Param, Post, Query } from '@nestjs/common'; +import { Body, Controller, Get, Logger, Param, Post, Query } from '@nestjs/common'; +import { env } from '../config/env'; import { GenerationService } from '../generation/generation.service'; import { GenerationQueue } from '../generation/generation.queue'; import { MerchantsService, UpsertMerchantDto } from './merchants.service'; @Controller('v1/merchants') export class MerchantsController { + private readonly logger = new Logger(MerchantsController.name); + constructor( private readonly merchants: MerchantsService, private readonly generation: GenerationService, @@ -21,15 +24,25 @@ export class MerchantsController { return this.merchants.findWithTaxonomy(id); } - /** o2o-site-AEO 사이트 발행 웹훅: 업체 등록 + 키워드 생성 예약 */ + /** 발행 웹훅 — 재빌드마다 불리므로 처음 보는 업체·기한 지난 업체만 생성해 LLM 요금을 묶는다 */ @Post('publish') async publish(@Body() dto: UpsertMerchantDto & { generate?: boolean; sync?: boolean }) { const merchant = await this.merchants.upsert(dto); if (dto.generate === false) return { merchant, generation: 'skipped' }; + if (!this.merchants.isStale(merchant, env.generation.refreshIntervalDays)) { + return { merchant, generation: 'fresh' }; + } - if (dto.sync) { - const stats = await this.generation.runForMerchant(merchant.id, 'published'); - return { merchant, generation: stats }; + // 첫 발행부터 새 키워드가 실리도록 처음 보는 업체만 기다린다. 실패해도 사전 키워드로 발행은 나간다. + if (dto.sync && merchant.last_generated_at === null) { + try { + const stats = await this.generation.runForMerchant(merchant.id, 'published'); + return { merchant, generation: { ...stats, details: undefined } }; + } catch (err) { + const message = err instanceof Error ? err.message : String(err); + this.logger.warn(`[${merchant.name}] 동기 생성 실패 — 사전 키워드로만 매칭된다: ${message}`); + return { merchant, generation: { failed: true, error: message } }; + } } const jobId = await this.queue.enqueue(merchant.id, 'published'); return { merchant, generation: { queued: true, jobId } }; diff --git a/ontology/src/merchants/merchants.service.ts b/ontology/src/merchants/merchants.service.ts index 1ca386f..cd6ffc4 100644 --- a/ontology/src/merchants/merchants.service.ts +++ b/ontology/src/merchants/merchants.service.ts @@ -83,6 +83,12 @@ export class MerchantsService { LIMIT ${limit}`; } + /** findStale 과 같은 기준을 행 하나에 — publish 가 생성 여부를 정할 때 쓴다 */ + isStale(merchant: Pick, intervalDays: number): boolean { + if (!merchant.last_generated_at) return true; + return Date.now() - new Date(merchant.last_generated_at).getTime() > intervalDays * 86_400_000; + } + async markGenerated(merchantId: string) { await this.sql`UPDATE merchant SET last_generated_at = now() WHERE id = ${merchantId}`; } diff --git a/ontology/src/serving/match.rules.ts b/ontology/src/serving/match.rules.ts index c7a491b..b92f0e1 100644 --- a/ontology/src/serving/match.rules.ts +++ b/ontology/src/serving/match.rules.ts @@ -114,7 +114,7 @@ export function checkAmenity(keyword: string, facts: MerchantFacts): AmenityVerd } // ──────────────────────────────────────────── 서브 질의 빌더 -const STAY_TYPE_HINTS = ['독채', '풀빌라', '스테이', '펜션', '글램핑', '카라반', '한옥', '민박', '감성']; +const STAY_TYPE_HINTS = ['독채', '풀빌라', '스테이', '펜션', '호텔', '리조트', '글램핑', '카라반', '한옥', '민박', '감성']; const CAPACITY_TOKEN = /\d+\s*인|기준|최대|소규모|중규모|대규모|수용/; /** 레인 설계 원칙 1. 레인끼리 겹치지 않게 한다. */ diff --git a/ontology/src/serving/match.service.ts b/ontology/src/serving/match.service.ts index 10df7b3..d0d197f 100644 --- a/ontology/src/serving/match.service.ts +++ b/ontology/src/serving/match.service.ts @@ -15,6 +15,8 @@ const LANE_DEPTH = 50; // 레인당 후보 깊이 — 깊을수록 generic const LANE_FLOOR = 0.80; // 이 코사인 미만은 그 레인에서 기여하지 않는다 // 매칭 후보로 인정하는 출처. const MATCH_SOURCES = ['dataset', 'nationwide', 'manual']; +// LLM 이 만든 키워드도 후보다. 단 intent=brand 는 남의 상호이므로 그 업체 것만 받는다. +const LLM_SOURCE = 'llm'; interface Hit { id: string; canonical: string; intent: string; kind: string; @@ -49,8 +51,9 @@ export class MatchService { const vectors = await this.embedder.embed(lanes.map((l) => l.text), 'query'); // 레인별 검색. + // 잎(stay.pension)이 아니라 뿌리(stay)로 좁힌다 — 펜션·호텔이 서로의 관련 키워드를 받고, 틀린 유형어는 호출측이 뺀다. const perLane = await Promise.all( - vectors.map((v) => this.laneSearch(v, LANE_DEPTH, merchant?.industry_id ?? null)), + vectors.map((v) => this.laneSearch(v, LANE_DEPTH, industryRoot(merchant?.industry_path), merchant?.id ?? null)), ); // 가중 RRF 융합 @@ -138,17 +141,22 @@ export class MatchService { } private async laneSearch( - embedding: number[], limit: number, industryId: string | null, + embedding: number[], limit: number, root: string | null, merchantId: string | null, ): Promise { const vec = toVector(embedding); const rows = await this.sql` - SELECT id, canonical, intent, kind, category, aliases, - 1 - (embedding <=> ${vec}::vector) AS score - FROM keyword - WHERE embedding IS NOT NULL - AND source = ANY(${MATCH_SOURCES}) - AND (${industryId}::text IS NULL OR industry_id IS NULL OR industry_id = ${industryId}) - ORDER BY embedding <=> ${vec}::vector + SELECT k.id, k.canonical, k.intent, k.kind, k.category, k.aliases, + 1 - (k.embedding <=> ${vec}::vector) AS score + FROM keyword k + LEFT JOIN industry i ON i.id = k.industry_id + WHERE k.embedding IS NOT NULL + AND (k.source = ANY(${MATCH_SOURCES}) + OR (k.source = ${LLM_SOURCE} + AND (k.intent <> 'brand' + OR EXISTS (SELECT 1 FROM merchant_keyword mk + WHERE mk.keyword_id = k.id AND mk.merchant_id = ${merchantId}::uuid)))) + AND (${root}::text IS NULL OR k.industry_id IS NULL OR i.path <@ ${root}::ltree) + ORDER BY k.embedding <=> ${vec}::vector LIMIT ${limit}`; return rows.map((r) => ({ ...r, score: Number(r.score) })); } @@ -185,11 +193,16 @@ export class MatchService { async dictionarySize() { const [row] = await this.sql>` SELECT count(*)::int AS n FROM keyword - WHERE embedding IS NOT NULL AND source = ANY(${MATCH_SOURCES})`; + WHERE embedding IS NOT NULL AND (source = ANY(${MATCH_SOURCES}) OR source = ${LLM_SOURCE})`; return row?.n ?? 0; } } +/** 'stay.pension' → 'stay'. 업종이 없으면 null(좁히지 않는다) */ +export function industryRoot(path: string | null | undefined): string | null { + return path ? path.split('.')[0] : null; +} + function str(v: unknown): string[] { return Array.isArray(v) ? v.filter((x): x is string => typeof x === 'string') : []; } diff --git a/ontology/src/serving/serving.service.ts b/ontology/src/serving/serving.service.ts index 5b25b54..2da743a 100644 --- a/ontology/src/serving/serving.service.ts +++ b/ontology/src/serving/serving.service.ts @@ -7,11 +7,11 @@ import { EmbeddingProvider } from '../embedding/types'; import { MerchantsService, MerchantWithTaxonomy } from '../merchants/merchants.service'; const LODGING_ROOT = 'stay'; -// 검색어 하나당 살펴볼 가까운 키워드 수. 한 숙소가 연결한 키워드는 15개 안팎이라 이 정도면 겹친다. +// 검색어 하나당 살펴볼 가까운 키워드 수(숙소에 연결된 키워드 중에서). const SEARCH_KEYWORD_DEPTH = 60; -// ★ 짧은 한글 키워드는 전혀 다른 말도 코사인이 높다(dedup.service 실측: '선유도 펜션' ↔ '새만금 펜션' = 0.936). -// 그래서 벡터 갈래는 이 값 이상만 "같은 뜻" 으로 치고, 지역 구분은 벡터가 아니라 위치 갈래에 맡긴다. -const SEARCH_FLOOR = 0.94; +// e5-small 에서 맞는 숙소는 0.87 이상, 동떨어진 숙소는 0.86 아래로 갈린다. 지역 구분은 이 값이 아니라 지역 이름으로 한다. +// ponytail: 고정 임계값 — 숙소가 늘어 경계가 흐려지면 1위 대비 상대 컷으로 바꾼다. +const SEARCH_FLOOR = 0.87; const SEARCH_SEO_KEYWORDS = 20; export interface SeoPayload { @@ -181,18 +181,7 @@ export class ServingService { return row?.n ?? 0; } - /** - * 검색어 → 그 말로 찾을 만한 숙소와 각 숙소의 SEO 메타. - * - * 두 갈래를 합친다. - * 키워드 — 검색어와 가까운 키워드(표기 변형은 aliases 로 흡수돼 있다)를 **가진** 숙소 - * 위치 — 검색어에 지역 이름이 들어 있으면 그 지역(하위 포함) 숙소 - * 위치가 맞는 숙소를 앞에 둔다. `군산 오션뷰` 로 찾는데 여수 오션뷰 숙소가 먼저 나오면 틀린 답이다. - * - * ★ 업종은 숙박 뿌리(stay.*)로 묶는다 — 펜션을 찾아도 관련 키워드를 가진 호텔이 함께 나온다(2026-09-30 지시). - * ★ 키워드 갈래는 merchant_keyword(생성 때 연결된 것)만 본다. /v1/match 의 추천은 저장되지 않으므로 - * 생성이 한 번도 안 돈 숙소는 위치 갈래로만 잡힌다. - */ + /** 검색어 → 맞는 숙소(숙박 전체)와 SEO 메타. 키워드 갈래 + 지역 갈래, 지역을 말하면 그 지역만 */ async searchLodgings(rawQuery: string, limit: number) { const query = rawQuery.trim(); if (!query) return { input: query, results: [] }; @@ -206,10 +195,11 @@ export class ServingService { region_hit: boolean; score: number | null; matched: string[] | null; }>>` WITH near AS ( - SELECT id, 1 - (embedding <=> ${vec}::vector) AS score - FROM keyword - WHERE embedding IS NOT NULL - ORDER BY embedding <=> ${vec}::vector + SELECT k.id, 1 - (k.embedding <=> ${vec}::vector) AS score + FROM keyword k + WHERE k.embedding IS NOT NULL + AND EXISTS (SELECT 1 FROM merchant_keyword mk WHERE mk.keyword_id = k.id AND mk.status = 'active') + ORDER BY k.embedding <=> ${vec}::vector LIMIT ${SEARCH_KEYWORD_DEPTH} ), hits AS ( SELECT id, score FROM near WHERE score >= ${SEARCH_FLOOR} @@ -221,7 +211,7 @@ export class ServingService { SELECT m.external_id, r.name AS region, i.name AS industry, COALESCE(bool_or(r.path <@ p.path), false) AS region_hit, max(h.score) AS score, - array_remove(array_agg(DISTINCT k.canonical), NULL) AS matched + array_remove(array_agg(k.canonical ORDER BY h.score DESC), NULL) AS matched FROM merchant m JOIN industry i ON i.id = m.industry_id AND i.path <@ ${LODGING_ROOT}::ltree LEFT JOIN region r ON r.id = m.region_id @@ -230,7 +220,8 @@ export class ServingService { LEFT JOIN hits h ON h.id = mk.keyword_id LEFT JOIN keyword k ON k.id = h.id GROUP BY m.id, r.name, i.name - HAVING max(h.score) IS NOT NULL OR bool_or(p.path IS NOT NULL) + HAVING (max(h.score) IS NOT NULL OR bool_or(p.path IS NOT NULL)) + AND (NOT EXISTS (SELECT 1 FROM places) OR bool_or(p.path IS NOT NULL)) ORDER BY region_hit DESC, score DESC NULLS LAST, m.name LIMIT ${limit}`; @@ -241,7 +232,7 @@ export class ServingService { industry: r.industry, regionMatched: r.region_hit, score: r.score === null ? null : Number(r.score), - matchedKeywords: r.matched ?? [], + matchedKeywords: [...new Set(r.matched ?? [])], seo: await this.seo(r.external_id, SEARCH_SEO_KEYWORDS), }); } diff --git a/solution/backend/services/external/site_ontology.py b/solution/backend/services/external/site_ontology.py index ed92f47..5de5dd6 100644 --- a/solution/backend/services/external/site_ontology.py +++ b/solution/backend/services/external/site_ontology.py @@ -7,6 +7,8 @@ from config.server_configs import external_api_config # 로컬 임베딩 모델이라 호출당 1~2초다. TIMEOUT_SEC = 15.0 +# 처음 보는 업체는 publish 가 OpenAI 키워드 생성을 기다린다(키워드 15개 + 질문 5개). +PUBLISH_TIMEOUT_SEC = 60.0 DEFAULT_LIMIT = 40 PUBLISH_PATH = "/v1/merchants/publish" @@ -25,8 +27,8 @@ def is_configured() -> bool: return bool(base_url()) -async def _post(client: httpx.AsyncClient, path: str, body: dict) -> dict: - res = await client.post(f"{base_url()}{path}", json=body) +async def _post(client: httpx.AsyncClient, path: str, body: dict, timeout: float = TIMEOUT_SEC) -> dict: + res = await client.post(f"{base_url()}{path}", json=body, timeout=timeout) if res.status_code >= 400: raise SiteOntologyError(f"{path} HTTP {res.status_code}: {res.text[:200]}") try: @@ -40,17 +42,18 @@ async def _post(client: httpx.AsyncClient, path: str, body: dict) -> dict: async def match_for_merchant(merchant: dict, limit: int = DEFAULT_LIMIT) -> dict: """업체를 저장하고, 그 업체로 해석된 추천 결과(/v1/match 응답)를 돌려준다.""" - body = {**merchant, "generate": False} + # sync 는 처음 보는 업체에만 먹는다 — 첫 발행부터 새 키워드가 실리고, 갱신은 저쪽 큐가 한다. + body = {**merchant, "generate": True, "sync": True} try: async with httpx.AsyncClient(timeout=TIMEOUT_SEC) as client: try: - await _post(client, PUBLISH_PATH, body) + await _post(client, PUBLISH_PATH, body, PUBLISH_TIMEOUT_SEC) except SiteOntologyError as ex: if not body.get("regionId"): raise LOG.w(f"[site-ontology] regionId={body['regionId']} 로 저장하지 못했다 — 지역 없이 다시 보낸다: {ex}") body = {**body, "regionId": None} - await _post(client, PUBLISH_PATH, body) + await _post(client, PUBLISH_PATH, body, PUBLISH_TIMEOUT_SEC) result = await _post(client, MATCH_PATH, {"query": merchant["externalId"], "limit": limit}) except httpx.HTTPError as ex: raise SiteOntologyError(f"{type(ex).__name__}: {ex}") from ex diff --git a/solution/backend/services/seo_keywords.py b/solution/backend/services/seo_keywords.py index d0d5133..0d423f8 100644 --- a/solution/backend/services/seo_keywords.py +++ b/solution/backend/services/seo_keywords.py @@ -9,6 +9,9 @@ from services.external import site_ontology from services.site_payload import _parse_address_parts LODGING_INDUSTRY = "stay.pension" +HOTEL_INDUSTRY = "stay.hotel" +# 업종별 숙박 유형어. 자기 유형어만 자료 없이 통과한다 — 호텔에 `군산 독채펜션` 이 붙지 않게. +_TYPE_WORDS = {LODGING_INDUSTRY: "펜션", HOTEL_INDUSTRY: "호텔"} MAX_KEYWORDS = 10 MAX_NEARBY = 8 @@ -57,9 +60,9 @@ _REGION_KEYS = { } # 자료에 없어도 되는 낱말 — 업종어와 "근처" 류. -_GENERIC_WORDS = frozenset({"펜션", "숙소", "숙박", "스테이", "근처", "가까운", "주변", "인근", "예약", "추천"}) +_GENERIC_WORDS = frozenset({"숙소", "숙박", "스테이", "근처", "가까운", "주변", "인근", "예약", "추천"}) # "독채펜션"·"감성숙소" 처럼 붙여 쓴 업종어는 떼고 앞부분만 자료와 대조한다. -_GENERIC_SUFFIXES = ("펜션", "숙소", "스테이") +_GENERIC_SUFFIXES = ("숙소", "스테이") # 제목에는 싣지 않는 낱말. _TITLE_BLOCKED_WORDS = frozenset({"예약", "추천"}) # 한 글자 낱말("봄"·"뷰")은 소개문 어딘가에 우연히 들어 있어 대조가 무의미하다 — 통과시키지 않는다. @@ -102,6 +105,12 @@ def _locality_word(*addresses: str | None) -> str: return word[:-1] if len(word) > 2 and word.endswith(("시", "군", "구")) else word +def industry_of(place: dict) -> str: + """외부 분류(없으면 상호)에 '호텔' 이 있으면 호텔, 아니면 펜션.""" + text = f"{_text(place.get('external_category'))} {_text(place.get('name'))}" + return HOTEL_INDUSTRY if "호텔" in text else LODGING_INDUSTRY + + def build_merchant(place_id: str, snapshot: dict) -> dict | None: """스냅샷 → SiteOntology publish 요청 본문.""" place = (snapshot or {}).get("place") or {} @@ -157,7 +166,7 @@ def build_merchant(place_id: str, snapshot: dict) -> dict | None: return { "externalId": place_id, "name": name, - "industryId": LODGING_INDUSTRY, + "industryId": industry_of(place), "regionId": region_key(road_address, address), "description": description, "profile": {key: value for key, value in profile.items() if value}, @@ -174,13 +183,14 @@ def _evidence(merchant: dict) -> str: return _compact(" ".join(_text(p) for p in parts if p)) -def _supported(keyword: str, evidence: str) -> bool: +def _supported(keyword: str, evidence: str, type_word: str = "펜션") -> bool: """모든 낱말이 자료에 있고, **자료로 확인한 낱말이 하나는 있어야** 한다.""" + suffixes = (*_GENERIC_SUFFIXES, type_word) specific = False for word in keyword.split(): - if word in _GENERIC_WORDS: + if word in _GENERIC_WORDS or word == type_word: continue - core = next((word[: -len(s)] for s in _GENERIC_SUFFIXES if word.endswith(s) and len(word) > len(s)), word) + core = next((word[: -len(s)] for s in suffixes if word.endswith(s) and len(word) > len(s)), word) core = _compact(core) if len(core) < _MIN_CORE_LEN or core not in evidence: return False @@ -193,15 +203,15 @@ def _is_question(item: dict) -> bool: return item.get("category") == "질문형" or "?" in canonical or canonical.endswith(("요", "까")) -def _usable(item, evidence: str) -> str | None: +def _usable(item, evidence: str, type_word: str) -> str | None: """메타 태그에 실어도 되는 추천이면 그 표기를, 아니면 None.""" if not isinstance(item, dict) or item.get("status") != "ok" or _is_question(item): return None canonical = _text(item.get("canonical")) - return canonical if canonical and _supported(canonical, evidence) else None + return canonical if canonical and _supported(canonical, evidence, type_word) else None -def _title_keyword(result: dict, evidence: str, locality: str) -> str | None: +def _title_keyword(result: dict, evidence: str, locality: str, type_word: str) -> str | None: """제목 업종어 자리에 넣을 대표 키워드 — 유형 레인에서 고른다.""" lane = next( (l for l in result.get("byLane") or [] if isinstance(l, dict) and l.get("key") == "type"), @@ -209,7 +219,7 @@ def _title_keyword(result: dict, evidence: str, locality: str) -> str | None: ) items = [i for i in lane.get("items") or [] if isinstance(i, dict)] for item in sorted(items, key=lambda i: i.get("category") != "코어"): - canonical = _usable(item, evidence) + canonical = _usable(item, evidence, type_word) if not canonical or _TITLE_BLOCKED_WORDS.intersection(canonical.split()): continue if locality and locality not in canonical: @@ -221,10 +231,11 @@ def _title_keyword(result: dict, evidence: str, locality: str) -> str | None: def select(result: dict, merchant: dict) -> dict: """/v1/match 응답 → {keywords, titleKeyword?}.""" evidence = _evidence(merchant) + type_word = _TYPE_WORDS.get(merchant.get("industryId"), _TYPE_WORDS[LODGING_INDUSTRY]) keywords: list[str] = [] seen: set[str] = set() for item in result.get("matches") or []: - canonical = _usable(item, evidence) + canonical = _usable(item, evidence, type_word) if not canonical or _compact(canonical) in seen: continue seen.add(_compact(canonical)) @@ -233,7 +244,7 @@ def select(result: dict, merchant: dict) -> dict: break seo: dict = {"keywords": keywords} - title = _title_keyword(result, evidence, _locality_word((merchant.get("profile") or {}).get("address"))) + title = _title_keyword(result, evidence, _locality_word((merchant.get("profile") or {}).get("address")), type_word) if title: seo["titleKeyword"] = title return seo diff --git a/solution/backend/services/snapshot.py b/solution/backend/services/snapshot.py index 4611a8b..afd4bc3 100644 --- a/solution/backend/services/snapshot.py +++ b/solution/backend/services/snapshot.py @@ -109,6 +109,8 @@ async def build_snapshot(place) -> dict: "name": place.name, "category": category.value, "category_name": schema.label, + # 숙박 안의 유형(호텔·펜션) 판정 근거 — services/seo_keywords.industry_of + "external_category": getattr(place, "external_category", None), "road_address": place.road_address, "address": place.address, "phone": place.phone, diff --git a/solution/backend/tests/test_seo_keywords.py b/solution/backend/tests/test_seo_keywords.py index 7490d2b..28c605a 100644 --- a/solution/backend/tests/test_seo_keywords.py +++ b/solution/backend/tests/test_seo_keywords.py @@ -177,8 +177,8 @@ async def test_업체를_저장한_뒤_그_업체로_추천을_받는다(ontolog await site_ontology.match_for_merchant(merchant) assert [url for url, _ in sent] == ["http://onto.test/v1/merchants/publish", "http://onto.test/v1/match"] - # generate:false 가 빠지면 SiteOntology 가 LLM 키워드 생성을 큐에 넣는다. - assert sent[0][1] == {**merchant, "generate": False} + # 처음 보는 업체는 생성을 기다려 첫 발행부터 새 키워드가 실린다(sync 는 저쪽이 처음 보는 업체에만 먹인다). + assert sent[0][1] == {**merchant, "generate": True, "sync": True} assert sent[1][1] == {"query": "place-1", "limit": site_ontology.DEFAULT_LIMIT} @@ -296,3 +296,32 @@ async def test_SiteOntology_가_죽어도_발행된다(auth_headers, client, db_ assert site["site"]["status"] == SiteStatus.PUBLISHED.value payload = json.loads(open(r["payload_path"], encoding="utf-8").read()) assert "seo" not in payload + + +# ── 호텔 ─────────────────────────────────────────────────────────────────── +def test_외부_분류에_호텔이_있으면_호텔_업종으로_보낸다(): + merchant = seo_keywords.build_merchant("place-1", _snapshot(external_category="여행 > 숙박 > 호텔")) + assert merchant["industryId"] == "stay.hotel" + + +def test_외부_분류가_없으면_상호로_가른다(): + assert seo_keywords.build_merchant("place-1", _snapshot(name="군산 베이 호텔"))["industryId"] == "stay.hotel" + assert seo_keywords.build_merchant("place-1", _snapshot())["industryId"] == "stay.pension" + + +def _item(canonical): + return {"canonical": canonical, "status": "ok", "category": "코어"} + + +def test_호텔에는_펜션_유형어를_싣지_않는다(): + merchant = seo_keywords.build_merchant("place-1", _snapshot(name="군산 베이 호텔")) + result = {"matches": [_item("군산 독채펜션"), _item("군산 독채 호텔"), _item("말랭이마을 숙소")]} + + assert seo_keywords.select(result, merchant)["keywords"] == ["군산 독채 호텔", "말랭이마을 숙소"] + + +def test_펜션에는_호텔_유형어를_싣지_않는다(): + merchant = seo_keywords.build_merchant("place-1", _snapshot()) + result = {"matches": [_item("군산 독채 호텔"), _item("군산 독채펜션")]} + + assert seo_keywords.select(result, merchant)["keywords"] == ["군산 독채펜션"]