o2o-site-AEO/ontology/src/keywords/dedup.service.ts
Mina Choi 01098835e9 [chore] docker-compose,ontology: 온톨로지를 이 레포로 들여 compose 한 벌로 띄운다 — 앱 Dockerfile 신설
발행이 SiteOntology 를 부르는데 서버는 따로 띄워야 했다. 실측(2026-09-14): 서버가 없으면
`[seo] SiteOntology 실패 — 키워드 없이 발행: ConnectError` 로 빌드는 성공하고 메타만 빈다 —
화면으로는 안 보이는 종류다. 한 벌로 묶어 "코드는 올라갔는데 서버가 없는" 상태를 없앤다.

- ontology/: gitea.o2o.kr/Web4ai/o2o-site-ontology 를 이 레포로 편입(그 원격은 그대로 남는다)
- ontology/Dockerfile(신규): 베이스는 node:22-slim. alpine 은 임베딩 런타임(onnxruntime)이
  musl 바이너리를 안 줘서 적재가 ERR_DLOPEN_FAILED 로 죽는다 — 빌드는 성공하고 실행에서만 터진다
- docker-compose.yml: ontology · ontology-postgres(pgvector) · ontology-redis 추가.
  자체 DB 를 쓰는 이유는 pgvector 확장 때문이다 — web4ai_db 를 남의 서비스 확장에 묶지 않는다
- 임베딩 모델(120MB)은 이미지에 굽지 않고 볼륨(ontology-model)에 남긴다
- 컨테이너끼리는 `http://ontology:3100` 으로 만난다. `.env` 의 127.0.0.1 은 컨테이너 자기 자신이라 안 닿는다

검증: 3개 기동 · 백엔드 컨테이너에서 ontology:3100/demo HTTP 200 · 마이그레이션·시드 완료

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
2026-09-14 17:41:03 +09:00

114 lines
3.7 KiB
TypeScript

import { Injectable, Logger } from '@nestjs/common';
import { env } from '../config/env';
import { KeywordIntent } from '../llm/types';
import { KeywordRepository } from './keyword.repository';
import { canonicalizeKeyword, isBanned, normalizeKeyword } from './normalize';
export type DedupAction =
| 'created' // 새 키워드
| 'matched_exact' // 1단계: 정규화 해시 일치
| 'matched_trigram' // 2단계: 표기 변형/오타
| 'matched_vector' // 3단계: 의미 중복 → alias 흡수
| 'rejected_banned'; // 금칙어
export interface DedupResult {
action: DedupAction;
keywordId: string | null;
canonical: string;
matchedTo?: string;
similarity?: number;
}
export interface ResolveInput {
raw: string;
intent: KeywordIntent;
embedding: number[];
locale: string;
industryId: string | null;
regionId: string | null;
}
/**
* 4단계 계단식 중복제거.
* 값비싼 벡터 비교는 마지막에, 후보 집합 안에서만 수행한다.
*/
@Injectable()
export class DedupService {
private readonly logger = new Logger(DedupService.name);
constructor(private readonly repo: KeywordRepository) {}
async resolve(input: ResolveInput): Promise<DedupResult> {
const canonical = canonicalizeKeyword(input.raw);
const normalized = normalizeKeyword(input.raw);
// 0단계 — 금칙어/과장광고 차단
if (!normalized || isBanned(canonical)) {
return { action: 'rejected_banned', keywordId: null, canonical };
}
// 1단계 — 정규화 완전 일치 (공백/구두점 차이 흡수)
const exact = await this.repo.findByNormalized(normalized, input.locale);
if (exact) {
await this.repo.absorbAlias(exact.id, canonical);
return {
action: 'matched_exact',
keywordId: exact.id,
canonical: exact.canonical,
matchedTo: exact.canonical,
similarity: 1,
};
}
// 2~3단계 — trigram 후보 + 벡터 ANN 후보를 모아 최고 유사도 판정
//
// 주의: 짧은 한글 키워드에서는 문장 임베딩의 절대 코사인이 변별력이 약하다.
// 실측(multilingual-e5-small): '선유도 펜션' ↔ '새만금 펜션' = 0.936,
// '군산 펜션' ↔ '군산 호텔' = 0.970 — 전혀 다른 키워드인데도 높게 나온다.
// 반면 어순만 바뀐 진짜 중복('군산 키즈룸 펜션' ↔ '군산 펜션 키즈룸')은 0.999 대에 몰린다.
// 그래서 임계값을 0.99 로 올려 잡고, 자동 병합의 주력은 1~2단계(어휘)에 둔다.
const candidates = await this.repo.findDedupCandidates(
input.embedding,
normalized,
input.locale,
env.dedup.candidateLimit,
);
const trigramHit = candidates.find((c) => c.trg >= env.dedup.trigramThreshold);
if (trigramHit) {
await this.repo.absorbAlias(trigramHit.id, canonical);
return {
action: 'matched_trigram',
keywordId: trigramHit.id,
canonical: trigramHit.canonical,
matchedTo: trigramHit.canonical,
similarity: trigramHit.trg,
};
}
const best = candidates[0];
if (best && best.cosine >= env.dedup.cosineThreshold) {
await this.repo.absorbAlias(best.id, canonical);
return {
action: 'matched_vector',
keywordId: best.id,
canonical: best.canonical,
matchedTo: best.canonical,
similarity: best.cosine,
};
}
// 4단계 — 신규 등록
const created = await this.repo.insert({
canonical,
normalized,
locale: input.locale,
intent: input.intent,
embedding: input.embedding,
industryId: input.industryId,
regionId: input.regionId,
});
return { action: 'created', keywordId: created.id, canonical: created.canonical };
}
}