여러 줄 주석이 설명보다 경위(예전·실측·지적)를 적고 있어 읽는 사람이 결론을 찾기 어려웠다. - ts·tsx·js·mjs·css·py 478개: 여러 줄 주석은 첫 문장 한 줄로, 과거형·날짜 문장은 삭제 - 주석 위치는 TypeScript 파서·파이썬 tokenize/ast 로 찾는다 — 문자열 안의 # · /* 는 건드리지 않는다 - eslint·ts·noqa·type: ignore 같은 지시 주석은 그대로 둔다 파이썬 275개 정리 전후 AST 동일, TS 298개 주석 뺀 토큰 동일(빈 JSX 주석 10곳만 차이). site·frontend·admin tsc, site vitest 105 passed Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
245 lines
9.1 KiB
TypeScript
245 lines
9.1 KiB
TypeScript
import { Inject, Injectable } from '@nestjs/common';
|
|
import { Sql, toVector } from '../db/db';
|
|
import { PG } from '../db/db.module';
|
|
import { EmbeddingProvider } from '../embedding/types';
|
|
import { normalizeKeyword } from '../keywords/normalize';
|
|
import { MerchantWithTaxonomy } from '../merchants/merchants.service';
|
|
import {
|
|
AreaGroup, Lane, MerchantFacts, buildLanes, checkAmenity, detectAreaGroup,
|
|
keywordAreaGroup, normalizeAmenities, violatesCapacity,
|
|
} from './match.rules';
|
|
|
|
// RRF 상수를 관례값 60 대신 20 으로 낮춘다.
|
|
const RRF_K = 20;
|
|
const LANE_DEPTH = 50; // 레인당 후보 깊이 — 깊을수록 generic 이 유리해진다
|
|
const LANE_FLOOR = 0.80; // 이 코사인 미만은 그 레인에서 기여하지 않는다
|
|
// 매칭 후보로 인정하는 출처.
|
|
const MATCH_SOURCES = ['dataset', 'nationwide', 'manual'];
|
|
|
|
interface Hit {
|
|
id: string; canonical: string; intent: string; kind: string;
|
|
category: string | null; aliases: string[]; score: number;
|
|
}
|
|
|
|
export interface MatchRow extends Hit {
|
|
rrf: number;
|
|
status: 'ok' | 'hold';
|
|
holdReason?: string;
|
|
lanes: Array<{ key: string; label: string; rank: number; score: number }>;
|
|
linked: boolean;
|
|
}
|
|
|
|
@Injectable()
|
|
export class MatchService {
|
|
constructor(
|
|
@Inject(PG) private readonly sql: Sql,
|
|
private readonly embedder: EmbeddingProvider,
|
|
) {}
|
|
|
|
/** 속성별 서브 질의 → 가중 RRF 융합 → 사실 기반 필터 */
|
|
async fusion(rawQuery: string, limit: number) {
|
|
const query = rawQuery.trim();
|
|
const merchant = await this.resolveMerchant(query);
|
|
const facts = merchant ? toFacts(merchant) : null;
|
|
|
|
const lanes: Lane[] = facts
|
|
? buildLanes(facts)
|
|
: [{ key: 'free', label: '입력문', weight: 1.0, text: query }];
|
|
|
|
const vectors = await this.embedder.embed(lanes.map((l) => l.text), 'query');
|
|
|
|
// 레인별 검색.
|
|
const perLane = await Promise.all(
|
|
vectors.map((v) => this.laneSearch(v, LANE_DEPTH, merchant?.industry_id ?? null)),
|
|
);
|
|
|
|
// 가중 RRF 융합
|
|
const acc = new Map<string, { hit: Hit; rrf: number; lanes: MatchRow['lanes'] }>();
|
|
perLane.forEach((hits, li) => {
|
|
const lane = lanes[li];
|
|
hits.forEach((hit, idx) => {
|
|
if (hit.score < LANE_FLOOR) return;
|
|
const rank = idx + 1;
|
|
const contrib = lane.weight / (RRF_K + rank);
|
|
const cur = acc.get(hit.id) ?? { hit, rrf: 0, lanes: [] };
|
|
cur.rrf += contrib;
|
|
cur.lanes.push({ key: lane.key, label: lane.label, rank, score: hit.score });
|
|
if (hit.score > cur.hit.score) cur.hit = hit;
|
|
acc.set(hit.id, cur);
|
|
});
|
|
});
|
|
|
|
// 사실 기반 필터
|
|
const kept: MatchRow[] = [];
|
|
const excluded: Array<{ canonical: string; reason: string }> = [];
|
|
for (const { hit, rrf, lanes: ls } of acc.values()) {
|
|
if (facts) {
|
|
if (violatesCapacity(hit.canonical, facts.capacityMax)) {
|
|
excluded.push({ canonical: hit.canonical, reason: `최대 ${facts.capacityMax}인 — 단체 키워드` });
|
|
continue;
|
|
}
|
|
const kwArea = keywordAreaGroup(hit.canonical);
|
|
if (kwArea && facts.areaGroup && kwArea !== facts.areaGroup) {
|
|
excluded.push({ canonical: hit.canonical, reason: `권역 불일치 — ${kwArea} (업체는 ${facts.areaGroup})` });
|
|
continue;
|
|
}
|
|
const am = checkAmenity(hit.canonical, facts);
|
|
if (!am.ok && !am.hold) {
|
|
excluded.push({ canonical: hit.canonical, reason: `미보유 시설 — ${am.amenity}` });
|
|
continue;
|
|
}
|
|
kept.push({
|
|
...hit, rrf,
|
|
lanes: ls.sort((a, b) => a.rank - b.rank),
|
|
status: am.ok ? 'ok' : 'hold',
|
|
holdReason: am.ok ? undefined : `${am.amenity} 미확인 — 사업자 확인 필요`,
|
|
linked: false,
|
|
});
|
|
} else {
|
|
kept.push({ ...hit, rrf, lanes: ls.sort((a, b) => a.rank - b.rank), status: 'ok', linked: false });
|
|
}
|
|
}
|
|
|
|
kept.sort((a, b) => b.rrf - a.rrf);
|
|
const top = kept.slice(0, limit);
|
|
|
|
// 레인별 상위 — SEO 페이지 배분은 평평한 순위가 아니라 이쪽을 쓴다.
|
|
const keptById = new Map(kept.map((k) => [k.id, k]));
|
|
const byLane = lanes.map((lane, li) => ({
|
|
key: lane.key, label: lane.label, weight: lane.weight, text: lane.text,
|
|
items: perLane[li]
|
|
.map((h) => keptById.get(h.id))
|
|
.filter((x): x is MatchRow => Boolean(x))
|
|
.slice(0, 8),
|
|
}));
|
|
|
|
await this.markLinked([...top, ...byLane.flatMap((l) => l.items)], merchant?.id ?? null);
|
|
|
|
return {
|
|
mode: 'fusion' as const,
|
|
input: query,
|
|
resolved: merchant ? publicMerchant(merchant) : null,
|
|
facts: facts && {
|
|
areaGroup: facts.areaGroup,
|
|
capacityMax: facts.capacityMax,
|
|
amenities: [...facts.amenities],
|
|
unverified: [...facts.unverified],
|
|
},
|
|
lanes: lanes.map((l, i) => ({
|
|
...l, top: perLane[i][0]?.canonical ?? null, topScore: perLane[i][0]?.score ?? null,
|
|
})),
|
|
embeddingProvider: this.embedder.name,
|
|
total: await this.dictionarySize(),
|
|
matches: top,
|
|
byLane,
|
|
excluded: excluded.slice(0, 40),
|
|
excludedTotal: excluded.length,
|
|
};
|
|
}
|
|
|
|
private async laneSearch(
|
|
embedding: number[], limit: number, industryId: string | null,
|
|
): Promise<Hit[]> {
|
|
const vec = toVector(embedding);
|
|
const rows = await this.sql<Hit[]>`
|
|
SELECT id, canonical, intent, kind, category, aliases,
|
|
1 - (embedding <=> ${vec}::vector) AS score
|
|
FROM keyword
|
|
WHERE embedding IS NOT NULL
|
|
AND source = ANY(${MATCH_SOURCES})
|
|
AND (${industryId}::text IS NULL OR industry_id IS NULL OR industry_id = ${industryId})
|
|
ORDER BY embedding <=> ${vec}::vector
|
|
LIMIT ${limit}`;
|
|
return rows.map((r) => ({ ...r, score: Number(r.score) }));
|
|
}
|
|
|
|
private async markLinked(rows: MatchRow[], merchantId: string | null) {
|
|
if (!merchantId || rows.length === 0) return;
|
|
const ids = rows.map((r) => r.id);
|
|
const linked = await this.sql<Array<{ keyword_id: string }>>`
|
|
SELECT keyword_id FROM merchant_keyword
|
|
WHERE merchant_id = ${merchantId} AND keyword_id = ANY(${ids}::uuid[])`;
|
|
const set = new Set(linked.map((l) => l.keyword_id));
|
|
for (const r of rows) r.linked = set.has(r.id);
|
|
}
|
|
|
|
async resolveMerchant(query: string) {
|
|
const norm = normalizeKeyword(query);
|
|
if (!norm) return null;
|
|
const rows = await this.sql<Array<MerchantWithTaxonomy & { sim: number }>>`
|
|
SELECT m.*,
|
|
i.name AS industry_name, i.path::text AS industry_path,
|
|
r.name AS region_name, r.path::text AS region_path,
|
|
similarity(regexp_replace(lower(m.name), '\\s', '', 'g'), ${norm}) AS sim
|
|
FROM merchant m
|
|
LEFT JOIN industry i ON i.id = m.industry_id
|
|
LEFT JOIN region r ON r.id = m.region_id
|
|
WHERE regexp_replace(lower(m.name), '\\s', '', 'g') = ${norm}
|
|
OR m.external_id = ${query}
|
|
OR similarity(regexp_replace(lower(m.name), '\\s', '', 'g'), ${norm}) >= 0.45
|
|
ORDER BY sim DESC NULLS LAST
|
|
LIMIT 1`;
|
|
return rows[0] ?? null;
|
|
}
|
|
|
|
async dictionarySize() {
|
|
const [row] = await this.sql<Array<{ n: number }>>`
|
|
SELECT count(*)::int AS n FROM keyword
|
|
WHERE embedding IS NOT NULL AND source = ANY(${MATCH_SOURCES})`;
|
|
return row?.n ?? 0;
|
|
}
|
|
}
|
|
|
|
function str(v: unknown): string[] {
|
|
return Array.isArray(v) ? v.filter((x): x is string => typeof x === 'string') : [];
|
|
}
|
|
|
|
function toFacts(m: MerchantWithTaxonomy): MerchantFacts {
|
|
const p = (m.profile ?? {}) as Record<string, unknown>;
|
|
const features = str(p['features']);
|
|
const services = str(p['services']);
|
|
const nearby = str(p['nearby']);
|
|
const address = typeof p['address'] === 'string' ? p['address'] : null;
|
|
const cap = p['capacity'] as { max?: number } | undefined;
|
|
|
|
const areaSource = [address ?? '', ...nearby, m.description].join(' ');
|
|
const areaGroup: AreaGroup | null = detectAreaGroup(areaSource);
|
|
|
|
return {
|
|
name: m.name,
|
|
region: m.region_name,
|
|
industry: m.industry_name,
|
|
description: m.description,
|
|
address,
|
|
areaGroup,
|
|
capacityMax: typeof cap?.max === 'number' ? cap.max : null,
|
|
services, features,
|
|
audiences: str(p['audiences']),
|
|
nearby,
|
|
amenities: normalizeAmenities([...features, ...services]),
|
|
unverified: new Set(str(p['unverified'])),
|
|
signals: collectSignals(p),
|
|
};
|
|
}
|
|
|
|
/** 고객 언어 신호를 모은다. */
|
|
function collectSignals(p: Record<string, unknown>): string[] {
|
|
const tags = str(p['hashtags']).map((t) => t.replace(/^#/, '').trim()).filter(Boolean);
|
|
const raw = Array.isArray(p['reviewSignals']) ? p['reviewSignals'] : [];
|
|
const reviews = raw
|
|
.filter((r): r is { term: string; count: number } =>
|
|
Boolean(r) && typeof (r as any).term === 'string')
|
|
.sort((a, b) => (b.count ?? 0) - (a.count ?? 0))
|
|
.map((r) => r.term.trim())
|
|
.filter(Boolean);
|
|
return [...new Set([...reviews, ...tags])];
|
|
}
|
|
|
|
function publicMerchant(m: MerchantWithTaxonomy) {
|
|
return {
|
|
id: m.id, externalId: m.external_id, name: m.name,
|
|
region: m.region_name, industry: m.industry_name,
|
|
description: m.description, siteUrl: m.site_url, profile: m.profile ?? {},
|
|
};
|
|
}
|