여러 줄 주석이 설명보다 경위(예전·실측·지적)를 적고 있어 읽는 사람이 결론을 찾기 어려웠다. - ts·tsx·js·mjs·css·py 478개: 여러 줄 주석은 첫 문장 한 줄로, 과거형·날짜 문장은 삭제 - 주석 위치는 TypeScript 파서·파이썬 tokenize/ast 로 찾는다 — 문자열 안의 # · /* 는 건드리지 않는다 - eslint·ts·noqa·type: ignore 같은 지시 주석은 그대로 둔다 파이썬 275개 정리 전후 AST 동일, TS 298개 주석 뺀 토큰 동일(빈 JSX 주석 10곳만 차이). site·frontend·admin tsc, site vitest 105 passed Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
81 lines
3.0 KiB
TypeScript
81 lines
3.0 KiB
TypeScript
/** 외부 연관키워드·검색량을 데이터셋에 병합한다. */
|
|
import { readFileSync, writeFileSync } from 'node:fs';
|
|
import { canonicalizeKeyword, isBanned, normalizeKeyword } from '../src/keywords/normalize';
|
|
|
|
const DATASET = 'data/gunsan-pension-keywords.json';
|
|
|
|
interface Related { keyword: string; volumePc: number; volumeMobile: number; competition: string | null }
|
|
|
|
function parse(path: string): Related[] {
|
|
const raw = readFileSync(path, 'utf8');
|
|
if (path.endsWith('.json')) {
|
|
return (JSON.parse(raw) as any[]).map(toRelated);
|
|
}
|
|
const lines = raw.split(/\r?\n/).filter((l) => l.trim() && !l.trimStart().startsWith('#'));
|
|
const head = lines.shift()!.split(',').map((h) => h.trim());
|
|
return lines.map((line) => {
|
|
const cells = line.split(',').map((c) => c.trim());
|
|
const o: Record<string, string> = {};
|
|
head.forEach((h, i) => (o[h] = cells[i] ?? ''));
|
|
return toRelated(o);
|
|
});
|
|
}
|
|
|
|
function toRelated(o: any): Related {
|
|
const num = (v: unknown) => {
|
|
const n = Number(String(v ?? '').replace(/[^0-9]/g, ''));
|
|
return Number.isFinite(n) ? n : 0;
|
|
};
|
|
return {
|
|
keyword: String(o.relKeyword ?? o.keyword ?? '').trim(),
|
|
volumePc: num(o.monthlyPcQcCnt),
|
|
volumeMobile: num(o.monthlyMobileQcCnt),
|
|
competition: o.compIdx ? String(o.compIdx).trim() : null,
|
|
};
|
|
}
|
|
|
|
function main() {
|
|
const file = process.argv[2];
|
|
const apply = process.argv.includes('--apply');
|
|
if (!file) { console.error('사용법: tsx scripts/import-related.ts <csv|json> [--apply]'); process.exit(1); }
|
|
|
|
const ds = JSON.parse(readFileSync(DATASET, 'utf8'));
|
|
const existing = new Map<string, any>(ds.items.map((i: any) => [normalizeKeyword(i.keyword), i]));
|
|
|
|
const rows = parse(file).filter((r) => r.keyword);
|
|
let added = 0, enriched = 0, skipped = 0;
|
|
const newItems: any[] = [];
|
|
|
|
for (const r of rows) {
|
|
const canonical = canonicalizeKeyword(r.keyword);
|
|
const norm = normalizeKeyword(canonical);
|
|
if (!norm || isBanned(canonical)) { skipped++; continue; }
|
|
const volume = r.volumePc + r.volumeMobile;
|
|
const hit = existing.get(norm);
|
|
if (hit) {
|
|
hit.volume = volume; hit.competition = r.competition; hit.volumeSource = 'naver-searchad';
|
|
enriched++;
|
|
} else {
|
|
const item = {
|
|
keyword: canonical, intent: 'local', kind: 'keyword', category: '연관',
|
|
relevance: 0.7, volume, competition: r.competition, volumeSource: 'naver-searchad',
|
|
};
|
|
newItems.push(item); existing.set(norm, item); added++;
|
|
}
|
|
}
|
|
|
|
console.log(`입력 ${rows.length}건 → 신규 ${added} · 기존 보강 ${enriched} · 제외 ${skipped}`);
|
|
if (newItems.length) {
|
|
console.log('\n신규 예시');
|
|
for (const i of newItems.slice(0, 8)) console.log(` ${i.keyword} (월 ${i.volume}, 경쟁 ${i.competition ?? '-'})`);
|
|
}
|
|
if (!apply) { console.log('\n파일에 쓰려면 --apply 를 붙일 것.'); return; }
|
|
|
|
ds.items = [...ds.items, ...newItems];
|
|
ds.count = ds.items.length;
|
|
writeFileSync(DATASET, JSON.stringify(ds, null, 2) + '\n');
|
|
console.log(`\n✅ ${DATASET} → ${ds.count}건`);
|
|
}
|
|
|
|
main();
|