Compare commits

..

No commits in common. "main" and "db-migration" have entirely different histories.

16 changed files with 113 additions and 268 deletions

2
.gitignore vendored
View File

@ -46,5 +46,3 @@ alembic/versions/*.pyc
test_results/ test_results/
app/test* app/test*
docker-compose.yml

View File

@ -16,7 +16,7 @@ class FirecrawlClient:
HTTPMethod.POST, HTTPMethod.POST,
url=f"{FIRECRAWL_BASE}/scrape", url=f"{FIRECRAWL_BASE}/scrape",
headers=self._headers(), headers=self._headers(),
json_body={"url": url, "formats": ["json", "links"], "jsonOptions": json_options, "waitFor": wait_for, "maxAge": 0}, json_body={"url": url, "formats": ["json", "links"], "jsonOptions": json_options, "waitFor": wait_for},
label="firecrawl-scrape", label="firecrawl-scrape",
) )
if not resp or not resp.is_success: if not resp or not resp.is_success:
@ -74,9 +74,9 @@ class FirecrawlClient:
headers=self._headers(), headers=self._headers(),
json_body={ json_body={
"url": url, "url": url,
"formats": ["json", "links", "rawHtml"], "formats": ["json", "links", "html"],
"jsonOptions": { "jsonOptions": {
"prompt": "Extract: 클리닉 이름 - clinicName (Korean), clinic name (English), address, phone with dash format, business hours, slogan, services offered, doctors with name/title/specialty, brand identity (primary/accent/background/text colors in hex, heading/body fonts, logo URL from the actual header/main <img> src, og:image from <meta property='og:image'> content, favicon URL)", "prompt": "Extract: clinic name (Korean), clinic name (English), address, phone with dash format, business hours, slogan, services offered, doctors with name/title/specialty, brand identity (primary/accent/background/text colors in hex, heading/body fonts, logo URL from the actual header/main <img> src, og:image from <meta property='og:image'> content, favicon URL)",
"schema": { "schema": {
"type": "object", "type": "object",
"properties": { "properties": {
@ -127,11 +127,8 @@ class FirecrawlClient:
}, },
}, },
"waitFor": 5000, "waitFor": 5000,
"maxAge": 0,
"proxy": "auto", # 기본 엔진이 차단되는 사이트(예: viewclinic.com)는 자동으로 stealth 프록시로 재시도.
"timeout": 120000, # proxy:auto면 60s로도 충분했지만(실측), 혹시 모르니 여유있게 120s.
}, },
timeout=150, # 위 Firecrawl 잡 타임아웃보다 길어야 우리 쪽 HTTP 클라이언트가 먼저 끊지 않음. timeout=60,
label="firecrawl-clinic-info", label="firecrawl-clinic-info",
) )
if not resp or not resp.is_success: if not resp or not resp.is_success:
@ -150,7 +147,7 @@ class FirecrawlClient:
# "socialMedia": info.get("socialMedia", {}), # "socialMedia": info.get("socialMedia", {}),
"branding": info.get("branding", {}), "branding": info.get("branding", {}),
"siteLinks": data.get("links", []), "siteLinks": data.get("links", []),
"html": data.get("rawHtml", "") or data.get("html", ""), # rawHtml = 가공 전 원본 — <script> 등 보존. "html": data.get("html", ""), # raw HTML — collect 단계에서 tracking 추출 후 raw_data 에는 저장 안 함.
"sourceUrl": url, "sourceUrl": url,
} }
@ -190,7 +187,6 @@ class FirecrawlClient:
}, },
}, },
"waitFor": 5000, "waitFor": 5000,
"maxAge": 0,
}, },
timeout=60, timeout=60,
label="firecrawl-gangnamunni", label="firecrawl-gangnamunni",

View File

@ -219,7 +219,7 @@ class VisionClient:
' "logo_symbol": "심볼이 묘사하는 대상 (예: \'잎사귀\', \'추상 곡선\'). 없으면 빈 문자열",\n' ' "logo_symbol": "심볼이 묘사하는 대상 (예: \'잎사귀\', \'추상 곡선\'). 없으면 빈 문자열",\n'
' "logo_text": "로고에 보이는 워드마크 텍스트 그대로 (한글/영문). 없으면 빈 문자열",\n' ' "logo_text": "로고에 보이는 워드마크 텍스트 그대로 (한글/영문). 없으면 빈 문자열",\n'
' "logo_colors_desc": "로고에 쓰인 색감을 사람이 부르는 이름으로 서술 (예: \'딥네이비 + 골드\')",\n' ' "logo_colors_desc": "로고에 쓰인 색감을 사람이 부르는 이름으로 서술 (예: \'딥네이비 + 골드\')",\n'
' "logo_colors_hex": ["로고에서 시각적으로 가장 두드러진 색 최대 2개의 hex 근사값 배열. 예: [\'#1A2B3C\', \'#D4A017\']. 색이 1개면 1개만, 강한 색이 1개도 없으면 빈 배열."]\n' ' "logo_colors_hex": ["로고에서 시각적으로 두드러진 색 정확히 5개의 hex 근사값 배열. 예: [\'#1A2B3C\', \'#D4A017\', \'#FFFFFF\', \'#9E5C2A\', \'#1F1F1F\']. 강한 색이 5개 안 되면 음영/명도 차이로 5개 채울 것. 빈 배열 금지."]\n'
"}\n" "}\n"
"주의: logo_colors_hex 는 시각 추정이라 정확도 떨어질 수 있음. CSS 추출이 우선이고 이건 fallback/보완 용.\n" "주의: logo_colors_hex 는 시각 추정이라 정확도 떨어질 수 있음. CSS 추출이 우선이고 이건 fallback/보완 용.\n"
"모든 설명/텍스트 값은 반드시 한국어로 작성하세요 (영어 금지)." "모든 설명/텍스트 값은 반드시 한국어로 작성하세요 (영어 금지)."
@ -229,9 +229,14 @@ class VisionClient:
return {} return {}
# logo_images는 우리가 직접 채움 (Vision은 묘사만) # logo_images는 우리가 직접 채움 (Vision은 묘사만)
result["logo_images"] = {"circle": None, "horizontal": logo_url, "korean": None} result["logo_images"] = {"circle": None, "horizontal": logo_url, "korean": None}
# logo_colors_hex 최대 2개로 제한. # logo_colors_hex 5개 강제 정규화 — LLM 이 4개나 6개 줄 수도 있어서 길이 fallback.
hex_list = [h for h in (result.get("logo_colors_hex") or []) if isinstance(h, str) and h.startswith("#")] hex_list = [h for h in (result.get("logo_colors_hex") or []) if isinstance(h, str) and h.startswith("#")]
result["logo_colors_hex"] = hex_list[:2] if hex_list:
while len(hex_list) < 5:
hex_list.append(hex_list[-1]) # 마지막 색 복제로 패딩
result["logo_colors_hex"] = hex_list[:5]
else:
result["logo_colors_hex"] = []
return result return result
async def describe_channel_logos( async def describe_channel_logos(

View File

@ -13,7 +13,7 @@ from integrations.llm.schemas.report import (
ScoresInput, ScoresOutput, ScoresInput, ScoresOutput,
OtherChannelsInput, OtherChannelsOutput OtherChannelsInput, OtherChannelsOutput
) )
from integrations.llm.schemas.plan import PlanInput, PlanOutput, SummarizeInput, SummarizeOutput from integrations.llm.schemas.plan import PlanInput, PlanOutput
from integrations.llm.schemas.market import ( from integrations.llm.schemas.market import (
MarketCompetitorsInput, MarketCompetitorsOutput, MarketCompetitorsInput, MarketCompetitorsOutput,
MarketKeywordsInput, MarketKeywordsOutput, MarketKeywordsInput, MarketKeywordsOutput,
@ -64,13 +64,6 @@ plan_prompt = Prompt(
output_class=PlanOutput, output_class=PlanOutput,
) )
summarize_prompt = Prompt(
file_name="summarize_prompt.txt",
prompt_model="PLAN_MODEL",
input_class=SummarizeInput,
output_class=SummarizeOutput,
)
market_competitors_prompt = Prompt( market_competitors_prompt = Prompt(
file_name="market_competitors_prompt.txt", file_name="market_competitors_prompt.txt",
prompt_model="MARKET_MODEL", prompt_model="MARKET_MODEL",

View File

@ -2,15 +2,6 @@ from typing import Literal
from pydantic import BaseModel from pydantic import BaseModel
class SummarizeInput(BaseModel):
label: str
data: str
class SummarizeOutput(BaseModel):
summary: str
class PlanInput(BaseModel): class PlanInput(BaseModel):
clinic_name: str | None = None clinic_name: str | None = None
clinic_name_en: str | None = None clinic_name_en: str | None = None

View File

@ -134,7 +134,7 @@ class YouTubeAudit(BaseModel):
channel_description: str channel_description: str
linked_urls: list[LinkedUrl] linked_urls: list[LinkedUrl]
playlists: list[str] playlists: list[str]
top_videos: list[TopVideo] = [] top_videos: list[TopVideo]
diagnosis: list[DiagnosisItem] diagnosis: list[DiagnosisItem]
@ -184,11 +184,11 @@ class FacebookPage(BaseModel):
followers: int followers: int
following: int following: int
category: str category: str
bio: str = "" bio: str
logo: str logo: str
logo_description: str logo_description: str
link: str link: str
linked_domain: str = "" linked_domain: str
reviews: int reviews: int
recent_post_age: str recent_post_age: str
has_whatsapp: bool | None = None has_whatsapp: bool | None = None
@ -245,7 +245,7 @@ class AdditionalDomain(BaseModel):
class WebsiteAudit(BaseModel): class WebsiteAudit(BaseModel):
primary_domain: str primary_domain: str
additional_domains: list[AdditionalDomain] = [] additional_domains: list[AdditionalDomain]
sns_links_on_site: bool sns_links_on_site: bool
sns_links_detail: list[SnsLink] | None = None sns_links_detail: list[SnsLink] | None = None
tracking_pixels: list[TrackingPixel] tracking_pixels: list[TrackingPixel]

View File

@ -1,7 +0,0 @@
다음은 "{label}" 원본 데이터입니다.
{data}
위 데이터를 마케팅 플랜 생성에 필요한 핵심 정보만 남기고 간결하게 요약하세요.
구체적인 수치·날짜·고유명사(채널명, 게시물 제목 등)는 그대로 보존하고, 중복되거나 플랜 작성에 불필요한 메타데이터는 제거하세요.
요약문 하나의 문자열로만 출력하세요.

View File

@ -48,7 +48,7 @@ class YouTubeClient:
resp = await http_request( resp = await http_request(
HTTPMethod.GET, HTTPMethod.GET,
url=f"{YT}/channels", url=f"{YT}/channels",
params={"part": "snippet,statistics,contentDetails", "id": channel_id, "key": self.api_key}, params={"part": "snippet,statistics", "id": channel_id, "key": self.api_key},
label="yt-channel", label="yt-channel",
) )
if not resp or not resp.is_success: if not resp or not resp.is_success:
@ -58,52 +58,26 @@ class YouTubeClient:
return None return None
channel = items[0] channel = items[0]
async def _video_details(video_ids: list[str]) -> list[dict]: video_ids: list[str] = []
"""video_ids 순서를 보존한 채 snippet/statistics/contentDetails 채워서 반환.""" resp = await http_request(
if not video_ids: HTTPMethod.GET,
return [] url=f"{YT}/search",
params={"part": "snippet", "channelId": channel_id, "order": "viewCount", "type": "video", "maxResults": 10, "key": self.api_key},
label="yt-search",
)
if resp and resp.is_success:
video_ids = [i["id"]["videoId"] for i in resp.json().get("items", []) if i.get("id", {}).get("videoId")]
videos: list[dict] = []
if video_ids:
resp = await http_request( resp = await http_request(
HTTPMethod.GET, HTTPMethod.GET,
url=f"{YT}/videos", url=f"{YT}/videos",
params={"part": "snippet,statistics,contentDetails", "id": ",".join(video_ids), "key": self.api_key}, params={"part": "snippet,statistics,contentDetails", "id": ",".join(video_ids), "key": self.api_key},
label="yt-videos", label="yt-videos",
) )
if not resp or not resp.is_success:
return []
video_map = {v["id"]: v for v in resp.json().get("items", [])}
return [video_map[vid] for vid in video_ids if vid in video_map]
# 인기 영상 top 10 (조회수순) — search index 기반, 정확한 정렬 보장.
resp = await http_request(
HTTPMethod.GET,
url=f"{YT}/search",
params={"part": "snippet", "channelId": channel_id, "order": "viewCount", "type": "video", "maxResults": 10, "key": self.api_key},
label="yt-search-top",
)
top_ids: list[str] = []
if resp and resp.is_success: if resp and resp.is_success:
top_ids = [i["id"]["videoId"] for i in resp.json().get("items", []) if i.get("id", {}).get("videoId")] videos = resp.json().get("items", [])[:10]
videos = await _video_details(top_ids)
# 최근 영상 10개 — search index는 누락이 흔해 채널의 실제 uploads 재생목록에서 직접 읽고
# publishedAt 기준으로 코드에서 직접 정렬 (재생목록 순서 자체는 보장되지 않으므로).
recents: list[dict] = []
uploads_id = (channel.get("contentDetails") or {}).get("relatedPlaylists", {}).get("uploads")
if uploads_id:
resp = await http_request(
HTTPMethod.GET,
url=f"{YT}/playlistItems",
params={"part": "snippet", "playlistId": uploads_id, "maxResults": 10, "key": self.api_key},
label="yt-uploads",
)
if resp and resp.is_success:
entries = resp.json().get("items", [])
entries.sort(key=lambda i: i.get("snippet", {}).get("publishedAt") or "", reverse=True)
recent_ids = [
vid for e in entries
if (vid := e.get("snippet", {}).get("resourceId", {}).get("videoId"))
]
recents = await _video_details(recent_ids)
playlists: list[dict] = [] playlists: list[dict] = []
resp = await http_request( resp = await http_request(
@ -115,19 +89,7 @@ class YouTubeClient:
if resp and resp.is_success: if resp and resp.is_success:
playlists = resp.json().get("items", []) playlists = resp.json().get("items", [])
return {"channelId": channel_id, "channel": channel, "videos": videos, "recents": recents, "playlists": playlists} return {"channelId": channel_id, "channel": channel, "videos": videos, "playlists": playlists}
@staticmethod
def _video_item(v: dict) -> dict:
return {
"title": v.get("snippet", {}).get("title"),
"views": int(v.get("statistics", {}).get("viewCount", 0)),
"likes": int(v.get("statistics", {}).get("likeCount", 0)),
"comments": int(v.get("statistics", {}).get("commentCount", 0)),
"date": v.get("snippet", {}).get("publishedAt"),
"duration": v.get("contentDetails", {}).get("duration"),
"url": f"https://www.youtube.com/watch?v={v['id']}",
}
async def get_channel(self, url: str) -> dict | None: async def get_channel(self, url: str) -> dict | None:
raw = await self.fetch_channel(url) raw = await self.fetch_channel(url)
@ -148,8 +110,18 @@ class YouTubeClient:
"subscribers": int(stats.get("subscriberCount", 0)), "subscribers": int(stats.get("subscriberCount", 0)),
"totalViews": int(stats.get("viewCount", 0)), "totalViews": int(stats.get("viewCount", 0)),
"totalVideos": int(stats.get("videoCount", 0)), "totalVideos": int(stats.get("videoCount", 0)),
"videos": [self._video_item(v) for v in raw["videos"]], "videos": [
"recents": [self._video_item(v) for v in raw["recents"]], {
"title": v.get("snippet", {}).get("title"),
"views": int(v.get("statistics", {}).get("viewCount", 0)),
"likes": int(v.get("statistics", {}).get("likeCount", 0)),
"comments": int(v.get("statistics", {}).get("commentCount", 0)),
"date": v.get("snippet", {}).get("publishedAt"),
"duration": v.get("contentDetails", {}).get("duration"),
"url": f"https://www.youtube.com/watch?v={v['id']}",
}
for v in raw["videos"]
],
"playlists": [ "playlists": [
p.get("snippet", {}).get("title") p.get("snippet", {}).get("title")
for p in raw["playlists"] for p in raw["playlists"]

View File

@ -130,7 +130,7 @@ class YouTubeAudit(CamelModel):
channel_description: str channel_description: str
linked_urls: list[LinkedUrl] linked_urls: list[LinkedUrl]
playlists: list[str] playlists: list[str]
top_videos: list[TopVideo] = [] top_videos: list[TopVideo]
diagnosis: list[DiagnosisItem] diagnosis: list[DiagnosisItem]
@ -176,11 +176,11 @@ class FacebookPage(CamelModel):
followers: int followers: int
following: int following: int
category: str category: str
bio: str = "" bio: str
logo: str logo: str
logo_description: str logo_description: str
link: str link: str
linked_domain: str = "" linked_domain: str
reviews: int reviews: int
recent_post_age: str recent_post_age: str
has_whatsapp: bool | None = None has_whatsapp: bool | None = None
@ -222,7 +222,7 @@ class AdditionalDomain(CamelModel):
class WebsiteAudit(CamelModel): class WebsiteAudit(CamelModel):
primary_domain: str primary_domain: str
additional_domains: list[AdditionalDomain] = [] additional_domains: list[AdditionalDomain]
sns_links_on_site: bool sns_links_on_site: bool
sns_links_detail: list[SnsLink] | None = None sns_links_detail: list[SnsLink] | None = None
tracking_pixels: list[TrackingPixel] tracking_pixels: list[TrackingPixel]

View File

@ -1,4 +1,3 @@
import asyncio
import json import json
import logging import logging
from urllib.parse import urlparse from urllib.parse import urlparse
@ -8,7 +7,7 @@ from common.db.run import update_run_report, update_run_plan, select_run_report_
from common.db.source import select_run_raw_data, select_mainpage_logo_url from common.db.source import select_run_raw_data, select_mainpage_logo_url
from common.db.market import select_market from common.db.market import select_market
from integrations.llm.llm_service import LLMService from integrations.llm.llm_service import LLMService
from integrations.llm.prompt import report_prompt, plan_prompt, summarize_prompt, youtube_diagnosis_prompt, brand_consistency_prompt, critical_issues_prompt, transformation_prompt, roadmap_prompt, scores_prompt, other_channels_prompt from integrations.llm.prompt import report_prompt, plan_prompt, youtube_diagnosis_prompt, brand_consistency_prompt, critical_issues_prompt, transformation_prompt, roadmap_prompt, scores_prompt, other_channels_prompt
from integrations.llm.schemas.report import ReportOutput, ClinicSnapshot, YouTubeAudit, BrandConsistencyOutput, CriticalIssuesOutput, DiagnosisItem, TransformationProposal, RoadmapOutput, RoadmapMonth, ScoresOutput, ChannelScore, WebsiteAudit, OtherChannelsOutput, OtherChannel from integrations.llm.schemas.report import ReportOutput, ClinicSnapshot, YouTubeAudit, BrandConsistencyOutput, CriticalIssuesOutput, DiagnosisItem, TransformationProposal, RoadmapOutput, RoadmapMonth, ScoresOutput, ChannelScore, WebsiteAudit, OtherChannelsOutput, OtherChannel
from services.branding import analyze_branding from services.branding import analyze_branding
from services.instagram_audit import build_instagram_audit from services.instagram_audit import build_instagram_audit
@ -50,26 +49,6 @@ async def generate_plan(analysis_run_id: str) -> PlanOutput:
def _json(v) -> str | None: def _json(v) -> str | None:
return json.dumps(v, ensure_ascii=False) if v else None return json.dumps(v, ensure_ascii=False) if v else None
# map: 큰 입력은 LLM으로 압축 요약해 100KB 초과 에러를 방지하고, 작은 입력은 그대로 둔다.
large_fields = {
"report": _json(report),
"market_competitors": _json(market.get("competitors")),
"market_keywords": _json(market.get("keywords")),
"market_trend": _json(market.get("trend")),
"market_target_audience": _json(market.get("target_audience")),
"tiktok": _json(tiktok),
"instagram": _json(instagram),
"facebook": _json(facebook),
"naver_cafe": _json(naver_cafe),
"channel_logos": _json(branding.get("channelLogos")),
"brand_assets": _json(branding.get("brandAssets")),
}
summarized = dict(zip(
large_fields.keys(),
await asyncio.gather(*(_summarize(label, data) for label, data in large_fields.items())),
))
# reduce: 요약된 입력을 모아 최종 플랜 생성.
input_data = { input_data = {
"clinic_name": clinic.get("clinicName"), "clinic_name": clinic.get("clinicName"),
"clinic_name_en": clinic.get("clinicNameEn"), "clinic_name_en": clinic.get("clinicNameEn"),
@ -78,24 +57,24 @@ async def generate_plan(analysis_run_id: str) -> PlanOutput:
"slogan": clinic.get("slogan"), "slogan": clinic.get("slogan"),
"services": json.dumps(clinic.get("services", []), ensure_ascii=False), "services": json.dumps(clinic.get("services", []), ensure_ascii=False),
"doctors": json.dumps(clinic.get("doctors", []), ensure_ascii=False), "doctors": json.dumps(clinic.get("doctors", []), ensure_ascii=False),
"report": _json(report),
"market_competitors": _json(market.get("competitors")),
"market_keywords": _json(market.get("keywords")),
"market_trend": _json(market.get("trend")),
"market_target_audience": _json(market.get("target_audience")),
"tiktok": _json(tiktok),
"instagram": _json(instagram),
"facebook": _json(facebook),
"naver_blog": _json(_naver_blog_summary(naver_blog)), "naver_blog": _json(_naver_blog_summary(naver_blog)),
"naver_cafe": _json(naver_cafe),
"kakao_talk": _json(kakaotalk), "kakao_talk": _json(kakaotalk),
**summarized, "channel_logos": _json(branding.get("channelLogos")),
"brand_assets": _json(branding.get("brandAssets")),
} }
return await LLMService(provider="perplexity").generate(plan_prompt, input_data) return await LLMService(provider="perplexity").generate(plan_prompt, input_data)
_SUMMARIZE_THRESHOLD = 4000 # 이 길이(문자 수)를 넘는 입력만 요약 LLM 호출 (불필요한 호출 방지)
async def _summarize(label: str, data: str | None) -> str | None:
if not data or len(data) <= _SUMMARIZE_THRESHOLD:
return data
result = await LLMService(provider="perplexity").generate(summarize_prompt, {"label": label, "data": data})
return result.summary
def _build_clinic_snapshot(mainpage: dict, gangnam_unni: dict, brand_assets: dict, logo_url: str | None) -> dict: def _build_clinic_snapshot(mainpage: dict, gangnam_unni: dict, brand_assets: dict, logo_url: str | None) -> dict:
snapshot: dict = {} snapshot: dict = {}
doctors = gangnam_unni.get("doctors", []) doctors = gangnam_unni.get("doctors", [])
@ -135,13 +114,12 @@ def _naver_blog_summary(blog: dict | None) -> dict | None:
async def _build_youtube_audit(youtube: dict) -> dict: # 기획상 1개의 input channel, 다중 채널은 기획에 없음. async def _build_youtube_audit(youtube: dict) -> dict: # 기획상 1개의 input channel, 다중 채널은 기획에 없음.
videos = youtube.get("videos", []) videos = youtube.get("videos", [])
recents = youtube.get("recents", [])
yt_patch: dict = { yt_patch: dict = {
"weekly_view_growth": {"absolute": 0, "percentage": 0.0}, "weekly_view_growth": {"absolute": 0, "percentage": 0.0},
"estimated_monthly_revenue": {"min": 0, "max": 0}, "estimated_monthly_revenue": {"min": 0, "max": 0},
"linked_urls": [], "linked_urls": [],
"avg_video_length": calc_avg_video_length(videos), "avg_video_length": calc_avg_video_length(videos),
"upload_frequency": calc_upload_frequency(recents), "upload_frequency": calc_upload_frequency(videos),
} }
if youtube.get("channelName"): yt_patch["channel_name"] = youtube["channelName"] if youtube.get("channelName"): yt_patch["channel_name"] = youtube["channelName"]
if youtube.get("handle"): yt_patch["handle"] = youtube["handle"] if youtube.get("handle"): yt_patch["handle"] = youtube["handle"]
@ -363,9 +341,8 @@ async def run_plan_task(analysis_run_id: str) -> None:
result = await generate_plan(analysis_run_id) result = await generate_plan(analysis_run_id)
# profile_photo 는 brand_assets.logo_description 으로 코드가 박음 (LLM "(가이드 미보유)" 같은 hallucination 차단). # profile_photo 는 brand_assets.logo_description 으로 코드가 박음 (LLM "(가이드 미보유)" 같은 hallucination 차단).
raw = await select_run_raw_data(analysis_run_id) raw = await select_run_raw_data(analysis_run_id)
branding_list = raw.get(SourceType.BRANDING) or [] branding = raw.get(SourceType.BRANDING) or []
branding_data = branding_list[0]["raw_data"] if branding_list else {} logo_desc = (((branding[0] if branding else {}).get("brandAssets") or {}).get("logo_description")) or ""
logo_desc = (branding_data.get("brandAssets") or {}).get("logo_description") or ""
result = _patch_plan(result, logo_desc) result = _patch_plan(result, logo_desc)
await update_run_plan(analysis_run_id, result.model_dump()) await update_run_plan(analysis_run_id, result.model_dump())
logger.info("[plan] done run=%s", analysis_run_id) logger.info("[plan] done run=%s", analysis_run_id)

View File

@ -39,11 +39,7 @@ async def _describe_logo(analysis_run_id: str, info_id: int, vc: VisionClient) -
if result: if result:
break break
if result: if result:
# collect_brand_basics 가 미리 채운 brand_colors/color_palette/color_source 보존하면서 logo_* 덧붙이기. await update_raw_info_merge(info_id, {"brandAssets": result})
raw = await select_run_raw_data(analysis_run_id)
existing = ((raw.get("branding") or [{}])[0].get("raw_data") or {}).get("brandAssets") or {}
merged = {**existing, **result}
await update_raw_info_merge(info_id, {"brandAssets": merged})
logger.info("[brand_logo] done keys=%s", list(result.keys()) if result else None) logger.info("[brand_logo] done keys=%s", list(result.keys()) if result else None)

View File

@ -86,15 +86,15 @@ async def collect_gangnam_unni(analysis_run_id: str, info_id: int, url: str) ->
logger.info("[gangnam_unni] done run=%s", analysis_run_id) logger.info("[gangnam_unni] done run=%s", analysis_run_id)
def _extract_website_audit(html: str, url: str) -> dict: def _extract_website_audit(html: str, site_links: list[str], url: str) -> dict:
"""raw HTML 에서 main CTA / tracking pixels / SNS / additional domains 정규식 추출.""" """raw HTML + siteLinks 에서 main CTA / tracking pixels / SNS / additional domains 정규식 추출."""
if not html: if not html:
return {} return {}
primary_host = urlparse(url).netloc primary_host = urlparse(url).netloc
result: dict = { result: dict = {
"trackingPixels": extract_tracking_pixels(html), "trackingPixels": extract_tracking_pixels(html),
"snsLinks": extract_sns_links(html), "snsLinks": extract_sns_links(site_links, html),
"additionalDomains": extract_additional_domains(html, primary_host), "additionalDomains": extract_additional_domains(site_links, primary_host),
} }
html_cta = extract_main_cta(html) html_cta = extract_main_cta(html)
if html_cta: if html_cta:
@ -115,7 +115,7 @@ async def collect_mainpage(analysis_run_id: str, info_id: int, hospital_id: str,
html = data.pop("html", "") or "" # raw_data 에는 저장 안 함 — 여기서만 사용하고 버림. html = data.pop("html", "") or "" # raw_data 에는 저장 안 함 — 여기서만 사용하고 버림.
# 홈페이지 URL 자체도 raw_data 에 박아둬야 brand_assets / 분석 단계에서 mainpage URL 재조회 없이 사용 가능. # 홈페이지 URL 자체도 raw_data 에 박아둬야 brand_assets / 분석 단계에서 mainpage URL 재조회 없이 사용 가능.
data = {**data, "sourceUrl": url} data = {**data, "sourceUrl": url}
data.update(_extract_website_audit(html, url)) data.update(_extract_website_audit(html, data.get("siteLinks", []), url))
await update_raw_info(info_id, data) await update_raw_info(info_id, data)
await update_hospital(hospital_id, data, analysis_run_id=analysis_run_id) await update_hospital(hospital_id, data, analysis_run_id=analysis_run_id)

View File

@ -89,8 +89,7 @@ def _logo_data(channel_logos: dict, channel: str) -> dict:
} }
def _page_patch(item: dict, channel_logos) -> dict: def _page_patch(item: dict, channel_logos) -> dict:
p: dict = {"page_name": "", "followers": 0, "category": "", "reviews": 0, p: dict = {}
"following": 0, "recent_post_age": "", "post_frequency": "", "engagement": ""}
fb = item["raw_data"] fb = item["raw_data"]
language = item.get("language") if item.get("language") else "KR" language = item.get("language") if item.get("language") else "KR"
label = "페이스북 " + language label = "페이스북 " + language

View File

@ -64,8 +64,8 @@ def build_kpi_dashboard(
print("facebook_kpis", facebook_kpis) print("facebook_kpis", facebook_kpis)
print("instagram_kpis", instagram_kpis) print("instagram_kpis", instagram_kpis)
kpis += [k for k in facebook_kpis if k] kpis += facebook_kpis
kpis += [k for k in instagram_kpis if k] kpis += instagram_kpis
for k in [ for k in [
_follower_kpi("YouTube 구독자", youtube.get("subscribers")), _follower_kpi("YouTube 구독자", youtube.get("subscribers")),
_follower_kpi("TikTok 팔로워", tiktok.get("followers")), _follower_kpi("TikTok 팔로워", tiktok.get("followers")),

View File

@ -1,11 +1,13 @@
"""HTML 정규식 기반 deterministic 추출 — tracking pixels / SNS / additional domains / main CTA.""" """collect 단계 - HTML / siteLinks 에서 tracking pixels / SNS links / additional domains 추출.
모두 정규식·도메인 매칭 기반 deterministic 추출 (LLM 미경유)."""
import re import re
from urllib.parse import urlparse from urllib.parse import urlparse
# ── tracking pixels ────────────────────────────────────────────────────────── # ── tracking pixels ──────────────────────────────────────────────────────────
# 픽셀별 시그니처: 하나라도 매치되면 installed, ID 는 첫 non-empty 그룹. # 픽셀별 시그니처: 이름 → 정규식 리스트. installed 판정은 OR (하나라도 매치되면 설치된 것).
# ID 캡처는 모든 패턴 시도 후 첫 non-empty 그룹 사용 (signature-only 패턴과 ID-capture 패턴 혼재 OK).
_TRACKING_PIXEL_PATTERNS: dict[str, list[re.Pattern]] = { _TRACKING_PIXEL_PATTERNS: dict[str, list[re.Pattern]] = {
"Google Analytics": [ "Google Analytics": [
re.compile(r"googletagmanager\.com/gtag/js\?id=(G-[A-Z0-9]+)", re.IGNORECASE), re.compile(r"googletagmanager\.com/gtag/js\?id=(G-[A-Z0-9]+)", re.IGNORECASE),
@ -37,7 +39,8 @@ _TRACKING_PIXEL_PATTERNS: dict[str, list[re.Pattern]] = {
def extract_tracking_pixels(html: str) -> list[dict]: def extract_tracking_pixels(html: str) -> list[dict]:
"""HTML 에서 트래킹 픽셀 설치 여부 + ID 추출.""" """HTML 에서 트래킹 픽셀 설치 여부 + ID 추출. 환각 0 — 정규식 매치만 신뢰.
모든 패턴 시도 하나라도 매치되면 installed. ID 캡처된 non-empty 그룹 사용."""
if not html: if not html:
return [] return []
pixels: list[dict] = [] pixels: list[dict] = []
@ -63,15 +66,9 @@ def extract_tracking_pixels(html: str) -> list[dict]:
# ── main CTA ───────────────────────────────────────────────────────────────── # ── main CTA ─────────────────────────────────────────────────────────────────
_CTA_KEYWORDS = re.compile( _CTA_KEYWORDS = re.compile(
r"(상담|예약|문의|신청|등록|Book\s*Now|Consult|Reservation|Contact|Apply)", r"(상담|예약|문의|신청|등록|진료|Book\s*Now|Consult|Reservation|Contact|Apply)",
re.IGNORECASE, re.IGNORECASE,
) )
# CTA 카테고리 우선순위 (전화>카톡>예약). 매치 없으면 결과에 포함 X.
_CTA_CATEGORIES: list[tuple[str, re.Pattern]] = [
("전화 상담", re.compile(r"전화|\bCall\b|\bPhone\b", re.IGNORECASE)),
("카카오톡 상담", re.compile(r"카(카오)?톡|Kakao", re.IGNORECASE)),
("온라인 예약", re.compile(r"예약|Reserv|Book", re.IGNORECASE)),
]
_BUTTON_TAG = re.compile(r"<button\b[^>]*>(.*?)</button>", re.IGNORECASE | re.DOTALL) _BUTTON_TAG = re.compile(r"<button\b[^>]*>(.*?)</button>", re.IGNORECASE | re.DOTALL)
_ANCHOR_TAG = re.compile(r"<a\b[^>]*>(.*?)</a>", re.IGNORECASE | re.DOTALL) _ANCHOR_TAG = re.compile(r"<a\b[^>]*>(.*?)</a>", re.IGNORECASE | re.DOTALL)
@ -81,37 +78,22 @@ def _clean_text(inner: str) -> str:
return re.sub(r"\s+", " ", text).strip() return re.sub(r"\s+", " ", text).strip()
_MAIN_CTA_LIMIT = 3
def _classify_cta(text: str) -> str | None:
"""CTA 텍스트 → 카테고리 (전화/카톡/예약). 매치 없으면 None."""
for label, pat in _CTA_CATEGORIES:
if pat.search(text):
return label
return None
def extract_main_cta(html: str) -> str: def extract_main_cta(html: str) -> str:
"""<button>·<a> 라벨을 카테고리로 분류 후 우선순위 순서로 최대 _MAIN_CTA_LIMIT 개 join.""" """HTML 에서 primary CTA 텍스트 추출. 우선순위: <button> 첫 매치 → <a> 첫 매치.
CTA 키워드 매칭 + 길이 1~20 (버튼 라벨 추정)."""
if not html: if not html:
return "" return ""
found: set[str] = set()
for pat in (_BUTTON_TAG, _ANCHOR_TAG): for pat in (_BUTTON_TAG, _ANCHOR_TAG):
for m in pat.finditer(html): for m in pat.finditer(html):
text = _clean_text(m.group(1)) text = _clean_text(m.group(1))
if text and 1 <= len(text) <= 20 and _CTA_KEYWORDS.search(text): if text and 1 <= len(text) <= 20 and _CTA_KEYWORDS.search(text):
cat = _classify_cta(text) return text
if cat: return ""
found.add(cat)
ordered_labels = [label for label, _ in _CTA_CATEGORIES]
result = [label for label in ordered_labels if label in found]
return " + ".join(result[:_MAIN_CTA_LIMIT])
# ── SNS links ──────────────────────────────────────────────────────────────── # ── SNS links ────────────────────────────────────────────────────────────────
# SNS 도메인 → 표준 platform 이름. # SNS 도메인 → 표준 platform 이름. siteLinks 필터링 + DOM 위치 판정에 공유.
_SNS_DOMAINS: dict[str, str] = { _SNS_DOMAINS: dict[str, str] = {
"facebook.com": "Facebook", "facebook.com": "Facebook",
"instagram.com": "Instagram", "instagram.com": "Instagram",
@ -128,16 +110,11 @@ _SNS_DOMAINS: dict[str, str] = {
_FOOTER_BLOCK = re.compile(r"<footer\b[^>]*>(.*?)</footer>", re.IGNORECASE | re.DOTALL) _FOOTER_BLOCK = re.compile(r"<footer\b[^>]*>(.*?)</footer>", re.IGNORECASE | re.DOTALL)
_HEADER_BLOCK = re.compile(r"<header\b[^>]*>(.*?)</header>", re.IGNORECASE | re.DOTALL) _HEADER_BLOCK = re.compile(r"<header\b[^>]*>(.*?)</header>", re.IGNORECASE | re.DOTALL)
# <a> 태그 안의 href 만 매치 — <link>/<script> 등 리소스 호출 제외 (CDN 노이즈 차단).
_ANCHOR_HREF_PATTERN = re.compile(
r"""<a\b[^>]*\bhref\s*=\s*['"](https?://[^'"\s]+)['"]""",
re.IGNORECASE,
)
def _platform_for(url: str) -> str | None: def _platform_for(url: str) -> str | None:
try: try:
host = urlparse(url).netloc.lower().removeprefix("www.") host = urlparse(url).netloc.lower().lstrip("www.")
except Exception: except Exception:
return None return None
for domain, name in _SNS_DOMAINS.items(): for domain, name in _SNS_DOMAINS.items():
@ -147,7 +124,7 @@ def _platform_for(url: str) -> str | None:
def _location_for(url: str, html: str) -> str: def _location_for(url: str, html: str) -> str:
"""url 이 <header>/<footer> 블록 안에 있는지 판정 — 둘 다 아니면 'body'.""" """url 이 HTML 의 <header>/<footer> 안에 들어있는지로 위치 판정. 둘 다 아니면 'body'."""
if not html: if not html:
return "" return ""
needle = re.escape(url) needle = re.escape(url)
@ -160,21 +137,11 @@ def _location_for(url: str, html: str) -> str:
return "body" return "body"
def _extract_anchor_hrefs(html: str) -> list[str]: def extract_sns_links(site_links: list[str], html: str = "") -> list[dict]:
"""<a href="http..."> 만 추출 — 순서 보존, 중복 제거.""" """siteLinks 에서 SNS 도메인 매칭. location 은 HTML 의 footer/header tag 블록으로 판정.
if not html: 중복 platform URL 유지."""
return []
seen: dict[str, None] = {}
for url in _ANCHOR_HREF_PATTERN.findall(html):
if url not in seen:
seen[url] = None
return list(seen.keys())
def extract_sns_links(html: str) -> list[dict]:
"""anchor href 에서 SNS 도메인 매칭. 중복 platform 은 첫 URL 만 유지."""
seen: dict[str, dict] = {} seen: dict[str, dict] = {}
for url in _extract_anchor_hrefs(html): for url in site_links or []:
platform = _platform_for(url) platform = _platform_for(url)
if not platform or platform in seen: if not platform or platform in seen:
continue continue
@ -186,63 +153,26 @@ def extract_sns_links(html: str) -> list[dict]:
return list(seen.values()) return list(seen.values())
# ── additional domains (글로벌/다국어 사이트만) ───────────────────────────────── # ── additional domains ───────────────────────────────────────────────────────
# 언어 코드 → 한국어 라벨. def extract_additional_domains(site_links: list[str], primary_host: str) -> list[dict]:
_LANG_LABEL: dict[str, str] = { """siteLinks 의 host 중 primary 도메인이 아닌 외부 도메인 모으기. SNS·맵 등 utility 도메인 제외.
"en": "영어", "eng": "영어", purpose 분류는 LLM 필요 여기선 문자열."""
"zh": "중국어", "cn": "중국어", "chn": "중국어", skip_domains = set(_SNS_DOMAINS) | {
"ja": "일본어", "jp": "일본어", "map.kakao.com", "map.naver.com", "maps.google.com", "goo.gl",
"ko": "한국어", "kor": "한국어", "kr": "한국어", }
"vi": "베트남어", "vn": "베트남어", primary = (primary_host or "").lower().lstrip("www.")
"th": "태국어", "thai": "태국어",
"ru": "러시아어",
"es": "스페인어",
"mn": "몽골어",
"ar": "아랍어", "arab": "아랍어",
"id": "인도네시아어",
"de": "독일어",
"fr": "프랑스어",
"pt": "포르투갈어",
}
_LANG_LI_PATTERN = re.compile(
r'<li\b[^>]*\bdata-lang\s*=\s*[\'"]([^\'"]+)[\'"][^>]*>(.*?)</li>',
re.IGNORECASE | re.DOTALL,
)
def extract_additional_domains(html: str, primary_host: str) -> list[dict]:
"""글로벌/다국어 사이트 anchor 수집 — data-lang 마커 + URL 서브도메인 prefix 로 식별."""
if not html:
return []
primary = (primary_host or "").lower().removeprefix("www.")
seen: dict[str, dict] = {} seen: dict[str, dict] = {}
for url in site_links or []:
def add(host: str, lang_kr: str): if not url or not url.startswith("http"):
host = host.lower().removeprefix("www.")
if not host or host == primary or host in seen:
return
seen[host] = {"domain": host, "purpose": lang_kr}
# 1) <li data-lang="xx"> 안의 anchor
for m in _LANG_LI_PATTERN.finditer(html):
lang_kr = _LANG_LABEL.get(m.group(1).lower())
if not lang_kr:
continue continue
a_m = re.search(r'<a[^>]*href\s*=\s*[\'"](https?://[^\'"\s]+)[\'"]', m.group(2), re.IGNORECASE) try:
if a_m: host = urlparse(url).netloc.lower().lstrip("www.")
add(urlparse(a_m.group(1)).netloc, lang_kr) except Exception:
continue
# 2) URL 서브도메인 prefix 가 언어 코드 if not host or host == primary or host in seen:
for url in _extract_anchor_hrefs(html): continue
host = urlparse(url).netloc.lower().removeprefix("www.") if any(host == d or host.endswith("." + d) for d in skip_domains):
parts = host.split(".") continue
if len(parts) >= 3: seen[host] = {"domain": host, "purpose": ""}
lang_kr = _LANG_LABEL.get(parts[0])
if lang_kr:
add(host, lang_kr)
return list(seen.values()) return list(seen.values())

View File

@ -12,8 +12,3 @@ services:
volumes: volumes:
- ./app:/app - ./app:/app
restart: unless-stopped restart: unless-stopped
networks:
- o2o-net
networks:
o2o-net:
external: true