2026-05-08 17:41:15 +09:00
|
|
|
|
from __future__ import annotations
|
|
|
|
|
|
|
|
|
|
|
|
import re
|
|
|
|
|
|
|
|
|
|
|
|
from crawler_platform.app.config.loader import ProjectConfig
|
|
|
|
|
|
from crawler_platform.app.core.extractor.base import ExtractedClaim, ExtractedEntity, ExtractionBundle
|
|
|
|
|
|
from crawler_platform.app.core.extractor.rule_based import GenericRuleBasedExtractor, find_price, first_non_empty_line
|
|
|
|
|
|
from crawler_platform.app.core.ontology.mapper import normalize_predicate
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
NOTE_LABELS = {
|
|
|
|
|
|
"top_notes": ["top notes", "top note", "opening notes", "탑 노트", "탑노트", "상단 노트"],
|
|
|
|
|
|
"middle_notes": ["middle notes", "heart notes", "미들 노트", "하트 노트", "미들노트"],
|
|
|
|
|
|
"base_notes": ["base notes", "base note", "베이스 노트", "베이스노트", "잔향"],
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
FIELD_TO_PREDICATE = {
|
|
|
|
|
|
"top_notes": "hasTopNote",
|
|
|
|
|
|
"middle_notes": "hasMiddleNote",
|
|
|
|
|
|
"base_notes": "hasBaseNote",
|
|
|
|
|
|
"accords": "hasAccord",
|
|
|
|
|
|
"mood_tags": "evokesMood",
|
|
|
|
|
|
"season_tags": "suitableForSeason",
|
|
|
|
|
|
"occasion_tags": "suitableForOccasion",
|
|
|
|
|
|
"review_keywords": "hasReviewKeyword",
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
MOOD_KEYWORDS = {
|
|
|
|
|
|
"Fresh": ["fresh", "clean", "상쾌", "깨끗", "청량"],
|
|
|
|
|
|
"Romantic": ["romantic", "soft", "로맨틱", "부드러운"],
|
|
|
|
|
|
"Elegant": ["elegant", "luxury", "우아", "고급"],
|
|
|
|
|
|
"Cozy": ["cozy", "warm", "포근", "따뜻"],
|
|
|
|
|
|
"Energetic": ["bright", "sparkling", "활기", "발랄"],
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
SEASON_KEYWORDS = {
|
|
|
|
|
|
"Spring": ["spring", "봄"],
|
|
|
|
|
|
"Summer": ["summer", "여름"],
|
|
|
|
|
|
"Autumn": ["autumn", "fall", "가을"],
|
|
|
|
|
|
"Winter": ["winter", "겨울"],
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
OCCASION_KEYWORDS = {
|
|
|
|
|
|
"Daily": ["daily", "everyday", "데일리", "매일"],
|
|
|
|
|
|
"Date": ["date", "데이트"],
|
|
|
|
|
|
"Office": ["office", "work", "오피스", "출근"],
|
|
|
|
|
|
"Evening": ["evening", "night", "저녁", "밤"],
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
ACCORD_KEYWORDS = [
|
|
|
|
|
|
"citrus",
|
|
|
|
|
|
"floral",
|
|
|
|
|
|
"woody",
|
|
|
|
|
|
"musky",
|
|
|
|
|
|
"amber",
|
|
|
|
|
|
"powdery",
|
|
|
|
|
|
"green",
|
|
|
|
|
|
"spicy",
|
|
|
|
|
|
"sweet",
|
|
|
|
|
|
"fresh",
|
|
|
|
|
|
"시트러스",
|
|
|
|
|
|
"플로럴",
|
|
|
|
|
|
"우디",
|
|
|
|
|
|
"머스크",
|
|
|
|
|
|
"앰버",
|
|
|
|
|
|
"파우더리",
|
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
|
|
REVIEW_KEYWORDS = [
|
|
|
|
|
|
"long lasting",
|
|
|
|
|
|
"compliment",
|
|
|
|
|
|
"too strong",
|
|
|
|
|
|
"soft",
|
|
|
|
|
|
"fresh",
|
|
|
|
|
|
"powdery",
|
|
|
|
|
|
"지속력",
|
|
|
|
|
|
"잔향",
|
|
|
|
|
|
"호불호",
|
|
|
|
|
|
"칭찬",
|
|
|
|
|
|
"은은",
|
|
|
|
|
|
"강한",
|
|
|
|
|
|
]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
class PerfumeRuleBasedExtractor(GenericRuleBasedExtractor):
|
|
|
|
|
|
name = "perfume_rule_based"
|
|
|
|
|
|
|
|
|
|
|
|
def extract_entities(self, page_text: str, project_config: ProjectConfig) -> list[ExtractedEntity]:
|
2026-05-11 13:02:11 +09:00
|
|
|
|
product_cards = extract_product_cards(page_text)
|
|
|
|
|
|
if product_cards:
|
|
|
|
|
|
brand = valid_brand(extract_brand(page_text, "")) or infer_site_brand(page_text)
|
|
|
|
|
|
entities: list[ExtractedEntity] = []
|
|
|
|
|
|
if brand:
|
|
|
|
|
|
entities.append(ExtractedEntity("Brand", brand, confidence=0.62))
|
|
|
|
|
|
for card in product_cards:
|
|
|
|
|
|
attrs: dict[str, object] = {"name": card["name"]}
|
|
|
|
|
|
if brand:
|
|
|
|
|
|
attrs["brand"] = brand
|
|
|
|
|
|
if card.get("price"):
|
|
|
|
|
|
attrs["price"] = card["price"]
|
|
|
|
|
|
entities.append(
|
|
|
|
|
|
ExtractedEntity(
|
|
|
|
|
|
"Perfume",
|
|
|
|
|
|
str(card["name"]),
|
|
|
|
|
|
attrs,
|
|
|
|
|
|
evidence_text=str(card.get("evidence") or card["name"]),
|
|
|
|
|
|
confidence=0.74,
|
|
|
|
|
|
metadata={"page_pattern": "product_listing"},
|
|
|
|
|
|
)
|
|
|
|
|
|
)
|
|
|
|
|
|
return dedupe_entities(entities)
|
|
|
|
|
|
|
2026-05-08 17:41:15 +09:00
|
|
|
|
product_name = extract_product_name(page_text)
|
|
|
|
|
|
attrs = {"name": product_name}
|
|
|
|
|
|
brand = extract_brand(page_text, product_name)
|
|
|
|
|
|
if brand:
|
|
|
|
|
|
attrs["brand"] = brand
|
|
|
|
|
|
price = find_price(page_text)
|
|
|
|
|
|
if price:
|
|
|
|
|
|
attrs["price"] = {k: v for k, v in price.items() if k != "evidence"}
|
|
|
|
|
|
entities = [ExtractedEntity("Perfume", product_name, attrs, confidence=0.68)]
|
|
|
|
|
|
if brand:
|
|
|
|
|
|
entities.append(ExtractedEntity("Brand", brand, confidence=0.62))
|
|
|
|
|
|
for field, entity_type in [
|
|
|
|
|
|
("top_notes", "Note"),
|
|
|
|
|
|
("middle_notes", "Note"),
|
|
|
|
|
|
("base_notes", "Note"),
|
|
|
|
|
|
("accords", "Accord"),
|
|
|
|
|
|
("mood_tags", "Mood"),
|
|
|
|
|
|
("season_tags", "Season"),
|
|
|
|
|
|
("occasion_tags", "Occasion"),
|
|
|
|
|
|
("review_keywords", "Review"),
|
|
|
|
|
|
]:
|
|
|
|
|
|
for value, evidence in extract_field_values(field, page_text):
|
|
|
|
|
|
entities.append(ExtractedEntity(entity_type, value, evidence_text=evidence, confidence=0.6))
|
|
|
|
|
|
return dedupe_entities(entities)
|
|
|
|
|
|
|
|
|
|
|
|
def extract_attributes(
|
|
|
|
|
|
self,
|
|
|
|
|
|
entity: ExtractedEntity,
|
|
|
|
|
|
page_text: str,
|
|
|
|
|
|
project_config: ProjectConfig,
|
|
|
|
|
|
) -> dict[str, object]:
|
|
|
|
|
|
if entity.entity_type != "Perfume":
|
|
|
|
|
|
return {}
|
|
|
|
|
|
attrs: dict[str, object] = {}
|
|
|
|
|
|
longevity = find_metric(page_text, ["longevity", "lasting", "지속력"])
|
|
|
|
|
|
sillage = find_metric(page_text, ["sillage", "projection", "확산력", "발향"])
|
|
|
|
|
|
if longevity:
|
|
|
|
|
|
attrs["longevity"] = longevity
|
|
|
|
|
|
if sillage:
|
|
|
|
|
|
attrs["sillage"] = sillage
|
|
|
|
|
|
gender_bias = find_gender_bias(page_text)
|
|
|
|
|
|
if gender_bias:
|
|
|
|
|
|
attrs["gender_bias"] = gender_bias
|
|
|
|
|
|
return attrs
|
|
|
|
|
|
|
|
|
|
|
|
def extract_relations(
|
|
|
|
|
|
self,
|
|
|
|
|
|
entities: list[ExtractedEntity],
|
|
|
|
|
|
page_text: str,
|
|
|
|
|
|
project_config: ProjectConfig,
|
|
|
|
|
|
) -> list[ExtractedClaim]:
|
2026-05-11 13:02:11 +09:00
|
|
|
|
product_cards = extract_product_cards(page_text)
|
|
|
|
|
|
if product_cards:
|
|
|
|
|
|
claims: list[ExtractedClaim] = []
|
|
|
|
|
|
brand = next((entity for entity in entities if entity.entity_type == "Brand"), None)
|
|
|
|
|
|
for card in product_cards:
|
|
|
|
|
|
if brand:
|
|
|
|
|
|
claims.append(
|
|
|
|
|
|
ExtractedClaim(
|
|
|
|
|
|
str(card["name"]),
|
|
|
|
|
|
"Perfume",
|
|
|
|
|
|
"hasBrand",
|
|
|
|
|
|
brand.name,
|
|
|
|
|
|
"Brand",
|
|
|
|
|
|
evidence_text=brand.evidence_text or brand.name,
|
|
|
|
|
|
confidence=0.72,
|
|
|
|
|
|
confidence_reason="site brand inferred from listing page",
|
|
|
|
|
|
)
|
|
|
|
|
|
)
|
|
|
|
|
|
if card.get("price"):
|
|
|
|
|
|
claims.append(
|
|
|
|
|
|
ExtractedClaim(
|
|
|
|
|
|
str(card["name"]),
|
|
|
|
|
|
"Perfume",
|
|
|
|
|
|
"hasPrice",
|
|
|
|
|
|
object_value=card["price"],
|
|
|
|
|
|
evidence_text=str(card.get("evidence") or card["name"]),
|
|
|
|
|
|
confidence=0.76,
|
|
|
|
|
|
confidence_reason="Korean product listing price pattern matched",
|
|
|
|
|
|
)
|
|
|
|
|
|
)
|
|
|
|
|
|
return claims
|
|
|
|
|
|
|
2026-05-08 17:41:15 +09:00
|
|
|
|
perfume = next((entity for entity in entities if entity.entity_type == "Perfume"), None)
|
|
|
|
|
|
if perfume is None:
|
|
|
|
|
|
return []
|
|
|
|
|
|
claims: list[ExtractedClaim] = []
|
|
|
|
|
|
brand = next((entity for entity in entities if entity.entity_type == "Brand"), None)
|
|
|
|
|
|
if brand:
|
|
|
|
|
|
claims.append(
|
|
|
|
|
|
ExtractedClaim(
|
|
|
|
|
|
perfume.name,
|
|
|
|
|
|
"Perfume",
|
|
|
|
|
|
"hasBrand",
|
|
|
|
|
|
brand.name,
|
|
|
|
|
|
"Brand",
|
|
|
|
|
|
evidence_text=brand.evidence_text or brand.name,
|
|
|
|
|
|
confidence=0.78,
|
|
|
|
|
|
confidence_reason="brand pattern matched",
|
|
|
|
|
|
)
|
|
|
|
|
|
)
|
|
|
|
|
|
for field, predicate in FIELD_TO_PREDICATE.items():
|
|
|
|
|
|
entity_type = field_entity_type(field)
|
|
|
|
|
|
for value, evidence in extract_field_values(field, page_text):
|
|
|
|
|
|
claims.append(
|
|
|
|
|
|
ExtractedClaim(
|
|
|
|
|
|
perfume.name,
|
|
|
|
|
|
"Perfume",
|
|
|
|
|
|
predicate,
|
|
|
|
|
|
value,
|
|
|
|
|
|
entity_type,
|
|
|
|
|
|
evidence_text=evidence,
|
|
|
|
|
|
evidence_summary=f"{field} includes {value}",
|
|
|
|
|
|
confidence=field_confidence(field),
|
|
|
|
|
|
confidence_reason=f"{field} rule matched",
|
|
|
|
|
|
)
|
|
|
|
|
|
)
|
|
|
|
|
|
price = find_price(page_text)
|
|
|
|
|
|
if price:
|
|
|
|
|
|
claims.append(
|
|
|
|
|
|
ExtractedClaim(
|
|
|
|
|
|
perfume.name,
|
|
|
|
|
|
"Perfume",
|
|
|
|
|
|
"hasPrice",
|
|
|
|
|
|
object_value={k: v for k, v in price.items() if k != "evidence"},
|
|
|
|
|
|
evidence_text=price["evidence"],
|
|
|
|
|
|
confidence=0.7,
|
|
|
|
|
|
confidence_reason="price pattern matched",
|
|
|
|
|
|
)
|
|
|
|
|
|
)
|
|
|
|
|
|
return claims
|
|
|
|
|
|
|
|
|
|
|
|
def normalize_to_ontology(self, bundle: ExtractionBundle, ontology: dict[str, object]) -> ExtractionBundle:
|
|
|
|
|
|
for claim in bundle.claims:
|
|
|
|
|
|
claim.predicate = normalize_predicate(claim.predicate, ontology)
|
|
|
|
|
|
return bundle
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def extract_product_name(page_text: str) -> str:
|
|
|
|
|
|
for line in page_text.splitlines()[:8]:
|
|
|
|
|
|
clean = line.strip()
|
2026-05-11 13:02:11 +09:00
|
|
|
|
if clean and not looks_like_navigation(clean) and not is_template_placeholder(clean):
|
2026-05-08 17:41:15 +09:00
|
|
|
|
return clean[:240]
|
|
|
|
|
|
return first_non_empty_line(page_text) or "Unknown Perfume"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def extract_brand(page_text: str, product_name: str) -> str | None:
|
|
|
|
|
|
patterns = [
|
|
|
|
|
|
r"(?:brand|브랜드)\s*[::]\s*(?P<brand>[A-Za-z0-9가-힣 '&.-]{2,80})",
|
|
|
|
|
|
r"by\s+(?P<brand>[A-Z][A-Za-z0-9 '&.-]{2,80})",
|
|
|
|
|
|
]
|
|
|
|
|
|
for pattern in patterns:
|
|
|
|
|
|
match = re.search(pattern, page_text, flags=re.IGNORECASE)
|
|
|
|
|
|
if match:
|
|
|
|
|
|
return cleanup_value(match.group("brand"))
|
|
|
|
|
|
lines = [line.strip() for line in page_text.splitlines() if line.strip()]
|
|
|
|
|
|
if len(lines) >= 2 and lines[1].lower() not in product_name.lower():
|
|
|
|
|
|
candidate = cleanup_value(lines[1])
|
|
|
|
|
|
if len(candidate) <= 80 and not looks_like_navigation(candidate):
|
|
|
|
|
|
return candidate
|
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-05-11 13:02:11 +09:00
|
|
|
|
def extract_product_cards(page_text: str) -> list[dict[str, object]]:
|
|
|
|
|
|
lines = [line.strip() for line in page_text.splitlines() if line.strip()]
|
|
|
|
|
|
cards: list[dict[str, object]] = []
|
|
|
|
|
|
idx = 0
|
|
|
|
|
|
while idx < len(lines):
|
|
|
|
|
|
if lines[idx] != "상품명":
|
|
|
|
|
|
idx += 1
|
|
|
|
|
|
continue
|
|
|
|
|
|
name, name_idx = next_value_after_label(lines, idx)
|
|
|
|
|
|
if not name or is_template_placeholder(name) or name in {":", "상품명"}:
|
|
|
|
|
|
idx += 1
|
|
|
|
|
|
continue
|
|
|
|
|
|
card: dict[str, object] = {"name": cleanup_value(name), "evidence": f"상품명: {name}"}
|
|
|
|
|
|
scan_end = next_label_index(lines, "상품명", name_idx + 1) or min(len(lines), name_idx + 12)
|
|
|
|
|
|
for price_label in ("할인판매가", "판매가", "price", "Price"):
|
|
|
|
|
|
label_idx = find_label_index(lines, price_label, name_idx + 1, scan_end)
|
|
|
|
|
|
if label_idx is None:
|
|
|
|
|
|
continue
|
|
|
|
|
|
raw_price, _price_idx = next_value_after_label(lines, label_idx)
|
|
|
|
|
|
parsed = parse_price_value(raw_price)
|
|
|
|
|
|
if parsed:
|
|
|
|
|
|
card["price"] = parsed
|
|
|
|
|
|
card["evidence"] = f"{card['evidence']} / {price_label}: {raw_price}"
|
|
|
|
|
|
break
|
|
|
|
|
|
cards.append(card)
|
|
|
|
|
|
idx = scan_end
|
|
|
|
|
|
return dedupe_product_cards(cards)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def next_value_after_label(lines: list[str], label_idx: int) -> tuple[str | None, int]:
|
|
|
|
|
|
for idx in range(label_idx + 1, min(len(lines), label_idx + 5)):
|
|
|
|
|
|
value = cleanup_value(lines[idx])
|
|
|
|
|
|
if not value or value == ":":
|
|
|
|
|
|
continue
|
|
|
|
|
|
return value, idx
|
|
|
|
|
|
return None, label_idx
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def next_label_index(lines: list[str], label: str, start: int) -> int | None:
|
|
|
|
|
|
for idx in range(start, len(lines)):
|
|
|
|
|
|
if lines[idx] == label:
|
|
|
|
|
|
return idx
|
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def find_label_index(lines: list[str], label: str, start: int, end: int) -> int | None:
|
|
|
|
|
|
lower_label = label.lower()
|
|
|
|
|
|
for idx in range(start, min(end, len(lines))):
|
|
|
|
|
|
if lines[idx].lower() == lower_label:
|
|
|
|
|
|
return idx
|
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def parse_price_value(raw_price: str | None) -> dict[str, object] | None:
|
|
|
|
|
|
if not raw_price:
|
|
|
|
|
|
return None
|
|
|
|
|
|
match = re.search(r"(?P<amount>\d{1,3}(?:,\d{3})*|\d+)\s*(?P<currency>원|KRW|₩|USD|\$)?", raw_price)
|
|
|
|
|
|
if not match:
|
|
|
|
|
|
return None
|
|
|
|
|
|
currency = match.group("currency") or "KRW"
|
|
|
|
|
|
if currency in {"원", "₩"}:
|
|
|
|
|
|
currency = "KRW"
|
|
|
|
|
|
return {"amount": float(match.group("amount").replace(",", "")), "currency": currency}
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def dedupe_product_cards(cards: list[dict[str, object]]) -> list[dict[str, object]]:
|
|
|
|
|
|
seen: set[str] = set()
|
|
|
|
|
|
result: list[dict[str, object]] = []
|
|
|
|
|
|
for card in cards:
|
|
|
|
|
|
key = str(card["name"]).strip().lower()
|
|
|
|
|
|
if key in seen:
|
|
|
|
|
|
continue
|
|
|
|
|
|
seen.add(key)
|
|
|
|
|
|
result.append(card)
|
|
|
|
|
|
return result
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def infer_site_brand(page_text: str) -> str | None:
|
|
|
|
|
|
if "912 공식 홈페이지" in page_text or "912" in page_text[:500]:
|
|
|
|
|
|
return "912"
|
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def valid_brand(value: str | None) -> str | None:
|
|
|
|
|
|
if not value or is_template_placeholder(value):
|
|
|
|
|
|
return None
|
|
|
|
|
|
return value
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-05-08 17:41:15 +09:00
|
|
|
|
def extract_field_values(field: str, page_text: str) -> list[tuple[str, str]]:
|
|
|
|
|
|
if field in NOTE_LABELS:
|
|
|
|
|
|
return extract_labeled_values(page_text, NOTE_LABELS[field])
|
|
|
|
|
|
if field == "accords":
|
|
|
|
|
|
return keyword_values(page_text, ACCORD_KEYWORDS)
|
|
|
|
|
|
if field == "mood_tags":
|
|
|
|
|
|
return taxonomy_keyword_values(page_text, MOOD_KEYWORDS)
|
|
|
|
|
|
if field == "season_tags":
|
|
|
|
|
|
return taxonomy_keyword_values(page_text, SEASON_KEYWORDS)
|
|
|
|
|
|
if field == "occasion_tags":
|
|
|
|
|
|
return taxonomy_keyword_values(page_text, OCCASION_KEYWORDS)
|
|
|
|
|
|
if field == "review_keywords":
|
|
|
|
|
|
return keyword_values(page_text, REVIEW_KEYWORDS)
|
|
|
|
|
|
return []
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def extract_labeled_values(page_text: str, labels: list[str]) -> list[tuple[str, str]]:
|
|
|
|
|
|
values: list[tuple[str, str]] = []
|
|
|
|
|
|
lines = page_text.splitlines()
|
|
|
|
|
|
for idx, line in enumerate(lines):
|
|
|
|
|
|
lower = line.lower()
|
|
|
|
|
|
if any(label.lower() in lower for label in labels):
|
|
|
|
|
|
evidence = line
|
|
|
|
|
|
raw = line.split(":", 1)[-1] if ":" in line else ""
|
|
|
|
|
|
if not raw and idx + 1 < len(lines):
|
|
|
|
|
|
raw = lines[idx + 1]
|
|
|
|
|
|
evidence = f"{line} {raw}"
|
|
|
|
|
|
for value in split_values(raw):
|
|
|
|
|
|
values.append((value, evidence[:1000]))
|
|
|
|
|
|
return values
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def split_values(raw: str) -> list[str]:
|
|
|
|
|
|
raw = re.sub(r"\([^)]*\)", "", raw)
|
|
|
|
|
|
parts = re.split(r"[,/|·ㆍ]+|\band\b| 및 | 그리고 ", raw, flags=re.IGNORECASE)
|
|
|
|
|
|
return [cleanup_value(part) for part in parts if 1 < len(cleanup_value(part)) <= 80]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def keyword_values(page_text: str, keywords: list[str]) -> list[tuple[str, str]]:
|
|
|
|
|
|
lower = page_text.lower()
|
|
|
|
|
|
found: list[tuple[str, str]] = []
|
|
|
|
|
|
for keyword in keywords:
|
|
|
|
|
|
if keyword.lower() in lower:
|
|
|
|
|
|
found.append((keyword.title() if keyword.isascii() else keyword, snippet_for(page_text, keyword)))
|
|
|
|
|
|
return found
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def taxonomy_keyword_values(page_text: str, taxonomy: dict[str, list[str]]) -> list[tuple[str, str]]:
|
|
|
|
|
|
lower = page_text.lower()
|
|
|
|
|
|
found: list[tuple[str, str]] = []
|
|
|
|
|
|
for label, keywords in taxonomy.items():
|
|
|
|
|
|
for keyword in keywords:
|
|
|
|
|
|
if keyword.lower() in lower:
|
|
|
|
|
|
found.append((label, snippet_for(page_text, keyword)))
|
|
|
|
|
|
break
|
|
|
|
|
|
return found
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def snippet_for(text: str, keyword: str, window: int = 160) -> str:
|
|
|
|
|
|
index = text.lower().find(keyword.lower())
|
|
|
|
|
|
if index < 0:
|
|
|
|
|
|
return keyword
|
|
|
|
|
|
start = max(index - window // 2, 0)
|
|
|
|
|
|
end = min(index + len(keyword) + window // 2, len(text))
|
|
|
|
|
|
return text[start:end].replace("\n", " ")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def cleanup_value(value: str) -> str:
|
|
|
|
|
|
return re.sub(r"\s+", " ", value.strip(" -::[]()")).strip()
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def looks_like_navigation(value: str) -> bool:
|
|
|
|
|
|
return value.lower() in {"home", "shop", "menu", "cart", "login", "검색", "장바구니", "홈"}
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-05-11 13:02:11 +09:00
|
|
|
|
def is_template_placeholder(value: str) -> bool:
|
|
|
|
|
|
clean = value.strip()
|
|
|
|
|
|
return clean.startswith("{#") or clean.endswith("}") or clean in {
|
|
|
|
|
|
"CLONE FRAGRANCE",
|
|
|
|
|
|
"NICHE FRAGRANCE",
|
|
|
|
|
|
"HOME FRAGRANCE",
|
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-05-08 17:41:15 +09:00
|
|
|
|
def field_entity_type(field: str) -> str:
|
|
|
|
|
|
return {
|
|
|
|
|
|
"top_notes": "Note",
|
|
|
|
|
|
"middle_notes": "Note",
|
|
|
|
|
|
"base_notes": "Note",
|
|
|
|
|
|
"accords": "Accord",
|
|
|
|
|
|
"mood_tags": "Mood",
|
|
|
|
|
|
"season_tags": "Season",
|
|
|
|
|
|
"occasion_tags": "Occasion",
|
|
|
|
|
|
"review_keywords": "Review",
|
|
|
|
|
|
}[field]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def field_confidence(field: str) -> float:
|
|
|
|
|
|
return {
|
|
|
|
|
|
"top_notes": 0.82,
|
|
|
|
|
|
"middle_notes": 0.82,
|
|
|
|
|
|
"base_notes": 0.82,
|
|
|
|
|
|
"accords": 0.66,
|
|
|
|
|
|
"mood_tags": 0.62,
|
|
|
|
|
|
"season_tags": 0.62,
|
|
|
|
|
|
"occasion_tags": 0.6,
|
|
|
|
|
|
"review_keywords": 0.58,
|
|
|
|
|
|
}[field]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def find_metric(page_text: str, labels: list[str]) -> str | None:
|
|
|
|
|
|
for label in labels:
|
|
|
|
|
|
match = re.search(rf"{label}\s*[::]?\s*(?P<value>\d(?:\.\d)?/5|moderate|strong|weak|long|short|좋음|강함|약함)", page_text, re.IGNORECASE)
|
|
|
|
|
|
if match:
|
|
|
|
|
|
return cleanup_value(match.group("value"))
|
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def find_gender_bias(page_text: str) -> str | None:
|
|
|
|
|
|
lower = page_text.lower()
|
|
|
|
|
|
if "unisex" in lower or "공용" in lower:
|
|
|
|
|
|
return "unisex"
|
|
|
|
|
|
if "for women" in lower or "여성" in lower:
|
|
|
|
|
|
return "feminine"
|
|
|
|
|
|
if "for men" in lower or "남성" in lower:
|
|
|
|
|
|
return "masculine"
|
|
|
|
|
|
return None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def dedupe_entities(entities: list[ExtractedEntity]) -> list[ExtractedEntity]:
|
|
|
|
|
|
seen: set[tuple[str, str]] = set()
|
|
|
|
|
|
result: list[ExtractedEntity] = []
|
|
|
|
|
|
for entity in entities:
|
|
|
|
|
|
key = (entity.entity_type, entity.name.strip().lower())
|
|
|
|
|
|
if key in seen:
|
|
|
|
|
|
continue
|
|
|
|
|
|
seen.add(key)
|
|
|
|
|
|
result.append(entity)
|
|
|
|
|
|
return result
|