# import re
# import logging
# import json
# import requests
# import tldextract
# from typing import Dict, List, Any
# from bs4 import BeautifulSoup
# from langchain_community.tools import DuckDuckGoSearchRun
# from langchain_google_genai import ChatGoogleGenerativeAI

# # -------------------- LOGGING --------------------
# logging.basicConfig(
#     level=logging.INFO,
#     format='%(levelname)s: %(message)s'
# )
# logger = logging.getLogger(__name__)

# # -------------------- ANALYZER CLASS --------------------
# class MarketingLeadAnalyzer:
#     def __init__(self, api_key: str):
#         # Initialize Gemini with JSON mode for structured output
#         self.model = ChatGoogleGenerativeAI(
#             model="gemini-3-flash-preview",
#             temperature=0.3,
#             google_api_key=api_key,

#             model_kwargs={"response_format": {"type": "json_object"}}
#         )
#         self.search_service = DuckDuckGoSearchRun()
        
#         self.system_instruction = """
#         You are an expert SEO and Marketing Consultant. 
#         Analyze the provided Website Metadata, Page Content, and Search Signals.
        
#         Return a VALID JSON object with exactly these keys:
#         {
#             "seo_score": 0-100,
#             "conversion_score": 0-100,
#             "strengths": [],
#             "weaknesses": [],
#             "missing_opportunities": [],
#             "recommended_actions": [],
#             "tech_stack_detected": [],
#             "business_type": "string",
#             "target_audience": "string"
#         }
#         """

#     # -------------------- EXTRACTION LOGIC --------------------

#     def _get_domain(self, url: str) -> str:
#         ext = tldextract.extract(url)
#         return f"{ext.domain}.{ext.suffix}"

#     def fetch_site_data(self, url: str) -> Dict[str, Any]:
#         """Uses BeautifulSoup to extract high-signal marketing data."""
#         try:
#             headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebkit/537.36"}
#             response = requests.get(url, headers=headers, timeout=15)
#             response.raise_for_status()
#             html = response.text
#             soup = BeautifulSoup(html, "html.parser")

#             # 1. Extract Meta Tags (SEO Intent)
#             meta_data = {
#                 "title": soup.title.string if soup.title else "",
#                 "description": "",
#                 "og_image": ""
#             }
#             desc_tag = soup.find("meta", attrs={"name": "description"}) or soup.find("meta", attrs={"property": "og:description"})
#             if desc_tag: meta_data["description"] = desc_tag.get("content", "")

#             # 2. Detect Tracking Pixels & Tools
#             scripts = str(soup.find_all("script"))
#             tools = {
#                 "google_analytics": "gtag" in scripts or "google-analytics" in scripts,
#                 "facebook_pixel": "fbevents" in scripts,
#                 "shopify": "shopify" in scripts.lower(),
#                 "hotjar": "hotjar" in scripts.lower()
#             }

#             # 3. Clean Main Content (Remove Nav, Footer, Scripts)
#             for noise in soup(["script", "style", "nav", "footer", "header", "noscript"]):
#                 noise.decompose()
            
#             # Get text from main body, filtering for substantial blocks
#             lines = [line.strip() for line in soup.get_text(separator="\n").splitlines() if len(line.strip()) > 30]
#             clean_text = "\n".join(lines)[:10000] # Token safety limit

#             return {
#                 "meta": meta_data,
#                 "tools": tools,
#                 "content": clean_text,
#                 "raw_html": html # Used for tool detection fallback
#             }
#         except Exception as e:
#             logger.error(f"Failed to fetch {url}: {e}")
#             return None

#     def get_market_context(self, domain: str) -> Dict[str, str]:
#         """Gathers external signals via Search."""
#         logger.info(f"Gathering search signals for {domain}...")
#         return {
#             "reviews": self._safe_search(f"{domain} customer reviews and complaints"),
#             "ads": self._safe_search(f"{domain} recent marketing campaigns and ads"),
#             "competitors": self._safe_search(f"main competitors for {domain}")
#         }

#     def _safe_search(self, query: str) -> str:
#         try:
#             return self.search_service.invoke(query)
#         except:
#             return "No data found."

#     # -------------------- ORCHESTRATION --------------------

#     def perform_audit(self, url: str) -> Dict:
#         domain = self._get_domain(url)
#         logger.info(f"--- Starting Audit: {domain} ---")

#         # 1. Scrape
#         site_data = self.fetch_site_data(url)
#         if not site_data:
#             return {"error": "Website unreachable"}

#         # 2. Search
#         market_data = self.get_market_context(domain)

#         # 3. Analyze with Gemini
#         prompt = f"""
#         DOMAIN: {domain}
        
#         WEBSITE METADATA:
#         {json.dumps(site_data['meta'])}
        
#         DETECTED TOOLS:
#         {json.dumps(site_data['tools'])}
        
#         EXTERNAL MARKET SIGNALS:
#         {json.dumps(market_data)}
        
#         PAGE CONTENT:
#         {site_data['content']}
#         """

#         try:
#             response = self.model.invoke([
#                 ("system", self.system_instruction),
#                 ("human", prompt)
#             ])
#             clean_json = re.sub(r'^```json\s*|```$', '', response.content.strip(), flags=re.MULTILINE)
#             return json.loads(clean_json)
#         except Exception as e:
#             logger.error(f"AI Analysis failed: {e}")
#             return {"error": "AI Processing failed"}

import logging
import json
import requests
import tldextract
from typing import Dict, List, Any, Optional
from pydantic import BaseModel, Field
from bs4 import BeautifulSoup
from langchain_community.tools import DuckDuckGoSearchRun
from langchain_google_genai import ChatGoogleGenerativeAI

# -------------------- STRUCTURED OUTPUT MODEL --------------------

class AuditOutput(BaseModel):
    """Structured output for the website marketing audit."""
    seo_score: int = Field(description="SEO performance score from 0-100", ge=0, le=100)
    conversion_score: int = Field(description="Conversion optimization score from 0-100", ge=0, le=100)
    strengths: List[str] = Field(description="List of positive marketing/SEO aspects found")
    weaknesses: List[str] = Field(description="List of technical or marketing flaws")
    missing_opportunities: List[str] = Field(description="High-value tactics the business is currently ignoring")
    recommended_actions: List[str] = Field(description="Prioritized list of steps to improve performance")
    tech_stack_detected: List[str] = Field(description="Detected CMS, tracking pixels, or marketing tools")
    business_type: str = Field(description="Specific niche or industry of the business")
    target_audience: str = Field(description="Primary demographic or persona the site targets")

# -------------------- ANALYZER CLASS --------------------

class MarketingLeadAnalyzer:
    def __init__(self, api_key: str):
        # Initialize Gemini with the Pydantic model for structured output
        # Using with_structured_output is the modern LangChain way to handle this
        llm = ChatGoogleGenerativeAI(
            model="gemini-3-flash-preview", # or gemini-3-flash-preview
            temperature=0.3,
            google_api_key=api_key
        )
        self.structured_llm = llm.with_structured_output(AuditOutput)
        
        self.search_service = DuckDuckGoSearchRun()
        
        self.system_instruction = """
        You are an expert SEO and Marketing Consultant. 
        Analyze the provided Website Metadata, Page Content, and Search Signals.
        Extract technical insights and provide a high-level strategic audit.
        """

    def _get_domain(self, url: str) -> str:
        ext = tldextract.extract(url)
        return f"{ext.domain}.{ext.suffix}"

    def fetch_site_data(self, url: str) -> Optional[Dict[str, Any]]:
        """Uses BeautifulSoup to extract high-signal marketing data."""
        try:
            headers = {"User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36"}
            response = requests.get(url, headers=headers, timeout=15)
            response.raise_for_status()
            html = response.text
            soup = BeautifulSoup(html, "html.parser")

            # Extract Meta Tags
            meta_data = {
                "title": soup.title.string if soup.title else "",
                "description": "",
                "og_image": ""
            }
            desc_tag = soup.find("meta", attrs={"name": "description"}) or soup.find("meta", attrs={"property": "og:description"})
            if desc_tag: meta_data["description"] = desc_tag.get("content", "")

            # Detect Tracking Pixels & Tools
            scripts = str(soup.find_all("script"))
            tools = {
                "google_analytics": "gtag" in scripts or "google-analytics" in scripts,
                "facebook_pixel": "fbevents" in scripts,
                "shopify": "shopify" in scripts.lower(),
                "hotjar": "hotjar" in scripts.lower()
            }

            for noise in soup(["script", "style", "nav", "footer", "header", "noscript"]):
                noise.decompose()
            
            lines = [line.strip() for line in soup.get_text(separator="\n").splitlines() if len(line.strip()) > 30]
            clean_text = "\n".join(lines)[:10000]

            return {
                "meta": meta_data,
                "tools": tools,
                "content": clean_text
            }
        except Exception as e:
            logging.error(f"Failed to fetch {url}: {e}")
            return None

    def get_market_context(self, domain: str) -> Dict[str, str]:
        """Gathers external signals via Search."""
        return {
            "reviews": self._safe_search(f"{domain} customer reviews and complaints"),
            "ads": self._safe_search(f"{domain} recent marketing campaigns and ads"),
            "competitors": self._safe_search(f"main competitors for {domain}")
        }

    def _safe_search(self, query: str) -> str:
        try:
            return self.search_service.invoke(query)
        except:
            return "No data found."

    def perform_audit(self, url: str) -> Dict:
        domain = self._get_domain(url)
        logging.info(f"--- Starting Audit: {domain} ---")

        site_data = self.fetch_site_data(url)
        if not site_data:
            return {"error": "Website unreachable"}

        market_data = self.get_market_context(domain)

        prompt = f"""
        DOMAIN: {domain}
        
        WEBSITE METADATA:
        {json.dumps(site_data['meta'])}
        
        DETECTED TOOLS:
        {json.dumps(site_data['tools'])}
        
        EXTERNAL MARKET SIGNALS:
        {json.dumps(market_data)}
        
        PAGE CONTENT:
        {site_data['content']}
        """

        try:
            # The structured_llm automatically returns a Pydantic object
            analysis: AuditOutput = self.structured_llm.invoke([
                ("system", self.system_instruction),
                ("human", prompt)
            ])
            
            # Convert Pydantic object to dictionary for response
            return analysis.dict()
            
        except Exception as e:
            logging.error(f"AI Analysis failed: {e}")
            return {"error": "AI Processing failed"}