"""Recolector local de anuncios públicos para Radar Vehicular.

Consulta únicamente páginas públicas de inventario y conserva marca, modelo,
precio publicado si existe, fuente, URL y fechas de primera/última observación.
No obtiene datos de vendedores ni de cuentas privadas.
"""
from __future__ import annotations
import html, json, re, threading, time, unicodedata
from datetime import datetime, timezone, timedelta
from html.parser import HTMLParser
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from urllib.parse import urljoin, urlparse, urlunparse, parse_qsl, urlencode
from urllib.robotparser import RobotFileParser
from urllib.request import Request, urlopen
from urllib.error import HTTPError

ROOT = Path(__file__).resolve().parent
PAGE = ROOT / "Radar.html"
DBFILE = ROOT / "Radar_datos_live.json"
HOST, PORT = "127.0.0.1", 8765
INTERVAL = 24 * 60 * 60
UA = "EDARadar/1.0 (public vehicle listing aggregates)"
LOCK = threading.RLock()
STATE = {"monitor_started": None, "last_refresh": None, "sources": {}, "listings": {}}
ROBOTS_DELAY = {}
HOST_FETCHED_AT = {}
LAST_MANUAL_TRIGGER = -60.0
BRANDS = ["mercedes benz", "land rover", "alfa romeo", "great wall", "range rover", "volkswagen", "chevrolet", "hyundai", "mitsubishi", "toyota", "nissan", "suzuki", "honda", "mazda", "subaru", "bmw", "audi", "ford", "fiat", "renault", "jeep", "chery", "geely", "jac", "volvo", "peugeot", "citroen", "dodge", "isuzu", "ssangyong", "byd", "kia", "ram", "baic", "haval", "foton", "maxus", "mini", "mitsubishi", "lexus", "porsche", "jaguar", "tesla", "mahindra", "jetour", "kgm", "great wall"]
MODEL_BRANDS={"tucson":"Hyundai","hb20":"Hyundai","accent":"Hyundai","santa":"Hyundai","sonet":"Kia","sportage":"Kia","picanto":"Kia","sorento":"Kia","corolla":"Toyota","prado":"Toyota","rav4":"Toyota","hilux":"Toyota","vitz":"Toyota","aqua":"Toyota","sienta":"Toyota","premio":"Toyota","allion":"Toyota","wish":"Toyota","auris":"Toyota","swift":"Suzuki","jimny":"Suzuki","vitara":"Suzuki","outlander":"Mitsubishi","lancer":"Mitsubishi","civic":"Honda","fit":"Honda","crv":"Honda","versa":"Nissan","kicks":"Nissan","march":"Nissan","xtrail":"Nissan","tracker":"Chevrolet","onix":"Chevrolet","ranger":"Ford","escape":"Ford","explorer":"Ford","focus":"Ford","cx5":"Mazda","cx3":"Mazda","forester":"Subaru","impreza":"Subaru","durango":"Dodge","x5":"Bmw","q7":"Audi","tiguan":"Volkswagen","golf":"Volkswagen","polo":"Volkswagen"}
STOP = {"vendo", "vendo", "vender", "oferta", "oportunidad", "vehiculo", "vehiculos", "auto", "automovil", "automovil", "nuevo", "usado", "usada", "recien", "importado", "importada", "ano", "año", "modelo", "full", "equipo", "unico", "dueno", "dueno", "impecable", "estado", "automatico", "automatica", "diesel", "nafta", "flex", "turbo", "motor", "km", "kms", "con", "garantia", "permuto", "financio", "financiacion"}

def now(): return datetime.now(timezone.utc).isoformat(timespec="seconds")
def norm(s):
    s = unicodedata.normalize("NFKD", str(s or "").lower()).encode("ascii", "ignore").decode()
    return re.sub(r"\s+", " ", re.sub(r"[^a-z0-9 ]", " ", s)).strip()
def load_state():
    global STATE
    try:
        data = json.loads(DBFILE.read_text(encoding="utf-8"))
        if isinstance(data, dict) and isinstance(data.get("listings"), dict): STATE.update(data)
    except (OSError, json.JSONDecodeError): pass
    if not STATE.get("monitor_started"): STATE["monitor_started"] = now()
def persist():
    temp = DBFILE.with_suffix(".tmp")
    temp.write_text(json.dumps(STATE, ensure_ascii=False, indent=2), encoding="utf-8")
    temp.replace(DBFILE)
def fetch(url, timeout=35):
    req = Request(url, headers={"User-Agent": UA, "Accept": "text/html,application/xml,application/json;q=0.9,*/*;q=0.8"})
    with urlopen(req, timeout=timeout) as res: return res.read().decode("utf-8", "replace")
def polite_fetch(url, timeout=35):
    host=urlparse(url).netloc.lower(); delay=ROBOTS_DELAY.get(host,0)
    elapsed=time.monotonic()-HOST_FETCHED_AT.get(host,0)
    if delay>elapsed: time.sleep(delay-elapsed)
    body=fetch(url,timeout); HOST_FETCHED_AT[host]=time.monotonic(); return body
def allowed(url):
    parts = urlparse(url); robots = f"{parts.scheme}://{parts.netloc}/robots.txt"
    rp = RobotFileParser()
    try: rp.parse(fetch(robots, timeout=15).splitlines())
    except HTTPError as e:
        if e.code==404:
            ROBOTS_DELAY[parts.netloc.lower()]=0
            return True, "robots.txt no existe (HTTP 404); solo se consulta la página pública"
        return False, "No se pudo verificar robots.txt (HTTP %s)"%e.code
    except Exception: return False, "No se pudo verificar robots.txt"
    ROBOTS_DELAY[parts.netloc.lower()]=rp.crawl_delay(UA) or rp.crawl_delay("*") or 0
    return rp.can_fetch(UA, url), "robots.txt permite el acceso" if rp.can_fetch(UA, url) else "robots.txt no permite el acceso"
def split_make_model(title, url=""):
    source = norm(title)
    if len(source) < 3:
        slug = urlparse(url).path.rsplit("/", 1)[-1]
        slug = re.sub(r"-\d{5,}$", "", slug)
        source = norm(slug.replace("-", " "))
    tokens = source.split()
    while tokens and tokens[0] in STOP: tokens.pop(0)
    if not tokens: return "Sin identificar", "Modelo sin identificar"
    brand = None; size = 0
    for b in BRANDS:
        bt = b.split()
        if tokens[:len(bt)] == bt and len(bt) > size: brand, size = b, len(bt)
    if brand:
        model_tokens = tokens[size:]
    else:
        inferred=next(((MODEL_BRANDS[t],i) for i,t in enumerate(tokens) if t in MODEL_BRANDS),None)
        if inferred:
            brand,index=inferred; model_tokens=tokens[index:]
        else:
            brand = tokens[0]
            model_tokens = tokens[1:]
    while model_tokens and model_tokens[0] in {"new", "all"}: model_tokens.pop(0)
    model = []
    for tok in model_tokens:
        if re.fullmatch(r"(?:19|20)\d{2}", tok): break
        if tok in STOP: break
        model.append(tok)
        if len(model) >= 2: break
    return brand.title(), (" ".join(model).title() if model else "Modelo sin identificar")
def listing(source, url, title, brand="", model="", price=None, currency=""):
    if not brand or not model: brand, model = split_make_model(title, url)
    title = re.sub(r"\s+", " ", html.unescape(str(title or ""))).strip()[:220]
    if not url or not brand or not model: return None
    return {"source": source, "url": url, "title": title, "brand": brand.strip(), "model": model.strip(), "price": price, "currency": currency, "first_seen": None, "last_seen": None, "baseline": False}
def parse_usadospy(page):
    m = re.search(r'<script[^>]+id=["\']itemlist-jsonld["\'][^>]*>(.*?)</script>', page, re.I|re.S)
    if not m: raise ValueError("No apareció el catálogo público estructurado")
    data = json.loads(m.group(1)); out=[]
    for row in data.get("itemListElement", []):
        v=row.get("item", {}); brand=(v.get("brand") or {}).get("name", "") if isinstance(v.get("brand"),dict) else v.get("brand", "")
        model=v.get("model", ""); url=v.get("url", ""); offer=v.get("offers", {}) or {}
        if url and not url.startswith("http"): url=urljoin("https://usadospy.com/",url)
        out.append(listing("UsadosPy",url,v.get("name",f"{brand} {model}"),brand,model,offer.get("price"),offer.get("priceCurrency","")))
    return [x for x in out if x]
def parse_links(page, source, base, path_prefix, id_pattern):
    candidates={}
    # Search-result pages publish listing titles as ordinary anchors. We keep only
    # canonical vehicle-listing paths and discard navigation/company/locality links.
    pat=r'<a\b[^>]*href=["\']([^"\']+)["\'][^>]*>(.*?)</a>'
    for href,body in re.findall(pat,page,re.I|re.S):
        url=urljoin(base,html.unescape(href).split("#")[0])
        path=urlparse(url).path
        if not path.startswith(path_prefix) or not re.search(id_pattern,path): continue
        text=html.unescape(re.sub(r"<[^>]+>"," ",body)); text=re.sub(r"\s+"," ",text).strip()
        cleaned=norm(text)
        if not cleaned or cleaned.startswith(("destacado ","compro ","busco ","buscamos ")) or cleaned in {"ver mas","ver mas anuncios"}: continue
        if any(bad in cleaned for bad in ("montacargas","maquinaria pesada","tractor agricola","repuesto automotor","alquilo ")): continue
        if url in candidates and len(candidates[url])>=len(text): continue
        candidates[url]=text
    out=[]
    for url,text in candidates.items():
        # If an image-only/featured anchor has no text, use the public URL slug.
        b,m=split_make_model(text,url)
        out.append(listing(source,url,text or f"{b} {m}",b,m))
    return out

def clean_listing_url(url):
    p=urlparse(url)
    query=[(k,v) for k,v in parse_qsl(p.query,keep_blank_values=True) if not k.lower().startswith(("utm_","fbclid","gclid","trk","from"))]
    return urlunparse((p.scheme,p.netloc,p.path,"",urlencode(query),""))
def parse_public_portal(page, source, base):
    """Lee tarjetas enlazadas en páginas públicas, sin API ni sesión privada."""
    host=urlparse(base).netloc.lower().removeprefix("www."); found={}
    pattern=r'<a\b[^>]*href=["\']([^"\']+)["\'][^>]*>(.*?)</a>'
    blocked_words=("moto ","motocicleta","motocross","scooter","maquinaria","tractor","repuesto","alquilo","camion grua")
    reserved={"","autos","auto","vehiculos","vehiculo","catalogo","catalogo-de-vehiculos","buscar","search","contacto","nosotros","login","registro","publicar","vender","financiar","favoritos"}
    for href,body in re.findall(pattern,page,re.I|re.S):
        url=clean_listing_url(urljoin(base,html.unescape(href).split("#")[0]))
        parsed=urlparse(url)
        if parsed.netloc.lower().removeprefix("www.")!=host: continue
        parts=[norm(x) for x in parsed.path.split("/") if x]
        if not parts or (len(parts)==1 and parts[0] in reserved): continue
        title=html.unescape(re.sub(r"<[^>]+>"," ",body)); title=re.sub(r"\s+"," ",title).strip()
        text=norm(title)
        if len(text)<8 or any(word in text for word in blocked_words): continue
        tokens=text.split(); start=None; brand_len=0
        for i in range(len(tokens)):
            for brand in BRANDS:
                bt=brand.split()
                if tokens[i:i+len(bt)]==bt and len(bt)>brand_len: start=i; brand_len=len(bt)
            if start is not None: break
        if start is None:
            inferred=next((i for i,t in enumerate(tokens) if t in MODEL_BRANDS),None)
            if inferred is None: continue
            start=inferred
        candidate=" ".join(tokens[start:])
        brand,model=split_make_model(candidate,url)
        if not model or model=="Modelo sin identificar": continue
        if norm(model) in {"crf","africa twin","leoncino","f 750"}: continue
        key=url.lower()
        if key not in found or len(title)>len(found[key][2]):
            found[key]=(url,brand,title,model)
    return [listing(source,url,title,brand,model) for url,brand,title,model in found.values()]

def collect_public_portal(source,url):
    ok,msg=allowed(url)
    if not ok: raise PermissionError(msg)
    rows=parse_public_portal(polite_fetch(url),source,url)
    if not rows: raise ValueError("La página permitió la consulta, pero no expuso anuncios legibles en HTML público")
    return rows,"Catálogo público; sujeto a robots.txt y accesibilidad del portal"
def collect_usadospy():
    url="https://usadospy.com/"; ok,msg=allowed(url)
    if not ok: raise PermissionError(msg)
    return parse_usadospy(polite_fetch(url)), "UsadosPy permite rastreo público; catálogo estructurado"
def collect_clasicar():
    url="https://clasicar.com.py/vehiculos"; ok,msg=allowed(url)
    if not ok: raise PermissionError(msg)
    page=polite_fetch(url)
    return parse_links(page,"ClasiCar",url,"/vehiculos/",r"-[a-f0-9]{8,}$"), "Página pública /vehiculos autorizada por robots.txt; sin consultar /api"
def collect_clasipar():
    base="https://clasipar.paraguay.com"; urls=[base+"/motor/autos",base+"/motor/autos/page-2"]; all_items=[]
    for i,url in enumerate(urls):
        ok,msg=allowed(url)
        if not ok: raise PermissionError(msg)
        if i: time.sleep(3)
        all_items.extend(parse_links(polite_fetch(url),"Clasipar",base,"/motor/autos/",r"-\d{5,}$"))
    return all_items, "Categoría pública /motor/autos, páginas 1 y 2 según robots.txt"
COLLECTORS=[
    ("UsadosPy",collect_usadospy),
    ("ClasiCar",collect_clasicar),
    ("Clasipar",collect_clasipar),
    ("Carden",lambda:collect_public_portal("Carden","https://carden.com.py/")),
    ("Autoya",lambda:collect_public_portal("Autoya","https://www.autoya.com.py/")),
    ("Paraguauto",lambda:collect_public_portal("Paraguauto","https://paraguauto.com/")),
    ("Motor.com.py",lambda:collect_public_portal("Motor.com.py","https://motor.com.py/")),
    ("Usados.com.py",lambda:collect_public_portal("Usados.com.py","https://www.usados.com.py/")),
    ("Automotor Seminuevos",lambda:collect_public_portal("Automotor Seminuevos","https://automotorseminuevos.com.py/catalogo/")),
    ("Autoclick",lambda:collect_public_portal("Autoclick","https://autoclick.com.py/")),
]

def refresh_all():
    with LOCK:
        # The first poll is saved as a live-inventory baseline, not as historical
        # publication dates. Later first observations form the monitored series.
        is_baseline = not bool(STATE["listings"])
        stamp=now(); totals=0
        for name,fn in COLLECTORS:
            result={"last_attempt":stamp,"ok":False,"count":0,"message":""}
            try:
                rows,note=fn(); result.update(ok=True,count=len(rows),message=note)
                for r in rows:
                    k=name+"|"+r["url"]
                    old=STATE["listings"].get(k)
                    if old:
                        old.update({"title":r["title"],"brand":r["brand"],"model":r["model"],"price":r["price"],"currency":r["currency"],"last_seen":stamp})
                    else:
                        r.update({"first_seen":stamp,"last_seen":stamp,"baseline":is_baseline})
                        STATE["listings"][k]=r
                    totals+=1
            except Exception as e:
                result["message"]=str(e)[:260]
            STATE["sources"][name]=result
        STATE["last_refresh"]=stamp
        STATE["last_count"]=totals
        STATE["next_refresh"]=(datetime.now(timezone.utc)+timedelta(seconds=INTERVAL)).isoformat(timespec="seconds")
        persist()
def scheduler():
    while True:
        refresh_all()
        time.sleep(INTERVAL)
def start_manual_refresh():
    global LAST_MANUAL_TRIGGER
    with LOCK:
        moment=time.monotonic()
        if moment-LAST_MANUAL_TRIGGER<60: return False
        LAST_MANUAL_TRIGGER=moment
    threading.Thread(target=refresh_all,daemon=True).start()
    return True
def first_seen_date(x):
    try: return datetime.fromisoformat(x["first_seen"].replace("Z","+00:00"))
    except Exception: return datetime.now(timezone.utc)
def build_response(months):
    with LOCK:
        cutoff=datetime.now(timezone.utc)-timedelta(days=30*months)
        rows=[dict(v) for v in STATE["listings"].values() if first_seen_date(v)>=cutoff]
        groups={}
        for r in rows:
            k=norm(r["brand"]+" "+r["model"]); g=groups.setdefault(k,{"brand":r["brand"],"model":r["model"],"offers":0,"sources":set()});g["offers"]+=1;g["sources"].add(r["source"])
        ranking=sorted(({**v,"sources":sorted(v["sources"])} for v in groups.values()),key=lambda x:(-x["offers"],x["brand"],x["model"]))
        brand_groups={}
        for r in rows:
            k=norm(r["brand"]); g=brand_groups.setdefault(k,{"brand":r["brand"],"offers":0}); g["offers"]+=1
        brand_ranking=sorted(brand_groups.values(),key=lambda x:(-x["offers"],x["brand"]))[:10]
        allrows=[dict(v) for v in STATE["listings"].values()]
        historical={}
        for r in allrows:
            if r.get("baseline"): continue
            d=first_seen_date(r); key=norm(r["brand"]+" "+r["model"]); m=d.strftime("%Y-%m")
            historical.setdefault(key,{"brand":r["brand"],"model":r["model"],"months":{}})["months"][m]=historical.setdefault(key,{"brand":r["brand"],"model":r["model"],"months":{}})["months"].get(m,0)+1
        coverage_months=len({first_seen_date(r).strftime("%Y-%m") for r in allrows if not r.get("baseline")})
        leaders=sorted(historical.values(),key=lambda x:sum(x["months"].values()),reverse=True)[:6]
        if not leaders:
            # Before a full month of observations, show only leading live-snapshot
            # models and explicitly mark the estimate provisional.
            leaders=[{"brand":x["brand"],"model":x["model"],"months":{}} for x in ranking[:6]]
        forecasts=[]; nowdt=datetime.now(timezone.utc)
        for offset in range(1,7):
            target=(nowdt.replace(day=1)+timedelta(days=32*offset)).replace(day=1); ym=target.strftime("%Y-%m")
            for g in leaders:
                vals=g.get("months",{}); counts=list(vals.values()); recent=sum(counts[-3:])/min(3,len(counts)) if counts else 0
                same=[v for k,v in vals.items() if k.endswith("-"+f"{target.month:02d}")]
                estimate=(.65*(sum(same)/len(same))+.35*recent) if len(same)>=2 else recent
                # With no post-baseline observations, order is a provisional copy
                # of the initial captured inventory mix, not a historical forecast.
                if not counts:
                    estimate=next((x["offers"] for x in ranking if norm(x["brand"]+" "+x["model"])==norm(g["brand"]+" "+g["model"])),0)
                forecasts.append({"month":ym,"brand":g["brand"],"model":g["model"],"estimate":round(estimate,1),"confidence":"Media" if coverage_months>=10 else "Baja" if coverage_months>=3 else "Muy baja","basis":"Mismo mes histórico + últimos 3 meses" if len(same)>=2 else "Tendencia capturada" if counts else "Línea base inicial; provisional"})
        sources=[{"name":k,**v} for k,v in STATE["sources"].items()]
        month_counts=[]; current=datetime.now(timezone.utc); current_index=current.year*12+current.month-1
        for offset in range(11,-1,-1):
            year,month0=divmod(current_index-offset,12); ym=f"{year}-{month0+1:02d}"
            count=sum(1 for x in allrows if not x.get("baseline") and first_seen_date(x).strftime("%Y-%m")==ym)
            month_counts.append({"month":ym,"offers":count})
        projection_groups={}
        for f in forecasts:
            k=norm(f["brand"]+" "+f["model"]); g=projection_groups.setdefault(k,{"brand":f["brand"],"model":f["model"],"offers_estimated":0.0,"confidence":f["confidence"]}); g["offers_estimated"]+=f["estimate"]
        model_projection=sorted(projection_groups.values(),key=lambda x:(-x["offers_estimated"],x["brand"],x["model"]))[:10]
        return {"ok":True,"period_months":months,"monitor_started":STATE.get("monitor_started"),"last_refresh":STATE.get("last_refresh"),"next_refresh":STATE.get("next_refresh"),"coverage_months":coverage_months,"sources_ok":sum(1 for x in sources if x.get("ok")),"sources_total":len(COLLECTORS),"sources":sources,"ranking":ranking[:30],"brand_ranking":brand_ranking,"total_offers":len(rows),"forecast":forecasts,"model_projection":model_projection,"monthly":month_counts,"tracking_note":"El historial comienza cuando se inicia este recolector. La primera lectura es una línea base de inventario actual, no la fecha real de publicación. Solo se cuentan anuncios públicos; no son ventas confirmadas."}

load_state()
class Handler(BaseHTTPRequestHandler):
    def log_message(self,fmt,*args): pass
    def send(self,code,body,ctype):
        raw=body.encode("utf-8") if isinstance(body,str) else body
        self.send_response(code);self.send_header("Content-Type",ctype);self.send_header("Content-Length",str(len(raw)));self.send_header("Cache-Control","no-store");self.end_headers();self.wfile.write(raw)
    def do_GET(self):
        path=urlparse(self.path)
        if path.path=="/":
            try:self.send(200,PAGE.read_text(encoding="utf-8"),"text/html; charset=utf-8")
            except OSError:self.send(500,"No se encontró Radar.html","text/plain; charset=utf-8")
        elif path.path=="/api/data":
            import urllib.parse
            q=urllib.parse.parse_qs(path.query); n=int(q.get("months",["1"])[0]); n=n if n in (1,3,6,12) else 1
            self.send(200,json.dumps(build_response(n),ensure_ascii=False),"application/json; charset=utf-8")
        else:self.send(404,"Not found","text/plain; charset=utf-8")
    def do_POST(self):
        if urlparse(self.path).path!="/api/refresh":self.send(404,"Not found","text/plain; charset=utf-8");return
        if not start_manual_refresh(): self.send(429,json.dumps({"ok":False,"message":"Espera un minuto antes de solicitar otra lectura."}),"application/json; charset=utf-8");return
        self.send(202,json.dumps({"ok":True,"message":"Actualización iniciada"}),"application/json; charset=utf-8")

if __name__=="__main__":
    print(f"Radar disponible en http://{HOST}:{PORT}",flush=True)
    threading.Thread(target=scheduler,daemon=True).start()
    ThreadingHTTPServer((HOST,PORT),Handler).serve_forever()

