import urllib.request, json, re, statistics

URLS = [
 "https://natashaskitchen.com/banana-bread-recipe-video/",
 "https://sallysbakingaddiction.com/best-banana-bread-recipe/",
 "https://www.alattefood.com/banana-bread/",
 "https://tastesbetterfromscratch.com/our-favorite-banana-bread/",
 "https://www.davidlebovitz.com/banana-bread-or-1/",
 "https://joyfoodsunshine.com/the-most-amazing-chocolate-chip-cookies/",
 "https://pinchofyum.com/the-best-soft-chocolate-chip-cookies",
 "https://sallysbakingaddiction.com/chewy-chocolate-chip-cookies/",
 "https://cookiesfordays.com/chocolate-chip-cookie-recipe/",
 "https://melissaknorris.com/easy-homemade-chocolate-chip-cookies/",
 "https://blog.wilton.com/how-to-make-homemade-chocolate-chip-cookies/",
 "https://asimplepalate.com/blog/chicken-parmigiana/",
 "https://natashaskitchen.com/chicken-parmesan-recipe/",
 "https://pinchofyum.com/chicken-parmesan",
 "https://coleycooks.com/chicken-parmesan/",
 "https://tastesbetterfromscratch.com/baked-parmesan-crusted-chicken/",
 "https://tasteandsee.com/parmesan-crusted-chicken/",
 "https://cafedelites.com/best-fluffy-pancakes/",
 "https://www.loveandlemons.com/pancakes-recipe/",
 "https://www.laurafuentes.com/fluffy-pancakes-recipe/",
 "https://www.inspiredtaste.net/24593/essential-pancake-recipe/",
 "https://www.mostlyhomemademom.com/homemade-pancakes/",
 "https://www.daphneoz.com/recipes/my-current-favorite-chocolate-chip-cookie/",
]
UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0.0.0 Safari/537.36"
AD_HOSTS = ["googlesyndication","doubleclick","adthrive","mediavine","adsbygoogle","googletagmanager","google-analytics","raptive","ezoic"]

def fetch(url):
    req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept": "text/html"})
    with urllib.request.urlopen(req, timeout=25) as r:
        return r.read().decode("utf-8", "ignore")

def visible(html):
    h = re.sub(r"(?is)<script.*?</script>", " ", html)
    h = re.sub(r"(?is)<style.*?</style>", " ", h)
    h = re.sub(r"(?s)<[^>]+>", " ", h)
    h = re.sub(r"&[a-z#0-9]+;", " ", h)
    return re.sub(r"\s+", " ", h).strip()

def norm(s): return re.sub(r"[^a-z0-9]", "", s.lower())

def flatten(data):
    out = []
    def walk(x):
        if isinstance(x, dict):
            if "@graph" in x:
                for g in x["@graph"]: walk(g)
            out.append(x)
        elif isinstance(x, list):
            for i in x: walk(i)
    walk(data)
    return out

def get_recipe(html):
    for m in re.findall(r'(?is)<script[^>]*application/ld\+json[^>]*>(.*?)</script>', html):
        try: data = json.loads(m.strip())
        except Exception: continue
        for obj in flatten(data):
            t = obj.get("@type")
            if t == "Recipe" or (isinstance(t, list) and "Recipe" in t):
                return obj
    return None

def words_before(words, first_ing):
    nfirst = norm(first_ing)[:24]
    if not nfirst: return None
    acc = 0; nprefix = []
    joined = norm("".join(words))
    idx = joined.find(nfirst)
    if idx < 0: return None
    for i, w in enumerate(words):
        acc += len(norm(w))
        if acc >= idx:
            return i
    return None

rows = []
for url in URLS:
    try:
        html = fetch(url)
    except Exception as e:
        rows.append({"url": url, "ok": False, "err": type(e).__name__}); continue
    words = visible(html).split()
    recipe = get_recipe(html)
    wb = None
    if recipe:
        ings = recipe.get("recipeIngredient") or recipe.get("ingredients")
        if isinstance(ings, str): ings = [ings]
        if ings: wb = words_before(words, ings[0])
    ads = sum(html.lower().count(h) for h in AD_HOSTS)
    rows.append({"url": url, "ok": True, "total": len(words), "before": wb,
                 "jump": "jump to recipe" in html.lower(), "ad": ads, "kb": len(html)//1024})

def med(xs): return round(statistics.median(xs)) if xs else None
def mean(xs): return round(statistics.mean(xs)) if xs else None

ok = [r for r in rows if r.get("ok")]
befores = [r["before"] for r in ok if r.get("before") is not None]
print(f"{'SITE':38} {'before':>7} {'total':>6} {'jump':>5} {'ads':>4} {'kb':>5}")
for r in rows:
    if not r["ok"]:
        print(f"{r['url'][:38]:38}  FETCH FAIL ({r['err']})"); continue
    dom = re.sub(r"^https?://(www\.)?", "", r["url"]).split("/")[0]
    print(f"{dom:38} {str(r['before']):>7} {r['total']:>6} {('Y' if r['jump'] else '-'):>5} {r['ad']:>4} {r['kb']:>5}")

print("\n=== AGGREGATES ===")
print(f"URLs attempted: {len(rows)}   fetched OK: {len(ok)}   with recipe schema: {sum(1 for r in ok if r['before'] is not None)}")
print(f"Words before the recipe   median: {med(befores)}   mean: {mean(befores)}   max: {max(befores) if befores else '-'}   min: {min(befores) if befores else '-'}")
print(f"Total words on page       median: {med([r['total'] for r in ok])}")
print(f"Jump-to-recipe button     {sum(1 for r in ok if r['jump'])}/{len(ok)} = {round(100*sum(1 for r in ok if r['jump'])/len(ok))}%")
print(f"Ad/tracker references     median: {med([r['ad'] for r in ok])}")
print(f"Page HTML weight (KB)     median: {med([r['kb'] for r in ok])}")
