import sys, re, collections, requests
sys.path.insert(0, "proxy")
from bs4 import BeautifulSoup
from psionproxy import profiles
from psionproxy.extract import extract

URLS = {
  "Wikipedia": "https://en.wikipedia.org/wiki/Psion_Series_7",
  "BBC News":  "https://www.bbc.co.uk/news",
  "Hacker News": "https://news.ycombinator.com/",
}
UA = ("Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
      "(KHTML, like Gecko) Chrome/140.0 Safari/537.36")

def report(label, prof, html, url):
    body, was_article = extract(html, url, profile=prof)
    s = BeautifulSoup(body, "html.parser")
    t = collections.Counter(x.name for x in s.find_all())
    n = len(body.encode(prof.charset, "replace"))
    over = "OVER" if n > prof.budget_html_hard else "ok"
    print(f"  {label:6} {n:>8,}B {over:>4}/{prof.budget_html_hard//1000}K "
          f"style={t.get('style',0)} link={t.get('link',0)} "
          f"style@={len(s.select('[style]')):>3} class@={len(s.select('[class]')):>4} "
          f"table={t.get('table',0):>2} img={t.get('img',0):>3} "
          f"b/i/u={t.get('b',0)+t.get('i',0)+t.get('u',0):>3} "
          f"font={t.get('font',0):>3} div={t.get('div',0):>3}")

for name, url in URLS.items():
    try:
        html = requests.get(url, headers={"User-Agent": UA}, timeout=25).text
    except Exception as e:
        print(f"{name}: fetch failed {e}"); continue
    print(f"=== {name}  (upstream {len(html):,} chars) ===")
    report("EPOC", profiles.EPOC, html, url)
    report("CE",   profiles.CE,   html, url)
    print()
