SysOps: voice-agy-webbuilder-backup — 2026-06-23 10:04 UTC

This commit is contained in:
sysops
2026-06-23 10:04:23 +00:00
parent 3bf15c4850
commit 26fe76afdd
165 changed files with 47427 additions and 1264 deletions
+529 -62
View File
@@ -1,14 +1,15 @@
from __future__ import annotations
import hashlib
import json
import re
import subprocess
from urllib.parse import urlparse
from typing import Any
from urllib.parse import urljoin, urlparse
import httpx
from bs4 import BeautifulSoup
from app.db import execute, fetch_one, get_connection
from app.db import execute, fetch_all, fetch_one, get_connection
USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36 FoodlinkkBot/2.0"
@@ -45,6 +46,59 @@ def _fetch_page(url: str) -> tuple[str, str, str, str] | None:
return None
def _extract_parse_info(
html: str,
text: str,
title: str,
final_url: str,
base_url: str,
) -> dict[str, Any]:
soup = BeautifulSoup(html or "", "html.parser")
headings: list[str] = []
for tag in soup.find_all(["h1", "h2", "h3"])[:12]:
t = re.sub(r"\s+", " ", (tag.get_text() or "").strip())
if t:
headings.append(t[:140])
links_sample: list[dict[str, str]] = []
seen_hrefs: set[str] = set()
for a in soup.find_all("a", href=True):
href = (a.get("href") or "").strip()
if not href or href.startswith("#") or href.lower().startswith("javascript:"):
continue
if not href.startswith("http"):
href = urljoin(base_url or final_url, href)
if href in seen_hrefs:
continue
seen_hrefs.add(href)
label = re.sub(r"\s+", " ", (a.get_text() or "").strip())[:90]
links_sample.append({"href": href, "label": label or href})
if len(links_sample) >= 10:
break
words = len(text.split()) if text else 0
excerpt = ""
if text:
excerpt = text[:320] + ("" if len(text) > 320 else "")
meta_desc = ""
md = soup.find("meta", attrs={"name": "description"})
if md and md.get("content"):
meta_desc = str(md["content"]).strip()[:240]
return {
"title": title or "(geen titel)",
"final_url": final_url,
"word_count": words,
"char_count": len(text or ""),
"excerpt": excerpt,
"meta_description": meta_desc,
"headings": headings,
"links_count": len(seen_hrefs) if seen_hrefs else len(soup.find_all("a", href=True)),
"links_sample": links_sample,
}
def get_page_hash(url: str) -> str | None:
fetched = _fetch_page(url)
if not fetched:
@@ -53,20 +107,31 @@ def get_page_hash(url: str) -> str | None:
return hashlib.md5(text.encode("utf-8")).hexdigest()
def _save_snapshot(site_id: int, url: str, final_url: str, title: str, text: str, html: str) -> int | None:
import json
def _save_snapshot(
site_id: int,
url: str,
final_url: str,
title: str,
text: str,
html: str,
parse_info: dict[str, Any] | None = None,
) -> int | None:
parse_info = parse_info or _extract_parse_info(html, text, title, final_url, url)
metadata = {"source": "monitor", "parse": parse_info}
links_json = json.dumps(parse_info.get("links_sample") or [])
try:
with get_connection() as conn:
with conn.cursor() as cur:
cur.execute(
"""
INSERT INTO crawled_pages (url, final_url, title, content, content_html, site_id, metadata, crawled_at)
VALUES (%s, %s, %s, %s, %s, %s, %s::jsonb, NOW())
INSERT INTO crawled_pages (url, final_url, title, content, content_html, site_id, metadata, links, crawled_at)
VALUES (%s, %s, %s, %s, %s, %s, %s::jsonb, %s::jsonb, NOW())
ON CONFLICT (url) DO UPDATE SET
final_url=EXCLUDED.final_url, title=EXCLUDED.title,
content=EXCLUDED.content, content_html=EXCLUDED.content_html,
site_id=EXCLUDED.site_id, crawled_at=NOW()
site_id=EXCLUDED.site_id, metadata=EXCLUDED.metadata,
links=EXCLUDED.links, crawled_at=NOW()
RETURNING id
""",
(
@@ -76,7 +141,8 @@ def _save_snapshot(site_id: int, url: str, final_url: str, title: str, text: str
text[:50000],
html[:100000],
site_id,
json.dumps({"source": "monitor"}),
json.dumps(metadata),
links_json,
),
)
page_id = cur.fetchone()[0]
@@ -125,7 +191,8 @@ def add_site(url: str, name: str) -> dict:
)
site_id = cur.fetchone()[0]
if fetched:
_save_snapshot(site_id, url, final_url, title, text, html)
parse_info = _extract_parse_info(html, text, title, final_url, url)
_save_snapshot(site_id, url, final_url, title, text, html, parse_info)
row = fetch_one(
"SELECT id, url, name, last_hash, last_crawled, last_title, is_active, last_snapshot_id FROM monitored_sites WHERE id = %s",
(site_id,),
@@ -142,61 +209,461 @@ def remove_site(site_id: int, soft: bool = True) -> None:
execute("DELETE FROM monitored_sites WHERE id = %s", (site_id,))
def trigger_crawl(site_id: int | None = None) -> dict:
def _crawl_one_site(site: dict[str, Any]) -> dict[str, Any]:
site_id = int(site["id"])
url = site["url"]
name = site.get("name") or url
result: dict[str, Any] = {
"site_id": site_id,
"url": url,
"name": name,
"status": "ERROR",
"changed": False,
"error": None,
}
fetched = _fetch_page(url)
if not fetched:
msg = f"Kan {url} niet bereiken"
execute(
"INSERT INTO crawl_logs (site_id, status, message) VALUES (%s, %s, %s)",
(site_id, "ERROR", msg),
)
result["error"] = msg
return result
final_url, title, text, html = fetched
parse_info = _extract_parse_info(html, text, title, final_url, url)
new_hash = hashlib.md5(text.encode("utf-8")).hexdigest()
old_hash = site.get("last_hash")
changed = bool(old_hash and old_hash != new_hash)
if changed:
execute(
"INSERT INTO page_changes (site_id, old_hash, new_hash) VALUES (%s, %s, %s)",
(site_id, old_hash, new_hash),
)
execute(
"INSERT INTO crawl_logs (site_id, status, message) VALUES (%s, %s, %s)",
(site_id, "CHANGE", f"Wijziging op {url}{title}"),
)
result["status"] = "CHANGE"
else:
execute(
"INSERT INTO crawl_logs (site_id, status, message) VALUES (%s, %s, %s)",
(site_id, "OK", f"Crawl OK — {title} ({parse_info['word_count']} woorden)"),
)
result["status"] = "OK"
execute(
"""
UPDATE monitored_sites SET last_hash=%s, last_crawled=NOW(), last_title=%s WHERE id=%s
""",
(new_hash, title, site_id),
)
snapshot_id = _save_snapshot(site_id, url, final_url, title, text, html, parse_info)
result.update(parse_info)
result["changed"] = changed
result["snapshot_id"] = snapshot_id
return result
def trigger_crawl(site_id: int | None = None) -> dict[str, Any]:
if site_id:
row = fetch_one(
"SELECT id, url, name, last_hash, last_title FROM monitored_sites WHERE id = %s AND is_active = TRUE",
(site_id,),
)
sites = [dict(row)] if row else []
else:
sites = [
dict(r)
for r in fetch_all(
"SELECT id, url, name, last_hash, last_title FROM monitored_sites WHERE is_active = TRUE ORDER BY id"
)
]
if not sites:
return {
"ok": True,
"method": "inline",
"sites": 0,
"changed": 0,
"errors": 0,
"results": [],
"message": "Geen actieve monitor-sites — voeg eerst een URL toe.",
}
results: list[dict[str, Any]] = []
changed = 0
errors = 0
for site in sites:
row = _crawl_one_site(site)
results.append(row)
if row.get("status") == "ERROR":
errors += 1
if row.get("changed"):
changed += 1
return {
"ok": errors < len(sites),
"method": "inline",
"sites": len(sites),
"changed": changed,
"errors": errors,
"results": results,
}
def list_parse_results(site_id: int | None = None, limit: int = 20) -> list[dict[str, Any]]:
limit = max(1, min(limit, 50))
if site_id:
rows = fetch_all(
"""
SELECT cp.id, cp.url, cp.final_url, cp.title, cp.content, cp.metadata, cp.links,
cp.crawled_at, cp.site_id, ms.name AS site_name
FROM crawled_pages cp
LEFT JOIN monitored_sites ms ON ms.id = cp.site_id
WHERE cp.site_id = %s
ORDER BY cp.crawled_at DESC NULLS LAST
LIMIT %s
""",
(site_id, limit),
)
else:
rows = fetch_all(
"""
SELECT cp.id, cp.url, cp.final_url, cp.title, cp.content, cp.metadata, cp.links,
cp.crawled_at, cp.site_id, ms.name AS site_name
FROM crawled_pages cp
LEFT JOIN monitored_sites ms ON ms.id = cp.site_id
ORDER BY cp.crawled_at DESC NULLS LAST
LIMIT %s
""",
(limit,),
)
out: list[dict[str, Any]] = []
for row in rows:
item = dict(row)
meta = item.get("metadata") or {}
if isinstance(meta, str):
try:
meta = json.loads(meta)
except Exception:
meta = {}
parse = (meta or {}).get("parse") or {}
content = item.get("content") or ""
if not parse.get("excerpt") and content:
parse["excerpt"] = content[:320] + ("" if len(content) > 320 else "")
if not parse.get("word_count") and content:
parse["word_count"] = len(str(content).split())
links = item.get("links") or []
if isinstance(links, str):
try:
links = json.loads(links)
except Exception:
links = []
if not parse.get("links_sample") and links:
parse["links_sample"] = links
crawled = item.get("crawled_at")
if crawled is not None and hasattr(crawled, "isoformat"):
item["crawled_at"] = crawled.isoformat()
out.append(
{
"id": item.get("id"),
"site_id": item.get("site_id"),
"site_name": item.get("site_name"),
"url": item.get("url"),
"final_url": item.get("final_url"),
"title": item.get("title") or parse.get("title"),
"crawled_at": item.get("crawled_at"),
"word_count": parse.get("word_count", 0),
"excerpt": parse.get("excerpt", ""),
"meta_description": parse.get("meta_description", ""),
"headings": parse.get("headings") or [],
"links_count": parse.get("links_count", len(links)),
"links_sample": parse.get("links_sample") or links[:10],
}
)
return out
NL_STOPWORDS = frozenset(
"""
de het een en van in op te dat die dit voor met als zij ze er maar om ook al naar dan wel
kan zo nog uit over bij tot door na ons uw u uw je jij mij hem haar hun was zijn worden wordt
heb hebt heeft hebben had deed doen done the and or is are was were be been being a an to of in
for on at by from with about into through during before after above below between under again
further then once here there when where why how all each few more most other some such no nor
not only own same so than too very just don should now naar website home pagina menu contact
service cookie cookies privacy login inloggen registreren meer lees read click klik
""".split()
)
def _tokens(text: str, min_len: int = 4) -> list[str]:
if not text:
return []
raw = re.findall(r"[a-zA-Zà-üÀ-Ü0-9][a-zA-Zà-üÀ-Ü0-9\-]{2,}", text.lower())
return [t for t in raw if len(t) >= min_len and t not in NL_STOPWORDS and not t.isdigit()]
def _normalize_page_row(row: dict[str, Any], *, content_limit: int = 8000) -> dict[str, Any]:
item = dict(row)
meta = item.get("metadata") or {}
if isinstance(meta, str):
try:
meta = json.loads(meta)
except Exception:
meta = {}
parse = (meta or {}).get("parse") or {}
content = str(item.get("content") or "")
if not parse.get("excerpt") and content:
parse["excerpt"] = content[:320] + ("" if len(content) > 320 else "")
if not parse.get("word_count") and content:
parse["word_count"] = len(content.split())
links = item.get("links") or []
if isinstance(links, str):
try:
links = json.loads(links)
except Exception:
links = []
if not parse.get("links_sample") and links:
parse["links_sample"] = links
crawled = item.get("crawled_at")
if crawled is not None and hasattr(crawled, "isoformat"):
crawled = crawled.isoformat()
content_read = content[:content_limit]
if len(content) > content_limit:
content_read += "\n\n[… tekst ingekort — open volledige pagina voor alles …]"
return {
"id": item.get("id"),
"site_id": item.get("site_id"),
"site_name": item.get("site_name"),
"url": item.get("url"),
"final_url": item.get("final_url"),
"title": item.get("title") or parse.get("title"),
"crawled_at": crawled,
"word_count": parse.get("word_count", 0),
"char_count": parse.get("char_count", len(content)),
"excerpt": parse.get("excerpt", ""),
"meta_description": parse.get("meta_description", ""),
"headings": parse.get("headings") or [],
"links_count": parse.get("links_count", len(links)),
"links_sample": parse.get("links_sample") or links[:15],
"content_read": content_read,
"content_length": len(content),
"has_full_content": len(content) > 0,
}
def get_parse_page(page_id: int) -> dict[str, Any] | None:
row = fetch_one(
"""
SELECT cp.id, cp.url, cp.final_url, cp.title, cp.content, cp.metadata, cp.links,
cp.crawled_at, cp.site_id, ms.name AS site_name
FROM crawled_pages cp
LEFT JOIN monitored_sites ms ON ms.id = cp.site_id
WHERE cp.id = %s
""",
(page_id,),
)
if not row:
return None
page = _normalize_page_row(dict(row), content_limit=50000)
page["content_full"] = str(dict(row).get("content") or "")
return page
def build_parse_intelligence(
site_id: int | None = None,
query: str | None = None,
limit: int = 30,
) -> dict[str, Any]:
"""Aggregate parsed pages for analysis — hype terms, trends, readable content."""
limit = max(1, min(limit, 100))
if site_id:
rows = fetch_all(
"""
SELECT DISTINCT ON (cp.site_id)
cp.id, cp.url, cp.final_url, cp.title, cp.content, cp.metadata, cp.links,
cp.crawled_at, cp.site_id, ms.name AS site_name
FROM crawled_pages cp
LEFT JOIN monitored_sites ms ON ms.id = cp.site_id
WHERE cp.site_id = %s
ORDER BY cp.site_id, cp.crawled_at DESC NULLS LAST
""",
(site_id,),
)
else:
rows = fetch_all(
"""
SELECT DISTINCT ON (cp.site_id)
cp.id, cp.url, cp.final_url, cp.title, cp.content, cp.metadata, cp.links,
cp.crawled_at, cp.site_id, ms.name AS site_name
FROM crawled_pages cp
LEFT JOIN monitored_sites ms ON ms.id = cp.site_id
WHERE cp.site_id IS NOT NULL
ORDER BY cp.site_id, cp.crawled_at DESC NULLS LAST
LIMIT %s
""",
(limit,),
)
pages = [_normalize_page_row(dict(r)) for r in rows]
q = (query or "").strip().lower()
if q:
pages = [
p
for p in pages
if q in (p.get("title") or "").lower()
or q in (p.get("content_read") or "").lower()
or q in (p.get("excerpt") or "").lower()
or any(q in h.lower() for h in p.get("headings") or [])
]
changed_site_ids: set[int] = set()
recent_changes: list[dict[str, Any]] = []
try:
cmd = ["docker", "exec", "foodlinkk_worker", "python", "-c", "import trigger"]
subprocess.run(cmd, capture_output=True, timeout=120, check=False)
return {"ok": True, "method": "worker"}
change_rows = fetch_all(
"""
SELECT pc.site_id, pc.changed_at, ms.name, ms.url
FROM page_changes pc
JOIN monitored_sites ms ON ms.id = pc.site_id
WHERE pc.changed_at >= NOW() - INTERVAL '7 days'
ORDER BY pc.changed_at DESC
LIMIT 30
"""
)
for cr in change_rows:
sid = int(cr["site_id"])
changed_site_ids.add(sid)
ts = cr.get("changed_at")
if ts is not None and hasattr(ts, "isoformat"):
ts = ts.isoformat()
recent_changes.append(
{
"site_id": sid,
"site_name": cr.get("name"),
"url": cr.get("url"),
"changed_at": ts,
}
)
except Exception:
pass
from app.db import fetch_all
term_scores: dict[str, dict[str, Any]] = {}
heading_counts: dict[str, dict[str, Any]] = {}
if site_id:
row = fetch_one(
"SELECT id, url, last_hash, last_title FROM monitored_sites WHERE id = %s AND is_active = TRUE",
(site_id,),
)
sites = [row] if row else []
else:
sites = fetch_all(
"SELECT id, url, last_hash, last_title FROM monitored_sites WHERE is_active = TRUE"
)
def bump_term(term: str, site_name: str, weight: int = 1) -> None:
if len(term) < 3:
return
bucket = term_scores.setdefault(term, {"term": term, "score": 0, "sites": set()})
bucket["score"] += weight
if site_name:
bucket["sites"].add(site_name)
changed = 0
with get_connection() as conn:
with conn.cursor() as cur:
for site in sites:
fetched = _fetch_page(site["url"])
if not fetched:
cur.execute(
"INSERT INTO crawl_logs (site_id, status, message) VALUES (%s, %s, %s)",
(site["id"], "ERROR", f"Cannot reach {site['url']}"),
)
continue
final_url, title, text, html = fetched
new_hash = hashlib.md5(text.encode("utf-8")).hexdigest()
old_hash = site.get("last_hash")
if old_hash and old_hash != new_hash:
cur.execute(
"INSERT INTO page_changes (site_id, old_hash, new_hash) VALUES (%s, %s, %s)",
(site["id"], old_hash, new_hash),
)
cur.execute(
"INSERT INTO crawl_logs (site_id, status, message) VALUES (%s, %s, %s)",
(site["id"], "CHANGE", f"Change detected on {site['url']}{title}"),
)
changed += 1
else:
cur.execute(
"INSERT INTO crawl_logs (site_id, status, message) VALUES (%s, %s, %s)",
(site["id"], "OK", f"Crawl OK — {title}"),
)
cur.execute(
"""
UPDATE monitored_sites SET last_hash=%s, last_crawled=NOW(), last_title=%s WHERE id=%s
""",
(new_hash, title, site["id"]),
)
_save_snapshot(site["id"], site["url"], final_url, title, text, html)
return {"ok": True, "method": "inline", "changes": changed, "sites": len(sites)}
for page in pages:
site_name = page.get("site_name") or str(page.get("site_id") or "")
for tok in _tokens(page.get("title") or "", min_len=3):
bump_term(tok, site_name, 3)
for h in page.get("headings") or []:
hnorm = re.sub(r"\s+", " ", h.strip())[:80]
if len(hnorm) < 3:
continue
hc = heading_counts.setdefault(hnorm.lower(), {"label": hnorm, "count": 0, "sites": set()})
hc["count"] += 1
hc["sites"].add(site_name)
for tok in _tokens(h, min_len=3):
bump_term(tok, site_name, 4)
for tok in _tokens(page.get("content_read") or ""):
bump_term(tok, site_name, 1)
for link in page.get("links_sample") or []:
for tok in _tokens(link.get("label") or "", min_len=3):
bump_term(tok, site_name, 2)
hype_terms: list[dict[str, Any]] = []
for term, data in term_scores.items():
if data["score"] < 4:
continue
sites_list = sorted(data["sites"])
hype_terms.append(
{
"term": term,
"score": data["score"],
"site_count": len(sites_list),
"sites": sites_list[:5],
"cross_site": len(sites_list) >= 2,
}
)
hype_terms.sort(key=lambda x: (-x["score"], -x["site_count"], x["term"]))
hype_terms = hype_terms[:40]
heading_trends = []
for _key, data in heading_counts.items():
if data["count"] < 1:
continue
heading_trends.append(
{
"label": data["label"],
"count": data["count"],
"sites": sorted(data["sites"])[:6],
"cross_site": len(data["sites"]) >= 2,
}
)
heading_trends.sort(key=lambda x: (-x["count"], -len(x["sites"]), x["label"]))
heading_trends = heading_trends[:25]
top_term_set = {t["term"] for t in hype_terms[:15]}
food_signals = frozenset(
"halal vegan plantaardig biologisch bio trend nieuw actie aanbieding kip rund vlees vis "
"groente fruit snack curry kebab burger protein eiwit alternatief duurzaam premium "
"supermarkt retail assortiment prijs private label merk".split()
)
for page in pages:
signals: list[str] = []
wc = int(page.get("word_count") or 0)
sid = page.get("site_id")
if wc >= 1500:
signals.append("Rijke pagina — veel te analyseren")
elif wc >= 400:
signals.append("Normale pagina-dichtheid")
if sid in changed_site_ids:
signals.append("Recent gewijzigd — mogelijke hype/shift")
page_terms = set(_tokens((page.get("content_read") or "") + " " + " ".join(page.get("headings") or [])))
matched_hype = [t for t in top_term_set if t in page_terms]
food_hits = [t for t in page_terms if t in food_signals]
for t in matched_hype[:4]:
signals.append(f"Trend-term: {t}")
for t in food_hits[:3]:
if f"Trend-term: {t}" not in signals:
signals.append(f"Food-signaal: {t}")
if page.get("meta_description"):
signals.append("SEO meta beschikbaar")
page["signals"] = signals[:8]
page["hype_score"] = len(matched_hype) * 10 + len(food_hits) * 5 + (20 if sid in changed_site_ids else 0) + min(wc // 200, 15)
page["recently_changed"] = sid in changed_site_ids
pages.sort(key=lambda p: (-(p.get("hype_score") or 0), -(p.get("word_count") or 0)))
total_words = sum(int(p.get("word_count") or 0) for p in pages)
return {
"summary": {
"pages": len(pages),
"total_words": total_words,
"themes_detected": len(hype_terms),
"headings_unique": len(heading_trends),
"changes_7d": len(recent_changes),
"query": q or None,
},
"hype_terms": hype_terms,
"heading_trends": heading_trends,
"recent_changes": recent_changes[:12],
"pages": pages,
}