diff --git a/backend/app/categorize.py b/backend/app/categorize.py new file mode 100644 index 0000000..ec33e80 --- /dev/null +++ b/backend/app/categorize.py @@ -0,0 +1,246 @@ +"""Assign a WordPress category to an article. + +The rules mirror the one-off backfill of 2026-07-31 that moved 955 existing +posts out of the catch-all "Allgemein" category, so newly published articles +land in the same buckets as the archive. + +Matching notes: +- Keywords match on a word boundary. Plain substring matching is wrong here: + "klage" would match the city name "Klagenfurt". +- A trailing "*" opts a keyword into German compound nouns, so "campingplatz*" + also matches "Campingplatzbetreiber". +- Tags weigh more than the title. Tags were chosen deliberately, while titles + often carry incidental place names. +""" + +from __future__ import annotations + +import re +from collections.abc import Sequence + +# Category key -> WordPress category slug. +CATEGORY_SLUGS: dict[str, str] = { + "angebote-rabatte": "angebote-rabatte", + "in-eigener-sache": "in-eigener-sache", + "recht-vorschriften": "recht-vorschriften", + "fahrzeug-technik": "fahrzeug", + "ausruestung-tests": "ausruestung-tests", + "campingplaetze": "campingplaetze", + "stellplaetze": "stellplaetze", + "news-branche": "news", + "reiseziele": "reiseziele", + "camping-tipps": "camping-tipps", + "vanlife": "vanlife", +} + +# Ordered by priority: on a points tie the earlier entry wins. +_RULES: tuple[tuple[str, tuple[tuple[str, int], ...]], ...] = ( + ("angebote-rabatte", ( + ("amazon sale", 3), ("amazon-sale", 3), ("amazon-angebote", 3), ("im sale", 3), + ("rabatt*", 3), ("schnaeppchen*", 3), ("prime day", 3), ("black friday", 3), + ("gutschein*", 3), ("sparangebot*", 3), ("top-angebot*", 3), ("sommerangebot*", 3), + ("knaller", 3), ("ausverkauf", 3), ("tiefpreis*", 3), ("rekordpreis*", 3), + ("preissturz", 3), ("bestpreis*", 3), ("deal", 2), ("deals", 2), + # Lidl and Aldi articles are always discount posts on this blog, so they + # carry enough weight to trigger the override below on their own. + ("lidl", 3), ("aldi", 3), ("im angebot", 3), + ("reduziert", 2), ("sale", 1), ("prozent auf", 3), ("guenstiger*", 1), + )), + ("in-eigener-sache", ( + ("vanityontour", 3), ("vanitycast", 3), ("expense logbook", 3), + ("jahresrueckblick", 3), ("rueckblick", 2), ("website", 3), ("in eigener sache", 3), + ("blog", 2), ("discord", 3), ("newsletter", 2), ("appstore", 2), ("app store", 2), + ("ausfall", 2), ("stoerung", 2), ("podcast", 2), ("campertag", 3), ("styyl", 3), + )), + ("recht-vorschriften", ( + ("stvo", 3), ("fuehrerschein*", 3), ("bussgeld*", 3), ("gesetz*", 3), + ("vorschrift*", 3), ("verkehrsrecht*", 3), ("versicherungsschutz", 3), + ("tempolimit", 3), ("promillegrenze", 3), ("maut", 3), ("urteil*", 3), + ("rechtslage", 3), ("abgemahnt", 3), ("bauantrag", 3), ("baurecht", 3), + ("genehmigung*", 2), ("verboten", 2), ("erlaubt", 2), ("strafe*", 2), ("haftung", 2), + ("illegal", 2), ("stellplatzverordnung", 3), ("datenschutz", 2), ("kurtaxe", 3), + ("uebernachtungssteuer", 3), ("kfz-versicherung", 3), ("kfz versicherung", 3), + ("versicherung*", 1), ("handyverbot", 3), ("regeln", 1), + ("landschaftsschutzgebiet", 2), + )), + ("fahrzeug-technik", ( + ("wohnmobil*", 3), ("wohnwagen*", 3), ("reisemobil*", 3), ("kastenwagen", 3), + ("camper van", 3), ("camper-van*", 3), ("truma", 3), ("klimaanlage*", 3), + ("standheizung*", 3), ("energieversorgung", 3), ("ecoflow", 3), ("jackery", 3), + ("powerstation*", 3), ("solar*", 3), ("batterie*", 3), ("wechselrichter", 3), + ("notstrom", 3), ("generator", 3), ("stromversorgung", 3), ("umbau", 3), ("ausbau", 3), + ("selbstausbau", 3), ("fahrverhalten", 3), ("reifen", 3), ("gasanlage*", 3), + ("gasflasche*", 3), ("werkstatt", 3), ("tuev", 3), ("caravaning", 3), ("caravan", 2), + ("dometic", 3), ("anhaengerkupplung", 3), ("chassis", 3), ("dieselheizung*", 3), + ("wasserpumpe*", 3), ("osram", 3), ("auffahrkeil*", 3), ("mover", 3), ("dachluke*", 3), + ("aufbau", 1), ("heizung", 2), ("strom", 2), ("motor", 2), ("led", 2), ("autark", 2), + ("ladedose", 3), ("bordtechnik", 3), ("gewicht", 1), ("auflastung", 3), + ("wiegeaktion", 3), ("gebrauchtwagen*", 3), ("kfz", 2), ("gasversorgung", 3), + ("flaschengas*", 3), ("marder*", 3), ("reparatur", 2), ("diy", 2), + )), + ("ausruestung-tests", ( + ("kuehlbox*", 3), ("kuehltasche*", 3), ("schlafsack*", 3), ("schlafsaecke", 3), + ("dachzelt*", 3), ("campingbett*", 3), ("zelt*", 3), ("gaskocher", 3), + ("campingkocher", 3), ("kopfkissen", 3), ("campingstuhl*", 3), ("campingstuehle", 3), + ("sonnensegel", 3), ("luftmatratze*", 3), ("isomatte*", 3), ("campingdusche*", 3), + ("produkttest*", 3), ("testbericht*", 3), ("im test", 3), ("getestet", 3), + ("decathlon", 3), ("coleman", 3), ("campingaz", 3), ("quechua", 3), ("ausruestung", 3), + ("campingausruestung", 3), ("gadget*", 3), ("markise*", 3), ("nachttisch*", 3), + ("geschirr", 3), ("grill*", 3), ("campingmoebel", 3), ("stirnlampe*", 3), + ("powerbank*", 3), ("wasserkanister", 3), ("vorzelt*", 3), ("haengematte*", 3), + ("campingkueche*", 3), ("campingtisch*", 3), ("camping-helfer", 3), + ("kaffeemaschine*", 3), ("campingtoilette*", 3), ("rucksack*", 3), ("wanderschuh*", 3), + ("thermacell", 3), ("kabeltrommel*", 3), ("fernglas", 3), ("fernglaeser", 3), + ("tarp", 3), ("lichterkette*", 3), ("klapptisch*", 3), ("uv-schutz", 2), + ("wasserdicht*", 2), ("belueftung", 2), ("kaufberatung", 3), ("marktcheck", 3), + ("mueckenschutz", 3), ("sonnenschutz", 2), ("ventilator*", 3), ("campinggeschirr", 3), + ("campingzelt*", 3), ("wurfzelt*", 3), ("tunnelzelt*", 3), ("familienzelt*", 3), + ("trekkingzelt*", 3), ("aufblasbare*", 2), ("trenntoilette*", 3), + ("kassettentoilette*", 3), ("toilette*", 2), ("gaswarner", 3), ("rauchmelder", 3), + ("feuerloescher", 3), ("co melder", 3), ("router", 3), ("mobilfunk", 3), ("wlan", 3), + ("lte", 3), ("5g", 3), ("internet", 2), ("buchtipp*", 2), + ("duschzelt*", 3), ("kuppelzelt*", 3), ("pop-up-zelt*", 3), ("messer", 3), + ("stuhl", 3), ("faltstuhl*", 3), ("klappstuhl*", 3), + )), + ("campingplaetze", ( + ("campingplatz*", 3), ("campingplaetze", 3), ("5-sterne*", 3), ("fuenf sterne", 3), + ("adac-superplatz", 3), ("adac bewertung", 3), ("adac-bewertung", 3), + ("superplatz*", 3), ("glamping", 3), ("wellness-camping", 3), ("suedsee-camp", 3), + ("wirthshof", 3), ("trekking-camp*", 3), ("naturcamping*", 3), ("campingpark*", 3), + ("ferienpark*", 3), ("sanitaer*", 3), ("platz des jahres", 3), ("campingdorf", 3), + ("campinganlage*", 3), ("campingplatzbetreiber", 3), ("campingfuehrer", 3), + ("pincamp", 3), ("dauercamp*", 3), ("camping-check", 3), ("campingresort", 3), + ("stammgaeste", 2), ("kinderanimation", 2), ("resort", 2), ("wellness", 2), + ("campingplatz-ranking", 3), ("platzbewertung", 3), + )), + ("stellplaetze", ( + ("stellplatz*", 3), ("stellplaetze", 3), ("freistehen", 3), ("wildcamp*", 3), + ("camping-car park", 3), ("wohnmobilstellplatz*", 3), ("wohnmobilstellplaetze", 3), + ("wohnmobilhafen", 3), ("wohnmobil-stellplaetze", 3), ("stellplatzfuehrer", 3), + ("uebernachtungsmoeglichkeit*", 2), ("stellplatz-radar", 3), ("parkplatz", 1), + ("parken", 1), ("church4night", 3), ("raststaette*", 2), ("autohof*", 2), + )), + ("news-branche", ( + ("bvcd", 3), ("bundesverband", 3), ("camping-boom", 3), ("campingboom", 3), + ("uebernachtungszahlen", 3), ("preisanalyse", 3), ("preisentwicklung", 3), + ("campingpreise", 3), ("preisanstieg", 3), ("preiserhoehung*", 3), ("destatis", 3), + ("camping-trends", 3), ("campingtrend*", 3), ("marktanalyse", 3), ("caravan salon", 3), + ("messe", 3), ("promobil", 3), ("insolvenz", 3), ("saisonstart", 3), + ("jahreszahlen", 3), ("uebernachtungen", 3), ("rekordzahl", 3), ("statistik*", 3), + ("umfrage", 3), ("studie", 3), ("branche", 3), ("investitionen", 2), ("tourismus", 2), + ("nachfrage", 2), ("bilanz", 2), ("rekord", 2), ("auszeichnung", 2), + ("ausgezeichnet", 2), ("preisvergleich", 2), ("eroeffnung", 2), ("eroeffnet", 2), + ("uebernimmt", 2), ("verkauft", 2), ("pressemeldung", 2), ("feuerwehr", 3), + ("polizei", 2), ("unfall", 3), ("verletzte", 3), ("hochwasser", 2), ("evakuiert", 3), + ("veranstaltung*", 2), + )), + ("reiseziele", ( + ("nordsee", 3), ("ostsee", 3), ("niedersachsen", 3), ("bodensee", 3), + ("lueneburger heide", 3), ("thueringen", 3), ("harz", 3), ("italien", 3), + ("kroatien", 3), ("schweiz", 3), ("oesterreich", 3), ("norwegen", 3), ("schweden", 3), + ("niederlande", 3), ("holland", 3), ("ruegen", 3), ("fehmarn", 3), ("gardasee", 3), + ("schwarzwald", 3), ("sauerland", 3), ("edersee", 3), ("bayern", 3), ("hessen", 3), + ("nrw", 3), ("mecklenburg*", 3), ("baden-wuerttemberg", 3), ("adriakueste", 3), + ("daenemark", 3), ("frankreich", 3), ("spanien", 3), ("portugal", 3), ("slowenien", 3), + ("weserradweg", 3), ("waldeck*", 3), ("sylt", 3), ("usedom", 3), ("allgaeu", 3), + ("eifel", 3), ("mosel", 3), ("alpen", 3), ("toskana", 3), ("brandenburg", 3), + ("sachsen", 3), ("schleswig-holstein", 3), ("rheinland-pfalz", 3), ("saarland", 3), + ("tirol", 3), ("suedtirol", 3), ("belgien", 3), ("tschechien", 3), ("polen", 3), + ("ungarn", 3), ("pyrenaeen", 3), ("nationalpark*", 3), ("reiseziel*", 3), + ("rundreise*", 3), ("roadtrip*", 3), ("reisebericht*", 3), ("ausflugsziel*", 3), + ("kurztrip*", 3), ("europa", 2), ("kueste", 2), ("region", 1), ("uckermark", 3), + ("lausitz", 3), ("schwaebische alb", 3), ("ausflug*", 2), ("geheimtipp*", 2), + ("uebersee", 2), ("kanada", 3), ("pfalz", 1), ("erzgebirge", 2), + ("spreewald", 2), ("vogtland", 2), ("rhoen", 2), ("odenwald", 2), + ("chiemsee", 2), ("ammersee", 2), ("muensterland", 2), ("ostfriesland", 2), + ("emsland", 2), ("altmuehltal", 2), ("teutoburger wald", 2), ("taunus", 2), + ("westerwald", 2), ("bergisches land", 2), ("nordwesten", 2), + )), + ("camping-tipps", ( + ("tipp", 3), ("tipps", 3), ("tricks", 3), ("ratgeber", 3), ("anleitung", 3), + ("checkliste", 3), ("packliste", 3), ("buchungstipp*", 3), ("reiseplanung", 3), + ("urlaubsplanung", 3), ("camping apps", 3), ("nebensaison", 3), ("wintercamping", 3), + ("campingsaison", 3), ("camping-saison", 3), ("angrillen", 3), ("reisetipp*", 3), + ("erste hilfe", 3), ("diebstahlschutz", 3), ("so geht", 2), ("so funktioniert", 2), + ("worauf", 2), ("das sollten", 2), ("muss man wissen", 2), ("wie man", 2), + ("vermeiden", 2), ("hygiene", 2), ("sicherheit", 2), ("vorbereitung", 2), ("hitze", 2), + ("wetter", 1), ("guide", 2), ("mythos", 2), ("mythen", 2), ("mit hund", 2), + ("hundefreundlich*", 2), + )), + ("vanlife", ( + ("vanlife", 3), ("van-life", 3), ("van life", 3), ("minimalismus", 3), + ("digitale nomaden", 3), ("aussteiger", 3), ("slow travel", 3), + ("campergemeinschaft", 3), ("camper-gemeinschaft", 3), ("lebensgefuehl", 3), + ("auszeit", 2), ("freiheit", 2), ("abenteuer", 2), ("gedanken", 2), + ("nachhaltigkeit", 2), + )), +) + +_PRIORITY = {key: i for i, (key, _) in enumerate(_RULES)} + +# A clear discount signal decides on its own: a tent on sale is first of all an +# offer. Otherwise "Ausruestung & Tests" wins on the sheer number of product +# tags and the offers category stays nearly empty. +_OFFER_KEY = "angebote-rabatte" +_OFFER_OVERRIDE_MIN_WEIGHT = 3 + + +def _normalise(text: str) -> str: + text = text.lower() + for src, dst in (("\u00e4", "ae"), ("\u00f6", "oe"), ("\u00fc", "ue"), ("\u00df", "ss")): + text = text.replace(src, dst) + text = text.replace("&", " ").replace("&", " ") + text = re.sub(r"[\u201e\u201c\u201d\u2018\u2019\u00ab\u00bb\"']", " ", text) + text = text.replace("\u2013", "-").replace("\u2014", "-") + return re.sub(r"\s+", " ", text).strip() + + +def _compile(keyword: str) -> re.Pattern[str]: + compound = keyword.endswith("*") + core = _normalise(keyword[:-1] if compound else keyword) + body = re.escape(core).replace(r"\ ", r"\s+") + tail = r"[a-z0-9]*" if compound else r"(?![a-z0-9])" + return re.compile(r"(? str | None: + """Return a category key, or None when no rule matches. + + Callers should leave the category unset in that case so the article stays + visible in WordPress' default category instead of being filed wrongly. + """ + title_text = _normalise(title or "") + tag_text = _normalise(" | ".join(tags or [])) + + scores: dict[str, int] = {} + for key, patterns in _PATTERNS.items(): + total = 0 + for pattern, weight in patterns: + if pattern.search(tag_text): + total += 3 * weight + if pattern.search(title_text): + total += 2 * weight + if total: + scores[key] = total + + if not scores: + return None + + for pattern, weight in _PATTERNS[_OFFER_KEY]: + if weight >= _OFFER_OVERRIDE_MIN_WEIGHT and ( + pattern.search(tag_text) or pattern.search(title_text) + ): + return _OFFER_KEY + + return max(scores.items(), key=lambda item: (item[1], -_PRIORITY[item[0]]))[0] + + +def category_slug(title: str, tags: Sequence[str] | None = None) -> str | None: + """Return the WordPress category slug for an article, or None.""" + key = classify(title, tags) + return CATEGORY_SLUGS.get(key) if key else None diff --git a/backend/app/config.py b/backend/app/config.py index 1a92e0c..a0c8f66 100644 --- a/backend/app/config.py +++ b/backend/app/config.py @@ -30,6 +30,11 @@ class Settings(BaseSettings): wordpress_username: str | None = Field(default=None, validation_alias=AliasChoices("WORDPRESS_USERNAME", "WP_USERNAME")) wordpress_app_password: str | None = Field(default=None, validation_alias=AliasChoices("WORDPRESS_APP_PASSWORD", "WP_PASSWORD")) wordpress_default_status: str = "draft" + # Tag hygiene: the rewriter proposes far more tags than a post needs, and + # every unknown one used to be created on the spot - that is how 955 posts + # accumulated 3.055 tags, 1.683 of them used exactly once. + wordpress_max_tags_per_post: int = 5 + wordpress_new_tag_min_proposals: int = 3 openai_api_key: str | None = Field(default=None, validation_alias=AliasChoices("OPENAI_API_KEY")) openai_model: str = "gpt-4o-mini" diff --git a/backend/app/db.py b/backend/app/db.py index b6ef898..6aa1e0f 100644 --- a/backend/app/db.py +++ b/backend/app/db.py @@ -118,6 +118,21 @@ def init_db() -> None: UNIQUE(source_url) ); + -- One row per (tag, article) the rewriter ever proposed, whether or + -- not the tag made it into WordPress. The number of rows per + -- name_key is the "is this tag established" signal that gates + -- creating the tag in WordPress. + CREATE TABLE IF NOT EXISTS tag_proposals ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + name_key TEXT NOT NULL, + name TEXT NOT NULL, + article_id INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL DEFAULT (datetime('now')), + UNIQUE(name_key, article_id) + ); + + CREATE INDEX IF NOT EXISTS idx_tag_proposals_name_key ON tag_proposals(name_key); + CREATE INDEX IF NOT EXISTS idx_articles_source_article_id ON articles(source_article_id); CREATE INDEX IF NOT EXISTS idx_articles_source_hash ON articles(source_hash); CREATE UNIQUE INDEX IF NOT EXISTS uq_articles_feed_source_article_id diff --git a/backend/app/repositories.py b/backend/app/repositories.py index cf38055..8032077 100644 --- a/backend/app/repositories.py +++ b/backend/app/repositories.py @@ -853,3 +853,41 @@ def list_articles(limit: int = 100, status_filter: str | None = None) -> list[di (safe_limit,), ).fetchall() return rows_to_dicts(rows) + + +def record_tag_proposals(names: list[str], article_id: int | None) -> dict[str, int]: + """Record that these tags were proposed for one article. + + Returns, per casefolded name, how many distinct articles have proposed it so + far - this one included. That count is what decides whether a tag is + established enough to be created in WordPress: a tag the rewriter invents + once for a single article should not become a permanent taxonomy entry. + + Re-publishing an article does not inflate the count, because the row is + keyed on (name_key, article_id). Articles without an id (ad-hoc publishing) + all share the sentinel 0 and therefore count once in total. + """ + counts: dict[str, int] = {} + if not names: + return counts + key_for_article = int(article_id or 0) + with get_conn() as conn: + for raw in names: + name = str(raw or "").strip() + if not name: + continue + name_key = name.casefold() + conn.execute( + """ + INSERT INTO tag_proposals (name_key, name, article_id) + VALUES (?, ?, ?) + ON CONFLICT(name_key, article_id) DO NOTHING + """, + (name_key, name, key_for_article), + ) + row = conn.execute( + "SELECT COUNT(*) AS n FROM tag_proposals WHERE name_key = ?", + (name_key,), + ).fetchone() + counts[name_key] = int(row["n"] if row else 0) + return counts diff --git a/backend/app/wordpress.py b/backend/app/wordpress.py index bb96198..32ec594 100644 --- a/backend/app/wordpress.py +++ b/backend/app/wordpress.py @@ -12,6 +12,8 @@ from html import unescape as _html_unescape from urllib.parse import quote_plus, urlparse from urllib.request import Request, urlopen +from . import categorize +from . import repositories from .config import get_settings @@ -90,7 +92,25 @@ def _selected_tags_from_meta(meta_json: str | None) -> list[str]: return tags -def _resolve_wp_tag_ids(*, base_url: str, auth_header: str, tags: list[str]) -> list[int]: +def _resolve_wp_tag_ids( + *, + base_url: str, + auth_header: str, + tags: list[str], + creatable: set[str], +) -> list[int]: + """Map tag names to WordPress tag IDs, creating only established ones. + + A name matches an existing tag only on an exact (case-insensitive) name. + The former fallback of taking the first search hit filed "Camping" under + the unrelated existing tag "Campingplatz" whenever the exact tag was + missing. + + `creatable` holds the casefolded names that cleared the proposal threshold + and may be created in WordPress. Everything else is dropped with a log line + rather than turned into a new tag - an empty set means this run adds nothing + to the taxonomy. + """ ids: list[int] = [] seen: set[int] = set() for tag in tags: @@ -112,12 +132,12 @@ def _resolve_wp_tag_ids(*, base_url: str, auth_header: str, tags: list[str]) -> if row_name.casefold() == name.casefold(): tag_id = rid break - if tag_id is None: - for row in result: - if isinstance(row, dict) and int(row.get("id", 0) or 0) > 0: - tag_id = int(row.get("id", 0)) - break if tag_id is None: + if name.casefold() not in creatable: + _logger.info( + "Schlagwort '%s' noch nicht etabliert - wird nicht angelegt", name + ) + continue created = _wp_request( base_url=base_url, auth_header=auth_header, @@ -129,6 +149,7 @@ def _resolve_wp_tag_ids(*, base_url: str, auth_header: str, tags: list[str]) -> rid = int(created.get("id", 0) or 0) if rid > 0: tag_id = rid + _logger.info("Schlagwort '%s' in WordPress neu angelegt (#%s)", name, rid) if tag_id is not None and tag_id > 0 and tag_id not in seen: seen.add(tag_id) ids.append(tag_id) @@ -137,6 +158,74 @@ def _resolve_wp_tag_ids(*, base_url: str, auth_header: str, tags: list[str]) -> return ids +def _creatable_tag_names(tags: list[str], article_id: Any) -> set[str]: + """Casefolded names that may be created as new WordPress tags. + + A proposed tag has to show up for `wordpress_new_tag_min_proposals` + different articles before it earns a taxonomy entry. Until then it is + counted and dropped, so one-off inventions stop accumulating. + + If the bookkeeping itself fails, nothing is creatable: existing tags are + still assigned, and the taxonomy simply does not grow that run. + """ + if not tags: + return set() + settings = get_settings() + threshold = max(1, settings.wordpress_new_tag_min_proposals) + try: + aid = int(article_id) if article_id is not None else None + except (TypeError, ValueError): + aid = None + try: + counts = repositories.record_tag_proposals(tags, aid) + except Exception as exc: + _logger.warning("Schlagwort-Zähler nicht verfügbar, lege keine neuen an: %s", exc) + return set() + return {name for name, seen in counts.items() if seen >= threshold} + + +_category_id_cache: dict[str, int | None] = {} + + +def _resolve_wp_category_id(*, base_url: str, auth_header: str, slug: str) -> int | None: + """Look up a WordPress category by slug. + + Unlike tags, categories are never created on the fly: the set is curated in + WordPress, so an unknown slug means the rules and the site drifted apart and + should be fixed there, not silently papered over with a new category. + """ + if slug in _category_id_cache: + return _category_id_cache[slug] + + try: + result = _wp_request( + base_url=base_url, + auth_header=auth_header, + method="GET", + endpoint=f"categories?slug={quote_plus(slug)}&per_page=1", + ) + except Exception as exc: + # Do not cache transient failures - the next article should retry. + _logger.warning("Kategorie-Abfrage für '%s' fehlgeschlagen: %s", slug, exc) + return None + + category_id: int | None = None + if isinstance(result, list): + for row in result: + if not isinstance(row, dict): + continue + rid = int(row.get("id", 0) or 0) + if rid > 0: + category_id = rid + break + + if category_id is None: + _logger.warning("Kategorie mit Slug '%s' existiert nicht in WordPress", slug) + + _category_id_cache[slug] = category_id + return category_id + + _BLOCKED_IMAGE_EXTS = {".svg", ".gif", ".ico", ".webp"} _logger = logging.getLogger(__name__) @@ -508,14 +597,37 @@ def publish_article_draft(article: dict[str, Any]) -> tuple[int, str | None]: pass wp_post_id = article.get("wp_post_id") + tag_names = _selected_tags_from_meta(article.get("meta_json")) + # Only the leading few tags reach WordPress; the full list still feeds the + # category rules below, which were validated against it. + wp_tag_names = tag_names[: max(0, settings.wordpress_max_tags_per_post)] tag_ids = _resolve_wp_tag_ids( base_url=settings.wordpress_base_url, auth_header=auth, - tags=_selected_tags_from_meta(article.get("meta_json")), + tags=wp_tag_names, + creatable=_creatable_tag_names(wp_tag_names, article.get("id")), ) if tag_ids: payload["tags"] = tag_ids + # Without an explicit category WordPress files everything under "Allgemein". + # Leaving it unset when no rule matches is deliberate: a wrong category is + # harder to spot later than a handful of posts in the catch-all. + slug = categorize.category_slug(title, tag_names) + if slug: + category_id = _resolve_wp_category_id( + base_url=settings.wordpress_base_url, + auth_header=auth, + slug=slug, + ) + if category_id: + payload["categories"] = [category_id] + else: + _logger.info( + "Keine Kategorie-Regel für Artikel #%s (%s) - bleibt in der Standardkategorie", + article.get("id"), title[:60], + ) + if wp_post_id: result = _wp_request( base_url=settings.wordpress_base_url, diff --git a/backend/tests/test_categorize.py b/backend/tests/test_categorize.py new file mode 100644 index 0000000..6696105 --- /dev/null +++ b/backend/tests/test_categorize.py @@ -0,0 +1,61 @@ +import unittest + +from backend.app.categorize import CATEGORY_SLUGS, category_slug, classify + + +class TestCategorize(unittest.TestCase): + def test_tags_decide_the_category(self) -> None: + self.assertEqual( + classify("Neuer Platz am Wasser", ["Campingplatz", "5-Sterne-Campingplätze"]), + "campingplaetze", + ) + self.assertEqual(classify("Unterwegs im Norden", ["Nordsee", "Niedersachsen"]), "reiseziele") + self.assertEqual(classify("Update vom Van", ["Wohnmobil", "Truma", "Solar"]), "fahrzeug-technik") + + def test_title_alone_is_enough(self) -> None: + self.assertEqual(classify("Dometic CFX5 45 im Test: Top-Kühlbox", []), "ausruestung-tests") + self.assertEqual(classify("Führerscheine müssen getauscht werden", []), "recht-vorschriften") + + def test_discount_signal_beats_the_product_topic(self) -> None: + # Without the override this lands in "Ausruestung & Tests", because the + # product tags outnumber the single discount signal. + self.assertEqual( + classify("Campingzelte bei Amazon im Sale: bis zu 55 %", ["Zelte", "Schlafsack", "Rabatt"]), + "angebote-rabatte", + ) + + def test_keywords_respect_word_boundaries(self) -> None: + # "klage" must not match the city name "Klagenfurt" - plain substring + # matching filed this campsite opening under "Recht & Vorschriften". + self.assertEqual( + classify("Falkensteiner Camping Wörthersee startet in die erste Saison", + ["Campingplatz", "Klagenfurt", "Saisonstart"]), + "campingplaetze", + ) + + def test_compound_nouns_only_match_with_opt_in(self) -> None: + # "campingplatz*" is marked as a compound, so this still matches. + self.assertEqual(classify("Was Campingplatzbetreiber jetzt planen", []), "campingplaetze") + + def test_returns_none_when_nothing_matches(self) -> None: + self.assertIsNone(classify("Weihnachten 2021", ["Weihnachten"])) + self.assertIsNone(category_slug("Weihnachten 2021", ["Weihnachten"])) + + def test_umlauts_and_entities_are_normalised(self) -> None: + self.assertEqual(classify("Stellplätze an der Küste", []), "stellplaetze") + self.assertEqual(classify("Ausrüstung", ["Schlafsack"]), "ausruestung-tests") + + def test_renamed_categories_keep_their_original_slug(self) -> None: + # Both were renamed in WordPress but kept the indexed archive URL. + self.assertEqual(CATEGORY_SLUGS["fahrzeug-technik"], "fahrzeug") + self.assertEqual(CATEGORY_SLUGS["news-branche"], "news") + self.assertEqual(category_slug("Update vom Van", ["Wohnmobil", "Truma"]), "fahrzeug") + + def test_every_rule_maps_to_a_known_slug(self) -> None: + self.assertEqual(len(CATEGORY_SLUGS), 11) + for key, slug in CATEGORY_SLUGS.items(): + self.assertTrue(slug and slug.islower(), f"{key} has a suspicious slug: {slug!r}") + + +if __name__ == "__main__": + unittest.main() diff --git a/backend/tests/test_wordpress.py b/backend/tests/test_wordpress.py index 20b0618..2f77c65 100644 --- a/backend/tests/test_wordpress.py +++ b/backend/tests/test_wordpress.py @@ -1,8 +1,12 @@ import os +import tempfile import unittest +from pathlib import Path from unittest.mock import patch from backend.app import config as config_module +from backend.app import wordpress as wordpress_module +from backend.app.db import init_db from backend.app.wordpress import publish_article_draft @@ -11,12 +15,28 @@ class TestWordpressPublish(unittest.TestCase): os.environ["WORDPRESS_BASE_URL"] = "https://example.org" os.environ["WORDPRESS_USERNAME"] = "wp-user" os.environ["WORDPRESS_APP_PASSWORD"] = "wp-pass" + # Publishing records tag proposals, so it needs a database of its own - + # otherwise the tests would write into the live one. + self.tmp_dir = tempfile.TemporaryDirectory() + os.environ["APP_DB_PATH"] = str(Path(self.tmp_dir.name) / "test.db") config_module.get_settings.cache_clear() + init_db() + # The category lookup is cached for the process lifetime; without this + # one test would resolve a slug that the next one expects to be missing. + wordpress_module._category_id_cache.clear() def tearDown(self) -> None: - for key in ("WORDPRESS_BASE_URL", "WORDPRESS_USERNAME", "WORDPRESS_APP_PASSWORD"): + for key in ( + "WORDPRESS_BASE_URL", + "WORDPRESS_USERNAME", + "WORDPRESS_APP_PASSWORD", + "APP_DB_PATH", + "WORDPRESS_MAX_TAGS_PER_POST", + "WORDPRESS_NEW_TAG_MIN_PROPOSALS", + ): os.environ.pop(key, None) config_module.get_settings.cache_clear() + self.tmp_dir.cleanup() @patch("backend.app.wordpress._upload_featured_media") @patch("backend.app.wordpress._wp_request") @@ -83,7 +103,7 @@ class TestWordpressPublish(unittest.TestCase): @patch("backend.app.wordpress._upload_featured_media") @patch("backend.app.wordpress._wp_request") - def test_publish_resolves_and_sets_tags(self, mock_wp_request, mock_upload_media) -> None: + def test_publish_resolves_existing_tags_and_skips_unknown_ones(self, mock_wp_request, mock_upload_media) -> None: def _fake_wp_request(**kwargs): endpoint = kwargs.get("endpoint", "") method = kwargs.get("method", "") @@ -91,17 +111,13 @@ class TestWordpressPublish(unittest.TestCase): if "Rheingas" in endpoint: return [{"id": 11, "name": "Rheingas"}] return [] - if method == "POST" and endpoint == "tags": - name = (kwargs.get("payload") or {}).get("name") - if name == "Gasflasche": - return {"id": 12, "name": "Gasflasche"} - return {"id": 13, "name": str(name)} if method == "POST" and endpoint == "posts": return {"id": 900, "link": "https://example.org/?p=900"} return {} mock_wp_request.side_effect = _fake_wp_request article = { + "id": 1, "title": "Tag Test", "content_raw": "Inhalt", "source_url": "https://example.com/source", @@ -113,7 +129,138 @@ class TestWordpressPublish(unittest.TestCase): post_calls = [call for call in mock_wp_request.call_args_list if call.kwargs.get("endpoint") == "posts"] self.assertEqual(len(post_calls), 1) payload = post_calls[0].kwargs.get("payload", {}) - self.assertEqual(payload.get("tags"), [11, 12]) + # "Gasflasche" was proposed for the first time and is not created yet. + self.assertEqual(payload.get("tags"), [11]) + self.assertFalse( + any(c.kwargs.get("method") == "POST" and c.kwargs.get("endpoint") == "tags" + for c in mock_wp_request.call_args_list) + ) + + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_new_tag_is_created_once_enough_articles_proposed_it(self, mock_wp_request, mock_upload_media) -> None: + created_names: list[str] = [] + + def _fake_wp_request(**kwargs): + endpoint = kwargs.get("endpoint", "") + method = kwargs.get("method", "") + if method == "GET" and endpoint.startswith("tags?search="): + return [] + if method == "POST" and endpoint == "tags": + created_names.append(str((kwargs.get("payload") or {}).get("name"))) + return {"id": 55, "name": "Wintercamping"} + if method == "POST" and endpoint == "posts": + return {"id": 910, "link": "https://example.org/?p=910"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + + def _article(article_id: int) -> dict: + return { + "id": article_id, + "title": f"Artikel {article_id}", + "content_raw": "Inhalt", + "source_url": f"https://example.com/source/{article_id}", + "canonical_url": f"https://example.com/source/{article_id}", + "meta_json": '{"generated_tags":["Wintercamping"]}', + } + + for article_id in (1, 2): + publish_article_draft(_article(article_id)) + self.assertEqual(created_names, []) + + publish_article_draft(_article(3)) + self.assertEqual(created_names, ["Wintercamping"]) + payload = [c for c in mock_wp_request.call_args_list if c.kwargs.get("endpoint") == "posts"][-1].kwargs["payload"] + self.assertEqual(payload.get("tags"), [55]) + + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_republishing_the_same_article_does_not_count_twice(self, mock_wp_request, mock_upload_media) -> None: + def _fake_wp_request(**kwargs): + endpoint = kwargs.get("endpoint", "") + method = kwargs.get("method", "") + if method == "GET" and endpoint.startswith("tags?search="): + return [] + if method == "POST" and endpoint == "tags": + return {"id": 66, "name": "Solaranlage"} + if method == "POST": + return {"id": 920, "link": "https://example.org/?p=920"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + article = { + "id": 7, + "title": "Immer wieder derselbe Artikel", + "content_raw": "Inhalt", + "source_url": "https://example.com/source", + "canonical_url": "https://example.com/source", + "meta_json": '{"generated_tags":["Solaranlage"]}', + } + for _ in range(4): + publish_article_draft(article) + self.assertFalse( + any(c.kwargs.get("method") == "POST" and c.kwargs.get("endpoint") == "tags" + for c in mock_wp_request.call_args_list) + ) + + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_publish_caps_the_number_of_tags_per_post(self, mock_wp_request, mock_upload_media) -> None: + looked_up: list[str] = [] + + def _fake_wp_request(**kwargs): + endpoint = kwargs.get("endpoint", "") + method = kwargs.get("method", "") + if method == "GET" and endpoint.startswith("tags?search="): + looked_up.append(endpoint) + return [] + if method == "POST" and endpoint == "posts": + return {"id": 930, "link": "https://example.org/?p=930"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + names = [f"Tag{i}" for i in range(1, 13)] + article = { + "id": 42, + "title": "Viele Schlagwörter", + "content_raw": "Inhalt", + "source_url": "https://example.com/source", + "canonical_url": "https://example.com/source", + "meta_json": '{"generated_tags":' + str(names).replace("'", '"') + "}", + } + publish_article_draft(article) + self.assertEqual(len(looked_up), 5) + + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_tag_lookup_ignores_near_misses(self, mock_wp_request, mock_upload_media) -> None: + """A search hit that is not the exact name must not be reused. + + Plain substring reuse filed "Camping" under the unrelated existing tag + "Campingplatz" whenever the exact tag was missing. + """ + def _fake_wp_request(**kwargs): + endpoint = kwargs.get("endpoint", "") + method = kwargs.get("method", "") + if method == "GET" and endpoint.startswith("tags?search="): + return [{"id": 99, "name": "Campingplatz"}] + if method == "POST" and endpoint == "posts": + return {"id": 940, "link": "https://example.org/?p=940"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + article = { + "id": 5, + "title": "Naher Treffer", + "content_raw": "Inhalt", + "source_url": "https://example.com/source", + "canonical_url": "https://example.com/source", + "meta_json": '{"generated_tags":["Camping"]}', + } + publish_article_draft(article) + payload = [c for c in mock_wp_request.call_args_list if c.kwargs.get("endpoint") == "posts"][0].kwargs["payload"] + self.assertNotIn("tags", payload) @patch("backend.app.wordpress._upload_featured_media") @patch("backend.app.wordpress._wp_request") @@ -134,6 +281,78 @@ class TestWordpressPublish(unittest.TestCase): self.assertIn("", content) self.assertNotIn("", content) + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_publish_sets_category_from_rules(self, mock_wp_request, mock_upload_media) -> None: + def _fake_wp_request(**kwargs): + endpoint = kwargs.get("endpoint", "") + method = kwargs.get("method", "") + if method == "GET" and endpoint.startswith("tags?search="): + return [{"id": 21, "name": "Kühlbox"}] + if method == "GET" and endpoint.startswith("categories?slug=ausruestung-tests"): + return [{"id": 77, "slug": "ausruestung-tests"}] + if method == "POST" and endpoint == "posts": + return {"id": 901, "link": "https://example.org/?p=901"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + article = { + "title": "Dometic CFX5 45 im Test: Top-Kühlbox für unterwegs", + "content_raw": "Inhalt", + "source_url": "https://example.com/source", + "canonical_url": "https://example.com/source", + "meta_json": '{"generated_tags":["Kühlbox"]}', + } + post_id, _ = publish_article_draft(article) + self.assertEqual(post_id, 901) + payload = [c for c in mock_wp_request.call_args_list if c.kwargs.get("endpoint") == "posts"][0].kwargs["payload"] + self.assertEqual(payload.get("categories"), [77]) + + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_publish_omits_category_when_no_rule_matches(self, mock_wp_request, mock_upload_media) -> None: + def _fake_wp_request(**kwargs): + if kwargs.get("method") == "POST" and kwargs.get("endpoint") == "posts": + return {"id": 902, "link": "https://example.org/?p=902"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + article = { + "title": "Weihnachten 2021", + "content_raw": "Inhalt", + "source_url": "https://example.com/source", + "canonical_url": "https://example.com/source", + "meta_json": '{"generated_tags":[]}', + } + publish_article_draft(article) + payload = [c for c in mock_wp_request.call_args_list if c.kwargs.get("endpoint") == "posts"][0].kwargs["payload"] + self.assertNotIn("categories", payload) + self.assertFalse(any("categories?slug=" in (c.kwargs.get("endpoint") or "") + for c in mock_wp_request.call_args_list)) + + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_publish_skips_category_when_slug_is_missing_in_wordpress(self, mock_wp_request, mock_upload_media) -> None: + def _fake_wp_request(**kwargs): + endpoint = kwargs.get("endpoint", "") + if kwargs.get("method") == "GET" and endpoint.startswith("categories?slug="): + return [] + if kwargs.get("method") == "POST" and endpoint == "posts": + return {"id": 903, "link": "https://example.org/?p=903"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + article = { + "title": "Dometic CFX5 45 im Test: Top-Kühlbox für unterwegs", + "content_raw": "Inhalt", + "source_url": "https://example.com/source", + "canonical_url": "https://example.com/source", + "meta_json": '{"generated_tags":[]}', + } + publish_article_draft(article) + payload = [c for c in mock_wp_request.call_args_list if c.kwargs.get("endpoint") == "posts"][0].kwargs["payload"] + self.assertNotIn("categories", payload) + if __name__ == "__main__": unittest.main()