From f0d10112adb39fd371b9836e0b32dd7a7e80b9f5 Mon Sep 17 00:00:00 2001 From: Oliver Giertz Date: Fri, 31 Jul 2026 08:00:32 +0200 Subject: [PATCH 1/3] feat(wordpress): set post category from title and tag rules Published articles carried no category, so WordPress filed all of them under the catch-all "Allgemein" - 941 of 955 posts by the time this was noticed. The rules here mirror the one-off backfill of 2026-07-31 and reproduce its result on all 955 posts exactly. Matching is done on word boundaries rather than plain substrings, which otherwise filed a campsite in Klagenfurt under "Recht & Vorschriften" via the keyword "klage". German compounds opt in explicitly with a trailing "*". A clear discount signal takes precedence over the product topic, because product tags otherwise outnumber it. Unknown slugs are never auto-created: unlike tags, the category set is curated in WordPress, so a mismatch should surface rather than spawn a new category. Articles that match no rule stay uncategorised on purpose. Co-Authored-By: Claude Opus 5 --- backend/app/categorize.py | 237 +++++++++++++++++++++++++++++++ backend/app/wordpress.py | 64 ++++++++- backend/tests/test_categorize.py | 61 ++++++++ backend/tests/test_wordpress.py | 76 ++++++++++ 4 files changed, 437 insertions(+), 1 deletion(-) create mode 100644 backend/app/categorize.py create mode 100644 backend/tests/test_categorize.py diff --git a/backend/app/categorize.py b/backend/app/categorize.py new file mode 100644 index 0000000..f96b4f4 --- /dev/null +++ b/backend/app/categorize.py @@ -0,0 +1,237 @@ +"""Assign a WordPress category to an article. + +The rules mirror the one-off backfill of 2026-07-31 that moved 955 existing +posts out of the catch-all "Allgemein" category, so newly published articles +land in the same buckets as the archive. + +Matching notes: +- Keywords match on a word boundary. Plain substring matching is wrong here: + "klage" would match the city name "Klagenfurt". +- A trailing "*" opts a keyword into German compound nouns, so "campingplatz*" + also matches "Campingplatzbetreiber". +- Tags weigh more than the title. Tags were chosen deliberately, while titles + often carry incidental place names. +""" + +from __future__ import annotations + +import re +from collections.abc import Sequence + +# Category key -> WordPress category slug. +CATEGORY_SLUGS: dict[str, str] = { + "angebote-rabatte": "angebote-rabatte", + "in-eigener-sache": "in-eigener-sache", + "recht-vorschriften": "recht-vorschriften", + "fahrzeug-technik": "fahrzeug", + "ausruestung-tests": "ausruestung-tests", + "campingplaetze": "campingplaetze", + "stellplaetze": "stellplaetze", + "news-branche": "news", + "reiseziele": "reiseziele", + "camping-tipps": "camping-tipps", + "vanlife": "vanlife", +} + +# Ordered by priority: on a points tie the earlier entry wins. +_RULES: tuple[tuple[str, tuple[tuple[str, int], ...]], ...] = ( + ("angebote-rabatte", ( + ("amazon sale", 3), ("amazon-sale", 3), ("amazon-angebote", 3), ("im sale", 3), + ("rabatt*", 3), ("schnaeppchen*", 3), ("prime day", 3), ("black friday", 3), + ("gutschein*", 3), ("sparangebot*", 3), ("top-angebot*", 3), ("sommerangebot*", 3), + ("knaller", 3), ("ausverkauf", 3), ("tiefpreis*", 3), ("rekordpreis*", 3), + ("preissturz", 3), ("bestpreis*", 3), ("deal", 2), ("deals", 2), ("lidl", 2), + ("aldi", 2), ("reduziert", 2), ("sale", 1), ("prozent auf", 3), ("guenstiger*", 1), + )), + ("in-eigener-sache", ( + ("vanityontour", 3), ("vanitycast", 3), ("expense logbook", 3), + ("jahresrueckblick", 3), ("rueckblick", 2), ("website", 3), ("in eigener sache", 3), + ("blog", 2), ("discord", 3), ("newsletter", 2), ("appstore", 2), ("app store", 2), + ("ausfall", 2), ("stoerung", 2), ("podcast", 2), ("campertag", 3), ("styyl", 3), + )), + ("recht-vorschriften", ( + ("stvo", 3), ("fuehrerschein*", 3), ("bussgeld*", 3), ("gesetz*", 3), + ("vorschrift*", 3), ("verkehrsrecht*", 3), ("versicherungsschutz", 3), + ("tempolimit", 3), ("promillegrenze", 3), ("maut", 3), ("urteil*", 3), + ("rechtslage", 3), ("abgemahnt", 3), ("bauantrag", 3), ("baurecht", 3), + ("genehmigung*", 2), ("verboten", 2), ("erlaubt", 2), ("strafe*", 2), ("haftung", 2), + ("illegal", 2), ("stellplatzverordnung", 3), ("datenschutz", 2), ("kurtaxe", 3), + ("uebernachtungssteuer", 3), ("kfz-versicherung", 3), ("kfz versicherung", 3), + ("versicherung*", 1), ("handyverbot", 3), ("regeln", 1), + ("landschaftsschutzgebiet", 2), + )), + ("fahrzeug-technik", ( + ("wohnmobil*", 3), ("wohnwagen*", 3), ("reisemobil*", 3), ("kastenwagen", 3), + ("camper van", 3), ("camper-van*", 3), ("truma", 3), ("klimaanlage*", 3), + ("standheizung*", 3), ("energieversorgung", 3), ("ecoflow", 3), ("jackery", 3), + ("powerstation*", 3), ("solar*", 3), ("batterie*", 3), ("wechselrichter", 3), + ("notstrom", 3), ("generator", 3), ("stromversorgung", 3), ("umbau", 3), ("ausbau", 3), + ("selbstausbau", 3), ("fahrverhalten", 3), ("reifen", 3), ("gasanlage*", 3), + ("gasflasche*", 3), ("werkstatt", 3), ("tuev", 3), ("caravaning", 3), ("caravan", 2), + ("dometic", 3), ("anhaengerkupplung", 3), ("chassis", 3), ("dieselheizung*", 3), + ("wasserpumpe*", 3), ("osram", 3), ("auffahrkeil*", 3), ("mover", 3), ("dachluke*", 3), + ("aufbau", 1), ("heizung", 2), ("strom", 2), ("motor", 2), ("led", 2), ("autark", 2), + ("ladedose", 3), ("bordtechnik", 3), ("gewicht", 1), ("auflastung", 3), + ("wiegeaktion", 3), ("gebrauchtwagen*", 3), ("kfz", 2), ("gasversorgung", 3), + ("flaschengas*", 3), ("marder*", 3), ("reparatur", 2), ("diy", 2), + )), + ("ausruestung-tests", ( + ("kuehlbox*", 3), ("kuehltasche*", 3), ("schlafsack*", 3), ("schlafsaecke", 3), + ("dachzelt*", 3), ("campingbett*", 3), ("zelt*", 3), ("gaskocher", 3), + ("campingkocher", 3), ("kopfkissen", 3), ("campingstuhl*", 3), ("campingstuehle", 3), + ("sonnensegel", 3), ("luftmatratze*", 3), ("isomatte*", 3), ("campingdusche*", 3), + ("produkttest*", 3), ("testbericht*", 3), ("im test", 3), ("getestet", 3), + ("decathlon", 3), ("coleman", 3), ("campingaz", 3), ("quechua", 3), ("ausruestung", 3), + ("campingausruestung", 3), ("gadget*", 3), ("markise*", 3), ("nachttisch*", 3), + ("geschirr", 3), ("grill*", 3), ("campingmoebel", 3), ("stirnlampe*", 3), + ("powerbank*", 3), ("wasserkanister", 3), ("vorzelt*", 3), ("haengematte*", 3), + ("campingkueche*", 3), ("campingtisch*", 3), ("camping-helfer", 3), + ("kaffeemaschine*", 3), ("campingtoilette*", 3), ("rucksack*", 3), ("wanderschuh*", 3), + ("thermacell", 3), ("kabeltrommel*", 3), ("fernglas", 3), ("fernglaeser", 3), + ("tarp", 3), ("lichterkette*", 3), ("klapptisch*", 3), ("uv-schutz", 2), + ("wasserdicht*", 2), ("belueftung", 2), ("kaufberatung", 3), ("marktcheck", 3), + ("mueckenschutz", 3), ("sonnenschutz", 2), ("ventilator*", 3), ("campinggeschirr", 3), + ("campingzelt*", 3), ("wurfzelt*", 3), ("tunnelzelt*", 3), ("familienzelt*", 3), + ("trekkingzelt*", 3), ("aufblasbare*", 2), ("trenntoilette*", 3), + ("kassettentoilette*", 3), ("toilette*", 2), ("gaswarner", 3), ("rauchmelder", 3), + ("feuerloescher", 3), ("co melder", 3), ("router", 3), ("mobilfunk", 3), ("wlan", 3), + ("lte", 3), ("5g", 3), ("internet", 2), ("buchtipp*", 2), + )), + ("campingplaetze", ( + ("campingplatz*", 3), ("campingplaetze", 3), ("5-sterne*", 3), ("fuenf sterne", 3), + ("adac-superplatz", 3), ("adac bewertung", 3), ("adac-bewertung", 3), + ("superplatz*", 3), ("glamping", 3), ("wellness-camping", 3), ("suedsee-camp", 3), + ("wirthshof", 3), ("trekking-camp*", 3), ("naturcamping*", 3), ("campingpark*", 3), + ("ferienpark*", 3), ("sanitaer*", 3), ("platz des jahres", 3), ("campingdorf", 3), + ("campinganlage*", 3), ("campingplatzbetreiber", 3), ("campingfuehrer", 3), + ("pincamp", 3), ("dauercamp*", 3), ("camping-check", 3), ("campingresort", 3), + ("stammgaeste", 2), ("kinderanimation", 2), ("resort", 2), ("wellness", 2), + ("campingplatz-ranking", 3), ("platzbewertung", 3), + )), + ("stellplaetze", ( + ("stellplatz*", 3), ("stellplaetze", 3), ("freistehen", 3), ("wildcamp*", 3), + ("camping-car park", 3), ("wohnmobilstellplatz*", 3), ("wohnmobilstellplaetze", 3), + ("wohnmobilhafen", 3), ("wohnmobil-stellplaetze", 3), ("stellplatzfuehrer", 3), + ("uebernachtungsmoeglichkeit*", 2), ("stellplatz-radar", 3), ("parkplatz", 1), + ("parken", 1), ("church4night", 3), ("raststaette*", 2), ("autohof*", 2), + )), + ("news-branche", ( + ("bvcd", 3), ("bundesverband", 3), ("camping-boom", 3), ("campingboom", 3), + ("uebernachtungszahlen", 3), ("preisanalyse", 3), ("preisentwicklung", 3), + ("campingpreise", 3), ("preisanstieg", 3), ("preiserhoehung*", 3), ("destatis", 3), + ("camping-trends", 3), ("campingtrend*", 3), ("marktanalyse", 3), ("caravan salon", 3), + ("messe", 3), ("promobil", 3), ("insolvenz", 3), ("saisonstart", 3), + ("jahreszahlen", 3), ("uebernachtungen", 3), ("rekordzahl", 3), ("statistik*", 3), + ("umfrage", 3), ("studie", 3), ("branche", 3), ("investitionen", 2), ("tourismus", 2), + ("nachfrage", 2), ("bilanz", 2), ("rekord", 2), ("auszeichnung", 2), + ("ausgezeichnet", 2), ("preisvergleich", 2), ("eroeffnung", 2), ("eroeffnet", 2), + ("uebernimmt", 2), ("verkauft", 2), ("pressemeldung", 2), ("feuerwehr", 3), + ("polizei", 2), ("unfall", 3), ("verletzte", 3), ("hochwasser", 2), ("evakuiert", 3), + ("veranstaltung*", 2), + )), + ("reiseziele", ( + ("nordsee", 3), ("ostsee", 3), ("niedersachsen", 3), ("bodensee", 3), + ("lueneburger heide", 3), ("thueringen", 3), ("harz", 3), ("italien", 3), + ("kroatien", 3), ("schweiz", 3), ("oesterreich", 3), ("norwegen", 3), ("schweden", 3), + ("niederlande", 3), ("holland", 3), ("ruegen", 3), ("fehmarn", 3), ("gardasee", 3), + ("schwarzwald", 3), ("sauerland", 3), ("edersee", 3), ("bayern", 3), ("hessen", 3), + ("nrw", 3), ("mecklenburg*", 3), ("baden-wuerttemberg", 3), ("adriakueste", 3), + ("daenemark", 3), ("frankreich", 3), ("spanien", 3), ("portugal", 3), ("slowenien", 3), + ("weserradweg", 3), ("waldeck*", 3), ("sylt", 3), ("usedom", 3), ("allgaeu", 3), + ("eifel", 3), ("mosel", 3), ("alpen", 3), ("toskana", 3), ("brandenburg", 3), + ("sachsen", 3), ("schleswig-holstein", 3), ("rheinland-pfalz", 3), ("saarland", 3), + ("tirol", 3), ("suedtirol", 3), ("belgien", 3), ("tschechien", 3), ("polen", 3), + ("ungarn", 3), ("pyrenaeen", 3), ("nationalpark*", 3), ("reiseziel*", 3), + ("rundreise*", 3), ("roadtrip*", 3), ("reisebericht*", 3), ("ausflugsziel*", 3), + ("kurztrip*", 3), ("europa", 2), ("kueste", 2), ("region", 1), ("uckermark", 3), + ("lausitz", 3), ("schwaebische alb", 3), ("ausflug*", 2), ("geheimtipp*", 2), + ("uebersee", 2), ("kanada", 3), + )), + ("camping-tipps", ( + ("tipp", 3), ("tipps", 3), ("tricks", 3), ("ratgeber", 3), ("anleitung", 3), + ("checkliste", 3), ("packliste", 3), ("buchungstipp*", 3), ("reiseplanung", 3), + ("urlaubsplanung", 3), ("camping apps", 3), ("nebensaison", 3), ("wintercamping", 3), + ("campingsaison", 3), ("camping-saison", 3), ("angrillen", 3), ("reisetipp*", 3), + ("erste hilfe", 3), ("diebstahlschutz", 3), ("so geht", 2), ("so funktioniert", 2), + ("worauf", 2), ("das sollten", 2), ("muss man wissen", 2), ("wie man", 2), + ("vermeiden", 2), ("hygiene", 2), ("sicherheit", 2), ("vorbereitung", 2), ("hitze", 2), + ("wetter", 1), ("guide", 2), ("mythos", 2), ("mythen", 2), ("mit hund", 2), + ("hundefreundlich*", 2), + )), + ("vanlife", ( + ("vanlife", 3), ("van-life", 3), ("van life", 3), ("minimalismus", 3), + ("digitale nomaden", 3), ("aussteiger", 3), ("slow travel", 3), + ("campergemeinschaft", 3), ("camper-gemeinschaft", 3), ("lebensgefuehl", 3), + ("auszeit", 2), ("freiheit", 2), ("abenteuer", 2), ("gedanken", 2), + ("nachhaltigkeit", 2), + )), +) + +_PRIORITY = {key: i for i, (key, _) in enumerate(_RULES)} + +# A clear discount signal decides on its own: a tent on sale is first of all an +# offer. Otherwise "Ausruestung & Tests" wins on the sheer number of product +# tags and the offers category stays nearly empty. +_OFFER_KEY = "angebote-rabatte" +_OFFER_OVERRIDE_MIN_WEIGHT = 3 + + +def _normalise(text: str) -> str: + text = text.lower() + for src, dst in (("\u00e4", "ae"), ("\u00f6", "oe"), ("\u00fc", "ue"), ("\u00df", "ss")): + text = text.replace(src, dst) + text = text.replace("&", " ").replace("&", " ") + text = re.sub(r"[\u201e\u201c\u201d\u2018\u2019\u00ab\u00bb\"']", " ", text) + text = text.replace("\u2013", "-").replace("\u2014", "-") + return re.sub(r"\s+", " ", text).strip() + + +def _compile(keyword: str) -> re.Pattern[str]: + compound = keyword.endswith("*") + core = _normalise(keyword[:-1] if compound else keyword) + body = re.escape(core).replace(r"\ ", r"\s+") + tail = r"[a-z0-9]*" if compound else r"(?![a-z0-9])" + return re.compile(r"(? str | None: + """Return a category key, or None when no rule matches. + + Callers should leave the category unset in that case so the article stays + visible in WordPress' default category instead of being filed wrongly. + """ + title_text = _normalise(title or "") + tag_text = _normalise(" | ".join(tags or [])) + + scores: dict[str, int] = {} + for key, patterns in _PATTERNS.items(): + total = 0 + for pattern, weight in patterns: + if pattern.search(tag_text): + total += 3 * weight + if pattern.search(title_text): + total += 2 * weight + if total: + scores[key] = total + + if not scores: + return None + + for pattern, weight in _PATTERNS[_OFFER_KEY]: + if weight >= _OFFER_OVERRIDE_MIN_WEIGHT and ( + pattern.search(tag_text) or pattern.search(title_text) + ): + return _OFFER_KEY + + return max(scores.items(), key=lambda item: (item[1], -_PRIORITY[item[0]]))[0] + + +def category_slug(title: str, tags: Sequence[str] | None = None) -> str | None: + """Return the WordPress category slug for an article, or None.""" + key = classify(title, tags) + return CATEGORY_SLUGS.get(key) if key else None diff --git a/backend/app/wordpress.py b/backend/app/wordpress.py index bb96198..0d81ae1 100644 --- a/backend/app/wordpress.py +++ b/backend/app/wordpress.py @@ -12,6 +12,7 @@ from html import unescape as _html_unescape from urllib.parse import quote_plus, urlparse from urllib.request import Request, urlopen +from . import categorize from .config import get_settings @@ -137,6 +138,48 @@ def _resolve_wp_tag_ids(*, base_url: str, auth_header: str, tags: list[str]) -> return ids +_category_id_cache: dict[str, int | None] = {} + + +def _resolve_wp_category_id(*, base_url: str, auth_header: str, slug: str) -> int | None: + """Look up a WordPress category by slug. + + Unlike tags, categories are never created on the fly: the set is curated in + WordPress, so an unknown slug means the rules and the site drifted apart and + should be fixed there, not silently papered over with a new category. + """ + if slug in _category_id_cache: + return _category_id_cache[slug] + + try: + result = _wp_request( + base_url=base_url, + auth_header=auth_header, + method="GET", + endpoint=f"categories?slug={quote_plus(slug)}&per_page=1", + ) + except Exception as exc: + # Do not cache transient failures - the next article should retry. + _logger.warning("Kategorie-Abfrage für '%s' fehlgeschlagen: %s", slug, exc) + return None + + category_id: int | None = None + if isinstance(result, list): + for row in result: + if not isinstance(row, dict): + continue + rid = int(row.get("id", 0) or 0) + if rid > 0: + category_id = rid + break + + if category_id is None: + _logger.warning("Kategorie mit Slug '%s' existiert nicht in WordPress", slug) + + _category_id_cache[slug] = category_id + return category_id + + _BLOCKED_IMAGE_EXTS = {".svg", ".gif", ".ico", ".webp"} _logger = logging.getLogger(__name__) @@ -508,14 +551,33 @@ def publish_article_draft(article: dict[str, Any]) -> tuple[int, str | None]: pass wp_post_id = article.get("wp_post_id") + tag_names = _selected_tags_from_meta(article.get("meta_json")) tag_ids = _resolve_wp_tag_ids( base_url=settings.wordpress_base_url, auth_header=auth, - tags=_selected_tags_from_meta(article.get("meta_json")), + tags=tag_names, ) if tag_ids: payload["tags"] = tag_ids + # Without an explicit category WordPress files everything under "Allgemein". + # Leaving it unset when no rule matches is deliberate: a wrong category is + # harder to spot later than a handful of posts in the catch-all. + slug = categorize.category_slug(title, tag_names) + if slug: + category_id = _resolve_wp_category_id( + base_url=settings.wordpress_base_url, + auth_header=auth, + slug=slug, + ) + if category_id: + payload["categories"] = [category_id] + else: + _logger.info( + "Keine Kategorie-Regel für Artikel #%s (%s) - bleibt in der Standardkategorie", + article.get("id"), title[:60], + ) + if wp_post_id: result = _wp_request( base_url=settings.wordpress_base_url, diff --git a/backend/tests/test_categorize.py b/backend/tests/test_categorize.py new file mode 100644 index 0000000..6696105 --- /dev/null +++ b/backend/tests/test_categorize.py @@ -0,0 +1,61 @@ +import unittest + +from backend.app.categorize import CATEGORY_SLUGS, category_slug, classify + + +class TestCategorize(unittest.TestCase): + def test_tags_decide_the_category(self) -> None: + self.assertEqual( + classify("Neuer Platz am Wasser", ["Campingplatz", "5-Sterne-Campingplätze"]), + "campingplaetze", + ) + self.assertEqual(classify("Unterwegs im Norden", ["Nordsee", "Niedersachsen"]), "reiseziele") + self.assertEqual(classify("Update vom Van", ["Wohnmobil", "Truma", "Solar"]), "fahrzeug-technik") + + def test_title_alone_is_enough(self) -> None: + self.assertEqual(classify("Dometic CFX5 45 im Test: Top-Kühlbox", []), "ausruestung-tests") + self.assertEqual(classify("Führerscheine müssen getauscht werden", []), "recht-vorschriften") + + def test_discount_signal_beats_the_product_topic(self) -> None: + # Without the override this lands in "Ausruestung & Tests", because the + # product tags outnumber the single discount signal. + self.assertEqual( + classify("Campingzelte bei Amazon im Sale: bis zu 55 %", ["Zelte", "Schlafsack", "Rabatt"]), + "angebote-rabatte", + ) + + def test_keywords_respect_word_boundaries(self) -> None: + # "klage" must not match the city name "Klagenfurt" - plain substring + # matching filed this campsite opening under "Recht & Vorschriften". + self.assertEqual( + classify("Falkensteiner Camping Wörthersee startet in die erste Saison", + ["Campingplatz", "Klagenfurt", "Saisonstart"]), + "campingplaetze", + ) + + def test_compound_nouns_only_match_with_opt_in(self) -> None: + # "campingplatz*" is marked as a compound, so this still matches. + self.assertEqual(classify("Was Campingplatzbetreiber jetzt planen", []), "campingplaetze") + + def test_returns_none_when_nothing_matches(self) -> None: + self.assertIsNone(classify("Weihnachten 2021", ["Weihnachten"])) + self.assertIsNone(category_slug("Weihnachten 2021", ["Weihnachten"])) + + def test_umlauts_and_entities_are_normalised(self) -> None: + self.assertEqual(classify("Stellplätze an der Küste", []), "stellplaetze") + self.assertEqual(classify("Ausrüstung", ["Schlafsack"]), "ausruestung-tests") + + def test_renamed_categories_keep_their_original_slug(self) -> None: + # Both were renamed in WordPress but kept the indexed archive URL. + self.assertEqual(CATEGORY_SLUGS["fahrzeug-technik"], "fahrzeug") + self.assertEqual(CATEGORY_SLUGS["news-branche"], "news") + self.assertEqual(category_slug("Update vom Van", ["Wohnmobil", "Truma"]), "fahrzeug") + + def test_every_rule_maps_to_a_known_slug(self) -> None: + self.assertEqual(len(CATEGORY_SLUGS), 11) + for key, slug in CATEGORY_SLUGS.items(): + self.assertTrue(slug and slug.islower(), f"{key} has a suspicious slug: {slug!r}") + + +if __name__ == "__main__": + unittest.main() diff --git a/backend/tests/test_wordpress.py b/backend/tests/test_wordpress.py index 20b0618..6ac591e 100644 --- a/backend/tests/test_wordpress.py +++ b/backend/tests/test_wordpress.py @@ -3,6 +3,7 @@ import unittest from unittest.mock import patch from backend.app import config as config_module +from backend.app import wordpress as wordpress_module from backend.app.wordpress import publish_article_draft @@ -12,6 +13,9 @@ class TestWordpressPublish(unittest.TestCase): os.environ["WORDPRESS_USERNAME"] = "wp-user" os.environ["WORDPRESS_APP_PASSWORD"] = "wp-pass" config_module.get_settings.cache_clear() + # The category lookup is cached for the process lifetime; without this + # one test would resolve a slug that the next one expects to be missing. + wordpress_module._category_id_cache.clear() def tearDown(self) -> None: for key in ("WORDPRESS_BASE_URL", "WORDPRESS_USERNAME", "WORDPRESS_APP_PASSWORD"): @@ -134,6 +138,78 @@ class TestWordpressPublish(unittest.TestCase): self.assertIn("", content) self.assertNotIn("", content) + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_publish_sets_category_from_rules(self, mock_wp_request, mock_upload_media) -> None: + def _fake_wp_request(**kwargs): + endpoint = kwargs.get("endpoint", "") + method = kwargs.get("method", "") + if method == "GET" and endpoint.startswith("tags?search="): + return [{"id": 21, "name": "Kühlbox"}] + if method == "GET" and endpoint.startswith("categories?slug=ausruestung-tests"): + return [{"id": 77, "slug": "ausruestung-tests"}] + if method == "POST" and endpoint == "posts": + return {"id": 901, "link": "https://example.org/?p=901"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + article = { + "title": "Dometic CFX5 45 im Test: Top-Kühlbox für unterwegs", + "content_raw": "Inhalt", + "source_url": "https://example.com/source", + "canonical_url": "https://example.com/source", + "meta_json": '{"generated_tags":["Kühlbox"]}', + } + post_id, _ = publish_article_draft(article) + self.assertEqual(post_id, 901) + payload = [c for c in mock_wp_request.call_args_list if c.kwargs.get("endpoint") == "posts"][0].kwargs["payload"] + self.assertEqual(payload.get("categories"), [77]) + + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_publish_omits_category_when_no_rule_matches(self, mock_wp_request, mock_upload_media) -> None: + def _fake_wp_request(**kwargs): + if kwargs.get("method") == "POST" and kwargs.get("endpoint") == "posts": + return {"id": 902, "link": "https://example.org/?p=902"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + article = { + "title": "Weihnachten 2021", + "content_raw": "Inhalt", + "source_url": "https://example.com/source", + "canonical_url": "https://example.com/source", + "meta_json": '{"generated_tags":[]}', + } + publish_article_draft(article) + payload = [c for c in mock_wp_request.call_args_list if c.kwargs.get("endpoint") == "posts"][0].kwargs["payload"] + self.assertNotIn("categories", payload) + self.assertFalse(any("categories?slug=" in (c.kwargs.get("endpoint") or "") + for c in mock_wp_request.call_args_list)) + + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_publish_skips_category_when_slug_is_missing_in_wordpress(self, mock_wp_request, mock_upload_media) -> None: + def _fake_wp_request(**kwargs): + endpoint = kwargs.get("endpoint", "") + if kwargs.get("method") == "GET" and endpoint.startswith("categories?slug="): + return [] + if kwargs.get("method") == "POST" and endpoint == "posts": + return {"id": 903, "link": "https://example.org/?p=903"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + article = { + "title": "Dometic CFX5 45 im Test: Top-Kühlbox für unterwegs", + "content_raw": "Inhalt", + "source_url": "https://example.com/source", + "canonical_url": "https://example.com/source", + "meta_json": '{"generated_tags":[]}', + } + publish_article_draft(article) + payload = [c for c in mock_wp_request.call_args_list if c.kwargs.get("endpoint") == "posts"][0].kwargs["payload"] + self.assertNotIn("categories", payload) + if __name__ == "__main__": unittest.main() From 4282759b2c1f4513c018000742fd00276a401406 Mon Sep 17 00:00:00 2001 From: Oliver Giertz Date: Fri, 31 Jul 2026 08:21:46 +0200 Subject: [PATCH 2/3] feat(categorize): close rule gaps found on scheduled posts The first pass only covered published posts, so 137 scheduled ones were still uncategorised. Seven of them matched no rule at all and exposed real gaps: the regions Pfalz and Erzgebirge, and the gear nouns in "Duschzelte", "Outdoor-Messer" and "Outdoor-Stuhl", which the existing compound keywords could not reach because the noun sits at the end. Adding those flipped two discount posts ("Lidl Zelt im Angebot", "Aldi Relax-Stuhl") into the gear category, because the new product keywords outweighed the weaker "lidl"/"aldi" signals. Both retailers only ever appear in discount posts here, so they now carry enough weight to trigger the offer override themselves. Region keywords are weight 2, not 3: at 3 they outranked an explicit "Campingplatz" in the title and pulled campsite articles into "Reiseziele". Bare "pfalz" is weight 1 because it double-counts inside "rheinland-pfalz". Scheduled posts now match a rule 136 of 137 times. Ten published posts change category as a result and were updated in WordPress to keep the archive and the rules in sync. Co-Authored-By: Claude Opus 5 --- backend/app/categorize.py | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/backend/app/categorize.py b/backend/app/categorize.py index f96b4f4..ec33e80 100644 --- a/backend/app/categorize.py +++ b/backend/app/categorize.py @@ -40,8 +40,11 @@ _RULES: tuple[tuple[str, tuple[tuple[str, int], ...]], ...] = ( ("rabatt*", 3), ("schnaeppchen*", 3), ("prime day", 3), ("black friday", 3), ("gutschein*", 3), ("sparangebot*", 3), ("top-angebot*", 3), ("sommerangebot*", 3), ("knaller", 3), ("ausverkauf", 3), ("tiefpreis*", 3), ("rekordpreis*", 3), - ("preissturz", 3), ("bestpreis*", 3), ("deal", 2), ("deals", 2), ("lidl", 2), - ("aldi", 2), ("reduziert", 2), ("sale", 1), ("prozent auf", 3), ("guenstiger*", 1), + ("preissturz", 3), ("bestpreis*", 3), ("deal", 2), ("deals", 2), + # Lidl and Aldi articles are always discount posts on this blog, so they + # carry enough weight to trigger the override below on their own. + ("lidl", 3), ("aldi", 3), ("im angebot", 3), + ("reduziert", 2), ("sale", 1), ("prozent auf", 3), ("guenstiger*", 1), )), ("in-eigener-sache", ( ("vanityontour", 3), ("vanitycast", 3), ("expense logbook", 3), @@ -96,6 +99,8 @@ _RULES: tuple[tuple[str, tuple[tuple[str, int], ...]], ...] = ( ("kassettentoilette*", 3), ("toilette*", 2), ("gaswarner", 3), ("rauchmelder", 3), ("feuerloescher", 3), ("co melder", 3), ("router", 3), ("mobilfunk", 3), ("wlan", 3), ("lte", 3), ("5g", 3), ("internet", 2), ("buchtipp*", 2), + ("duschzelt*", 3), ("kuppelzelt*", 3), ("pop-up-zelt*", 3), ("messer", 3), + ("stuhl", 3), ("faltstuhl*", 3), ("klappstuhl*", 3), )), ("campingplaetze", ( ("campingplatz*", 3), ("campingplaetze", 3), ("5-sterne*", 3), ("fuenf sterne", 3), @@ -145,7 +150,11 @@ _RULES: tuple[tuple[str, tuple[tuple[str, int], ...]], ...] = ( ("rundreise*", 3), ("roadtrip*", 3), ("reisebericht*", 3), ("ausflugsziel*", 3), ("kurztrip*", 3), ("europa", 2), ("kueste", 2), ("region", 1), ("uckermark", 3), ("lausitz", 3), ("schwaebische alb", 3), ("ausflug*", 2), ("geheimtipp*", 2), - ("uebersee", 2), ("kanada", 3), + ("uebersee", 2), ("kanada", 3), ("pfalz", 1), ("erzgebirge", 2), + ("spreewald", 2), ("vogtland", 2), ("rhoen", 2), ("odenwald", 2), + ("chiemsee", 2), ("ammersee", 2), ("muensterland", 2), ("ostfriesland", 2), + ("emsland", 2), ("altmuehltal", 2), ("teutoburger wald", 2), ("taunus", 2), + ("westerwald", 2), ("bergisches land", 2), ("nordwesten", 2), )), ("camping-tipps", ( ("tipp", 3), ("tipps", 3), ("tricks", 3), ("ratgeber", 3), ("anleitung", 3), From 682755c6a09ede4aafdca2790adfc61340298bbe Mon Sep 17 00:00:00 2001 From: Oliver G Date: Fri, 31 Jul 2026 08:21:53 +0200 Subject: [PATCH 3/3] feat(wordpress): only assign tags that are established _resolve_wp_tag_ids created a WordPress tag for every keyword the rewriter invented, up to 12 per post. That is where the 3.055 tags for 955 posts came from - 1.683 of them used exactly once, 473 attached to no post at all. The categories are stable now, but the tags would simply grow back. Three changes, all on the write path: A proposed tag has to appear for wordpress_new_tag_min_proposals (3) different articles before it is created. Proposals are counted in the new tag_proposals table, keyed on (name, article), so re-publishing an article does not inflate its own count. Tags that already exist in WordPress are assigned as before - the gate only guards creation. Only the first wordpress_max_tags_per_post (5) tags reach WordPress. The full list still feeds the category rules, which were validated against it. The lookup fallback of reusing the first search hit is gone. It filed "Camping" under the unrelated existing tag "Campingplatz" whenever the exact tag was missing, which quietly produced wrong tags rather than none. If the proposal bookkeeping fails, nothing is creatable that run: existing tags still get assigned and the taxonomy stays put, rather than falling back to creating everything. Co-Authored-By: Claude Opus 5 --- backend/app/config.py | 5 + backend/app/db.py | 15 +++ backend/app/repositories.py | 38 ++++++++ backend/app/wordpress.py | 64 +++++++++++-- backend/tests/test_wordpress.py | 159 ++++++++++++++++++++++++++++++-- 5 files changed, 266 insertions(+), 15 deletions(-) diff --git a/backend/app/config.py b/backend/app/config.py index 1a92e0c..a0c8f66 100644 --- a/backend/app/config.py +++ b/backend/app/config.py @@ -30,6 +30,11 @@ class Settings(BaseSettings): wordpress_username: str | None = Field(default=None, validation_alias=AliasChoices("WORDPRESS_USERNAME", "WP_USERNAME")) wordpress_app_password: str | None = Field(default=None, validation_alias=AliasChoices("WORDPRESS_APP_PASSWORD", "WP_PASSWORD")) wordpress_default_status: str = "draft" + # Tag hygiene: the rewriter proposes far more tags than a post needs, and + # every unknown one used to be created on the spot - that is how 955 posts + # accumulated 3.055 tags, 1.683 of them used exactly once. + wordpress_max_tags_per_post: int = 5 + wordpress_new_tag_min_proposals: int = 3 openai_api_key: str | None = Field(default=None, validation_alias=AliasChoices("OPENAI_API_KEY")) openai_model: str = "gpt-4o-mini" diff --git a/backend/app/db.py b/backend/app/db.py index b6ef898..6aa1e0f 100644 --- a/backend/app/db.py +++ b/backend/app/db.py @@ -118,6 +118,21 @@ def init_db() -> None: UNIQUE(source_url) ); + -- One row per (tag, article) the rewriter ever proposed, whether or + -- not the tag made it into WordPress. The number of rows per + -- name_key is the "is this tag established" signal that gates + -- creating the tag in WordPress. + CREATE TABLE IF NOT EXISTS tag_proposals ( + id INTEGER PRIMARY KEY AUTOINCREMENT, + name_key TEXT NOT NULL, + name TEXT NOT NULL, + article_id INTEGER NOT NULL DEFAULT 0, + created_at TEXT NOT NULL DEFAULT (datetime('now')), + UNIQUE(name_key, article_id) + ); + + CREATE INDEX IF NOT EXISTS idx_tag_proposals_name_key ON tag_proposals(name_key); + CREATE INDEX IF NOT EXISTS idx_articles_source_article_id ON articles(source_article_id); CREATE INDEX IF NOT EXISTS idx_articles_source_hash ON articles(source_hash); CREATE UNIQUE INDEX IF NOT EXISTS uq_articles_feed_source_article_id diff --git a/backend/app/repositories.py b/backend/app/repositories.py index cf38055..8032077 100644 --- a/backend/app/repositories.py +++ b/backend/app/repositories.py @@ -853,3 +853,41 @@ def list_articles(limit: int = 100, status_filter: str | None = None) -> list[di (safe_limit,), ).fetchall() return rows_to_dicts(rows) + + +def record_tag_proposals(names: list[str], article_id: int | None) -> dict[str, int]: + """Record that these tags were proposed for one article. + + Returns, per casefolded name, how many distinct articles have proposed it so + far - this one included. That count is what decides whether a tag is + established enough to be created in WordPress: a tag the rewriter invents + once for a single article should not become a permanent taxonomy entry. + + Re-publishing an article does not inflate the count, because the row is + keyed on (name_key, article_id). Articles without an id (ad-hoc publishing) + all share the sentinel 0 and therefore count once in total. + """ + counts: dict[str, int] = {} + if not names: + return counts + key_for_article = int(article_id or 0) + with get_conn() as conn: + for raw in names: + name = str(raw or "").strip() + if not name: + continue + name_key = name.casefold() + conn.execute( + """ + INSERT INTO tag_proposals (name_key, name, article_id) + VALUES (?, ?, ?) + ON CONFLICT(name_key, article_id) DO NOTHING + """, + (name_key, name, key_for_article), + ) + row = conn.execute( + "SELECT COUNT(*) AS n FROM tag_proposals WHERE name_key = ?", + (name_key,), + ).fetchone() + counts[name_key] = int(row["n"] if row else 0) + return counts diff --git a/backend/app/wordpress.py b/backend/app/wordpress.py index 0d81ae1..32ec594 100644 --- a/backend/app/wordpress.py +++ b/backend/app/wordpress.py @@ -13,6 +13,7 @@ from urllib.parse import quote_plus, urlparse from urllib.request import Request, urlopen from . import categorize +from . import repositories from .config import get_settings @@ -91,7 +92,25 @@ def _selected_tags_from_meta(meta_json: str | None) -> list[str]: return tags -def _resolve_wp_tag_ids(*, base_url: str, auth_header: str, tags: list[str]) -> list[int]: +def _resolve_wp_tag_ids( + *, + base_url: str, + auth_header: str, + tags: list[str], + creatable: set[str], +) -> list[int]: + """Map tag names to WordPress tag IDs, creating only established ones. + + A name matches an existing tag only on an exact (case-insensitive) name. + The former fallback of taking the first search hit filed "Camping" under + the unrelated existing tag "Campingplatz" whenever the exact tag was + missing. + + `creatable` holds the casefolded names that cleared the proposal threshold + and may be created in WordPress. Everything else is dropped with a log line + rather than turned into a new tag - an empty set means this run adds nothing + to the taxonomy. + """ ids: list[int] = [] seen: set[int] = set() for tag in tags: @@ -113,12 +132,12 @@ def _resolve_wp_tag_ids(*, base_url: str, auth_header: str, tags: list[str]) -> if row_name.casefold() == name.casefold(): tag_id = rid break - if tag_id is None: - for row in result: - if isinstance(row, dict) and int(row.get("id", 0) or 0) > 0: - tag_id = int(row.get("id", 0)) - break if tag_id is None: + if name.casefold() not in creatable: + _logger.info( + "Schlagwort '%s' noch nicht etabliert - wird nicht angelegt", name + ) + continue created = _wp_request( base_url=base_url, auth_header=auth_header, @@ -130,6 +149,7 @@ def _resolve_wp_tag_ids(*, base_url: str, auth_header: str, tags: list[str]) -> rid = int(created.get("id", 0) or 0) if rid > 0: tag_id = rid + _logger.info("Schlagwort '%s' in WordPress neu angelegt (#%s)", name, rid) if tag_id is not None and tag_id > 0 and tag_id not in seen: seen.add(tag_id) ids.append(tag_id) @@ -138,6 +158,32 @@ def _resolve_wp_tag_ids(*, base_url: str, auth_header: str, tags: list[str]) -> return ids +def _creatable_tag_names(tags: list[str], article_id: Any) -> set[str]: + """Casefolded names that may be created as new WordPress tags. + + A proposed tag has to show up for `wordpress_new_tag_min_proposals` + different articles before it earns a taxonomy entry. Until then it is + counted and dropped, so one-off inventions stop accumulating. + + If the bookkeeping itself fails, nothing is creatable: existing tags are + still assigned, and the taxonomy simply does not grow that run. + """ + if not tags: + return set() + settings = get_settings() + threshold = max(1, settings.wordpress_new_tag_min_proposals) + try: + aid = int(article_id) if article_id is not None else None + except (TypeError, ValueError): + aid = None + try: + counts = repositories.record_tag_proposals(tags, aid) + except Exception as exc: + _logger.warning("Schlagwort-Zähler nicht verfügbar, lege keine neuen an: %s", exc) + return set() + return {name for name, seen in counts.items() if seen >= threshold} + + _category_id_cache: dict[str, int | None] = {} @@ -552,10 +598,14 @@ def publish_article_draft(article: dict[str, Any]) -> tuple[int, str | None]: wp_post_id = article.get("wp_post_id") tag_names = _selected_tags_from_meta(article.get("meta_json")) + # Only the leading few tags reach WordPress; the full list still feeds the + # category rules below, which were validated against it. + wp_tag_names = tag_names[: max(0, settings.wordpress_max_tags_per_post)] tag_ids = _resolve_wp_tag_ids( base_url=settings.wordpress_base_url, auth_header=auth, - tags=tag_names, + tags=wp_tag_names, + creatable=_creatable_tag_names(wp_tag_names, article.get("id")), ) if tag_ids: payload["tags"] = tag_ids diff --git a/backend/tests/test_wordpress.py b/backend/tests/test_wordpress.py index 6ac591e..2f77c65 100644 --- a/backend/tests/test_wordpress.py +++ b/backend/tests/test_wordpress.py @@ -1,9 +1,12 @@ import os +import tempfile import unittest +from pathlib import Path from unittest.mock import patch from backend.app import config as config_module from backend.app import wordpress as wordpress_module +from backend.app.db import init_db from backend.app.wordpress import publish_article_draft @@ -12,15 +15,28 @@ class TestWordpressPublish(unittest.TestCase): os.environ["WORDPRESS_BASE_URL"] = "https://example.org" os.environ["WORDPRESS_USERNAME"] = "wp-user" os.environ["WORDPRESS_APP_PASSWORD"] = "wp-pass" + # Publishing records tag proposals, so it needs a database of its own - + # otherwise the tests would write into the live one. + self.tmp_dir = tempfile.TemporaryDirectory() + os.environ["APP_DB_PATH"] = str(Path(self.tmp_dir.name) / "test.db") config_module.get_settings.cache_clear() + init_db() # The category lookup is cached for the process lifetime; without this # one test would resolve a slug that the next one expects to be missing. wordpress_module._category_id_cache.clear() def tearDown(self) -> None: - for key in ("WORDPRESS_BASE_URL", "WORDPRESS_USERNAME", "WORDPRESS_APP_PASSWORD"): + for key in ( + "WORDPRESS_BASE_URL", + "WORDPRESS_USERNAME", + "WORDPRESS_APP_PASSWORD", + "APP_DB_PATH", + "WORDPRESS_MAX_TAGS_PER_POST", + "WORDPRESS_NEW_TAG_MIN_PROPOSALS", + ): os.environ.pop(key, None) config_module.get_settings.cache_clear() + self.tmp_dir.cleanup() @patch("backend.app.wordpress._upload_featured_media") @patch("backend.app.wordpress._wp_request") @@ -87,7 +103,7 @@ class TestWordpressPublish(unittest.TestCase): @patch("backend.app.wordpress._upload_featured_media") @patch("backend.app.wordpress._wp_request") - def test_publish_resolves_and_sets_tags(self, mock_wp_request, mock_upload_media) -> None: + def test_publish_resolves_existing_tags_and_skips_unknown_ones(self, mock_wp_request, mock_upload_media) -> None: def _fake_wp_request(**kwargs): endpoint = kwargs.get("endpoint", "") method = kwargs.get("method", "") @@ -95,17 +111,13 @@ class TestWordpressPublish(unittest.TestCase): if "Rheingas" in endpoint: return [{"id": 11, "name": "Rheingas"}] return [] - if method == "POST" and endpoint == "tags": - name = (kwargs.get("payload") or {}).get("name") - if name == "Gasflasche": - return {"id": 12, "name": "Gasflasche"} - return {"id": 13, "name": str(name)} if method == "POST" and endpoint == "posts": return {"id": 900, "link": "https://example.org/?p=900"} return {} mock_wp_request.side_effect = _fake_wp_request article = { + "id": 1, "title": "Tag Test", "content_raw": "Inhalt", "source_url": "https://example.com/source", @@ -117,7 +129,138 @@ class TestWordpressPublish(unittest.TestCase): post_calls = [call for call in mock_wp_request.call_args_list if call.kwargs.get("endpoint") == "posts"] self.assertEqual(len(post_calls), 1) payload = post_calls[0].kwargs.get("payload", {}) - self.assertEqual(payload.get("tags"), [11, 12]) + # "Gasflasche" was proposed for the first time and is not created yet. + self.assertEqual(payload.get("tags"), [11]) + self.assertFalse( + any(c.kwargs.get("method") == "POST" and c.kwargs.get("endpoint") == "tags" + for c in mock_wp_request.call_args_list) + ) + + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_new_tag_is_created_once_enough_articles_proposed_it(self, mock_wp_request, mock_upload_media) -> None: + created_names: list[str] = [] + + def _fake_wp_request(**kwargs): + endpoint = kwargs.get("endpoint", "") + method = kwargs.get("method", "") + if method == "GET" and endpoint.startswith("tags?search="): + return [] + if method == "POST" and endpoint == "tags": + created_names.append(str((kwargs.get("payload") or {}).get("name"))) + return {"id": 55, "name": "Wintercamping"} + if method == "POST" and endpoint == "posts": + return {"id": 910, "link": "https://example.org/?p=910"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + + def _article(article_id: int) -> dict: + return { + "id": article_id, + "title": f"Artikel {article_id}", + "content_raw": "Inhalt", + "source_url": f"https://example.com/source/{article_id}", + "canonical_url": f"https://example.com/source/{article_id}", + "meta_json": '{"generated_tags":["Wintercamping"]}', + } + + for article_id in (1, 2): + publish_article_draft(_article(article_id)) + self.assertEqual(created_names, []) + + publish_article_draft(_article(3)) + self.assertEqual(created_names, ["Wintercamping"]) + payload = [c for c in mock_wp_request.call_args_list if c.kwargs.get("endpoint") == "posts"][-1].kwargs["payload"] + self.assertEqual(payload.get("tags"), [55]) + + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_republishing_the_same_article_does_not_count_twice(self, mock_wp_request, mock_upload_media) -> None: + def _fake_wp_request(**kwargs): + endpoint = kwargs.get("endpoint", "") + method = kwargs.get("method", "") + if method == "GET" and endpoint.startswith("tags?search="): + return [] + if method == "POST" and endpoint == "tags": + return {"id": 66, "name": "Solaranlage"} + if method == "POST": + return {"id": 920, "link": "https://example.org/?p=920"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + article = { + "id": 7, + "title": "Immer wieder derselbe Artikel", + "content_raw": "Inhalt", + "source_url": "https://example.com/source", + "canonical_url": "https://example.com/source", + "meta_json": '{"generated_tags":["Solaranlage"]}', + } + for _ in range(4): + publish_article_draft(article) + self.assertFalse( + any(c.kwargs.get("method") == "POST" and c.kwargs.get("endpoint") == "tags" + for c in mock_wp_request.call_args_list) + ) + + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_publish_caps_the_number_of_tags_per_post(self, mock_wp_request, mock_upload_media) -> None: + looked_up: list[str] = [] + + def _fake_wp_request(**kwargs): + endpoint = kwargs.get("endpoint", "") + method = kwargs.get("method", "") + if method == "GET" and endpoint.startswith("tags?search="): + looked_up.append(endpoint) + return [] + if method == "POST" and endpoint == "posts": + return {"id": 930, "link": "https://example.org/?p=930"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + names = [f"Tag{i}" for i in range(1, 13)] + article = { + "id": 42, + "title": "Viele Schlagwörter", + "content_raw": "Inhalt", + "source_url": "https://example.com/source", + "canonical_url": "https://example.com/source", + "meta_json": '{"generated_tags":' + str(names).replace("'", '"') + "}", + } + publish_article_draft(article) + self.assertEqual(len(looked_up), 5) + + @patch("backend.app.wordpress._upload_featured_media") + @patch("backend.app.wordpress._wp_request") + def test_tag_lookup_ignores_near_misses(self, mock_wp_request, mock_upload_media) -> None: + """A search hit that is not the exact name must not be reused. + + Plain substring reuse filed "Camping" under the unrelated existing tag + "Campingplatz" whenever the exact tag was missing. + """ + def _fake_wp_request(**kwargs): + endpoint = kwargs.get("endpoint", "") + method = kwargs.get("method", "") + if method == "GET" and endpoint.startswith("tags?search="): + return [{"id": 99, "name": "Campingplatz"}] + if method == "POST" and endpoint == "posts": + return {"id": 940, "link": "https://example.org/?p=940"} + return {} + + mock_wp_request.side_effect = _fake_wp_request + article = { + "id": 5, + "title": "Naher Treffer", + "content_raw": "Inhalt", + "source_url": "https://example.com/source", + "canonical_url": "https://example.com/source", + "meta_json": '{"generated_tags":["Camping"]}', + } + publish_article_draft(article) + payload = [c for c in mock_wp_request.call_args_list if c.kwargs.get("endpoint") == "posts"][0].kwargs["payload"] + self.assertNotIn("tags", payload) @patch("backend.app.wordpress._upload_featured_media") @patch("backend.app.wordpress._wp_request")