_resolve_wp_tag_ids created a WordPress tag for every keyword the rewriter invented, up to 12 per post. That is where the 3.055 tags for 955 posts came from - 1.683 of them used exactly once, 473 attached to no post at all. The categories are stable now, but the tags would simply grow back. Three changes, all on the write path: A proposed tag has to appear for wordpress_new_tag_min_proposals (3) different articles before it is created. Proposals are counted in the new tag_proposals table, keyed on (name, article), so re-publishing an article does not inflate its own count. Tags that already exist in WordPress are assigned as before - the gate only guards creation. Only the first wordpress_max_tags_per_post (5) tags reach WordPress. The full list still feeds the category rules, which were validated against it. The lookup fallback of reusing the first search hit is gone. It filed "Camping" under the unrelated existing tag "Campingplatz" whenever the exact tag was missing, which quietly produced wrong tags rather than none. If the proposal bookkeeping fails, nothing is creatable that run: existing tags still get assigned and the taxonomy stays put, rather than falling back to creating everything. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
73 lines
3.2 KiB
Python
73 lines
3.2 KiB
Python
from functools import lru_cache
|
|
from pathlib import Path
|
|
|
|
from dotenv import load_dotenv
|
|
from pydantic import AliasChoices, Field
|
|
from pydantic_settings import BaseSettings, SettingsConfigDict
|
|
|
|
|
|
class Settings(BaseSettings):
|
|
# Prefer backend-specific env file to avoid collisions with legacy root .env
|
|
model_config = SettingsConfigDict(
|
|
env_file=("backend/.env", ".env"),
|
|
env_file_encoding="utf-8",
|
|
extra="ignore",
|
|
)
|
|
|
|
app_env: str = "development"
|
|
app_name: str = "rss-news-backend"
|
|
app_secret_key: str = "replace-with-a-long-random-secret"
|
|
|
|
app_admin_username: str = "admin"
|
|
app_admin_password: str = "change-me"
|
|
|
|
session_cookie_name: str = "rss_news_session"
|
|
session_max_age_seconds: int = 28800
|
|
|
|
app_db_path: str = "backend/data/rss_news.db"
|
|
|
|
wordpress_base_url: str | None = Field(default=None, validation_alias=AliasChoices("WORDPRESS_BASE_URL", "WP_BASE_URL"))
|
|
wordpress_username: str | None = Field(default=None, validation_alias=AliasChoices("WORDPRESS_USERNAME", "WP_USERNAME"))
|
|
wordpress_app_password: str | None = Field(default=None, validation_alias=AliasChoices("WORDPRESS_APP_PASSWORD", "WP_PASSWORD"))
|
|
wordpress_default_status: str = "draft"
|
|
# Tag hygiene: the rewriter proposes far more tags than a post needs, and
|
|
# every unknown one used to be created on the spot - that is how 955 posts
|
|
# accumulated 3.055 tags, 1.683 of them used exactly once.
|
|
wordpress_max_tags_per_post: int = 5
|
|
wordpress_new_tag_min_proposals: int = 3
|
|
openai_api_key: str | None = Field(default=None, validation_alias=AliasChoices("OPENAI_API_KEY"))
|
|
openai_model: str = "gpt-4o-mini"
|
|
|
|
# Telegram Bot
|
|
telegram_bot_token: str | None = Field(default=None, validation_alias=AliasChoices("TELEGRAM_BOT_TOKEN"))
|
|
telegram_chat_id: str | None = Field(default=None, validation_alias=AliasChoices("TELEGRAM_CHAT_ID"))
|
|
telegram_webhook_secret: str | None = Field(default=None, validation_alias=AliasChoices("TELEGRAM_WEBHOOK_SECRET"))
|
|
|
|
# N8N API authentication
|
|
n8n_api_key: str | None = Field(default=None, validation_alias=AliasChoices("N8N_API_KEY"))
|
|
|
|
# Pipeline behaviour
|
|
pipeline_relevance_auto: int = 80 # >= this: auto-process
|
|
pipeline_relevance_warn: int = 60 # >= this: Telegram warning, else reject
|
|
pipeline_max_drafts_per_day: int = 4
|
|
pipeline_publish_hours: str = "9,12,15,18" # comma-separated preferred publish hours (CET)
|
|
pipeline_publish_start_hour: int = 9
|
|
pipeline_publish_end_hour: int = 19
|
|
pipeline_publish_min_gap_hours: int = 3
|
|
pipeline_min_words_raw: int = 120 # minimum words in raw content before rewrite (else reject)
|
|
pipeline_min_words_rewritten: int = 150 # minimum words in rewritten content (else reject)
|
|
pipeline_max_article_age_days: int = 7 # skip articles older than N days during ingestion (0 = no limit)
|
|
|
|
|
|
@lru_cache(maxsize=1)
|
|
def get_settings() -> Settings:
|
|
# Prefer shared legacy env from the original rss-news workspace if present.
|
|
env_candidates = (
|
|
Path("/Users/oliver/Documents/rss-news/.env"),
|
|
Path("backend/.env"),
|
|
Path(".env"),
|
|
)
|
|
for env_path in env_candidates:
|
|
if env_path.exists():
|
|
load_dotenv(env_path, override=False)
|
|
return Settings()
|