"""Content Service — Generates full SEO-optimized blog articles with Claude."""

import json
import logging
import re

import anthropic
import markdown
import yaml
from django.conf import settings

from blogs.models import BlogArticle, BlogArticleImage, BlogGenerationLog, BlogSettings

logger = logging.getLogger(__name__)

DEFAULT_SEO_RULES = {
    "meta_title_max": 60,
    "meta_description_max": 160,
    "keyword_density_min": 5,
    "keyword_density_max": 8,
    "word_count_min": 1500,
    "word_count_max": 2500,
    "faq_word_budget_min": 400,
    "faq_word_budget_max": 600,
    "min_main_h2_count": 5,
    "min_faq_if_present_short": 3,
    "min_faq_if_present_mid": 4,
    "min_faq_if_present_long": 5,
    "word_band_short": 1400,
    "word_band_long": 2000,
    "max_output_tokens": 12000,
}


def _seo_copy(language: str) -> dict[str, str]:
    """Localized section headings for SEO body structure."""
    if (language or "de").lower().startswith("en"):
        return {
            "takeaways_h2": "Key takeaways",
            "faq_h2": "Frequently asked questions",
            "mistakes_h2": "Common mistakes to avoid",
            "cta_hint": "End with ## [Brand] — [short action headline]",
            "takeaways_alt": "Key takeaways",
            "faq_alt": "Frequently asked questions",
        }
    return {
        "takeaways_h2": "Wichtigste Erkenntnisse",
        "faq_h2": "Häufig gestellte Fragen",
        "mistakes_h2": "Häufige Fehler vermeiden",
        "cta_hint": "Ende mit ## Jetzt bei [Brand] starten (oder passende CTA-Ueberschrift)",
        "takeaways_alt": "Key takeaways",
        "faq_alt": "Frequently asked questions",
    }


def _log_step(article, step, status, details=None):
    BlogGenerationLog.objects.create(
        article=article, step=step, status=status, details=details or {}
    )


def _build_system_prompt(website, style_analysis=None):
    """Build the dynamic system prompt from BlogSettings / website configuration."""
    existing_articles = website.get("existing_articles", [])
    links_text = (
        "\n".join(
            f"  - {a['title']} -> {a['url']}"
            for a in existing_articles
            if isinstance(a, dict) and a.get("url")
        )
        if existing_articles
        else "  Keine vorhanden"
    )

    seo_rules = DEFAULT_SEO_RULES
    keyword_density_min = seo_rules.get("keyword_density_min", 5)
    keyword_density_max = seo_rules.get("keyword_density_max", 8)
    meta_title_max = seo_rules.get("meta_title_max", 60)
    meta_desc_max = seo_rules.get("meta_description_max", 160)

    style_section = ""
    if style_analysis:
        style_section = f"""
Stil-Referenz (aus Analyse bestehender Artikel):
- Ton: {style_analysis.get('tone', 'Nicht analysiert')}
- Vokabular: {style_analysis.get('vocabulary_level', 'mittel')}
- Satzstruktur: {style_analysis.get('sentence_style', 'Gemischt')}
- Absatzlaenge: {style_analysis.get('paragraph_length', 'kurz')}
- Besonderheiten: {style_analysis.get('unique_characteristics', 'Keine')}
- Typische Elemente: {', '.join(style_analysis.get('typical_elements', []))}
- Formatierung: {', '.join(style_analysis.get('formatting_preferences', []))}
"""

    language = website.get("language", "de")
    lang_instruction = "Sprache: Deutsch" if language == "de" else "Sprache: Englisch"
    domain = website.get("domain", "")
    cta_link_hint = f"+ Link: {domain}" if domain else ""
    copy = _seo_copy(language)

    return f"""Du bist ein SEO-Content-Autor fuer {website['name']} ({domain}).
Du erstellst Blog-Artikel, die organischen Traffic generieren und Leser zu Aktionen konvertieren.

Ueber {website['name']}:
- Was: {website.get('description', 'Keine Beschreibung')}
- Zielgruppe: {website.get('target_audience', 'Nicht definiert')}
- Website: {domain}
- Tonalitaet: {website.get('tone_description', 'Professionell und informativ')}
{style_section}

Format & Struktur (Pflicht-Reihenfolge im Markdown-Body):
- Gesamtlaenge: 1.500-2.500 Woerter (Opening + Takeaways + Hauptteil + optional FAQ + CTA)
- {lang_instruction}
- KEIN # H1 im Body (Titel kommt aus Frontmatter/planning; Body startet mit Fliesstext)
- Absaetze: Kurz und scanbar. Bullet Points wo sinnvoll.

Pflicht-Gliederung (exakte Reihenfolge nach YAML-Frontmatter):
1) Opening: 1-2 kurze Absaetze (max ~120 Woerter). Haupt-Keyword im ersten Satz.
2) ## {copy['takeaways_h2']}: 4-6 Bullet-Punkte, je ein praegnanter Satz.
3) Hauptteil: 6-10 Abschnitte als ## H2 (Frage-Titel bevorzugt, z.B. "Was ist ...?", "Wie funktioniert ...?").
   Mindestens ein H2 mit Haupt-Keyword. Nutze ### H3 fuer Unterpunkte.
   Pflicht-H2: "## {copy['mistakes_h2']}" (oder sehr ähnliche Formulierung).
   Mindestens 2 interne Links in diesem Hauptteil (nicht nur in FAQ).
   Wenn FAQ geplant ist: Hauptteil etwas kuerzer halten, damit das Gesamtwortlimit eingehalten wird.
4) FAQ (NUR wenn include_faq: true im Frontmatter — siehe unten):
   ## {copy['faq_h2']} mit 3-6 Eintraegen (### Frage mit ?, Antwort 2-4 Saetze).
   Wortbudget fuer FAQ: ca. 400-600 Woerter. Bei Gesamtlaenge ~1.500 Woerter: 3-4 FAQs;
   bei ~2.000+ Woertern: 4-6 FAQs.
   include_faq: true bei Erklaer-/Ratgeber-Themen (Was ist, Wie funktioniert, Guide, Tipps, Vergleich).
   include_faq: false bei sehr kurzen/promo-artigen Beitraegen oder wenn Nutzerfragen redundant waeren.
5) CTA: Letzter Abschnitt als ## mit kurzer Ueberschrift + Absatz + Link zu {website['name']}.
   {copy['cta_hint']} {cta_link_hint}

- Ueberschriften: H2 und H3, keyword-optimiert
- Bilder: 3-6 Bildplatzhalter mit detaillierter Beschreibung im Alt-Text.
  Format: [IMAGE: Alt-Text-Beschreibung]
  Das ERSTE [IMAGE: ...] ist nur das Beitragsbild (Featured Image) und darf NICHT im Fliesstext stehen.
  Weitere [IMAGE: ...] gehoeren in den Artikeltext.
  WICHTIG: Beschreibungen sollen KEINE Text-Elemente, Labels oder Produktnamen enthalten
  (AI-Bildgeneratoren halluzinieren Text). Stattdessen: Nahaufnahmen von Zutaten,
  abstrakte Illustrationen, Lifestyle-Fotos, Haende mit Produkten (ohne Label sichtbar).
- CTA: Jeder Artikel endet mit einem Call-to-Action zu {website['name']}
  {cta_link_hint}

SEO-Pflichtfelder (am Anfang des Artikels als YAML-Frontmatter):
- meta_title: Max {meta_title_max} Zeichen, Haupt-Keyword am Anfang
- meta_description: {meta_desc_max - 10}-{meta_desc_max} Zeichen, mit Keyword und CTA
- include_faq: true oder false (Pflichtfeld — entscheide nach Thema)
- tags: 5-8 relevante Tags, kommasepariert
- slug: URL-freundlich, kurz, keyword-relevant

Kein Breadcrumb, keine Sidebar, kein Autorenblock, kein "Weitere Artikel"-Block im Body.

SEO-Checkliste (immer einhalten):
- Haupt-Keyword in: Titel (H1), Meta-Title, Meta-Description, erstem Absatz, mindestens einer H2
- 2-3 sekundaere Keywords natuerlich im Text verteilt
- Interne Verlinkung: Mindestens 2 Links zu bestehenden Artikeln
- Keyword-Dichte: Natuerlich, nicht spammy. Haupt-Keyword ca. {keyword_density_min}-{keyword_density_max}x im gesamten Text.

Bestehende Artikel fuer interne Verlinkung:
{links_text}

CTA-Text: Jetzt bei {website['name']} starten {cta_link_hint}"""


def _build_user_prompt(article):
    settings_obj = BlogSettings.load()
    copy = _seo_copy(settings_obj.language or article.locale or "de")
    secondary_kw = article.secondary_keywords or []
    secondary_text = (
        f"\nSekundaere Keywords: {', '.join(secondary_kw)}" if secondary_kw else ""
    )

    return f"""Erstelle den vollstaendigen Artikel zum Thema:

Titel/Thema: {article.title or 'Kein Titel'}
Haupt-Keyword: {article.target_keyword or ''}{secondary_text}

Anforderungen:
1. Pflicht-Reihenfolge: Opening -> {copy['takeaways_h2']} -> H2-Hauptteil inkl. {copy['mistakes_h2']} -> (FAQ nur wenn include_faq: true) -> CTA
2. Kein # H1 im Markdown-Body
3. 3-6 Bildplatzhalter im Hauptteil (KEINE Text-Elemente in Bildbeschreibungen!)
4. Mindestens 2 interne Links im Hauptteil
5. Setze include_faq: true nur wenn echte Nutzerfragen sinnvoll sind; dann ## {copy['faq_h2']} mit 3-6 ###-Fragen
6. Gesamt 1.500-2.500 Woerter; bei FAQ das Hauptteil-Wortbudget reduzieren, FAQ nicht weglassen wegen Laenge wenn include_faq: true

WICHTIG: Beginne mit YAML-Frontmatter:
```
---
meta_title: "..."
meta_description: "..."
include_faq: true
tags: ["tag1", "tag2", ...]
slug: "..."
---
```

Dann der Markdown-Body in der Pflicht-Reihenfolge.
WICHTIG: Keine ``` Zeilen vor oder nach dem Body (kein Code-Fence um Frontmatter oder Artikel)."""


def _strip_code_fences(text):
    text = (text or "").strip()
    fenced = re.match(
        r"^```(?:yaml|yml|markdown|md)?\s*\n(.*)\n```\s*$",
        text,
        re.DOTALL | re.IGNORECASE,
    )
    if fenced:
        return fenced.group(1).strip()
    if text.startswith("```"):
        lines = text.split("\n")
        if lines[0].startswith("```"):
            lines = lines[1:]
        if lines and lines[-1].strip() == "```":
            lines = lines[:-1]
        return "\n".join(lines).strip()
    return text


def _strip_orphan_fence_lines(body: str) -> str:
    """
    Remove leftover ``` lines Claude often adds after YAML frontmatter.
    Without this, markdown renders them as <p>```</p> at the top of the article.
    """
    lines = (body or "").splitlines()
    while lines and re.match(r"^```\w*\s*$", lines[0].strip()):
        lines.pop(0)
    while lines and lines[-1].strip() == "```":
        lines.pop()
    return "\n".join(lines).strip()


def _parse_article_response(raw: str) -> dict:
    """Parse Claude response: YAML frontmatter + markdown body."""
    result = {
        "body": raw,
        "meta_title": "",
        "meta_description": "",
        "tags": [],
        "slug": "",
        "title": "",
        "include_faq": None,
    }

    raw = _strip_code_fences(raw)
    frontmatter_match = re.match(r"^---\s*\n(.*?)\n---\s*\n(.*)", raw, re.DOTALL)
    if frontmatter_match:
        frontmatter_text = frontmatter_match.group(1)
        result["body"] = _strip_orphan_fence_lines(frontmatter_match.group(2))
        try:
            data = yaml.safe_load(frontmatter_text)
            if isinstance(data, dict):
                result["meta_title"] = str(data.get("meta_title") or "").strip()
                result["meta_description"] = str(
                    data.get("meta_description") or ""
                ).strip()
                result["slug"] = str(data.get("slug") or "").strip().strip("/")
                tags = data.get("tags")
                if isinstance(tags, list):
                    result["tags"] = [str(t).strip() for t in tags if str(t).strip()]
                elif isinstance(tags, str) and tags.strip():
                    result["tags"] = [
                        t.strip() for t in tags.split(",") if t.strip()
                    ]
                result["include_faq"] = _parse_include_faq_value(
                    data.get("include_faq")
                )
        except yaml.YAMLError:
            logger.warning("YAML frontmatter parse failed, using line fallback")

        if not result["meta_title"]:
            for line in frontmatter_text.split("\n"):
                line = line.strip()
                if line.startswith("meta_title:"):
                    result["meta_title"] = (
                        line.split(":", 1)[1].strip().strip("\"'")
                    )
                elif line.startswith("meta_description:"):
                    result["meta_description"] = (
                        line.split(":", 1)[1].strip().strip("\"'")
                    )
                elif line.startswith("slug:"):
                    result["slug"] = line.split(":", 1)[1].strip().strip("\"'")
                elif line.startswith("include_faq:"):
                    result["include_faq"] = _parse_include_faq_value(
                        line.split(":", 1)[1].strip().strip("\"'")
                    )
                elif line.startswith("tags:"):
                    tags_str = line.split(":", 1)[1].strip()
                    if tags_str.startswith("["):
                        try:
                            result["tags"] = json.loads(tags_str)
                        except json.JSONDecodeError:
                            result["tags"] = [
                                t.strip().strip("\"'")
                                for t in tags_str.strip("[]").split(",")
                            ]
                    else:
                        result["tags"] = [
                            t.strip() for t in tags_str.split(",") if t.strip()
                        ]
    else:
        result["body"] = _strip_orphan_fence_lines(raw)

    h1_match = re.search(r"^#\s+(.+)$", result["body"], re.MULTILINE)
    if h1_match:
        result["title"] = h1_match.group(1).strip()

    return result


def _parse_include_faq_value(raw) -> bool | None:
    if raw is None or raw == "":
        return None
    if isinstance(raw, bool):
        return raw
    s = str(raw).strip().lower()
    if s in ("true", "yes", "1", "ja"):
        return True
    if s in ("false", "no", "0", "nein"):
        return False
    return None


def _count_markdown_words(body_md: str) -> int:
    text = re.sub(r"^#+\s+.+$", "", body_md or "", flags=re.MULTILINE)
    text = re.sub(r"\[IMAGE:\s*.+?\]", "", text)
    text = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", text)
    text = re.sub(r"[*_`#>|]", "", text)
    return len(text.split())


def _count_faq_items(body_md: str, language: str) -> int:
    count = 0
    in_faq = False
    for line in (body_md or "").splitlines():
        if line.startswith("## "):
            in_faq = _heading_matches(line[3:].strip(), _faq_heading_patterns(language))
            continue
        if in_faq and line.startswith("### "):
            count += 1
    return count


def _has_faq_section(body_md: str, language: str) -> bool:
    for line in (body_md or "").splitlines():
        if line.startswith("## "):
            if _heading_matches(line[3:].strip(), _faq_heading_patterns(language)):
                return True
    return False


def _min_faq_for_word_count(word_count: int, seo_rules: dict | None = None) -> int:
    rules = seo_rules or DEFAULT_SEO_RULES
    if word_count >= rules.get("word_band_long", 2000):
        return rules.get("min_faq_if_present_long", 5)
    if word_count >= rules.get("word_band_short", 1400):
        return rules.get("min_faq_if_present_mid", 4)
    return rules.get("min_faq_if_present_short", 3)


def _excerpt_from_body(body, max_len):
    text = re.sub(r"^#+\s+.+$", "", body, flags=re.MULTILINE)
    text = re.sub(r"\[IMAGE:\s*.+?\]", "", text)
    text = re.sub(r"\[([^\]]+)\]\([^)]+\)", r"\1", text)
    text = re.sub(r"[*_`#>|]", "", text)
    para = next((p.strip() for p in text.split("\n\n") if p.strip()), "")
    return para[:max_len]


def _fill_missing_metadata(parsed, article, seo_rules):
    """Derive SEO fields when the model omits or malforms frontmatter."""
    fallbacks = []
    title = parsed.get("title") or article.title or ""
    tmax = int(seo_rules.get("meta_title_max", 60))
    dmax = int(seo_rules.get("meta_description_max", 160))

    if not parsed.get("meta_title"):
        parsed["meta_title"] = (title or article.target_keyword or "Blog")[:tmax]
        fallbacks.append("meta_title")
    if not parsed.get("slug"):
        parsed["slug"] = _generate_slug(
            title or article.target_keyword or f"article-{article.pk}"
        )
        fallbacks.append("slug")
    if not parsed.get("meta_description"):
        parsed["meta_description"] = _excerpt_from_body(
            parsed.get("body") or "", dmax
        ) or (article.target_keyword or title or "")[:dmax]
        fallbacks.append("meta_description")
    if not parsed.get("title"):
        parsed["title"] = title
    if not parsed.get("tags"):
        tags = [article.target_keyword] if article.target_keyword else []
        tags.extend(article.secondary_keywords or [])
        parsed["tags"] = [t for t in tags if t][:8]
        if tags:
            fallbacks.append("tags")
    return fallbacks


def _validate_meta(parsed, article, seo_rules):
    fallbacks = _fill_missing_metadata(parsed, article, seo_rules)
    if not parsed.get("body", "").strip():
        raise ValueError("Model returned empty article body")
    tmax = int(seo_rules.get("meta_title_max", 60))
    dmax = int(seo_rules.get("meta_description_max", 160))
    parsed["meta_title"] = parsed["meta_title"][:tmax]
    parsed["meta_description"] = parsed["meta_description"][:dmax]
    parsed["slug"] = _generate_slug(parsed["slug"] or parsed.get("title") or "")
    if fallbacks:
        parsed["_metadata_fallbacks"] = fallbacks
    return parsed


def _normalize_heading(text: str) -> str:
    return re.sub(r"\s+", " ", (text or "").strip().lower())


def _heading_matches(text: str, patterns: tuple[str, ...]) -> bool:
    norm = _normalize_heading(text)
    return any(p in norm or norm in p for p in patterns)


def _takeaway_heading_patterns(language: str) -> tuple[str, ...]:
    copy = _seo_copy(language)
    return (
        _normalize_heading(copy["takeaways_h2"]),
        _normalize_heading(copy["takeaways_alt"]),
        "wichtigste erkenntnisse",
        "key takeaways",
    )


def _faq_heading_patterns(language: str) -> tuple[str, ...]:
    copy = _seo_copy(language)
    return (
        _normalize_heading(copy["faq_h2"]),
        _normalize_heading(copy["faq_alt"]),
        "häufig gestellte fragen",
        "frequently asked questions",
        "faq",
    )


def _mistakes_heading_patterns(language: str) -> tuple[str, ...]:
    copy = _seo_copy(language)
    return (
        _normalize_heading(copy["mistakes_h2"]),
        "common mistakes",
        "häufige fehler",
    )


def _validate_content_structure_md(
    body_md: str,
    language: str,
    *,
    include_faq: bool | None = None,
    seo_rules: dict | None = None,
) -> list[str]:
    """Return warnings for SEO body structure (FAQ rules depend on include_faq)."""
    rules = seo_rules or DEFAULT_SEO_RULES
    warnings = []
    if not (body_md or "").strip():
        warnings.append("empty_body")
        return warnings

    word_count = _count_markdown_words(body_md)
    wmin = rules.get("word_count_min", 1500)
    wmax = rules.get("word_count_max", 2500)
    if word_count < wmin - 200:
        warnings.append(f"short_article:{word_count}<{wmin}")
    elif word_count > wmax + 300:
        warnings.append(f"long_article:{word_count}>{wmax}")

    if re.search(r"^#\s+[^#]", body_md, re.MULTILINE):
        warnings.append("h1_in_body")

    has_faq = _has_faq_section(body_md, language)
    faq_count = _count_faq_items(body_md, language) if has_faq else 0

    if not re.search(r"(?m)^##\s+.+", body_md):
        warnings.append("no_h2_sections")
    else:
        h2_texts = re.findall(r"(?m)^##\s+(.+)$", body_md)
        if not any(
            _heading_matches(t, _takeaway_heading_patterns(language)) for t in h2_texts
        ):
            warnings.append("missing_key_takeaways")
        if not any(
            _heading_matches(t, _mistakes_heading_patterns(language)) for t in h2_texts
        ):
            warnings.append("missing_mistakes_section")
        main_h2 = [
            t
            for t in h2_texts
            if not _heading_matches(t, _takeaway_heading_patterns(language))
            and not _heading_matches(t, _faq_heading_patterns(language))
        ]
        min_h2 = rules.get("min_main_h2_count", 5)
        if len(main_h2) < min_h2:
            warnings.append(f"few_main_sections:{len(main_h2)}<{min_h2}")

    if include_faq is True:
        if not has_faq:
            warnings.append("missing_faq_section")
        else:
            min_faq = _min_faq_for_word_count(word_count, rules)
            if faq_count < min_faq:
                warnings.append(f"few_faq_items:{faq_count}<{min_faq}")
    elif include_faq is not False and has_faq:
        min_faq = _min_faq_for_word_count(word_count, rules)
        if faq_count < min_faq:
            warnings.append(f"few_faq_items:{faq_count}<{min_faq}")

    return warnings


def _faq_structure_needs_repair(warnings: list[str]) -> bool:
    return any(
        w == "missing_faq_section" or w.startswith("few_faq_items:")
        for w in warnings
    )


def _build_faq_repair_prompt(body_md: str, language: str, min_faq: int) -> str:
    copy = _seo_copy(language)
    budget = DEFAULT_SEO_RULES.get("faq_word_budget_min", 400)
    budget_max = DEFAULT_SEO_RULES.get("faq_word_budget_max", 600)
    body_tail = body_md[-6000:] if len(body_md) > 6000 else body_md
    return f"""Der Blog-Artikel unten hat include_faq: true, aber der FAQ-Abschnitt fehlt oder ist unvollstaendig.

Aufgabe: Gib den VOLLSTAENDIGEN Markdown-Body zurueck (ohne YAML-Frontmatter, kein # H1).
- Fuege oder vervollstaendige ## {copy['faq_h2']} mit mindestens {min_faq} FAQ-Eintraegen.
- Jede Frage als ### (vollstaendige Frage mit ?), Antwort 2-4 Saetze.
- FAQ vor dem letzten CTA-##-Abschnitt einfuegen.
- Wortbudget fuer FAQ: ca. {budget}-{budget_max} Woerter; passe den Hauptteil leicht an, Gesamt 1.500-2.500 Woerter.
- Alle anderen Pflicht-Abschnitte (Takeaways, Hauptteil, Fehler, CTA) beibehalten.

Artikel:
{body_tail}
"""


def _claude_message(client, model: str, system: str, user: str, max_tokens: int) -> str:
    response = client.messages.create(
        model=model,
        max_tokens=max_tokens,
        messages=[{"role": "user", "content": user}],
        system=system,
        temperature=0.5,
    )
    return response.content[0].text.strip()


def _split_html_by_h2(html: str) -> list[tuple[str, str]]:
    """Return list of (h2_html, following_content_html)."""
    parts = re.split(r"(?=<h2[\s>])", html, flags=re.IGNORECASE)
    blocks = []
    for part in parts:
        part = part.strip()
        if not part:
            continue
        m = re.match(r"(<h2[^>]*>.*?</h2>)(.*)", part, re.DOTALL | re.IGNORECASE)
        if m:
            blocks.append((m.group(1), m.group(2)))
        elif not blocks:
            blocks.append(("", part))
    return blocks


def _h2_inner_text(h2_html: str) -> str:
    return re.sub(r"<[^>]+>", "", h2_html or "").strip()


def _add_h2_ids(html: str) -> str:
    def repl(match):
        attrs = match.group(1) or ""
        inner = match.group(2)
        if re.search(r"\bid\s*=", attrs, re.IGNORECASE):
            return match.group(0)
        slug = _generate_slug(_h2_inner_text(f"<span>{inner}</span>")) or "section"
        return f'<h2 id="{slug}"{attrs}>{inner}</h2>'

    return re.sub(
        r"<h2(\s[^>]*)?>(.*?)</h2>",
        repl,
        html,
        flags=re.IGNORECASE | re.DOTALL,
    )


def _wrap_seo_sections(html: str, language: str) -> str:
    """Wrap key takeaways and FAQ blocks for theme styling."""
    blocks = _split_html_by_h2(html)
    if not blocks:
        return html

    prefix = blocks[0][1] if blocks[0][0] == "" else ""
    if blocks[0][0] == "":
        blocks = blocks[1:]

    out = [prefix] if prefix else []
    takeaway_patterns = _takeaway_heading_patterns(language)
    faq_patterns = _faq_heading_patterns(language)

    for h2_html, body_html in blocks:
        title = _h2_inner_text(h2_html)
        chunk = h2_html + body_html
        if _heading_matches(title, takeaway_patterns):
            out.append(f'<div class="blog-key-takeaways">{chunk}</div>')
        elif _heading_matches(title, faq_patterns):
            out.append(f'<section class="blog-faq">{chunk}</section>')
        else:
            out.append(chunk)

    return "".join(out)


def _finalize_article_html(html: str, language: str) -> str:
    html = _add_h2_ids(html)
    return _wrap_seo_sections(html, language)


def _markdown_to_html(md_text: str, language: str = "de") -> str:
    """Convert markdown to HTML, preserving image placeholders."""
    md_text = re.sub(
        r"\[IMAGE:\s*(.+?)\]",
        r'<div class="image-placeholder" data-alt="\1"><!-- IMAGE: \1 --></div>',
        md_text,
    )
    html = markdown.markdown(
        md_text,
        extensions=["extra", "sane_lists", "smarty"],
    )
    return _finalize_article_html(html, language)


def _extract_image_placeholders(md_text: str) -> list:
    return re.findall(r"\[IMAGE:\s*(.+?)\]", md_text)


def _body_without_featured_placeholder(md_text: str) -> str:
    """Body markdown for HTML: excludes the first [IMAGE] (featured-only, not in content)."""
    return re.sub(r"\[IMAGE:\s*.+?\]\s*", "", md_text, count=1)


def _extract_internal_links(html: str, existing_articles: list) -> list:
    links = []
    for art in existing_articles or []:
        if not isinstance(art, dict):
            continue
        url = art.get("url", "")
        if url and url in html:
            links.append({"title": art.get("title", ""), "url": url})
    return links


def _generate_slug(title: str) -> str:
    slug = (title or "").lower()
    replacements = {
        "ä": "ae",
        "ö": "oe",
        "ü": "ue",
        "ß": "ss",
    }
    for old, new in replacements.items():
        slug = slug.replace(old, new)
    slug = re.sub(r"[^a-z0-9\-]", "-", slug)
    slug = re.sub(r"-+", "-", slug).strip("-")
    return slug[:80]


def _persist_generated_article(
    article,
    parsed: dict,
    website: dict,
    seo_rules: dict,
    *,
    meta_fallbacks=None,
    structure_warnings=None,
    faq_repaired=False,
):
    """Convert parsed markdown to HTML and save article + image slots."""
    image_placeholders = _extract_image_placeholders(parsed["body"])
    header_alt = image_placeholders[0] if image_placeholders else None
    inline_alts = image_placeholders[1:]

    body_md = parsed["body"]
    if header_alt:
        body_md = _body_without_featured_placeholder(body_md)

    language = website.get("language", article.locale or "de")
    html_content = _markdown_to_html(body_md, language=language)
    plain_text = re.sub(r"<[^>]+>", "", html_content)
    word_count = len(plain_text.split())

    keyword = (article.target_keyword or "").lower()
    keyword_count = plain_text.lower().count(keyword) if keyword else 0
    keyword_density = round(keyword_count / max(word_count, 1) * 100, 2)
    internal_links = _extract_internal_links(
        html_content, website.get("existing_articles", [])
    )

    article.title = parsed.get("title") or article.title
    article.slug = parsed.get("slug") or _generate_slug(
        parsed.get("title", article.title)
    )
    article.content = html_content
    article.meta_title = parsed["meta_title"]
    article.meta_description = parsed["meta_description"]
    article.tags = parsed.get("tags") or []
    article.word_count = word_count
    article.keyword_density = keyword_density
    article.internal_links = internal_links
    article.generation_error = ""
    article.save(
        update_fields=[
            "title",
            "slug",
            "content",
            "meta_title",
            "meta_description",
            "tags",
            "word_count",
            "keyword_density",
            "internal_links",
            "generation_error",
            "updated_at",
        ]
    )

    article.images.all().delete()
    if header_alt:
        BlogArticleImage.objects.create(
            article=article,
            alt_text=header_alt,
            position="header",
            generation_status="pending",
        )
    for i, alt in enumerate(inline_alts, start=1):
        BlogArticleImage.objects.create(
            article=article,
            alt_text=alt,
            position=f"inline-{i}",
            generation_status="pending",
        )

    log_details = {
        "word_count": word_count,
        "keyword_density": keyword_density,
        "image_placeholders": len(image_placeholders),
        "include_faq": parsed.get("include_faq"),
        "faq_count": _count_faq_items(body_md, language),
    }
    if meta_fallbacks:
        log_details["metadata_fallbacks"] = meta_fallbacks
    if structure_warnings:
        log_details["structure_warnings"] = structure_warnings
    if faq_repaired:
        log_details["faq_repaired"] = True
    return article, log_details, len(image_placeholders)


def generate_article_content(article_id):
    """Generate a full blog article using Claude (Django worker entrypoint)."""
    article = BlogArticle.objects.get(pk=article_id)
    website = BlogSettings.load().as_prompt_config()
    seo_rules = DEFAULT_SEO_RULES

    article.status = BlogArticle.STATUS_GENERATING
    article.save(update_fields=["status", "updated_at"])
    _log_step(article, "content_generation", "started")

    key = getattr(settings, "ANTHROPIC_API_KEY", "")
    if not key:
        raise RuntimeError("ANTHROPIC_API_KEY not configured")
    model = getattr(settings, "ANTHROPIC_MODEL", "claude-sonnet-4-20250514")
    max_tokens = int(
        getattr(settings, "BLOG_CONTENT_MAX_TOKENS", None)
        or seo_rules.get("max_output_tokens", 12000)
    )

    system_prompt = _build_system_prompt(website)
    user_prompt = _build_user_prompt(article)

    try:
        client = anthropic.Anthropic(api_key=key)
        raw_content = _claude_message(
            client, model, system_prompt, user_prompt, max_tokens
        )
        _log_step(
            article,
            "content_generation",
            "completed",
            {"raw_length": len(raw_content)},
        )

        parsed = _validate_meta(
            _parse_article_response(raw_content), article, seo_rules
        )
        meta_fallbacks = parsed.pop("_metadata_fallbacks", None)

        language = website.get("language", article.locale or "de")
        include_faq = parsed.get("include_faq")
        body_md = parsed["body"]
        structure_warnings = _validate_content_structure_md(
            body_md,
            language,
            include_faq=include_faq,
            seo_rules=seo_rules,
        )

        faq_repaired = False
        if include_faq is True and _faq_structure_needs_repair(structure_warnings):
            logger.info(
                "Article %s FAQ repair triggered: %s",
                article_id,
                structure_warnings,
            )
            _log_step(article, "faq_repair", "started", {"warnings": structure_warnings})
            min_faq = _min_faq_for_word_count(_count_markdown_words(body_md), seo_rules)
            repair_raw = _claude_message(
                client,
                model,
                system_prompt,
                _build_faq_repair_prompt(body_md, language, min_faq),
                max_tokens,
            )
            repaired = _parse_article_response(repair_raw)
            parsed["body"] = repaired.get("body") or repair_raw
            if repaired.get("include_faq") is not None:
                parsed["include_faq"] = repaired["include_faq"]
            faq_repaired = True
            structure_warnings = _validate_content_structure_md(
                parsed["body"],
                language,
                include_faq=True,
                seo_rules=seo_rules,
            )
            _log_step(
                article,
                "faq_repair",
                "completed",
                {"warnings_after": structure_warnings},
            )

        article, log_details, img_count = _persist_generated_article(
            article,
            parsed,
            website,
            seo_rules,
            meta_fallbacks=meta_fallbacks,
            structure_warnings=structure_warnings,
            faq_repaired=faq_repaired,
        )

        if structure_warnings:
            logger.warning(
                "Article %s structure warnings: %s",
                article_id,
                structure_warnings,
            )
        _log_step(article, "content_complete", "completed", log_details)

        logger.info(
            "Article %s generated: %s words, %s images, include_faq=%s",
            article_id,
            log_details.get("word_count"),
            img_count,
            log_details.get("include_faq"),
        )
        return article

    except Exception as exc:
        logger.exception("Article generation failed for %s", article_id)
        article.status = BlogArticle.STATUS_FAILED
        article.generation_error = str(exc)
        article.save(update_fields=["status", "generation_error", "updated_at"])
        _log_step(article, "content_generation", "failed", {"error": str(exc)})
        raise
