"""Stage 7: transform the timestamped transcript into a structured, AEO-friendly tutorial. Uses an HF Inference Providers chat model (default DeepSeek-V3) billed to the user's token. The model returns strict JSON so the downstream weighted indicator and the .docx builder have stable fields to work with. The output follows answer-engine-optimization (AEO) / AI-citation best practices: an answer-first paragraph near the top, descriptive title + meta description + URL slug, clear H2 step headings, and an FAQ section. SEO keyword placement is explicit (see ``_keyword_block``). """ from __future__ import annotations import json import re from huggingface_hub import InferenceClient DEFAULT_LLM = "deepseek-ai/DeepSeek-V3" # Rough char budget to stay clear of context limits on the free path. Long transcripts # are truncated (with a marker); good enough for a tutorial summary. MAX_TRANSCRIPT_CHARS = 24000 _SYSTEM = ( "You are a technical writer who optimizes for answer engines (ChatGPT, Claude, " "Gemini, Perplexity, Google AI). You convert a timestamped video transcript into a " "clear, citable, step-by-step written tutorial. You ALWAYS respond with a single JSON " "object and no prose outside it." ) _INSTRUCTIONS = """\ Turn the transcript below into a tutorial blog post. Return ONLY a JSON object with this schema: { "title": "string - H1 / title tag; concise and descriptive", "slug": "string - lowercase, hyphenated URL slug", "meta_description": "string - ~150-160 chars, compelling, search-friendly", "answer": "string - 1-3 sentence DIRECT answer/definition of the topic, placed first so " "AI assistants can quote it on its own", "intro": "string - 2-4 sentence overview that builds on the answer", "steps": [ { "heading": "string - short, descriptive H2 step title", "body": "string - 1-3 paragraphs in your own words; specific and verifiable", "quote": "string - a short, near-verbatim snippet (<=120 chars) copied from the " "transcript line this step is based on, used to locate the moment", "t_llm": number, // best timestamp IN SECONDS for an illustrative screenshot "importance": number // 0..1, how worth screenshotting this step is } ], "faqs": [ { "q": "string - a natural question a user might ask an AI assistant", "a": "string - a concise, factual 1-3 sentence answer" } ] } Rules: - 4 to 10 steps. Keep quotes copied from the transcript so they can be matched back. - t_llm must be within the transcript's time range. - Provide 3-5 FAQs phrased as real questions ("Can ...?", "How do I ...?", "What is ...?"). - Be specific and factual (citation-worthy); avoid marketing fluff. Note prerequisites or limitations where relevant. - Output valid JSON only. No markdown, no comments in the actual output. {brief_block} {keyword_block} Transcript (each line is "[mm:ss] text"): --- {transcript} --- """ def _brief_block(brief: str | None) -> str: """Render the user's 'what the content must cover' brief into the prompt.""" brief = (brief or "").strip() if not brief: return ("Content brief: none provided — cover the video's key steps faithfully " "based on the transcript.") return ( "Content brief — REQUIRED COVERAGE. The tutorial MUST address every point below. " "Ground each point in the transcript where possible; if the transcript doesn't " "cover a required point, still cover it accurately and concisely and don't invent " "specifics that contradict the video. Organise the steps so these points are " "clearly covered:\n" + brief ) def _keyword_block(keywords: dict | None) -> str: """Render the SEO/keyword placement guidance injected into the prompt. ``keywords`` = ``{"primary": str, "secondary": [str, ...]}``. """ keywords = keywords or {} primary = (keywords.get("primary") or "").strip() secondary = [s for s in (keywords.get("secondary") or []) if s.strip()] if not primary and not secondary: return "SEO keywords: none specified." lines = ["SEO / keyword requirements (keep everything natural - never keyword-stuff):"] if primary: lines += [ f'- PRIMARY keyword: "{primary}".', f' * Use "{primary}" naturally about 3 times in the body text.', f' * Also place "{primary}" in: the title (H1), the URL slug, the meta ' f'description, within the FIRST 100 words of the intro, and in one or two ' f'H2 step headings where it fits naturally.', ] if secondary: joined = ", ".join(f'"{s}"' for s in secondary) lines.append( f"- SECONDARY keywords ({joined}): use each one EXACTLY once, naturally, " "somewhere in the body." ) return "\n".join(lines) def _slugify(text: str) -> str: text = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-") return text[:80] or "tutorial" def _extract_json(text: str) -> dict: """Parse the model output into a dict, tolerating code fences / stray prose.""" text = text.strip() if text.startswith("```"): text = re.sub(r"^```(?:json)?\s*|\s*```$", "", text, flags=re.DOTALL).strip() try: return json.loads(text) except json.JSONDecodeError: start, end = text.find("{"), text.rfind("}") if start != -1 and end != -1 and end > start: return json.loads(text[start:end + 1]) raise def _normalize(data: dict) -> dict: """Coerce/clean fields so downstream stages never crash on missing keys.""" steps = [] for s in data.get("steps", []) or []: try: t = float(s.get("t_llm", 0) or 0) except (TypeError, ValueError): t = 0.0 try: imp = float(s.get("importance", 0.5) or 0.5) except (TypeError, ValueError): imp = 0.5 steps.append({ "heading": str(s.get("heading", "Step")).strip() or "Step", "body": str(s.get("body", "")).strip(), "quote": str(s.get("quote", "")).strip(), "t_llm": max(0.0, t), "importance": min(1.0, max(0.0, imp)), }) if not steps: raise RuntimeError("The LLM returned no usable steps.") faqs = [] for f in data.get("faqs", []) or []: q = str(f.get("q", "")).strip() a = str(f.get("a", "")).strip() if q and a: faqs.append({"q": q, "a": a}) title = str(data.get("title", "Tutorial")).strip() or "Tutorial" intro = str(data.get("intro", "")).strip() answer = str(data.get("answer", "")).strip() meta = str(data.get("meta_description", "")).strip() or (answer or intro)[:160] slug = _slugify(str(data.get("slug", "")).strip() or title) return { "title": title, "slug": slug, "meta_description": meta, "answer": answer, "intro": intro, "steps": steps, "faqs": faqs, } def count_keyword(tutorial: dict, keyword: str) -> int: """Count case-insensitive whole-keyword occurrences across all post text. Used only for transparency in the UI status — placement isn't hard-enforced. """ keyword = (keyword or "").strip() if not keyword: return 0 parts = [tutorial.get("title", ""), tutorial.get("slug", "").replace("-", " "), tutorial.get("meta_description", ""), tutorial.get("answer", ""), tutorial.get("intro", "")] for s in tutorial.get("steps", []): parts += [s.get("heading", ""), s.get("body", "")] for f in tutorial.get("faqs", []): parts += [f.get("q", ""), f.get("a", "")] blob = "\n".join(parts).lower() return len(re.findall(re.escape(keyword.lower()), blob)) def generate_tutorial(transcript: str, hf_token: str, model: str = DEFAULT_LLM, keywords: dict | None = None, brief: str | None = None) -> dict: """Call the chat model and return a normalized tutorial dict. Returned keys: ``title, slug, meta_description, answer, intro, steps, faqs``. ``keywords`` is ``{"primary": str, "secondary": [str, ...]}`` for SEO placement. ``brief`` is the user's free-text 'what the content must cover' requirement. """ if not hf_token: raise ValueError("An HF token is required for the tutorial LLM (billed to your key).") truncated = transcript[:MAX_TRANSCRIPT_CHARS] if len(transcript) > MAX_TRANSCRIPT_CHARS: truncated += "\n[... transcript truncated for length ...]" prompt = (_INSTRUCTIONS .replace("{brief_block}", _brief_block(brief)) .replace("{keyword_block}", _keyword_block(keywords)) .replace("{transcript}", truncated)) client = InferenceClient(token=hf_token) try: resp = client.chat.completions.create( model=model, messages=[ {"role": "system", "content": _SYSTEM}, {"role": "user", "content": prompt}, ], temperature=0.3, max_tokens=4000, ) except Exception as exc: raise RuntimeError(f"Tutorial LLM call failed ({model}): {exc}") from exc content = resp.choices[0].message.content or "" try: data = _extract_json(content) except Exception as exc: raise RuntimeError( f"Could not parse JSON from the LLM. First 400 chars:\n{content[:400]}" ) from exc return _normalize(data)