File size: 9,567 Bytes
01abd01
f6a6455
 
01abd01
 
 
 
 
 
 
f6a6455
 
 
 
 
 
 
 
 
 
 
 
 
 
 
01abd01
 
 
 
f6a6455
 
 
01abd01
f6a6455
 
01abd01
 
 
 
 
 
f6a6455
 
01abd01
 
f6a6455
 
 
 
 
01abd01
 
 
 
f6a6455
 
 
 
 
 
01abd01
 
 
f6a6455
 
21ba408
 
01fb241
 
f6a6455
 
 
 
 
 
 
21ba408
 
 
 
 
 
 
 
 
 
 
 
 
 
 
01abd01
 
 
 
 
 
 
 
 
01fb241
01abd01
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
01fb241
 
 
01abd01
 
 
 
 
f6a6455
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
01abd01
 
 
 
 
 
 
 
 
 
 
 
 
f6a6455
01abd01
 
 
 
 
f6a6455
01abd01
f6a6455
 
 
01abd01
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
01fb241
21ba408
01abd01
01fb241
01abd01
 
21ba408
01fb241
f6a6455
 
 
 
 
 
01fb241
21ba408
01fb241
 
f6a6455
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
"""Stage 7: transform the timestamped transcript into a structured, AEO-friendly tutorial.

Uses an HF Inference Providers chat model (default DeepSeek-V3) billed to the user's
token. The model returns strict JSON so the downstream weighted indicator and the .docx
builder have stable fields to work with.

The output follows answer-engine-optimization (AEO) / AI-citation best practices: an
answer-first paragraph near the top, descriptive title + meta description + URL slug,
clear H2 step headings, and an FAQ section. SEO keyword placement is explicit (see
``_keyword_block``).
"""
from __future__ import annotations

import json
import re

from huggingface_hub import InferenceClient

DEFAULT_LLM = "deepseek-ai/DeepSeek-V3"

# Rough char budget to stay clear of context limits on the free path. Long transcripts
# are truncated (with a marker); good enough for a tutorial summary.
MAX_TRANSCRIPT_CHARS = 24000

_SYSTEM = (
    "You are a technical writer who optimizes for answer engines (ChatGPT, Claude, "
    "Gemini, Perplexity, Google AI). You convert a timestamped video transcript into a "
    "clear, citable, step-by-step written tutorial. You ALWAYS respond with a single JSON "
    "object and no prose outside it."
)

_INSTRUCTIONS = """\
Turn the transcript below into a tutorial blog post. Return ONLY a JSON object with this schema:

{
  "title": "string - H1 / title tag; concise and descriptive",
  "slug": "string - lowercase, hyphenated URL slug",
  "meta_description": "string - ~150-160 chars, compelling, search-friendly",
  "answer": "string - 1-3 sentence DIRECT answer/definition of the topic, placed first so "
            "AI assistants can quote it on its own",
  "intro": "string - 2-4 sentence overview that builds on the answer",
  "steps": [
    {
      "heading": "string - short, descriptive H2 step title",
      "body": "string - 1-3 paragraphs in your own words; specific and verifiable",
      "quote": "string - a short, near-verbatim snippet (<=120 chars) copied from the "
               "transcript line this step is based on, used to locate the moment",
      "t_llm": number,        // best timestamp IN SECONDS for an illustrative screenshot
      "importance": number    // 0..1, how worth screenshotting this step is
    }
  ],
  "faqs": [
    { "q": "string - a natural question a user might ask an AI assistant",
      "a": "string - a concise, factual 1-3 sentence answer" }
  ]
}

Rules:
- 4 to 10 steps. Keep quotes copied from the transcript so they can be matched back.
- t_llm must be within the transcript's time range.
- Provide 3-5 FAQs phrased as real questions ("Can ...?", "How do I ...?", "What is ...?").
- Be specific and factual (citation-worthy); avoid marketing fluff. Note prerequisites or
  limitations where relevant.
- Output valid JSON only. No markdown, no comments in the actual output.

{brief_block}

{keyword_block}

Transcript (each line is "[mm:ss] text"):
---
{transcript}
---
"""


def _brief_block(brief: str | None) -> str:
    """Render the user's 'what the content must cover' brief into the prompt."""
    brief = (brief or "").strip()
    if not brief:
        return ("Content brief: none provided — cover the video's key steps faithfully "
                "based on the transcript.")
    return (
        "Content brief — REQUIRED COVERAGE. The tutorial MUST address every point below. "
        "Ground each point in the transcript where possible; if the transcript doesn't "
        "cover a required point, still cover it accurately and concisely and don't invent "
        "specifics that contradict the video. Organise the steps so these points are "
        "clearly covered:\n" + brief
    )


def _keyword_block(keywords: dict | None) -> str:
    """Render the SEO/keyword placement guidance injected into the prompt.

    ``keywords`` = ``{"primary": str, "secondary": [str, ...]}``.
    """
    keywords = keywords or {}
    primary = (keywords.get("primary") or "").strip()
    secondary = [s for s in (keywords.get("secondary") or []) if s.strip()]
    if not primary and not secondary:
        return "SEO keywords: none specified."

    lines = ["SEO / keyword requirements (keep everything natural - never keyword-stuff):"]
    if primary:
        lines += [
            f'- PRIMARY keyword: "{primary}".',
            f'  * Use "{primary}" naturally about 3 times in the body text.',
            f'  * Also place "{primary}" in: the title (H1), the URL slug, the meta '
            f'description, within the FIRST 100 words of the intro, and in one or two '
            f'H2 step headings where it fits naturally.',
        ]
    if secondary:
        joined = ", ".join(f'"{s}"' for s in secondary)
        lines.append(
            f"- SECONDARY keywords ({joined}): use each one EXACTLY once, naturally, "
            "somewhere in the body."
        )
    return "\n".join(lines)


def _slugify(text: str) -> str:
    text = re.sub(r"[^a-z0-9]+", "-", text.lower()).strip("-")
    return text[:80] or "tutorial"


def _extract_json(text: str) -> dict:
    """Parse the model output into a dict, tolerating code fences / stray prose."""
    text = text.strip()
    if text.startswith("```"):
        text = re.sub(r"^```(?:json)?\s*|\s*```$", "", text, flags=re.DOTALL).strip()
    try:
        return json.loads(text)
    except json.JSONDecodeError:
        start, end = text.find("{"), text.rfind("}")
        if start != -1 and end != -1 and end > start:
            return json.loads(text[start:end + 1])
        raise


def _normalize(data: dict) -> dict:
    """Coerce/clean fields so downstream stages never crash on missing keys."""
    steps = []
    for s in data.get("steps", []) or []:
        try:
            t = float(s.get("t_llm", 0) or 0)
        except (TypeError, ValueError):
            t = 0.0
        try:
            imp = float(s.get("importance", 0.5) or 0.5)
        except (TypeError, ValueError):
            imp = 0.5
        steps.append({
            "heading": str(s.get("heading", "Step")).strip() or "Step",
            "body": str(s.get("body", "")).strip(),
            "quote": str(s.get("quote", "")).strip(),
            "t_llm": max(0.0, t),
            "importance": min(1.0, max(0.0, imp)),
        })
    if not steps:
        raise RuntimeError("The LLM returned no usable steps.")

    faqs = []
    for f in data.get("faqs", []) or []:
        q = str(f.get("q", "")).strip()
        a = str(f.get("a", "")).strip()
        if q and a:
            faqs.append({"q": q, "a": a})

    title = str(data.get("title", "Tutorial")).strip() or "Tutorial"
    intro = str(data.get("intro", "")).strip()
    answer = str(data.get("answer", "")).strip()
    meta = str(data.get("meta_description", "")).strip() or (answer or intro)[:160]
    slug = _slugify(str(data.get("slug", "")).strip() or title)
    return {
        "title": title,
        "slug": slug,
        "meta_description": meta,
        "answer": answer,
        "intro": intro,
        "steps": steps,
        "faqs": faqs,
    }


def count_keyword(tutorial: dict, keyword: str) -> int:
    """Count case-insensitive whole-keyword occurrences across all post text.

    Used only for transparency in the UI status — placement isn't hard-enforced.
    """
    keyword = (keyword or "").strip()
    if not keyword:
        return 0
    parts = [tutorial.get("title", ""), tutorial.get("slug", "").replace("-", " "),
             tutorial.get("meta_description", ""), tutorial.get("answer", ""),
             tutorial.get("intro", "")]
    for s in tutorial.get("steps", []):
        parts += [s.get("heading", ""), s.get("body", "")]
    for f in tutorial.get("faqs", []):
        parts += [f.get("q", ""), f.get("a", "")]
    blob = "\n".join(parts).lower()
    return len(re.findall(re.escape(keyword.lower()), blob))


def generate_tutorial(transcript: str, hf_token: str, model: str = DEFAULT_LLM,
                      keywords: dict | None = None, brief: str | None = None) -> dict:
    """Call the chat model and return a normalized tutorial dict.

    Returned keys: ``title, slug, meta_description, answer, intro, steps, faqs``.
    ``keywords`` is ``{"primary": str, "secondary": [str, ...]}`` for SEO placement.
    ``brief`` is the user's free-text 'what the content must cover' requirement.
    """
    if not hf_token:
        raise ValueError("An HF token is required for the tutorial LLM (billed to your key).")

    truncated = transcript[:MAX_TRANSCRIPT_CHARS]
    if len(transcript) > MAX_TRANSCRIPT_CHARS:
        truncated += "\n[... transcript truncated for length ...]"
    prompt = (_INSTRUCTIONS
              .replace("{brief_block}", _brief_block(brief))
              .replace("{keyword_block}", _keyword_block(keywords))
              .replace("{transcript}", truncated))

    client = InferenceClient(token=hf_token)
    try:
        resp = client.chat.completions.create(
            model=model,
            messages=[
                {"role": "system", "content": _SYSTEM},
                {"role": "user", "content": prompt},
            ],
            temperature=0.3,
            max_tokens=4000,
        )
    except Exception as exc:
        raise RuntimeError(f"Tutorial LLM call failed ({model}): {exc}") from exc

    content = resp.choices[0].message.content or ""
    try:
        data = _extract_json(content)
    except Exception as exc:
        raise RuntimeError(
            f"Could not parse JSON from the LLM. First 400 chars:\n{content[:400]}"
        ) from exc
    return _normalize(data)