File size: 6,562 Bytes
48a842f
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
"""Batch-label tweets as engagement farming with a strong LLM teacher.

Usage:
    LABEL_API_KEY=... python label.py tweets.json --model kimi-k3

Any OpenAI-compatible endpoint works; --base-url overrides it
(default: OpenAI's API). Batches 20 tweets per request with
--workers batches in parallel. Writes labeled.json after every batch
(resumable: rerun skips labeled ids and retries null labels). Row shape:
{id, author, text, label, confidence}. train.py applies the confidence
cutoff (--min-confidence, default 0.85) at training time.
"""
import argparse
import json
import os
import sys
import threading
import time
from concurrent.futures import ThreadPoolExecutor

from openai import OpenAI

INSTRUCTIONS = """You label tweets for an anti-engagement-farming product.

Engagement-farming tweets exist to drive likes, replies, retweets, shares, tags
or clicks rather than to convey genuine information, storytelling, opinion or
value. This includes:
1. Explicit calls to action: "Like if you agree!", "Retweet to bless someone!"
2. Implicit prompts with no substance: open-ended questions solely for replies
3. Giveaways or incentives: "RT and I'll follow back!", "Reply to win a prize!"
4. Emotional hooks with no new content: "This made me cry... thoughts?"
5. Low-effort filler or greetings: "Good morning everyone!"
6. Clickbait teasers: "You won't believe what happened next..."
7. Tag-a-friend / chain posts / polls purely for engagement
8. Any other tactic where the primary goal is metric boosting

You receive a JSON array of {"i": index, "text": tweet}. For each element, judge
whether the tweet's PRIMARY PURPOSE is engagement farming and how confident you
are. Respond with ONLY a JSON object:
{"results": [{"i": 0, "label": true, "confidence": 0.9}, ...]}
label is true for engagement farming, false for genuine, confidence in [0, 1].
Every input index must appear exactly once.

Examples:
"Like this if you love pizza" -> label true, confidence 0.99
"Which do you prefer, summer or winter? Vote in the replies!" -> label true, confidence 0.9
"I just rescued a kitten from the shelter and she's settling in." -> label false, confidence 0.95
"Our Q2 earnings report is now available: link." -> label false, confidence 0.9"""


def parse_json(text):
    try:
        return json.loads(text)
    except json.JSONDecodeError:
        start, end = text.find("{"), text.rfind("}")
        if start >= 0 and end > start:
            return json.loads(text[start : end + 1])
        raise


def coerce_label(raw):
    if isinstance(raw, bool):
        return raw
    if isinstance(raw, str):
        return {"true": True, "false": False}.get(raw.strip().lower())
    return None


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("tweets", nargs="?", default="tweets.json")
    ap.add_argument("--out", default="labeled.json")
    ap.add_argument("--base-url", default=os.environ.get("LABEL_BASE_URL", "https://api.openai.com/v1"))
    ap.add_argument("--model", default=os.environ.get("LABEL_MODEL", "kimi-k3"))
    ap.add_argument("--batch-size", type=int, default=20)
    ap.add_argument("--workers", type=int, default=6)
    args = ap.parse_args()

    key = os.environ.get("LABEL_API_KEY") or os.environ.get("OPENAI_API_KEY")
    if not key:
        raise SystemExit("set LABEL_API_KEY (or OPENAI_API_KEY)")
    client = OpenAI(base_url=args.base_url, api_key=key)
    with open(args.tweets) as f:
        tweets = json.load(f)
    done = {}
    if os.path.exists(args.out):
        with open(args.out) as f:
            done = {r["id"]: r for r in json.load(f)}
    todo = [
        t
        for t in tweets
        if t.get("id") and t.get("text") and (t["id"] not in done or done[t["id"]].get("label") is None)
    ]
    print(f"{len(done)} already labeled, {len(todo)} to go", file=sys.stderr)

    batches = [todo[i : i + args.batch_size] for i in range(0, len(todo), args.batch_size)]
    lock = threading.Lock()

    def label_batch(batch):
        for attempt in (1, 2):
            try:
                resp = client.chat.completions.create(
                    model=args.model,
                    messages=[
                        {"role": "system", "content": INSTRUCTIONS},
                        {
                            "role": "user",
                            "content": json.dumps(
                                [{"i": i, "text": t["text"]} for i, t in enumerate(batch)]
                            ),
                        },
                    ],
                )
                data = parse_json(resp.choices[0].message.content or "")
                return (
                    data.get("results", [])
                    if isinstance(data, dict)
                    else data
                    if isinstance(data, list)
                    else []
                )
            except Exception as e:
                if attempt == 2:
                    print(f"batch failed, skipping (rerun to retry): {e}", file=sys.stderr)
                    return []
                time.sleep(3)
        return []

    def handle(pair):
        index, batch = pair
        results = label_batch(batch)
        with lock:
            for r in results:
                i = r.get("i")
                if not isinstance(i, int) or not 0 <= i < len(batch):
                    continue
                t = batch[i]
                conf = min(max(float(r.get("confidence", 0)), 0.0), 1.0)
                done[t["id"]] = {
                    "id": t["id"],
                    "author": t["author"],
                    "text": t["text"],
                    "src": t.get("src"),
                    "label": coerce_label(r.get("label")),
                    "confidence": conf,
                }
            with open(args.out, "w") as f:
                json.dump(list(done.values()), f, indent=1)
            kept = sum(1 for r in done.values() if r["label"] is not None)
            print(
                f"[{min((index + 1) * args.batch_size, len(todo))}/{len(todo)}] {kept} usable",
                file=sys.stderr,
            )

    with ThreadPoolExecutor(max_workers=args.workers) as pool:
        list(pool.map(handle, enumerate(batches)))

    farming = sum(1 for r in done.values() if r["label"] is True)
    genuine = sum(1 for r in done.values() if r["label"] is False)
    print(
        f"done: {len(done)} labeled, {farming} farming, {genuine} genuine, "
        f"{len(done) - farming - genuine} unlabeled",
        file=sys.stderr,
    )


if __name__ == "__main__":
    main()