File size: 7,129 Bytes
bf73d26
 
 
 
 
 
 
 
f97e0c5
bf73d26
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
f97e0c5
 
 
 
 
 
bf73d26
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
import asyncio
import json
import time
from datetime import datetime, timezone
from pathlib import Path

import httpx
from fastapi import FastAPI, HTTPException, Query
from fastapi.middleware.cors import CORSMiddleware
from fastapi.staticfiles import StaticFiles


APP_DIR = Path(__file__).resolve().parent
SEC_AGENT = "Robinhood open research prototype contact: https://huggingface.co/pumpfun/Robinhood"
SEC_HEADERS = {"User-Agent": SEC_AGENT, "Accept": "application/json"}
TICKERS_URL = "https://www.sec.gov/files/company_tickers.json"
SUBMISSIONS_URL = "https://data.sec.gov/submissions/CIK{cik}.json"
FACTS_URL = "https://data.sec.gov/api/xbrl/companyfacts/CIK{cik}.json"
FORMS = {"10-K", "10-K/A", "10-Q", "10-Q/A", "8-K", "8-K/A", "20-F", "20-F/A", "6-K", "6-K/A"}

METRICS = [
    ("revenue", "Revenue", "营业收入", ["RevenueFromContractWithCustomerExcludingAssessedTax", "Revenues", "SalesRevenueNet"]),
    ("net_income", "Net income", "净利润", ["NetIncomeLoss", "ProfitLoss"]),
    ("assets", "Total assets", "总资产", ["Assets"]),
    ("liabilities", "Total liabilities", "总负债", ["Liabilities"]),
    ("cash", "Cash & equivalents", "现金及等价物", ["CashAndCashEquivalentsAtCarryingValue", "CashCashEquivalentsRestrictedCashAndRestrictedCashEquivalents"]),
]

app = FastAPI(title="Robinhood public company research", docs_url=None, redoc_url=None)
app.add_middleware(
    CORSMiddleware,
    allow_origins=["https://pumpfun-robinhood-web.static.hf.space"],
    allow_methods=["GET"],
    allow_headers=["*"],
)
_ticker_cache = {"expires": 0.0, "rows": []}


def fetch_json(url: str) -> dict:
    try:
        response = httpx.get(url, headers=SEC_HEADERS, timeout=15, follow_redirects=True)
        response.raise_for_status()
        return response.json()
    except httpx.HTTPStatusError as exc:
        raise RuntimeError(f"SEC returned HTTP {exc.response.status_code}") from exc
    except (httpx.HTTPError, json.JSONDecodeError) as exc:
        raise RuntimeError("SEC data is temporarily unavailable") from exc


def ticker_rows() -> list[dict]:
    now = time.time()
    if _ticker_cache["rows"] and _ticker_cache["expires"] > now:
        return _ticker_cache["rows"]
    payload = fetch_json(TICKERS_URL)
    rows = list(payload.values())
    _ticker_cache.update(rows=rows, expires=now + 21600)
    return rows


def resolve_company(query: str) -> dict | None:
    needle = query.strip().upper()
    rows = ticker_rows()
    exact_ticker = next((row for row in rows if row["ticker"].upper() == needle), None)
    if exact_ticker:
        return exact_ticker
    exact_name = next((row for row in rows if row["title"].upper() == needle), None)
    if exact_name:
        return exact_name
    starts = [row for row in rows if row["title"].upper().startswith(needle)]
    if starts:
        return min(starts, key=lambda row: len(row["title"]))
    contains = [row for row in rows if needle in row["title"].upper()]
    return min(contains, key=lambda row: len(row["title"])) if contains else None


def filing_rows(submissions: dict) -> list[dict]:
    recent = submissions.get("filings", {}).get("recent", {})
    keys = ("accessionNumber", "filingDate", "reportDate", "form", "primaryDocument", "primaryDocDescription")
    count = len(recent.get("form", []))
    filings = []
    cik_plain = str(int(submissions["cik"]))
    for index in range(count):
        row = {key: recent.get(key, [""] * count)[index] for key in keys}
        if row["form"] not in FORMS or not row["primaryDocument"]:
            continue
        accession_plain = row["accessionNumber"].replace("-", "")
        filings.append({
            "form": row["form"],
            "filed": row["filingDate"],
            "period": row["reportDate"],
            "description": row["primaryDocDescription"] or row["form"],
            "url": f"https://www.sec.gov/Archives/edgar/data/{cik_plain}/{accession_plain}/{row['primaryDocument']}",
        })
        if len(filings) == 8:
            break
    return filings


def latest_metric(facts: dict, metric: tuple) -> dict | None:
    key, label_en, label_zh, tags = metric
    us_gaap = facts.get("facts", {}).get("us-gaap", {})
    for tag in tags:
        concept = us_gaap.get(tag)
        if not concept:
            continue
        units = concept.get("units", {})
        points = units.get("USD") or next(iter(units.values()), [])
        valid = [point for point in points if point.get("form") in FORMS and point.get("val") is not None and point.get("filed")]
        if not valid:
            continue
        point = max(valid, key=lambda item: (item.get("filed", ""), item.get("end", "")))
        return {
            "key": key,
            "label_en": label_en,
            "label_zh": label_zh,
            "value": point["val"],
            "unit": "USD",
            "period_start": point.get("start"),
            "period": point.get("end"),
            "filed": point.get("filed"),
            "form": point.get("form"),
            "taxonomy_tag": tag,
        }
    return None


def build_research(query: str) -> dict:
    match = resolve_company(query)
    if not match:
        raise HTTPException(status_code=404, detail="Company not found. Try a U.S. ticker such as AAPL or MSFT.")
    cik = str(match["cik_str"]).zfill(10)
    submissions = fetch_json(SUBMISSIONS_URL.format(cik=cik))
    facts = fetch_json(FACTS_URL.format(cik=cik))
    metrics = [result for metric in METRICS if (result := latest_metric(facts, metric))]
    return {
        "company": {
            "name": submissions.get("name") or match["title"],
            "ticker": (submissions.get("tickers") or [match["ticker"]])[0],
            "exchange": (submissions.get("exchanges") or [""])[0],
            "cik": cik,
            "sic": submissions.get("sic"),
            "sic_description": submissions.get("sicDescription"),
            "fiscal_year_end": submissions.get("fiscalYearEnd"),
            "state": submissions.get("stateOfIncorporation"),
        },
        "metrics": metrics,
        "filings": filing_rows(submissions),
        "retrieved_at": datetime.now(timezone.utc).isoformat(),
        "source": "U.S. Securities and Exchange Commission EDGAR",
        "limitations": {
            "en": "Figures reflect the latest matching standardized XBRL facts, whose periods may differ. Coverage and tagging vary by issuer. Verify decisions in the linked filings.",
            "zh": "数字来自最近匹配到的标准化 XBRL 数据,各指标期间可能不同;不同公司的覆盖和标签质量也有差异。请在原始申报中核实。",
        },
    }


@app.get("/api/health")
def health():
    return {"status": "ok"}


@app.get("/api/research")
async def research(q: str = Query(min_length=1, max_length=80)):
    clean = " ".join(q.split())
    try:
        return await asyncio.to_thread(build_research, clean)
    except HTTPException:
        raise
    except RuntimeError as exc:
        raise HTTPException(status_code=502, detail=str(exc)) from exc


app.mount("/", StaticFiles(directory=APP_DIR, html=True), name="site")