BTS / src /streamlit_app.py
Varriety's picture
Update src/streamlit_app.py
b21cc56 verified
Raw History Blame Contribute Delete
69.3 kB
import streamlit as st
import pandas as pd
import numpy as np
import re
import io
import time
import requests
import matplotlib.pyplot as plt
import matplotlib.gridspec as gridspec
import seaborn as sns
from datetime import datetime, timezone
from textblob import TextBlob
from scipy.stats import pearsonr
import nltk
from nltk.corpus import stopwords
from nltk.sentiment.vader import SentimentIntensityAnalyzer
from transformers import pipeline
import os
import streamlit.components.v1 as components
from langdetect import detect, DetectorFactory
DetectorFactory.seed = 0
# ==============================
# SETTING PATH ABSOLUT GAMBAR
# ==============================
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
img_hero = os.path.join(BASE_DIR, "bitcoin1.gif")
img_batch = os.path.join(BASE_DIR, "bitcoin2.gif")
# ==============================
# KONFIGURASI HALAMAN & STATE NAVIGASI
# ==============================
st.set_page_config(
page_title="Bitcoin Volatility Sentiment",
page_icon="β‚Ώ",
layout="wide",
initial_sidebar_state="collapsed"
)
if 'page' not in st.session_state:
st.session_state.page = "uji_kalimat"
# ==============================
# GLOBAL CSS
# ==============================
st.markdown("""
<style>
/* ── Google Fonts ── */
@import url('https://fonts.googleapis.com/css2?family=Inter:wght@400;500;600;700;800&display=swap');
/* ── Reset Streamlit chrome ── */
#MainMenu, footer, header { visibility: hidden; }
.block-container {
padding-top: 1rem !important;
padding-bottom: 0 !important;
max-width: 100% !important;
}
html, body, [class*="css"] {
font-family: 'Inter', -apple-system, BlinkMacSystemFont, "Segoe UI", Roboto, "Helvetica Neue", Arial, sans-serif !important;
color: #202630 !important;
}
.stApp {
background-color: #FAFAFA !important;
}
/* ── Custom Scrollbar ── */
::-webkit-scrollbar { width: 6px; }
::-webkit-scrollbar-track { background: transparent; }
::-webkit-scrollbar-thumb { background: #eaecef; border-radius: 3px; }
::-webkit-scrollbar-thumb:hover { background: #10b981; }
/* ── NAVBAR WRAPPER ── */
.vbc-logo {
font-weight: 700;
font-size: 1.25rem;
color: #0f172a !important;
display: flex;
align-items: center;
gap: 10px;
}
.vbc-logo-icon {
background: #10b981;
color: white;
width: 36px;
height: 36px;
border-radius: 8px;
display: inline-flex;
align-items: center;
justify-content: center;
font-size: 1.1rem;
font-weight: 800;
}
/* ── HERO SECTION (Uji Kalimat) ── */
.hero-wrap {
background: #FAFAFA;
min-height: auto;
display: flex;
align-items: center;
position: relative;
overflow: hidden;
padding: 0.5rem 3rem 2rem 3rem;
}
.hero-badge {
display: inline-block;
background: #e6fff1;
color: #1aa64a;
font-size: 0.75rem;
font-weight: 600;
padding: 6px 12px;
border-radius: 4px;
margin-bottom: 1.5rem;
}
.hero-title {
font-size: 3rem;
font-weight: 800;
line-height: 1.2;
color: #0f172a !important;
margin: 0 0 1rem;
}
.hero-title span {
color: #10b981;
}
.hero-sub {
font-size: 1rem;
color: #64748b !important;
max-width: 500px;
line-height: 1.6;
margin-bottom: 2rem;
}
.hero-card {
background: #FFFFFF;
border: 1px solid #e2e8f0;
border-radius: 50px;
padding: 10px 20px;
display: inline-flex;
align-items: center;
gap: 10px;
margin-bottom: 2rem;
box-shadow: 0 2px 4px rgba(0,0,0,0.02);
}
.hero-card-dot {
width: 8px; height: 8px;
border-radius: 50%;
background: #10b981;
flex-shrink: 0;
}
.hero-card p {
margin: 0;
font-size: 0.85rem;
color: #334155 !important;
font-weight: 600;
}
/* ── BATCH SECTION ── */
.batch-wrap {
background: #FAFAFA;
min-height: auto;
padding: 0.5rem 3rem 2rem 3rem;
}
.batch-eyebrow {
font-size: 0.85rem;
font-weight: 700;
color: #10b981 !important;
margin-bottom: 0.5rem;
}
.batch-title {
font-size: 2.5rem;
font-weight: 800;
color: #0f172a !important;
line-height: 1.2;
margin-bottom: 1rem;
}
.batch-sub {
font-size: 1rem;
color: #64748b !important;
max-width: 480px;
line-height: 1.6;
margin-bottom: 2rem;
}
/* ── RESULT / DASHBOARD SECTION ── */
.result-wrap {
background: #FFFFFF;
padding: 2rem 3rem 3rem 3rem;
border: 1px solid #e2e8f0;
border-radius: 16px;
box-shadow: 0 4px 6px rgba(0,0,0,0.01);
margin-bottom: 3rem;
}
.section-label {
font-size: 0.85rem;
font-weight: 700;
color: #10b981 !important;
margin-bottom: 0.5rem;
}
.section-title {
font-size: 1.5rem;
font-weight: 800;
color: #0f172a !important;
margin-bottom: 1.5rem;
}
/* ── METRIC CARDS ── */
div[data-testid="stMetric"] {
background: #FFFFFF;
border: 1px solid #e2e8f0;
border-radius: 12px;
padding: 1rem 1.2rem !important;
box-shadow: 0 2px 4px rgba(0,0,0,0.02);
}
div[data-testid="stMetricLabel"] > div {
color: #64748b !important;
font-size: 0.85rem !important;
font-weight: 600 !important;
}
div[data-testid="stMetricValue"] > div {
color: #0f172a !important;
font-weight: 800 !important;
font-size: 1.8rem !important;
}
/* ── BUTTONS ── */
div[data-testid="stButton"] > button {
font-weight: 600 !important;
font-size: 0.9rem !important;
border-radius: 50px !important;
padding: 0.5rem 1.2rem !important;
height: 42px !important;
transition: all 0.2s ease-in-out !important;
}
div[data-testid="stButton"] > button:focus:not(:active) {
box-shadow: none !important;
}
/* Primary CTA */
.btn-primary div[data-testid="stButton"] > button {
background: #10b981 !important;
color: #FFFFFF !important;
border: none !important;
}
.btn-primary div[data-testid="stButton"] > button:hover {
background: #059669 !important;
}
/* Secondary outline */
.btn-outline-white div[data-testid="stButton"] > button {
background: #FFFFFF !important;
color: #0f172a !important;
border: 1px solid #e2e8f0 !important;
}
.btn-outline-white div[data-testid="stButton"] > button:hover {
border-color: #10b981 !important;
color: #10b981 !important;
}
/* Active nav */
.btn-orange div[data-testid="stButton"] > button {
background: #e6fff1 !important;
color: #10b981 !important;
border: none !important;
}
/* Ghost nav β€” inactive */
.btn-ghost div[data-testid="stButton"] > button {
background: transparent !important;
color: #64748b !important;
border: 1px solid transparent !important;
}
.btn-ghost div[data-testid="stButton"] > button:hover {
color: #0f172a !important;
background: #f1f5f9 !important;
}
/* ── TEXT INPUT / TEXTAREA ── */
.stTextArea textarea {
background-color: #FFFFFF !important;
color: #0f172a !important;
border: 1px solid #e2e8f0 !important;
border-radius: 12px !important;
font-size: 0.95rem !important;
padding: 0.8rem 1rem !important;
transition: border-color 0.2s !important;
}
.stTextArea textarea:focus {
border-color: #10b981 !important;
box-shadow: 0 0 0 1px #10b981 !important;
}
.stTextArea label {
color: #334155 !important;
font-size: 0.85rem !important;
font-weight: 600 !important;
}
/* ── DATA TABLE ── */
div[data-testid="stDataFrame"] {
border: 1px solid #e2e8f0 !important;
border-radius: 12px !important;
}
/* ── FILE UPLOADER ── */
div[data-testid="stFileUploader"] {
border: 1px dashed #cbd5e1 !important;
border-radius: 12px !important;
background: #FFFFFF !important;
padding: 1.5rem !important;
}
div[data-testid="stFileUploader"]:hover {
border-color: #10b981 !important;
}
/* ── EXPANDER ── */
div[data-testid="stExpander"] {
border: 1px solid #e2e8f0 !important;
border-radius: 12px !important;
background: #FFFFFF !important;
}
/* ── DOWNLOAD BUTTON ── */
div[data-testid="stDownloadButton"] > button {
background: #FFFFFF !important;
color: #0f172a !important;
border-radius: 50px !important;
font-weight: 600 !important;
border: 1px solid #e2e8f0 !important;
padding: 0.5rem 1.2rem !important;
transition: all 0.2s !important;
}
div[data-testid="stDownloadButton"] > button:hover {
border-color: #10b981 !important;
color: #10b981 !important;
}
/* ── DIVIDER ── */
.vbc-divider {
border: none;
border-top: 1px solid #e2e8f0;
margin: 2rem 0;
}
/* ── LAG TABLE HIGHLIGHT ── */
.lag-best {
background: #e6fff1 !important;
font-weight: 700 !important;
color: #1aa64a !important;
}
</style>
""", unsafe_allow_html=True)
# ==============================
# FUNGSI AUTO-SCROLL
# ==============================
def scroll_to_target(target_id):
js_code = f"""
<script>
var target = window.parent.document.getElementById('{target_id}');
if(target) {{
target.scrollIntoView({{behavior: 'smooth', block: 'start'}});
}}
</script>
"""
components.html(js_code, height=0, width=0)
# ==============================
# HEADER / NAVBAR
# ==============================
def set_page(page_name):
st.session_state.page = page_name
col_logo, col_space, col_btn1, col_btn2 = st.columns([5, 3, 2, 2], vertical_alignment="center")
with col_logo:
st.markdown("""
<div class="vbc-logo" style="padding-left: 2rem;">
<span class="vbc-logo-icon">β‚Ώ</span>
Bitcoin Volatility Sentiment
</div>
""", unsafe_allow_html=True)
with col_btn1:
is_uji = st.session_state.page == "uji_kalimat"
css_class = "btn-orange" if is_uji else "btn-ghost"
st.markdown(f'<div class="{css_class}">', unsafe_allow_html=True)
if st.button("Uji Kalimat", use_container_width=True, key="nav_uji"):
set_page("uji_kalimat"); st.rerun()
st.markdown('</div>', unsafe_allow_html=True)
with col_btn2:
is_batch = st.session_state.page == "analisis_batch"
css_class = "btn-orange" if is_batch else "btn-ghost"
st.markdown(f'<div class="{css_class}">', unsafe_allow_html=True)
if st.button("Analisis Batch", use_container_width=True, key="nav_batch"):
set_page("analisis_batch"); st.rerun()
st.markdown('</div>', unsafe_allow_html=True)
st.markdown("<hr style='margin-top: 0.5rem; margin-bottom: 0.5rem; border: none; border-bottom: 1px solid #e2e8f0;'>", unsafe_allow_html=True)
# ==============================
# DOWNLOAD RESOURCES & LOAD MODELS
# ==============================
@st.cache_resource
def download_nltk_resources():
nltk.download('stopwords', quiet=True)
nltk.download('vader_lexicon', quiet=True)
nltk.download('punkt', quiet=True)
nltk.download('omw-1.4', quiet=True)
download_nltk_resources()
stop_words = set(stopwords.words('english'))
@st.cache_resource
def load_all_models():
vader = SentimentIntensityAnalyzer()
bertweet = pipeline("sentiment-analysis", model="finiteautomata/bertweet-base-sentiment-analysis", device=-1, truncation=True, max_length=128)
roberta = pipeline("sentiment-analysis", model="cardiffnlp/twitter-roberta-base-sentiment", device=-1, truncation=True, max_length=512)
roberta_large = pipeline("sentiment-analysis", model="siebert/sentiment-roberta-large-english", device=-1, truncation=True, max_length=512)
return vader, bertweet, roberta, roberta_large
with st.spinner('Mempersiapkan model AI...'):
vader, bertweet, roberta, roberta_large = load_all_models()
# ==============================================================================
# FUNGSI CLEAN TEXT
# ==============================================================================
def clean_text(text):
text = str(text)
# Hapus prefix metadata Twitter/X
text = re.sub(
r'^.*?Β·\s*\d+\s*(?:dtk|mnt|jam|s|h|sec|min)\s*(?:Membalas\s+@\w+\s*)?',
'',
text,
flags=re.IGNORECASE
)
# Hapus "Tampilkan lebih banyak" (artefak UI Twitter)
text = re.sub(r'Tampilkan lebih banyak.*$', '', text, flags=re.IGNORECASE)
# Hapus angka trailing engagement (like/retweet count)
text = re.sub(r'(\s+\d+)+\s*$', '', text).strip()
# Cleaning standar
text = text.lower()
text = re.sub(r"http\S+", "", text) # hapus URL
text = re.sub(r"@\w+", "", text) # hapus @mention
text = re.sub(r"#\w+", "", text) # hapus #hashtag
text = re.sub(r"[^\w\s]", "", text) # hapus tanda baca
text = re.sub(r"\b\d+\b", "", text) # hapus angka sisa
text = re.sub(r"\s+", " ", text).strip()
# Hapus stopwords
tokens = text.split()
tokens = [word for word in tokens if word not in stop_words]
return " ".join(tokens)
# ==============================================================================
# THRESHOLD TEXTBLOB
# ==============================================================================
TEXTBLOB_THRESHOLD = 0.10
def classify_tb(score):
if score > TEXTBLOB_THRESHOLD: return 'positive'
if score < -TEXTBLOB_THRESHOLD: return 'negative'
return 'neutral'
def map_roberta(label):
return {"LABEL_0": "negative", "LABEL_1": "neutral", "LABEL_2": "positive"}.get(label, "neutral")
def map_bertweet(label):
return {"pos": "positive", "neu": "neutral", "neg": "negative"}.get(label.lower(), "neutral")
def get_daily_label(score):
if score > TEXTBLOB_THRESHOLD: return 'Positive'
elif score < -TEXTBLOB_THRESHOLD: return 'Negative'
else: return 'Neutral'
# ==============================================================================
# HALAMAN 1 β€” UJI KALIMAT
# ==============================================================================
if st.session_state.page == "uji_kalimat":
st.markdown('<div class="hero-wrap">', unsafe_allow_html=True)
col_text, col_img = st.columns([1.1, 1], gap="large")
with col_text:
st.markdown("""
<div class="hero-badge">Website ini bukanlah alat prediksi harga Bitcoin real time, melainkan instrumen untuk melakukan analisis sentimen publik secara batch</div>
<h1 class="hero-title">
Bitcoin Volatility<br>
<span>vs Public</span> Sentiment
</h1>
<p class="hero-sub">
Analisis Volatilitas Harga Bitcoin Terhadap Sentimen Publik
Pada Platform X Berbasis Python.
</p>
<div class="hero-card">
<div class="hero-card-dot"></div>
<p><b>Peneliti:</b> Arya Galuh Saputra &nbsp;Β·&nbsp; H1D022022</p>
</div>
""", unsafe_allow_html=True)
user_input = st.text_area(
"Masukkan Tweet (Bahasa Inggris):",
"Great, Bitcoin just fly another 10% today.",
height=120
)
st.markdown("<br>", unsafe_allow_html=True)
col_btn1, col_btn2 = st.columns([1.6, 1])
with col_btn1:
st.markdown('<div class="btn-primary">', unsafe_allow_html=True)
analyze_btn = st.button("Proses Uji Kalimat", use_container_width=True)
st.markdown('</div>', unsafe_allow_html=True)
with col_img:
st.markdown("<div style='margin-top: 4rem;'></div>", unsafe_allow_html=True)
try:
st.image(img_hero, use_container_width=True)
except Exception:
st.markdown("""
<div style="background:#f5f5f5;border:1px dashed #cbd5e1;
border-radius:12px;height:320px;display:flex;align-items:center;
justify-content:center;color:#64748b;font-size:0.9rem;
text-align:center;padding:2rem;">
πŸ–ΌοΈ Gambar Tidak Ditemukan<br>Pastikan file <code>bitcoin1.gif</code> ada di direktori
</div>""", unsafe_allow_html=True)
st.markdown('</div>', unsafe_allow_html=True)
st.markdown('<div id="target-uji-kalimat"></div>', unsafe_allow_html=True)
if analyze_btn:
scroll_to_target("target-uji-kalimat")
col_space_left, col_center_output, col_space_right = st.columns([1, 4, 1])
with col_center_output:
st.markdown("""
<div class="result-wrap" style="padding-bottom: 2rem; margin-bottom: 1.5rem;">
<p class="section-label">Output Analisis</p>
<p class="section-title" style="margin-bottom: 0;">Hasil Deteksi Sentimen</p>
</div>
""", unsafe_allow_html=True)
try:
if detect(user_input) != 'en':
st.warning("⚠️ Teks sepertinya bukan bahasa Inggris. Hasil prediksi mungkin memiliki bias.")
except:
pass
text = clean_text(user_input)
with st.spinner("Mengekstraksi sentimen dengan 5 Model..."):
time.sleep(0.5)
try:
v_compound = vader.polarity_scores(text)['compound']
v_label = "positive" if v_compound > 0.05 else ("negative" if v_compound < -0.05 else "neutral")
except:
v_compound = 0.0
v_label = "neutral"
try:
t_label = classify_tb(TextBlob(text).sentiment.polarity)
except:
t_label = "neutral"
try:
b_label = map_bertweet(bertweet(text)[0]['label'])
except:
b_label = "neutral"
try:
r_label = map_roberta(roberta(text)[0]['label'])
except:
r_label = "neutral"
try:
rl_label = roberta_large(text)[0]['label'].lower()
except:
rl_label = "neutral"
def badge_color(label):
return {"positive": "#e6fff1", "negative": "#fef1f2", "neutral": "#f1f5f9"}[label]
def badge_text_color(label):
return {"positive": "#10b981", "negative": "#f43f5e", "neutral": "#64748b"}[label]
results = [
("VADER", v_label, f"compound: {v_compound:.4f}"),
("TextBlob", t_label, f"threshold: Β±{TEXTBLOB_THRESHOLD}"),
("BERTweet", b_label, ""),
("RoBERTa Base", r_label, ""),
("RoBERTa Large", rl_label, ""),
]
col_a, col_b = st.columns(2)
for i, (method, label, detail) in enumerate(results):
col = col_a if i % 2 == 0 else col_b
bg = badge_color(label)
tc = badge_text_color(label)
icon = "β†—" if label == "positive" else ("β†˜" if label == "negative" else "β†’")
detail_html = (
'<div style="font-size:0.7rem;color:#94a3b8;margin-top:2px;">'
+ detail +
'</div>'
) if detail else ''
if label == 'positive':
border_color = '#10b981'
elif label == 'negative':
border_color = '#f43f5e'
else:
border_color = '#cbd5e1'
html_card = (
'<div style="background:#FAFAFA;border:1px solid #e2e8f0;'
'border-left:4px solid ' + border_color + ';'
'border-radius:12px;padding:1rem 1.2rem;margin-bottom:1rem;'
'display:flex;align-items:center;justify-content:space-between;'
'box-shadow:0 2px 4px rgba(0,0,0,0.02);">'
'<div>'
'<div style="font-weight:600;font-size:0.75rem;color:#64748b;margin-bottom:4px;">'
+ method +
'</div>'
'<div style="font-weight:800;font-size:1.05rem;color:#0f172a;">'
+ label.capitalize() +
'</div>'
+ detail_html +
'</div>'
'<div style="background:' + bg + ';color:' + tc + ';'
'font-size:0.75rem;font-weight:700;'
'padding:6px 12px;border-radius:50px;">'
+ icon + ' ' + label.upper() +
'</div>'
'</div>'
)
with col:
st.markdown(html_card, unsafe_allow_html=True)
with st.expander("πŸ” Lihat teks setelah preprocessing"):
st.code(text if text.strip() else "(kosong setelah dibersihkan)", language=None)
# ==============================================================================
# HALAMAN 2 β€” ANALISIS BATCH
# ==============================================================================
elif st.session_state.page == "analisis_batch":
plt.style.use('default')
sns.set_theme(style="whitegrid", rc={
"axes.facecolor": "#FFFFFF",
"figure.facecolor": "#FAFAFA",
"axes.edgecolor": "#e2e8f0",
"text.color": "#0f172a",
"xtick.color": "#64748b",
"ytick.color": "#64748b",
"grid.color": "#f1f5f9",
})
st.markdown('<div class="batch-wrap">', unsafe_allow_html=True)
col_upload, col_img_b = st.columns([1.4, 1], gap="large")
with col_upload:
st.markdown("""
<p class="batch-eyebrow">Analisis Batch Processing</p>
<h2 class="batch-title">Volatilitas Harga Bitcoin Vs Sentimen Publik<br>Kolerasi Multi-Metode Analisis Sentimen</h2>
<p class="batch-sub">
Unggah file tweets (.txt) untuk diekstraksi dan
dianalisis terhadap volatilitas harga Bitcoin.
</p>""", unsafe_allow_html=True)
tweet_files = st.file_uploader(
"Pilih file Tweet (.txt)",
type=['txt'],
accept_multiple_files=True
)
with st.expander("Format TXT yang Didukung"):
st.code(
"username | 2024-03-01 14:00:00\n"
"Isi tweet baris pertama di sini\n\n"
"username2 | 2024-03-01 15:30:00\n"
"Isi tweet baris kedua di sini",
language="text"
)
st.markdown("<br>", unsafe_allow_html=True)
st.markdown('<div class="btn-primary">', unsafe_allow_html=True)
analyze_batch_btn = st.button("Eksekusi Analisis", key="batch_btn", use_container_width=False)
st.markdown('</div>', unsafe_allow_html=True)
with col_img_b:
st.markdown("<div style='margin-top: 4rem;'></div>", unsafe_allow_html=True)
try:
st.image(img_batch, use_container_width=True)
except Exception:
st.markdown("""
<div style="background:#f5f5f5;border:1px dashed #cbd5e1;
border-radius:12px;height:280px;display:flex;align-items:center;
justify-content:center;color:#64748b;font-size:0.9rem;
text-align:center;padding:2rem;">
πŸ–ΌοΈ Gambar Tidak Ditemukan<br>Pastikan file <code>bitcoin2.gif</code> ada di direktori
</div>""", unsafe_allow_html=True)
st.markdown('</div>', unsafe_allow_html=True)
# ==============================================================================
# SECTION TUTORIAL PENGUMPULAN DATA TWEET
# ==============================================================================
st.markdown("""
<div style="background:#FFFFFF;border:1px solid #e2e8f0;border-radius:16px;
padding:2rem 2.5rem;margin:0 0 2rem 0;box-shadow:0 2px 4px rgba(0,0,0,0.02);">
<div style="display:flex;align-items:center;gap:10px;margin-bottom:0.5rem;">
<span style="background:#e6fff1;color:#10b981;font-size:0.75rem;font-weight:700;
padding:4px 10px;border-radius:4px;">Panduan</span>
</div>
<h3 style="font-size:1.25rem;font-weight:800;color:#0f172a;margin:0 0 0.4rem;">
πŸ“₯ Cara Mengumpulkan Data Tweet
</h3>
<p style="font-size:0.9rem;color:#64748b;margin:0;">
Sebelum mengunggah file, kumpulkan data tweet dari platform X menggunakan
skrip <code>collect.js</code> yang dijalankan langsung di browser. Ikuti langkah-langkah berikut.
</p>
</div>
""", unsafe_allow_html=True)
# ── Tab Tutorial ──────────────────────────────────────────────────────────────
tab_langkah, tab_script, tab_format, tab_tips = st.tabs([
"πŸ“‹ Langkah-Langkah",
"πŸ’» Skrip collect.js",
"πŸ“„ Format File .txt",
"πŸ’‘ Tips & Catatan"
])
with tab_langkah:
st.markdown("""
<div style="padding:0.5rem 0;">
""", unsafe_allow_html=True)
langkah_data = [
("1", "#10b981", "Login ke Platform X",
"Buka <strong>x.com</strong> di browser Chrome/Edge/Firefox. Pastikan sudah login ke akun X Anda. Gunakan akun aktif agar tidak kena pembatasan akses.",
"🌐"),
("2", "#10b981", "Cari Keyword 'Bitcoin'",
"Di kolom pencarian X, ketik <strong>Bitcoin</strong> lalu tekan Enter. Pilih tab <strong>Latest</strong> (Terbaru) β€” bukan Top β€” agar hasil terurut cronologis dan lebih representatif untuk analisis harian.",
"πŸ”"),
("3", "#10b981", "Filter Tanggal (Opsional)",
"Untuk scraping per hari tertentu, gunakan filter pencarian lanjutan X: <strong>until:YYYY-MM-DD since:YYYY-MM-DD</strong>. Contoh: <code>Bitcoin since:2026-04-16 until:2026-04-17</code>. Ini memastikan data per file sesuai satu hari.",
"πŸ“…"),
("4", "#10b981", "Buka Developer Tools",
"Tekan <strong>F12</strong> (atau klik kanan β†’ Inspect) untuk membuka DevTools browser. Pilih tab <strong>Console</strong>. Pastikan tidak ada peringatan keamanan β€” beberapa browser meminta konfirmasi teks sebelum menjalankan skrip.",
"πŸ› οΈ"),
("5", "#10b981", "Jalankan Skrip collect.js",
"Copy seluruh isi skrip <code>collect.js</code> dari tab <strong>Skrip collect.js</strong> di atas, paste ke kolom Console, lalu tekan <strong>Enter</strong>. Skrip akan mulai men-scrape tweet yang tampil di halaman.",
"▢️"),
("6", "#10b981", "Scroll Halaman untuk Load Lebih Banyak Tweet",
"Setelah skrip aktif, <strong>scroll ke bawah</strong> perlahan pada halaman X untuk me-load lebih banyak tweet. Skrip akan otomatis mendeteksi tweet baru yang muncul. Targetkan minimal 200–300 tweet per hari untuk hasil analisis yang valid.",
"⬇️"),
("7", "#10b981", "Download File .txt",
"Ketik perintah <code>downloadTweets()</code> di Console lalu tekan Enter. File .txt akan otomatis terunduh. Rename file sesuai urutan hari: <strong>1.txt</strong> untuk hari pertama, <strong>2.txt</strong> untuk hari kedua, dst.",
"πŸ’Ύ"),
("8", "#10b981", "Ulangi untuk Setiap Hari",
"Ulangi langkah 2–7 untuk setiap hari yang ingin dianalisis. Pastikan minimal <strong>30 hari</strong> data agar korelasi Pearson memiliki kekuatan statistik yang cukup (r β‰₯ 0.35 pada n=30).",
"πŸ”"),
("9", "#10b981", "Upload Semua File ke Website",
"Setelah semua file siap (1.txt, 2.txt, ..., 30.txt), upload sekaligus ke kolom unggah di bawah ini, lalu klik <strong>Eksekusi Analisis</strong>.",
"πŸš€"),
]
for num, color, title, desc, icon in langkah_data:
st.markdown(f"""
<div style="display:flex;gap:14px;align-items:flex-start;
background:#FAFAFA;border:1px solid #e2e8f0;border-radius:12px;
padding:1rem 1.25rem;margin-bottom:10px;">
<div style="min-width:36px;height:36px;background:{color};color:#fff;
border-radius:50%;display:flex;align-items:center;justify-content:center;
font-weight:800;font-size:0.9rem;flex-shrink:0;">{num}</div>
<div>
<div style="font-weight:700;font-size:0.95rem;color:#0f172a;margin-bottom:4px;">
{icon} {title}
</div>
<div style="font-size:0.85rem;color:#475569;line-height:1.6;">{desc}</div>
</div>
</div>
""", unsafe_allow_html=True)
st.markdown("</div>", unsafe_allow_html=True)
with tab_script:
st.markdown("""
<div style="background:#f1f5f9;border-radius:8px;padding:0.75rem 1rem;
margin-bottom:1rem;font-size:0.82rem;color:#475569;line-height:1.6;">
<strong>πŸ“Œ Cara pakai:</strong> Copy seluruh skrip di bawah β†’ Paste di Console browser (F12)
saat berada di halaman pencarian X β†’ Tekan Enter β†’ Scroll halaman untuk load tweet β†’
Ketik <code>downloadTweets()</code> β†’ File .txt terunduh otomatis.
</div>
""", unsafe_allow_html=True)
collect_js = '''// ============================================================
// collect.js β€” X Tweet Scraper
// Jalankan di Console browser saat berada di halaman pencarian X
// Keyword yang digunakan: "Bitcoin" (tab: Latest)
// ============================================================
(function() {
// Menyimpan semua tweet yang sudah dikumpulkan (Set mencegah duplikat)
window._collectedTweets = window._collectedTweets || new Set();
window._tweetList = window._tweetList || [];
// Fungsi utama: scrape semua tweet yang saat ini tampil di halaman
function scrapeTweets() {
// Selector untuk artikel tweet di X.com
const tweetArticles = document.querySelectorAll('article[data-testid="tweet"]');
let newCount = 0;
tweetArticles.forEach(article => {
try {
// ── Ambil username (handle @...) ──────────────────────────
const userEl = article.querySelector('[data-testid="User-Name"]');
const username = userEl
? userEl.innerText.replace(/\n/g, ' ').trim()
: 'unknown';
// ── Ambil timestamp ───────────────────────────────────────
const timeEl = article.querySelector('time');
const datetime = timeEl
? timeEl.getAttribute('datetime') // format ISO: 2026-04-16T14:30:00.000Z
: new Date().toISOString();
// ── Ambil teks tweet ──────────────────────────────────────
const textEl = article.querySelector('[data-testid="tweetText"]');
const tweetText = textEl
? textEl.innerText.trim()
: '';
// Skip tweet kosong
if (!tweetText) return;
// Buat unique key untuk mencegah duplikat
const key = username + '|' + datetime + '|' + tweetText.substring(0, 50);
if (!window._collectedTweets.has(key)) {
window._collectedTweets.add(key);
// Format sesuai yang diharapkan source code Python:
// "username | datetime"
// "isi tweet"
const dateFormatted = datetime.replace('T', ' ').replace(/\\.\\d+Z$/, '').replace('Z', '');
window._tweetList.push({
meta: username + ' | ' + dateFormatted,
content: tweetText
});
newCount++;
}
} catch (e) {
// Skip tweet yang gagal diproses
}
});
console.log(`[collect.js] +${newCount} tweet baru | Total: ${window._tweetList.length}`);
}
// ── Auto-scrape setiap 2 detik saat halaman di-scroll ────────────────
if (window._scrapeInterval) {
clearInterval(window._scrapeInterval);
}
window._scrapeInterval = setInterval(scrapeTweets, 2000);
// Jalankan sekali langsung saat skrip diload
scrapeTweets();
// ── Fungsi download β€” ketik downloadTweets() di Console ──────────────
window.downloadTweets = function(filename) {
if (window._tweetList.length === 0) {
console.warn('[collect.js] Belum ada tweet terkumpul. Scroll halaman lebih banyak dulu.');
return;
}
// Susun konten file: setiap tweet dipisah baris kosong
const lines = window._tweetList.map(t => t.meta + '\\n' + t.content);
const content = lines.join('\\n\\n');
// Tentukan nama file otomatis berdasarkan tanggal tweet pertama
if (!filename) {
const firstDate = window._tweetList[0].meta.split(' | ')[1];
const dateStr = firstDate ? firstDate.split(' ')[0].replace(/-/g, '') : 'tweets';
filename = dateStr + '_bitcoin.txt';
}
// Buat blob dan trigger download
const blob = new Blob([content], { type: 'text/plain;charset=utf-8' });
const url = URL.createObjectURL(blob);
const a = document.createElement('a');
a.href = url;
a.download = filename;
a.click();
URL.revokeObjectURL(url);
console.log(`[collect.js] βœ… Download: ${filename} (${window._tweetList.length} tweet)`);
};
// ── Fungsi reset β€” ketik resetTweets() untuk mulai hari baru ─────────
window.resetTweets = function() {
window._collectedTweets = new Set();
window._tweetList = [];
console.log('[collect.js] πŸ”„ Data direset. Siap scraping hari baru.');
};
// ── Fungsi status ─────────────────────────────────────────────────────
window.statusTweets = function() {
console.log(`[collect.js] πŸ“Š Total tweet: ${window._tweetList.length}`);
if (window._tweetList.length > 0) {
console.log(' Pertama:', window._tweetList[0].meta);
console.log(' Terakhir:', window._tweetList[window._tweetList.length - 1].meta);
}
};
console.log('[collect.js] βœ… Skrip aktif!');
console.log(' β†’ Scroll halaman X untuk load lebih banyak tweet');
console.log(' β†’ Ketik downloadTweets() untuk download file .txt');
console.log(' β†’ Ketik resetTweets() untuk mulai hari baru');
console.log(' β†’ Ketik statusTweets() untuk cek jumlah tweet');
})();'''
st.code(collect_js, language="javascript")
st.markdown("""
<div style="background:#fffbeb;border:1px solid #fbbf24;border-radius:8px;
padding:0.75rem 1rem;font-size:0.82rem;color:#92400e;margin-top:0.5rem;">
<strong>⚠️ Perhatian:</strong> Beberapa browser (terutama Chrome) menampilkan peringatan
saat paste skrip ke Console. Jika diminta, ketik <code>allow pasting</code> lalu tekan Enter,
kemudian paste ulang skripnya.
</div>
""", unsafe_allow_html=True)
with tab_format:
st.markdown("""
<div style="font-size:0.9rem;color:#475569;line-height:1.7;margin-bottom:1rem;">
File <code>.txt</code> yang diunggah harus mengikuti format berikut agar dapat dibaca
oleh sistem. Setiap tweet terdiri dari <strong>2 baris</strong> dan dipisahkan
oleh <strong>satu baris kosong</strong>.
</div>
""", unsafe_allow_html=True)
col_f1, col_f2 = st.columns(2)
with col_f1:
st.markdown("**βœ… Format yang benar:**")
st.code(
"username | 2026-04-16 14:30:00\n"
"Bitcoin is looking bullish today! Great news for crypto holders.\n\n"
"another_user | 2026-04-16 15:45:00\n"
"BTC just hit 75k, incredible run. When moon?\n\n"
"crypto_analyst | 2026-04-16 16:20:00\n"
"Bearish divergence on BTC 4H chart. Be careful traders.",
language="text"
)
with col_f2:
st.markdown("**❌ Format yang salah:**")
st.code(
"# Jangan ada header CSV\n"
"date,user,tweet\n\n"
"# Jangan ada spasi ganda antar tweet\n\n\n"
"user | 2026-04-16\n"
"tweet...\n\n\n"
"# Tanggal harus ada jamnya\n"
"user | 2026-04-16\n"
"tweet tanpa jam...",
language="text"
)
st.markdown("""
<div style="background:#f8fafc;border:1px solid #e2e8f0;border-radius:10px;
padding:1rem 1.25rem;margin-top:1rem;">
<p style="font-weight:700;font-size:0.9rem;color:#0f172a;margin:0 0 8px;">
πŸ“ Konvensi Penamaan File
</p>
<div style="font-size:0.85rem;color:#475569;line-height:1.8;">
β€’ Satu file = satu hari data<br>
β€’ Nama file: <strong>1.txt</strong> (hari ke-1), <strong>2.txt</strong> (hari ke-2), dst.<br>
β€’ Sistem mengurutkan file secara alfanumerik sebelum diproses<br>
β€’ Tidak ada batasan jumlah tweet per file<br>
β€’ Encoding: <strong>UTF-8</strong> (default output collect.js)
</div>
</div>
""", unsafe_allow_html=True)
with tab_tips:
tips_data = [
("🎯", "Target Minimal Data",
"Gunakan minimal <strong>30 hari</strong> data untuk hasil korelasi yang bermakna secara statistik. Dengan n=30, nilai r β‰₯ 0.35 sudah signifikan pada p < 0.05. Makin banyak hari, makin kuat reliabilitas temuan."),
("πŸ“Š", "Jumlah Tweet per Hari",
"Targetkan <strong>200–500 tweet per hari</strong>. Terlalu sedikit (< 50 tweet) membuat rata-rata sentimen harian tidak representatif. Scroll halaman X selama 1–2 menit per hari untuk mendapatkan jumlah yang cukup."),
("πŸ”", "Kata Kunci yang Tepat",
"Penelitian ini menggunakan kata kunci tunggal <strong>\"Bitcoin\"</strong> (tanpa tanda petik di X). Pastikan memilih tab <strong>Latest</strong>, bukan Top, agar distribusi temporal merata dan tidak bias ke tweet viral."),
("πŸ“…", "Konsistensi Periode Waktu",
"Gunakan filter tanggal X untuk memastikan setiap file hanya berisi tweet dari <strong>satu hari kalender</strong>. Contoh: <code>Bitcoin since:2026-04-16 until:2026-04-17</code>. Ini penting agar agregasi harian akurat."),
("🌐", "Bahasa Tweet",
"Sistem otomatis memfilter tweet non-Inggris menggunakan <code>langdetect</code>. Dari pengalaman penelitian ini, sekitar <strong>13–14% tweet diskip</strong> karena bukan bahasa Inggris. Ini normal dan sudah diperhitungkan."),
("πŸ’Ύ", "Reset Antar Hari",
"Setelah download file untuk satu hari, selalu ketik <code>resetTweets()</code> di Console sebelum pindah ke hari berikutnya. Ini mencegah tweet dari hari sebelumnya ikut masuk ke file hari berikutnya."),
("⚑", "Performa Browser",
"Tutup tab lain yang tidak diperlukan saat scraping untuk mencegah browser melambat. Jika halaman X berhenti load tweet setelah scroll panjang, refresh halaman dan jalankan ulang collect.js (data sebelumnya akan hilang)."),
("πŸ”’", "Batas Scraping X",
"Platform X membatasi scraping agresif. Jika halaman tiba-tiba tidak menampilkan tweet baru meski di-scroll, tunggu 5–10 menit sebelum melanjutkan. Alternatif: gunakan akun berbeda atau ganti IP."),
]
col_t1, col_t2 = st.columns(2)
for i, (icon, title, desc) in enumerate(tips_data):
col = col_t1 if i % 2 == 0 else col_t2
with col:
st.markdown(f"""
<div style="background:#FAFAFA;border:1px solid #e2e8f0;border-radius:10px;
padding:0.9rem 1.1rem;margin-bottom:10px;">
<div style="font-weight:700;font-size:0.88rem;color:#0f172a;margin-bottom:4px;">
{icon} {title}
</div>
<div style="font-size:0.82rem;color:#64748b;line-height:1.6;">{desc}</div>
</div>
""", unsafe_allow_html=True)
st.markdown("<hr style='border:none;border-top:1px solid #e2e8f0;margin:1.5rem 0 2rem;'>",
unsafe_allow_html=True)
st.markdown('<div id="target-analisis-batch"></div>', unsafe_allow_html=True)
if tweet_files and analyze_batch_btn:
scroll_to_target("target-analisis-batch")
col_b_space1, col_b_content, col_b_space2 = st.columns([1, 8, 1])
with col_b_content:
st.markdown("""
<div class="result-wrap" style="padding-bottom: 2rem; margin-bottom: 1.5rem;">
<p class="section-label">Hasil Pemrosesan</p>
<p class="section-title" style="margin-bottom: 0;">Dashboard Analisis</p>
</div>
""", unsafe_allow_html=True)
tweet_files = sorted(tweet_files, key=lambda x: x.name)
data = []
with st.status("πŸ”„ Memproses data sentimen...", expanded=True) as status:
progress_bar = st.progress(0, text="Mengekstrak sentimen dari data...")
total_tweets_uploaded = 0
total_tweets_skipped = 0
for idx, file in enumerate(tweet_files):
content = file.getvalue().decode("utf-8").replace("\r\n", "\n").strip()
tweets = content.split("\n\n")
for tweet in tweets:
parts = tweet.strip().split("\n", 1)
if len(parts) != 2: continue
meta, text_raw = parts
try:
DetectorFactory.seed = 0
lang = detect(text_raw)
if lang != 'en':
total_tweets_skipped += 1
continue
except:
total_tweets_skipped += 1
continue
username, date_val = meta.split(" | ") if " | " in meta else ("unknown", "unknown")
short_date = date_val[:10]
text = clean_text(text_raw)
if not text.strip():
total_tweets_skipped += 1
continue
try:
v_compound = vader.polarity_scores(text)['compound']
vader_label = "positive" if v_compound > 0.05 else ("negative" if v_compound < -0.05 else "neutral")
except:
v_compound = 0.0
vader_label = "neutral"
try:
tb_polarity = TextBlob(text).sentiment.polarity
tb_label = classify_tb(tb_polarity)
except:
tb_polarity = 0.0
tb_label = "neutral"
try:
bertweet_label = map_bertweet(bertweet(text)[0]['label'])
except:
bertweet_label = "neutral"
try:
roberta_label = map_roberta(roberta(text)[0]['label'])
except:
roberta_label = "neutral"
try:
roberta_large_label = roberta_large(text)[0]['label'].lower()
except:
roberta_large_label = "neutral"
data.append({
"date": short_date,
"raw_tweet": text_raw.strip(),
"cleaned_tweet": text,
"vader": vader_label,
"textblob": tb_label,
"bertweet": bertweet_label,
"roberta": roberta_label,
"roberta_large": roberta_large_label,
"vader_score": v_compound,
"tb_score": tb_polarity,
})
total_tweets_uploaded += 1
progress_bar.progress((idx + 1) / len(tweet_files),
text=f"Memproses file {idx+1} dari {len(tweet_files)}")
status.update(label="βœ… Pemrosesan sentimen teks selesai!", state="complete", expanded=False)
df = pd.DataFrame(data)
if df.empty:
st.error("❌ Data kosong. Pastikan format TXT benar dan tweet berbahasa Inggris.")
else:
col_m1, col_m2, col_m3 = st.columns(3)
col_m1.metric("Tweet Diproses", f"{total_tweets_uploaded}", border=True)
col_m2.metric("Tweet Diabaikan (Non-EN)", f"{total_tweets_skipped}", border=True)
col_m3.metric("Model", "5 Model", border=True)
target_dates = sorted(df['date'].unique())
start_unix = int(datetime.strptime(target_dates[0], "%Y-%m-%d").replace(tzinfo=timezone.utc).timestamp()) - 86400
end_unix = int(datetime.strptime(target_dates[-1], "%Y-%m-%d").replace(tzinfo=timezone.utc).timestamp()) + 86400
# ==============================================================
# FUNGSI FETCH HARGA BTC
# ==============================================================
def fetch_via_coingecko(start_ts, end_ts):
"""Coba CoinGecko dengan retry eksponensial. Return list [[ts_ms, price]] atau raise."""
url = "https://api.coingecko.com/api/v3/coins/bitcoin/market_chart/range"
params = {"vs_currency": "usd", "from": start_ts, "to": end_ts}
headers = {"accept": "application/json", "User-Agent": "Mozilla/5.0"}
wait_times = [5, 15, 30]
for attempt, wait in enumerate(wait_times, 1):
time.sleep(wait)
res = requests.get(url, params=params, headers=headers, timeout=20)
if res.status_code == 200:
data_j = res.json()
if "prices" in data_j:
return data_j["prices"]
raise ValueError("Key 'prices' tidak ada di respons CoinGecko.")
if res.status_code == 429:
if attempt < len(wait_times):
continue # coba lagi dengan backoff lebih lama
raise ConnectionError(f"CoinGecko 429 setelah {attempt} percobaan.")
raise ConnectionError(f"CoinGecko error {res.status_code}: {res.text[:200]}")
raise ConnectionError("CoinGecko gagal setelah semua percobaan.")
def fetch_via_binance(start_ts, end_ts):
"""
Fallback: Binance Public API β€” BTCUSDT daily klines.
Endpoint bebas API key, limit 1000 candles per request.
"""
url = "https://api.binance.com/api/v3/klines"
prices = []
cur_start = start_ts * 1000
end_ms = end_ts * 1000
while cur_start < end_ms:
params = {
"symbol": "BTCUSDT",
"interval": "1d",
"startTime": cur_start,
"endTime": end_ms,
"limit": 1000,
}
res = requests.get(url, params=params, timeout=20)
if res.status_code != 200:
raise ConnectionError(f"Binance error {res.status_code}: {res.text[:200]}")
batch = res.json()
if not batch:
break
for candle in batch:
open_time = int(candle[0]) # ms
close_price = float(candle[4]) # close price
prices.append([open_time, close_price])
cur_start = int(batch[-1][0]) + 1
if len(batch) < 1000:
break
if not prices:
raise ValueError("Binance tidak mengembalikan data.")
return prices
def build_df_price(raw_prices, target_date_list):
"""Bersihkan raw [[ts_ms, price]] β†’ df_price dengan log_return."""
df_p = pd.DataFrame(raw_prices, columns=["timestamp", "price"])
df_p["date"] = pd.to_datetime(df_p["timestamp"], unit="ms").dt.date
df_p = df_p.groupby("date")["price"].mean().reset_index()
df_p["pct_change"] = df_p["price"].pct_change() * 100
df_p["log_return"] = np.log(df_p["price"] / df_p["price"].shift(1))
df_p.dropna(inplace=True)
df_p = df_p[df_p["date"].isin(pd.to_datetime(target_date_list).date)]
return df_p
raw_prices = None
api_source = None
with st.spinner("πŸ“‘ Mengambil data harga Bitcoin..."):
try:
raw_prices = fetch_via_coingecko(start_unix, end_unix)
api_source = "CoinGecko"
except Exception as cg_err:
st.warning(
f"⚠️ CoinGecko tidak tersedia ({cg_err}). "
"Beralih ke **Binance Public API** sebagai fallback..."
)
try:
raw_prices = fetch_via_binance(start_unix, end_unix)
api_source = "Binance"
except Exception as bn_err:
st.error(
f"❌ Kedua sumber data gagal.\n"
f"- CoinGecko: {cg_err}\n"
f"- Binance: {bn_err}\n\n"
"Coba lagi beberapa menit kemudian atau periksa koneksi internet."
)
if raw_prices is not None:
try:
df_price = build_df_price(raw_prices, target_dates)
st.info(f"βœ… Data harga BTC berhasil diambil dari **{api_source}**.")
if df_price.empty:
st.warning("⚠️ Data Harga API kosong. Pastikan rentang tanggal di .txt sesuai (yyyy-mm-dd).")
else:
st.markdown("<hr class='vbc-divider'>", unsafe_allow_html=True)
# ── Tabel Data Sentimen ──────────────────────────────
st.markdown("πŸ—£οΈ Data Sentimen")
raw_display_cols = ["date","raw_tweet","vader","textblob","bertweet","roberta","roberta_large","vader_score","tb_score"]
st.dataframe(df[raw_display_cols], use_container_width=True, hide_index=True)
# ==================================================
# AGREGASI HARIAN DUAL-MODE
# Mode A (Kategorik): konversi {pos:1, neu:0, neg:-1}
# Mode B (Numerik): rata-rata vader_score & tb_score
# ==================================================
sentiment_map = {"positive": 1, "neutral": 0, "negative": -1}
df_score = df.copy()
models_cat = ["vader","textblob","bertweet","roberta","roberta_large"]
for col in models_cat:
df_score[col] = df_score[col].map(sentiment_map)
# Agregasi kategorik (βˆ’1/0/1 mean)
df_sentiment_daily = df_score.groupby("date")[models_cat].mean().reset_index()
df_sentiment_daily["date"] = pd.to_datetime(df_sentiment_daily["date"]).dt.date
# Agregasi numerik VADER compound & TextBlob polarity
df_numeric_daily = df.groupby("date")[["vader_score","tb_score"]].mean().reset_index()
df_numeric_daily["date"] = pd.to_datetime(df_numeric_daily["date"]).dt.date
for col in models_cat:
df_sentiment_daily[f"{col}_label"] = df_sentiment_daily[col].apply(get_daily_label)
daily_display_cols = ["date"]
for col in models_cat:
daily_display_cols.extend([col, f"{col}_label"])
# ── Tabel Harga Bitcoin ───────────────────────────
st.markdown("β‚Ώ Data Harga & Volatilitas Bitcoin")
st.dataframe(df_price[["date","price","pct_change","log_return"]], use_container_width=True, hide_index=True)
# Merge data
df_merged = pd.merge(df_price, df_sentiment_daily, on="date", how="inner")
df_merged = pd.merge(df_merged, df_numeric_daily, on="date", how="inner")
# ── Tabel Data Final ──────────────────────────────
st.markdown("πŸ—‚οΈ Data Final")
final_display_cols = (
["date","price","pct_change","log_return"]
+ [c for c in daily_display_cols if c != "date"]
+ ["vader_score","tb_score"]
)
st.dataframe(df_merged[final_display_cols], use_container_width=True, hide_index=True)
# Download buttons
col_dl1, col_dl2, _ = st.columns([1, 1, 3])
csv_data = df_merged.to_csv(index=False).encode('utf-8')
col_dl1.download_button("πŸ“₯ Unduh CSV", data=csv_data, file_name="bitcoin_volatility_sentiment.csv", mime="text/csv", use_container_width=True)
buffer = io.BytesIO()
with pd.ExcelWriter(buffer, engine='xlsxwriter') as writer:
df_merged.to_excel(writer, index=False)
col_dl2.download_button("πŸ“₯ Unduh Excel", data=buffer.getvalue(), file_name="bitcoin_volatility_sentiment.xlsx", mime="application/vnd.ms-excel", use_container_width=True)
st.markdown("<hr class='vbc-divider'>", unsafe_allow_html=True)
# ==================================================
# UJI KORELASI PEARSON DENGAN LAG
# Lag 0 : sentimen hari t vs harga hari t (same-day)
# Lag +1: sentimen hari t vs harga hari t+1
# Lag +2: sentimen hari t vs harga hari t+2
#
# Kolom korelasi yang diuji:
# - vader_score (numerik compound, βˆ’1 s.d. 1)
# - tb_score (numerik polarity, βˆ’1 s.d. 1)
# - bertweet (kategorik βˆ’1/0/1 mean)
# - roberta (kategorik βˆ’1/0/1 mean)
# - roberta_large(kategorik βˆ’1/0/1 mean)
# ==================================================
st.subheader("πŸ”¬ Uji Korelasi Pearson")
st.caption(
"Menganalisis hubungan statistik antara skor sentimen harian dan "
"volatilitas log-return BTC pada lag 0 (hari sama), +1 hari, dan +2 hari. "
"VADER & TextBlob menggunakan skor numerik kontinu; model BERT menggunakan "
"rata-rata kategorikal (βˆ’1/0/1)."
)
corr_columns = {
"VADER (numerik)": "vader_score",
"TextBlob (numerik)": "tb_score",
"BERTweet (kategorik)": "bertweet",
"RoBERTa Base (kategorik)": "roberta",
"RoBERTa Large (kategorik)":"roberta_large",
}
corr_rows = []
raw_results = []
n_obs = len(df_merged)
for label_name, col_key in corr_columns.items():
for lag in [0, 1, 2]:
# shift(-lag): harga mundur lag hari ke depan
# artinya: sentimen t berkorelasi dengan harga t+lag
log_ret_shifted = df_merged["log_return"].shift(-lag)
valid_mask = log_ret_shifted.notna()
x_vals = df_merged.loc[valid_mask, col_key]
y_vals = log_ret_shifted[valid_mask]
if len(x_vals) < 4:
corr, pval = np.nan, np.nan
else:
try:
corr, pval = pearsonr(x_vals, y_vals)
except Exception:
corr, pval = np.nan, np.nan
arah = "Positif" if (corr is not np.nan and corr > 0) else "Negatif"
sig = "βœ… Signifikan" if (pval is not np.nan and pval < 0.05) else "Tidak Signifikan"
corr_rows.append({
"Metode": label_name,
"Lag": f"t+{lag}",
"r": f"{corr:.4f}" if not np.isnan(corr) else "N/A",
"Arah": arah,
"p-value": f"{pval:.4f}" if not np.isnan(pval) else "N/A",
"Signifikansi": sig,
})
raw_results.append({
"metode": label_name,
"col": col_key,
"lag": lag,
"r": corr if not np.isnan(corr) else 0.0,
"p": pval if not np.isnan(pval) else 1.0,
})
df_corr_table = pd.DataFrame(corr_rows)
# Tandai baris terbaik per metode (|r| terbesar)
def highlight_best(row):
try:
r_abs = abs(float(row["r"]))
except:
r_abs = 0
sig_ok = "βœ…" in str(row["Signifikansi"])
if sig_ok:
return ["background-color:#e6fff1; font-weight:600"] * len(row)
return [""] * len(row)
styled = df_corr_table.style.apply(highlight_best, axis=1)
st.dataframe(styled, use_container_width=True, hide_index=True)
# ── Scatter Plot ──────────────────────────────────
st.subheader("πŸ”΅ Pola Distribusi Scatter Plot (Lag t+0)")
scatter_cols = list(corr_columns.values())
cols_sc = st.columns(3)
for idx2, (disp_name, col_key) in enumerate(corr_columns.items()):
with cols_sc[idx2 % 3]:
fig_s, ax_s = plt.subplots(figsize=(5, 4))
sns.regplot(
data=df_merged, x=col_key, y="log_return", ax=ax_s,
scatter_kws={"s": 40, "color": "#10b981", "alpha": 0.5},
line_kws={"color": "#0f172a", "linewidth": 2}
)
# Tampilkan r dan p pada plot
try:
r_val, p_val = pearsonr(df_merged[col_key], df_merged["log_return"])
ax_s.set_title(f"{disp_name.split(' ')[0]}\nr={r_val:.3f}, p={p_val:.3f}", fontweight='bold', fontsize=9)
except:
ax_s.set_title(disp_name.split(' ')[0], fontweight='bold')
ax_s.set_xlabel("Sentimen Score")
ax_s.set_ylabel("Log Return")
plt.tight_layout()
st.pyplot(fig_s)
# ── Line Chart ────────────────────────────────────
st.subheader("πŸ“ˆ Trend Analisis: Sentiment vs BTC Volatility")
fig_line, ax_line = plt.subplots(figsize=(14, 6))
ax_line.plot(
df_merged["date"], df_merged["log_return"],
label="BTC Log Return", color="#f7931a", linewidth=3
)
colors_line = ["#3B82F6","#10B981","#EC4899","#14B8A6","#6366F1"]
for i, (disp_name, col_key) in enumerate(corr_columns.items()):
ax_line.plot(
df_merged["date"], df_merged[col_key],
label=f"Sentimen: {disp_name.split(' ')[0]}",
color=colors_line[i], linewidth=1.5, linestyle="--", alpha=0.8
)
ax_line.set_title("Pergerakan Sentimen vs Log Return Bitcoin", fontsize=14, pad=15, fontweight='bold')
ax_line.set_xlabel("Tanggal", fontsize=11)
ax_line.set_ylabel("Nilai Metrik", fontsize=11)
ax_line.legend(loc='upper left', bbox_to_anchor=(1, 1), frameon=True)
plt.tight_layout()
st.pyplot(fig_line)
# ── KESIMPULAN ────────────────────────────────────
st.markdown("<hr class='vbc-divider'>", unsafe_allow_html=True)
st.subheader("πŸ“ Kesimpulan")
max_idx = df_merged["log_return"].idxmax()
min_idx = df_merged["log_return"].idxmin()
date_max = df_merged.loc[max_idx, "date"]
date_min = df_merged.loc[min_idx, "date"]
st.write(
f"Puncak lonjakan positif (*max log return*) terjadi pada **{date_max}**, "
f"sedangkan penurunan ekstrem terjadi pada **{date_min}**. "
f"Dataset mencakup **{n_obs} hari** pengamatan."
)
# Cari lag & metode terbaik (|r| terbesar + signifikan)
sig_results = [r for r in raw_results if r["p"] < 0.05]
all_results = raw_results
if sig_results:
best = max(sig_results, key=lambda x: abs(x["r"]))
arah_text = "berbanding lurus (positif)" if best["r"] > 0 else "berbanding terbalik (negatif)"
sig_summary = ", ".join(
set(f"{r['metode'].split(' ')[0]} (lag t+{r['lag']})" for r in sig_results)
)
st.success(f"""
**Hipotesis Diterima (H1):** Ditemukan korelasi linier yang signifikan (*p-value* < 0.05) pada: **{sig_summary}**.
Metode & lag dengan korelasi terkuat adalah **{best['metode']} (lag t+{best['lag']})** dengan r = **{best['r']:.4f}**, sifat hubungan **{arah_text}**.
""")
else:
best_overall = max(all_results, key=lambda x: abs(x["r"]))
st.warning(f"""
**Hipotesis Ditolak (H0 Diterima):** Belum ditemukan bukti empiris korelasi linier yang signifikan pada seluruh metode dan lag yang diuji (*p-value* β‰₯ 0.05).
Korelasi terbesar ditemukan pada **{best_overall['metode']} (lag t+{best_overall['lag']})** dengan r = **{best_overall['r']:.4f}**. Volatilitas harga kemungkinan dipengaruhi faktor teknikal/fundamental di luar sentimen X, atau jumlah observasi belum mencukupi (n = {n_obs}, butuh minimal 30 hari agar r β‰₯ 0.35 bisa signifikan).
""")
except Exception as e:
st.error(f"⚠️ Terjadi kesalahan saat memproses data harga Bitcoin: {e}")
elif analyze_batch_btn and not tweet_files:
st.warning("⚠️ Silakan unggah minimal satu file .txt terlebih dahulu.")