import streamlit as st
import pandas as pd
import numpy as np
import re
import io
import time
import requests
import matplotlib.pyplot as plt
import matplotlib.gridspec as gridspec
import seaborn as sns
from datetime import datetime, timezone
from textblob import TextBlob
from scipy.stats import pearsonr
import nltk
from nltk.corpus import stopwords
from nltk.sentiment.vader import SentimentIntensityAnalyzer
from transformers import pipeline
import os
import streamlit.components.v1 as components
from langdetect import detect, DetectorFactory
DetectorFactory.seed = 0
# ==============================
# SETTING PATH ABSOLUT GAMBAR
# ==============================
BASE_DIR = os.path.dirname(os.path.abspath(__file__))
img_hero = os.path.join(BASE_DIR, "bitcoin1.gif")
img_batch = os.path.join(BASE_DIR, "bitcoin2.gif")
# ==============================
# KONFIGURASI HALAMAN & STATE NAVIGASI
# ==============================
st.set_page_config(
page_title="Bitcoin Volatility Sentiment",
page_icon="₿",
layout="wide",
initial_sidebar_state="collapsed"
)
if 'page' not in st.session_state:
st.session_state.page = "uji_kalimat"
# ==============================
# GLOBAL CSS
# ==============================
st.markdown("""
""", unsafe_allow_html=True)
# ==============================
# FUNGSI AUTO-SCROLL
# ==============================
def scroll_to_target(target_id):
js_code = f"""
"""
components.html(js_code, height=0, width=0)
# ==============================
# HEADER / NAVBAR
# ==============================
def set_page(page_name):
st.session_state.page = page_name
col_logo, col_space, col_btn1, col_btn2 = st.columns([5, 3, 2, 2], vertical_alignment="center")
with col_logo:
st.markdown("""
₿
Bitcoin Volatility Sentiment
""", unsafe_allow_html=True)
with col_btn1:
is_uji = st.session_state.page == "uji_kalimat"
css_class = "btn-orange" if is_uji else "btn-ghost"
st.markdown(f'', unsafe_allow_html=True)
if st.button("Uji Kalimat", use_container_width=True, key="nav_uji"):
set_page("uji_kalimat"); st.rerun()
st.markdown('
', unsafe_allow_html=True)
with col_btn2:
is_batch = st.session_state.page == "analisis_batch"
css_class = "btn-orange" if is_batch else "btn-ghost"
st.markdown(f'', unsafe_allow_html=True)
if st.button("Analisis Batch", use_container_width=True, key="nav_batch"):
set_page("analisis_batch"); st.rerun()
st.markdown('
', unsafe_allow_html=True)
st.markdown("
", unsafe_allow_html=True)
# ==============================
# DOWNLOAD RESOURCES & LOAD MODELS
# ==============================
@st.cache_resource
def download_nltk_resources():
nltk.download('stopwords', quiet=True)
nltk.download('vader_lexicon', quiet=True)
nltk.download('punkt', quiet=True)
nltk.download('omw-1.4', quiet=True)
download_nltk_resources()
stop_words = set(stopwords.words('english'))
@st.cache_resource
def load_all_models():
vader = SentimentIntensityAnalyzer()
bertweet = pipeline("sentiment-analysis", model="finiteautomata/bertweet-base-sentiment-analysis", device=-1, truncation=True, max_length=128)
roberta = pipeline("sentiment-analysis", model="cardiffnlp/twitter-roberta-base-sentiment", device=-1, truncation=True, max_length=512)
roberta_large = pipeline("sentiment-analysis", model="siebert/sentiment-roberta-large-english", device=-1, truncation=True, max_length=512)
return vader, bertweet, roberta, roberta_large
with st.spinner('Mempersiapkan model AI...'):
vader, bertweet, roberta, roberta_large = load_all_models()
# ==============================================================================
# FUNGSI CLEAN TEXT
# ==============================================================================
def clean_text(text):
text = str(text)
# Hapus prefix metadata Twitter/X
text = re.sub(
r'^.*?·\s*\d+\s*(?:dtk|mnt|jam|s|h|sec|min)\s*(?:Membalas\s+@\w+\s*)?',
'',
text,
flags=re.IGNORECASE
)
# Hapus "Tampilkan lebih banyak" (artefak UI Twitter)
text = re.sub(r'Tampilkan lebih banyak.*$', '', text, flags=re.IGNORECASE)
# Hapus angka trailing engagement (like/retweet count)
text = re.sub(r'(\s+\d+)+\s*$', '', text).strip()
# Cleaning standar
text = text.lower()
text = re.sub(r"http\S+", "", text) # hapus URL
text = re.sub(r"@\w+", "", text) # hapus @mention
text = re.sub(r"#\w+", "", text) # hapus #hashtag
text = re.sub(r"[^\w\s]", "", text) # hapus tanda baca
text = re.sub(r"\b\d+\b", "", text) # hapus angka sisa
text = re.sub(r"\s+", " ", text).strip()
# Hapus stopwords
tokens = text.split()
tokens = [word for word in tokens if word not in stop_words]
return " ".join(tokens)
# ==============================================================================
# THRESHOLD TEXTBLOB
# ==============================================================================
TEXTBLOB_THRESHOLD = 0.10
def classify_tb(score):
if score > TEXTBLOB_THRESHOLD: return 'positive'
if score < -TEXTBLOB_THRESHOLD: return 'negative'
return 'neutral'
def map_roberta(label):
return {"LABEL_0": "negative", "LABEL_1": "neutral", "LABEL_2": "positive"}.get(label, "neutral")
def map_bertweet(label):
return {"pos": "positive", "neu": "neutral", "neg": "negative"}.get(label.lower(), "neutral")
def get_daily_label(score):
if score > TEXTBLOB_THRESHOLD: return 'Positive'
elif score < -TEXTBLOB_THRESHOLD: return 'Negative'
else: return 'Neutral'
# ==============================================================================
# HALAMAN 1 — UJI KALIMAT
# ==============================================================================
if st.session_state.page == "uji_kalimat":
st.markdown('', unsafe_allow_html=True)
col_text, col_img = st.columns([1.1, 1], gap="large")
with col_text:
st.markdown("""
Website ini bukanlah alat prediksi harga Bitcoin real time, melainkan instrumen untuk melakukan analisis sentimen publik secara batch
Bitcoin Volatility
vs Public Sentiment
Analisis Volatilitas Harga Bitcoin Terhadap Sentimen Publik
Pada Platform X Berbasis Python.
Peneliti: Arya Galuh Saputra · H1D022022
""", unsafe_allow_html=True)
user_input = st.text_area(
"Masukkan Tweet (Bahasa Inggris):",
"Great, Bitcoin just fly another 10% today.",
height=120
)
st.markdown("
", unsafe_allow_html=True)
col_btn1, col_btn2 = st.columns([1.6, 1])
with col_btn1:
st.markdown('
', unsafe_allow_html=True)
analyze_btn = st.button("Proses Uji Kalimat", use_container_width=True)
st.markdown('
', unsafe_allow_html=True)
with col_img:
st.markdown("
", unsafe_allow_html=True)
try:
st.image(img_hero, use_container_width=True)
except Exception:
st.markdown("""
🖼️ Gambar Tidak Ditemukan
Pastikan file bitcoin1.gif ada di direktori
""", unsafe_allow_html=True)
st.markdown('
', unsafe_allow_html=True)
st.markdown('', unsafe_allow_html=True)
if analyze_btn:
scroll_to_target("target-uji-kalimat")
col_space_left, col_center_output, col_space_right = st.columns([1, 4, 1])
with col_center_output:
st.markdown("""
Output Analisis
Hasil Deteksi Sentimen
""", unsafe_allow_html=True)
try:
if detect(user_input) != 'en':
st.warning("⚠️ Teks sepertinya bukan bahasa Inggris. Hasil prediksi mungkin memiliki bias.")
except:
pass
text = clean_text(user_input)
with st.spinner("Mengekstraksi sentimen dengan 5 Model..."):
time.sleep(0.5)
try:
v_compound = vader.polarity_scores(text)['compound']
v_label = "positive" if v_compound > 0.05 else ("negative" if v_compound < -0.05 else "neutral")
except:
v_compound = 0.0
v_label = "neutral"
try:
t_label = classify_tb(TextBlob(text).sentiment.polarity)
except:
t_label = "neutral"
try:
b_label = map_bertweet(bertweet(text)[0]['label'])
except:
b_label = "neutral"
try:
r_label = map_roberta(roberta(text)[0]['label'])
except:
r_label = "neutral"
try:
rl_label = roberta_large(text)[0]['label'].lower()
except:
rl_label = "neutral"
def badge_color(label):
return {"positive": "#e6fff1", "negative": "#fef1f2", "neutral": "#f1f5f9"}[label]
def badge_text_color(label):
return {"positive": "#10b981", "negative": "#f43f5e", "neutral": "#64748b"}[label]
results = [
("VADER", v_label, f"compound: {v_compound:.4f}"),
("TextBlob", t_label, f"threshold: ±{TEXTBLOB_THRESHOLD}"),
("BERTweet", b_label, ""),
("RoBERTa Base", r_label, ""),
("RoBERTa Large", rl_label, ""),
]
col_a, col_b = st.columns(2)
for i, (method, label, detail) in enumerate(results):
col = col_a if i % 2 == 0 else col_b
bg = badge_color(label)
tc = badge_text_color(label)
icon = "↗" if label == "positive" else ("↘" if label == "negative" else "→")
detail_html = (
''
+ detail +
'
'
) if detail else ''
if label == 'positive':
border_color = '#10b981'
elif label == 'negative':
border_color = '#f43f5e'
else:
border_color = '#cbd5e1'
html_card = (
''
'
'
'
'
+ method +
'
'
'
'
+ label.capitalize() +
'
'
+ detail_html +
'
'
'
'
+ icon + ' ' + label.upper() +
'
'
'
'
)
with col:
st.markdown(html_card, unsafe_allow_html=True)
with st.expander("🔍 Lihat teks setelah preprocessing"):
st.code(text if text.strip() else "(kosong setelah dibersihkan)", language=None)
# ==============================================================================
# HALAMAN 2 — ANALISIS BATCH
# ==============================================================================
elif st.session_state.page == "analisis_batch":
plt.style.use('default')
sns.set_theme(style="whitegrid", rc={
"axes.facecolor": "#FFFFFF",
"figure.facecolor": "#FAFAFA",
"axes.edgecolor": "#e2e8f0",
"text.color": "#0f172a",
"xtick.color": "#64748b",
"ytick.color": "#64748b",
"grid.color": "#f1f5f9",
})
st.markdown('', unsafe_allow_html=True)
col_upload, col_img_b = st.columns([1.4, 1], gap="large")
with col_upload:
st.markdown("""
Analisis Batch Processing
Volatilitas Harga Bitcoin Vs Sentimen Publik
Kolerasi Multi-Metode Analisis Sentimen
Unggah file tweets (.txt) untuk diekstraksi dan
dianalisis terhadap volatilitas harga Bitcoin.
""", unsafe_allow_html=True)
tweet_files = st.file_uploader(
"Pilih file Tweet (.txt)",
type=['txt'],
accept_multiple_files=True
)
with st.expander("Format TXT yang Didukung"):
st.code(
"username | 2024-03-01 14:00:00\n"
"Isi tweet baris pertama di sini\n\n"
"username2 | 2024-03-01 15:30:00\n"
"Isi tweet baris kedua di sini",
language="text"
)
st.markdown("
", unsafe_allow_html=True)
st.markdown('
', unsafe_allow_html=True)
analyze_batch_btn = st.button("Eksekusi Analisis", key="batch_btn", use_container_width=False)
st.markdown('
', unsafe_allow_html=True)
with col_img_b:
st.markdown("
", unsafe_allow_html=True)
try:
st.image(img_batch, use_container_width=True)
except Exception:
st.markdown("""
🖼️ Gambar Tidak Ditemukan
Pastikan file bitcoin2.gif ada di direktori
""", unsafe_allow_html=True)
st.markdown('
', unsafe_allow_html=True)
# ==============================================================================
# SECTION TUTORIAL PENGUMPULAN DATA TWEET
# ==============================================================================
st.markdown("""
Panduan
📥 Cara Mengumpulkan Data Tweet
Sebelum mengunggah file, kumpulkan data tweet dari platform X menggunakan
skrip collect.js yang dijalankan langsung di browser. Ikuti langkah-langkah berikut.
""", unsafe_allow_html=True)
# ── Tab Tutorial ──────────────────────────────────────────────────────────────
tab_langkah, tab_script, tab_format, tab_tips = st.tabs([
"📋 Langkah-Langkah",
"💻 Skrip collect.js",
"📄 Format File .txt",
"💡 Tips & Catatan"
])
with tab_langkah:
st.markdown("""
""", unsafe_allow_html=True)
langkah_data = [
("1", "#10b981", "Login ke Platform X",
"Buka
x.com di browser Chrome/Edge/Firefox. Pastikan sudah login ke akun X Anda. Gunakan akun aktif agar tidak kena pembatasan akses.",
"🌐"),
("2", "#10b981", "Cari Keyword 'Bitcoin'",
"Di kolom pencarian X, ketik
Bitcoin lalu tekan Enter. Pilih tab
Latest (Terbaru) — bukan Top — agar hasil terurut cronologis dan lebih representatif untuk analisis harian.",
"🔍"),
("3", "#10b981", "Filter Tanggal (Opsional)",
"Untuk scraping per hari tertentu, gunakan filter pencarian lanjutan X:
until:YYYY-MM-DD since:YYYY-MM-DD. Contoh:
Bitcoin since:2026-04-16 until:2026-04-17. Ini memastikan data per file sesuai satu hari.",
"📅"),
("4", "#10b981", "Buka Developer Tools",
"Tekan
F12 (atau klik kanan → Inspect) untuk membuka DevTools browser. Pilih tab
Console. Pastikan tidak ada peringatan keamanan — beberapa browser meminta konfirmasi teks sebelum menjalankan skrip.",
"🛠️"),
("5", "#10b981", "Jalankan Skrip collect.js",
"Copy seluruh isi skrip
collect.js dari tab
Skrip collect.js di atas, paste ke kolom Console, lalu tekan
Enter. Skrip akan mulai men-scrape tweet yang tampil di halaman.",
"▶️"),
("6", "#10b981", "Scroll Halaman untuk Load Lebih Banyak Tweet",
"Setelah skrip aktif,
scroll ke bawah perlahan pada halaman X untuk me-load lebih banyak tweet. Skrip akan otomatis mendeteksi tweet baru yang muncul. Targetkan minimal 200–300 tweet per hari untuk hasil analisis yang valid.",
"⬇️"),
("7", "#10b981", "Download File .txt",
"Ketik perintah
downloadTweets() di Console lalu tekan Enter. File .txt akan otomatis terunduh. Rename file sesuai urutan hari:
1.txt untuk hari pertama,
2.txt untuk hari kedua, dst.",
"💾"),
("8", "#10b981", "Ulangi untuk Setiap Hari",
"Ulangi langkah 2–7 untuk setiap hari yang ingin dianalisis. Pastikan minimal
30 hari data agar korelasi Pearson memiliki kekuatan statistik yang cukup (r ≥ 0.35 pada n=30).",
"🔁"),
("9", "#10b981", "Upload Semua File ke Website",
"Setelah semua file siap (1.txt, 2.txt, ..., 30.txt), upload sekaligus ke kolom unggah di bawah ini, lalu klik
Eksekusi Analisis.",
"🚀"),
]
for num, color, title, desc, icon in langkah_data:
st.markdown(f"""
""", unsafe_allow_html=True)
st.markdown("
", unsafe_allow_html=True)
with tab_script:
st.markdown("""
📌 Cara pakai: Copy seluruh skrip di bawah → Paste di Console browser (F12)
saat berada di halaman pencarian X → Tekan Enter → Scroll halaman untuk load tweet →
Ketik downloadTweets() → File .txt terunduh otomatis.
""", unsafe_allow_html=True)
collect_js = '''// ============================================================
// collect.js — X Tweet Scraper
// Jalankan di Console browser saat berada di halaman pencarian X
// Keyword yang digunakan: "Bitcoin" (tab: Latest)
// ============================================================
(function() {
// Menyimpan semua tweet yang sudah dikumpulkan (Set mencegah duplikat)
window._collectedTweets = window._collectedTweets || new Set();
window._tweetList = window._tweetList || [];
// Fungsi utama: scrape semua tweet yang saat ini tampil di halaman
function scrapeTweets() {
// Selector untuk artikel tweet di X.com
const tweetArticles = document.querySelectorAll('article[data-testid="tweet"]');
let newCount = 0;
tweetArticles.forEach(article => {
try {
// ── Ambil username (handle @...) ──────────────────────────
const userEl = article.querySelector('[data-testid="User-Name"]');
const username = userEl
? userEl.innerText.replace(/\n/g, ' ').trim()
: 'unknown';
// ── Ambil timestamp ───────────────────────────────────────
const timeEl = article.querySelector('time');
const datetime = timeEl
? timeEl.getAttribute('datetime') // format ISO: 2026-04-16T14:30:00.000Z
: new Date().toISOString();
// ── Ambil teks tweet ──────────────────────────────────────
const textEl = article.querySelector('[data-testid="tweetText"]');
const tweetText = textEl
? textEl.innerText.trim()
: '';
// Skip tweet kosong
if (!tweetText) return;
// Buat unique key untuk mencegah duplikat
const key = username + '|' + datetime + '|' + tweetText.substring(0, 50);
if (!window._collectedTweets.has(key)) {
window._collectedTweets.add(key);
// Format sesuai yang diharapkan source code Python:
// "username | datetime"
// "isi tweet"
const dateFormatted = datetime.replace('T', ' ').replace(/\\.\\d+Z$/, '').replace('Z', '');
window._tweetList.push({
meta: username + ' | ' + dateFormatted,
content: tweetText
});
newCount++;
}
} catch (e) {
// Skip tweet yang gagal diproses
}
});
console.log(`[collect.js] +${newCount} tweet baru | Total: ${window._tweetList.length}`);
}
// ── Auto-scrape setiap 2 detik saat halaman di-scroll ────────────────
if (window._scrapeInterval) {
clearInterval(window._scrapeInterval);
}
window._scrapeInterval = setInterval(scrapeTweets, 2000);
// Jalankan sekali langsung saat skrip diload
scrapeTweets();
// ── Fungsi download — ketik downloadTweets() di Console ──────────────
window.downloadTweets = function(filename) {
if (window._tweetList.length === 0) {
console.warn('[collect.js] Belum ada tweet terkumpul. Scroll halaman lebih banyak dulu.');
return;
}
// Susun konten file: setiap tweet dipisah baris kosong
const lines = window._tweetList.map(t => t.meta + '\\n' + t.content);
const content = lines.join('\\n\\n');
// Tentukan nama file otomatis berdasarkan tanggal tweet pertama
if (!filename) {
const firstDate = window._tweetList[0].meta.split(' | ')[1];
const dateStr = firstDate ? firstDate.split(' ')[0].replace(/-/g, '') : 'tweets';
filename = dateStr + '_bitcoin.txt';
}
// Buat blob dan trigger download
const blob = new Blob([content], { type: 'text/plain;charset=utf-8' });
const url = URL.createObjectURL(blob);
const a = document.createElement('a');
a.href = url;
a.download = filename;
a.click();
URL.revokeObjectURL(url);
console.log(`[collect.js] ✅ Download: ${filename} (${window._tweetList.length} tweet)`);
};
// ── Fungsi reset — ketik resetTweets() untuk mulai hari baru ─────────
window.resetTweets = function() {
window._collectedTweets = new Set();
window._tweetList = [];
console.log('[collect.js] 🔄 Data direset. Siap scraping hari baru.');
};
// ── Fungsi status ─────────────────────────────────────────────────────
window.statusTweets = function() {
console.log(`[collect.js] 📊 Total tweet: ${window._tweetList.length}`);
if (window._tweetList.length > 0) {
console.log(' Pertama:', window._tweetList[0].meta);
console.log(' Terakhir:', window._tweetList[window._tweetList.length - 1].meta);
}
};
console.log('[collect.js] ✅ Skrip aktif!');
console.log(' → Scroll halaman X untuk load lebih banyak tweet');
console.log(' → Ketik downloadTweets() untuk download file .txt');
console.log(' → Ketik resetTweets() untuk mulai hari baru');
console.log(' → Ketik statusTweets() untuk cek jumlah tweet');
})();'''
st.code(collect_js, language="javascript")
st.markdown("""
⚠️ Perhatian: Beberapa browser (terutama Chrome) menampilkan peringatan
saat paste skrip ke Console. Jika diminta, ketik allow pasting lalu tekan Enter,
kemudian paste ulang skripnya.
""", unsafe_allow_html=True)
with tab_format:
st.markdown("""
File .txt yang diunggah harus mengikuti format berikut agar dapat dibaca
oleh sistem. Setiap tweet terdiri dari 2 baris dan dipisahkan
oleh satu baris kosong.
""", unsafe_allow_html=True)
col_f1, col_f2 = st.columns(2)
with col_f1:
st.markdown("**✅ Format yang benar:**")
st.code(
"username | 2026-04-16 14:30:00\n"
"Bitcoin is looking bullish today! Great news for crypto holders.\n\n"
"another_user | 2026-04-16 15:45:00\n"
"BTC just hit 75k, incredible run. When moon?\n\n"
"crypto_analyst | 2026-04-16 16:20:00\n"
"Bearish divergence on BTC 4H chart. Be careful traders.",
language="text"
)
with col_f2:
st.markdown("**❌ Format yang salah:**")
st.code(
"# Jangan ada header CSV\n"
"date,user,tweet\n\n"
"# Jangan ada spasi ganda antar tweet\n\n\n"
"user | 2026-04-16\n"
"tweet...\n\n\n"
"# Tanggal harus ada jamnya\n"
"user | 2026-04-16\n"
"tweet tanpa jam...",
language="text"
)
st.markdown("""
📁 Konvensi Penamaan File
• Satu file = satu hari data
• Nama file: 1.txt (hari ke-1), 2.txt (hari ke-2), dst.
• Sistem mengurutkan file secara alfanumerik sebelum diproses
• Tidak ada batasan jumlah tweet per file
• Encoding: UTF-8 (default output collect.js)
""", unsafe_allow_html=True)
with tab_tips:
tips_data = [
("🎯", "Target Minimal Data",
"Gunakan minimal 30 hari data untuk hasil korelasi yang bermakna secara statistik. Dengan n=30, nilai r ≥ 0.35 sudah signifikan pada p < 0.05. Makin banyak hari, makin kuat reliabilitas temuan."),
("📊", "Jumlah Tweet per Hari",
"Targetkan 200–500 tweet per hari. Terlalu sedikit (< 50 tweet) membuat rata-rata sentimen harian tidak representatif. Scroll halaman X selama 1–2 menit per hari untuk mendapatkan jumlah yang cukup."),
("🔍", "Kata Kunci yang Tepat",
"Penelitian ini menggunakan kata kunci tunggal \"Bitcoin\" (tanpa tanda petik di X). Pastikan memilih tab Latest, bukan Top, agar distribusi temporal merata dan tidak bias ke tweet viral."),
("📅", "Konsistensi Periode Waktu",
"Gunakan filter tanggal X untuk memastikan setiap file hanya berisi tweet dari satu hari kalender. Contoh: Bitcoin since:2026-04-16 until:2026-04-17. Ini penting agar agregasi harian akurat."),
("🌐", "Bahasa Tweet",
"Sistem otomatis memfilter tweet non-Inggris menggunakan langdetect. Dari pengalaman penelitian ini, sekitar 13–14% tweet diskip karena bukan bahasa Inggris. Ini normal dan sudah diperhitungkan."),
("💾", "Reset Antar Hari",
"Setelah download file untuk satu hari, selalu ketik resetTweets() di Console sebelum pindah ke hari berikutnya. Ini mencegah tweet dari hari sebelumnya ikut masuk ke file hari berikutnya."),
("⚡", "Performa Browser",
"Tutup tab lain yang tidak diperlukan saat scraping untuk mencegah browser melambat. Jika halaman X berhenti load tweet setelah scroll panjang, refresh halaman dan jalankan ulang collect.js (data sebelumnya akan hilang)."),
("🔒", "Batas Scraping X",
"Platform X membatasi scraping agresif. Jika halaman tiba-tiba tidak menampilkan tweet baru meski di-scroll, tunggu 5–10 menit sebelum melanjutkan. Alternatif: gunakan akun berbeda atau ganti IP."),
]
col_t1, col_t2 = st.columns(2)
for i, (icon, title, desc) in enumerate(tips_data):
col = col_t1 if i % 2 == 0 else col_t2
with col:
st.markdown(f"""
""", unsafe_allow_html=True)
st.markdown("
",
unsafe_allow_html=True)
st.markdown('', unsafe_allow_html=True)
if tweet_files and analyze_batch_btn:
scroll_to_target("target-analisis-batch")
col_b_space1, col_b_content, col_b_space2 = st.columns([1, 8, 1])
with col_b_content:
st.markdown("""
Hasil Pemrosesan
Dashboard Analisis
""", unsafe_allow_html=True)
tweet_files = sorted(tweet_files, key=lambda x: x.name)
data = []
with st.status("🔄 Memproses data sentimen...", expanded=True) as status:
progress_bar = st.progress(0, text="Mengekstrak sentimen dari data...")
total_tweets_uploaded = 0
total_tweets_skipped = 0
for idx, file in enumerate(tweet_files):
content = file.getvalue().decode("utf-8").replace("\r\n", "\n").strip()
tweets = content.split("\n\n")
for tweet in tweets:
parts = tweet.strip().split("\n", 1)
if len(parts) != 2: continue
meta, text_raw = parts
try:
DetectorFactory.seed = 0
lang = detect(text_raw)
if lang != 'en':
total_tweets_skipped += 1
continue
except:
total_tweets_skipped += 1
continue
username, date_val = meta.split(" | ") if " | " in meta else ("unknown", "unknown")
short_date = date_val[:10]
text = clean_text(text_raw)
if not text.strip():
total_tweets_skipped += 1
continue
try:
v_compound = vader.polarity_scores(text)['compound']
vader_label = "positive" if v_compound > 0.05 else ("negative" if v_compound < -0.05 else "neutral")
except:
v_compound = 0.0
vader_label = "neutral"
try:
tb_polarity = TextBlob(text).sentiment.polarity
tb_label = classify_tb(tb_polarity)
except:
tb_polarity = 0.0
tb_label = "neutral"
try:
bertweet_label = map_bertweet(bertweet(text)[0]['label'])
except:
bertweet_label = "neutral"
try:
roberta_label = map_roberta(roberta(text)[0]['label'])
except:
roberta_label = "neutral"
try:
roberta_large_label = roberta_large(text)[0]['label'].lower()
except:
roberta_large_label = "neutral"
data.append({
"date": short_date,
"raw_tweet": text_raw.strip(),
"cleaned_tweet": text,
"vader": vader_label,
"textblob": tb_label,
"bertweet": bertweet_label,
"roberta": roberta_label,
"roberta_large": roberta_large_label,
"vader_score": v_compound,
"tb_score": tb_polarity,
})
total_tweets_uploaded += 1
progress_bar.progress((idx + 1) / len(tweet_files),
text=f"Memproses file {idx+1} dari {len(tweet_files)}")
status.update(label="✅ Pemrosesan sentimen teks selesai!", state="complete", expanded=False)
df = pd.DataFrame(data)
if df.empty:
st.error("❌ Data kosong. Pastikan format TXT benar dan tweet berbahasa Inggris.")
else:
col_m1, col_m2, col_m3 = st.columns(3)
col_m1.metric("Tweet Diproses", f"{total_tweets_uploaded}", border=True)
col_m2.metric("Tweet Diabaikan (Non-EN)", f"{total_tweets_skipped}", border=True)
col_m3.metric("Model", "5 Model", border=True)
target_dates = sorted(df['date'].unique())
start_unix = int(datetime.strptime(target_dates[0], "%Y-%m-%d").replace(tzinfo=timezone.utc).timestamp()) - 86400
end_unix = int(datetime.strptime(target_dates[-1], "%Y-%m-%d").replace(tzinfo=timezone.utc).timestamp()) + 86400
# ==============================================================
# FUNGSI FETCH HARGA BTC
# ==============================================================
def fetch_via_coingecko(start_ts, end_ts):
"""Coba CoinGecko dengan retry eksponensial. Return list [[ts_ms, price]] atau raise."""
url = "https://api.coingecko.com/api/v3/coins/bitcoin/market_chart/range"
params = {"vs_currency": "usd", "from": start_ts, "to": end_ts}
headers = {"accept": "application/json", "User-Agent": "Mozilla/5.0"}
wait_times = [5, 15, 30]
for attempt, wait in enumerate(wait_times, 1):
time.sleep(wait)
res = requests.get(url, params=params, headers=headers, timeout=20)
if res.status_code == 200:
data_j = res.json()
if "prices" in data_j:
return data_j["prices"]
raise ValueError("Key 'prices' tidak ada di respons CoinGecko.")
if res.status_code == 429:
if attempt < len(wait_times):
continue # coba lagi dengan backoff lebih lama
raise ConnectionError(f"CoinGecko 429 setelah {attempt} percobaan.")
raise ConnectionError(f"CoinGecko error {res.status_code}: {res.text[:200]}")
raise ConnectionError("CoinGecko gagal setelah semua percobaan.")
def fetch_via_binance(start_ts, end_ts):
"""
Fallback: Binance Public API — BTCUSDT daily klines.
Endpoint bebas API key, limit 1000 candles per request.
"""
url = "https://api.binance.com/api/v3/klines"
prices = []
cur_start = start_ts * 1000
end_ms = end_ts * 1000
while cur_start < end_ms:
params = {
"symbol": "BTCUSDT",
"interval": "1d",
"startTime": cur_start,
"endTime": end_ms,
"limit": 1000,
}
res = requests.get(url, params=params, timeout=20)
if res.status_code != 200:
raise ConnectionError(f"Binance error {res.status_code}: {res.text[:200]}")
batch = res.json()
if not batch:
break
for candle in batch:
open_time = int(candle[0]) # ms
close_price = float(candle[4]) # close price
prices.append([open_time, close_price])
cur_start = int(batch[-1][0]) + 1
if len(batch) < 1000:
break
if not prices:
raise ValueError("Binance tidak mengembalikan data.")
return prices
def build_df_price(raw_prices, target_date_list):
"""Bersihkan raw [[ts_ms, price]] → df_price dengan log_return."""
df_p = pd.DataFrame(raw_prices, columns=["timestamp", "price"])
df_p["date"] = pd.to_datetime(df_p["timestamp"], unit="ms").dt.date
df_p = df_p.groupby("date")["price"].mean().reset_index()
df_p["pct_change"] = df_p["price"].pct_change() * 100
df_p["log_return"] = np.log(df_p["price"] / df_p["price"].shift(1))
df_p.dropna(inplace=True)
df_p = df_p[df_p["date"].isin(pd.to_datetime(target_date_list).date)]
return df_p
raw_prices = None
api_source = None
with st.spinner("📡 Mengambil data harga Bitcoin..."):
try:
raw_prices = fetch_via_coingecko(start_unix, end_unix)
api_source = "CoinGecko"
except Exception as cg_err:
st.warning(
f"⚠️ CoinGecko tidak tersedia ({cg_err}). "
"Beralih ke **Binance Public API** sebagai fallback..."
)
try:
raw_prices = fetch_via_binance(start_unix, end_unix)
api_source = "Binance"
except Exception as bn_err:
st.error(
f"❌ Kedua sumber data gagal.\n"
f"- CoinGecko: {cg_err}\n"
f"- Binance: {bn_err}\n\n"
"Coba lagi beberapa menit kemudian atau periksa koneksi internet."
)
if raw_prices is not None:
try:
df_price = build_df_price(raw_prices, target_dates)
st.info(f"✅ Data harga BTC berhasil diambil dari **{api_source}**.")
if df_price.empty:
st.warning("⚠️ Data Harga API kosong. Pastikan rentang tanggal di .txt sesuai (yyyy-mm-dd).")
else:
st.markdown("
", unsafe_allow_html=True)
# ── Tabel Data Sentimen ──────────────────────────────
st.markdown("🗣️ Data Sentimen")
raw_display_cols = ["date","raw_tweet","vader","textblob","bertweet","roberta","roberta_large","vader_score","tb_score"]
st.dataframe(df[raw_display_cols], use_container_width=True, hide_index=True)
# ==================================================
# AGREGASI HARIAN DUAL-MODE
# Mode A (Kategorik): konversi {pos:1, neu:0, neg:-1}
# Mode B (Numerik): rata-rata vader_score & tb_score
# ==================================================
sentiment_map = {"positive": 1, "neutral": 0, "negative": -1}
df_score = df.copy()
models_cat = ["vader","textblob","bertweet","roberta","roberta_large"]
for col in models_cat:
df_score[col] = df_score[col].map(sentiment_map)
# Agregasi kategorik (−1/0/1 mean)
df_sentiment_daily = df_score.groupby("date")[models_cat].mean().reset_index()
df_sentiment_daily["date"] = pd.to_datetime(df_sentiment_daily["date"]).dt.date
# Agregasi numerik VADER compound & TextBlob polarity
df_numeric_daily = df.groupby("date")[["vader_score","tb_score"]].mean().reset_index()
df_numeric_daily["date"] = pd.to_datetime(df_numeric_daily["date"]).dt.date
for col in models_cat:
df_sentiment_daily[f"{col}_label"] = df_sentiment_daily[col].apply(get_daily_label)
daily_display_cols = ["date"]
for col in models_cat:
daily_display_cols.extend([col, f"{col}_label"])
# ── Tabel Harga Bitcoin ───────────────────────────
st.markdown("₿ Data Harga & Volatilitas Bitcoin")
st.dataframe(df_price[["date","price","pct_change","log_return"]], use_container_width=True, hide_index=True)
# Merge data
df_merged = pd.merge(df_price, df_sentiment_daily, on="date", how="inner")
df_merged = pd.merge(df_merged, df_numeric_daily, on="date", how="inner")
# ── Tabel Data Final ──────────────────────────────
st.markdown("🗂️ Data Final")
final_display_cols = (
["date","price","pct_change","log_return"]
+ [c for c in daily_display_cols if c != "date"]
+ ["vader_score","tb_score"]
)
st.dataframe(df_merged[final_display_cols], use_container_width=True, hide_index=True)
# Download buttons
col_dl1, col_dl2, _ = st.columns([1, 1, 3])
csv_data = df_merged.to_csv(index=False).encode('utf-8')
col_dl1.download_button("📥 Unduh CSV", data=csv_data, file_name="bitcoin_volatility_sentiment.csv", mime="text/csv", use_container_width=True)
buffer = io.BytesIO()
with pd.ExcelWriter(buffer, engine='xlsxwriter') as writer:
df_merged.to_excel(writer, index=False)
col_dl2.download_button("📥 Unduh Excel", data=buffer.getvalue(), file_name="bitcoin_volatility_sentiment.xlsx", mime="application/vnd.ms-excel", use_container_width=True)
st.markdown("
", unsafe_allow_html=True)
# ==================================================
# UJI KORELASI PEARSON DENGAN LAG
# Lag 0 : sentimen hari t vs harga hari t (same-day)
# Lag +1: sentimen hari t vs harga hari t+1
# Lag +2: sentimen hari t vs harga hari t+2
#
# Kolom korelasi yang diuji:
# - vader_score (numerik compound, −1 s.d. 1)
# - tb_score (numerik polarity, −1 s.d. 1)
# - bertweet (kategorik −1/0/1 mean)
# - roberta (kategorik −1/0/1 mean)
# - roberta_large(kategorik −1/0/1 mean)
# ==================================================
st.subheader("🔬 Uji Korelasi Pearson")
st.caption(
"Menganalisis hubungan statistik antara skor sentimen harian dan "
"volatilitas log-return BTC pada lag 0 (hari sama), +1 hari, dan +2 hari. "
"VADER & TextBlob menggunakan skor numerik kontinu; model BERT menggunakan "
"rata-rata kategorikal (−1/0/1)."
)
corr_columns = {
"VADER (numerik)": "vader_score",
"TextBlob (numerik)": "tb_score",
"BERTweet (kategorik)": "bertweet",
"RoBERTa Base (kategorik)": "roberta",
"RoBERTa Large (kategorik)":"roberta_large",
}
corr_rows = []
raw_results = []
n_obs = len(df_merged)
for label_name, col_key in corr_columns.items():
for lag in [0, 1, 2]:
# shift(-lag): harga mundur lag hari ke depan
# artinya: sentimen t berkorelasi dengan harga t+lag
log_ret_shifted = df_merged["log_return"].shift(-lag)
valid_mask = log_ret_shifted.notna()
x_vals = df_merged.loc[valid_mask, col_key]
y_vals = log_ret_shifted[valid_mask]
if len(x_vals) < 4:
corr, pval = np.nan, np.nan
else:
try:
corr, pval = pearsonr(x_vals, y_vals)
except Exception:
corr, pval = np.nan, np.nan
arah = "Positif" if (corr is not np.nan and corr > 0) else "Negatif"
sig = "✅ Signifikan" if (pval is not np.nan and pval < 0.05) else "Tidak Signifikan"
corr_rows.append({
"Metode": label_name,
"Lag": f"t+{lag}",
"r": f"{corr:.4f}" if not np.isnan(corr) else "N/A",
"Arah": arah,
"p-value": f"{pval:.4f}" if not np.isnan(pval) else "N/A",
"Signifikansi": sig,
})
raw_results.append({
"metode": label_name,
"col": col_key,
"lag": lag,
"r": corr if not np.isnan(corr) else 0.0,
"p": pval if not np.isnan(pval) else 1.0,
})
df_corr_table = pd.DataFrame(corr_rows)
# Tandai baris terbaik per metode (|r| terbesar)
def highlight_best(row):
try:
r_abs = abs(float(row["r"]))
except:
r_abs = 0
sig_ok = "✅" in str(row["Signifikansi"])
if sig_ok:
return ["background-color:#e6fff1; font-weight:600"] * len(row)
return [""] * len(row)
styled = df_corr_table.style.apply(highlight_best, axis=1)
st.dataframe(styled, use_container_width=True, hide_index=True)
# ── Scatter Plot ──────────────────────────────────
st.subheader("🔵 Pola Distribusi Scatter Plot (Lag t+0)")
scatter_cols = list(corr_columns.values())
cols_sc = st.columns(3)
for idx2, (disp_name, col_key) in enumerate(corr_columns.items()):
with cols_sc[idx2 % 3]:
fig_s, ax_s = plt.subplots(figsize=(5, 4))
sns.regplot(
data=df_merged, x=col_key, y="log_return", ax=ax_s,
scatter_kws={"s": 40, "color": "#10b981", "alpha": 0.5},
line_kws={"color": "#0f172a", "linewidth": 2}
)
# Tampilkan r dan p pada plot
try:
r_val, p_val = pearsonr(df_merged[col_key], df_merged["log_return"])
ax_s.set_title(f"{disp_name.split(' ')[0]}\nr={r_val:.3f}, p={p_val:.3f}", fontweight='bold', fontsize=9)
except:
ax_s.set_title(disp_name.split(' ')[0], fontweight='bold')
ax_s.set_xlabel("Sentimen Score")
ax_s.set_ylabel("Log Return")
plt.tight_layout()
st.pyplot(fig_s)
# ── Line Chart ────────────────────────────────────
st.subheader("📈 Trend Analisis: Sentiment vs BTC Volatility")
fig_line, ax_line = plt.subplots(figsize=(14, 6))
ax_line.plot(
df_merged["date"], df_merged["log_return"],
label="BTC Log Return", color="#f7931a", linewidth=3
)
colors_line = ["#3B82F6","#10B981","#EC4899","#14B8A6","#6366F1"]
for i, (disp_name, col_key) in enumerate(corr_columns.items()):
ax_line.plot(
df_merged["date"], df_merged[col_key],
label=f"Sentimen: {disp_name.split(' ')[0]}",
color=colors_line[i], linewidth=1.5, linestyle="--", alpha=0.8
)
ax_line.set_title("Pergerakan Sentimen vs Log Return Bitcoin", fontsize=14, pad=15, fontweight='bold')
ax_line.set_xlabel("Tanggal", fontsize=11)
ax_line.set_ylabel("Nilai Metrik", fontsize=11)
ax_line.legend(loc='upper left', bbox_to_anchor=(1, 1), frameon=True)
plt.tight_layout()
st.pyplot(fig_line)
# ── KESIMPULAN ────────────────────────────────────
st.markdown("
", unsafe_allow_html=True)
st.subheader("📝 Kesimpulan")
max_idx = df_merged["log_return"].idxmax()
min_idx = df_merged["log_return"].idxmin()
date_max = df_merged.loc[max_idx, "date"]
date_min = df_merged.loc[min_idx, "date"]
st.write(
f"Puncak lonjakan positif (*max log return*) terjadi pada **{date_max}**, "
f"sedangkan penurunan ekstrem terjadi pada **{date_min}**. "
f"Dataset mencakup **{n_obs} hari** pengamatan."
)
# Cari lag & metode terbaik (|r| terbesar + signifikan)
sig_results = [r for r in raw_results if r["p"] < 0.05]
all_results = raw_results
if sig_results:
best = max(sig_results, key=lambda x: abs(x["r"]))
arah_text = "berbanding lurus (positif)" if best["r"] > 0 else "berbanding terbalik (negatif)"
sig_summary = ", ".join(
set(f"{r['metode'].split(' ')[0]} (lag t+{r['lag']})" for r in sig_results)
)
st.success(f"""
**Hipotesis Diterima (H1):** Ditemukan korelasi linier yang signifikan (*p-value* < 0.05) pada: **{sig_summary}**.
Metode & lag dengan korelasi terkuat adalah **{best['metode']} (lag t+{best['lag']})** dengan r = **{best['r']:.4f}**, sifat hubungan **{arah_text}**.
""")
else:
best_overall = max(all_results, key=lambda x: abs(x["r"]))
st.warning(f"""
**Hipotesis Ditolak (H0 Diterima):** Belum ditemukan bukti empiris korelasi linier yang signifikan pada seluruh metode dan lag yang diuji (*p-value* ≥ 0.05).
Korelasi terbesar ditemukan pada **{best_overall['metode']} (lag t+{best_overall['lag']})** dengan r = **{best_overall['r']:.4f}**. Volatilitas harga kemungkinan dipengaruhi faktor teknikal/fundamental di luar sentimen X, atau jumlah observasi belum mencukupi (n = {n_obs}, butuh minimal 30 hari agar r ≥ 0.35 bisa signifikan).
""")
except Exception as e:
st.error(f"⚠️ Terjadi kesalahan saat memproses data harga Bitcoin: {e}")
elif analyze_batch_btn and not tweet_files:
st.warning("⚠️ Silakan unggah minimal satu file .txt terlebih dahulu.")