import streamlit as st import pandas as pd import numpy as np import re import io import time import requests import matplotlib.pyplot as plt import matplotlib.gridspec as gridspec import seaborn as sns from datetime import datetime, timezone from textblob import TextBlob from scipy.stats import pearsonr import nltk from nltk.corpus import stopwords from nltk.sentiment.vader import SentimentIntensityAnalyzer from transformers import pipeline import os import streamlit.components.v1 as components from langdetect import detect, DetectorFactory DetectorFactory.seed = 0 # ============================== # SETTING PATH ABSOLUT GAMBAR # ============================== BASE_DIR = os.path.dirname(os.path.abspath(__file__)) img_hero = os.path.join(BASE_DIR, "bitcoin1.gif") img_batch = os.path.join(BASE_DIR, "bitcoin2.gif") # ============================== # KONFIGURASI HALAMAN & STATE NAVIGASI # ============================== st.set_page_config( page_title="Bitcoin Volatility Sentiment", page_icon="₿", layout="wide", initial_sidebar_state="collapsed" ) if 'page' not in st.session_state: st.session_state.page = "uji_kalimat" # ============================== # GLOBAL CSS # ============================== st.markdown(""" """, unsafe_allow_html=True) # ============================== # FUNGSI AUTO-SCROLL # ============================== def scroll_to_target(target_id): js_code = f""" """ components.html(js_code, height=0, width=0) # ============================== # HEADER / NAVBAR # ============================== def set_page(page_name): st.session_state.page = page_name col_logo, col_space, col_btn1, col_btn2 = st.columns([5, 3, 2, 2], vertical_alignment="center") with col_logo: st.markdown(""" """, unsafe_allow_html=True) with col_btn1: is_uji = st.session_state.page == "uji_kalimat" css_class = "btn-orange" if is_uji else "btn-ghost" st.markdown(f'
', unsafe_allow_html=True) if st.button("Uji Kalimat", use_container_width=True, key="nav_uji"): set_page("uji_kalimat"); st.rerun() st.markdown('
', unsafe_allow_html=True) with col_btn2: is_batch = st.session_state.page == "analisis_batch" css_class = "btn-orange" if is_batch else "btn-ghost" st.markdown(f'
', unsafe_allow_html=True) if st.button("Analisis Batch", use_container_width=True, key="nav_batch"): set_page("analisis_batch"); st.rerun() st.markdown('
', unsafe_allow_html=True) st.markdown("
", unsafe_allow_html=True) # ============================== # DOWNLOAD RESOURCES & LOAD MODELS # ============================== @st.cache_resource def download_nltk_resources(): nltk.download('stopwords', quiet=True) nltk.download('vader_lexicon', quiet=True) nltk.download('punkt', quiet=True) nltk.download('omw-1.4', quiet=True) download_nltk_resources() stop_words = set(stopwords.words('english')) @st.cache_resource def load_all_models(): vader = SentimentIntensityAnalyzer() bertweet = pipeline("sentiment-analysis", model="finiteautomata/bertweet-base-sentiment-analysis", device=-1, truncation=True, max_length=128) roberta = pipeline("sentiment-analysis", model="cardiffnlp/twitter-roberta-base-sentiment", device=-1, truncation=True, max_length=512) roberta_large = pipeline("sentiment-analysis", model="siebert/sentiment-roberta-large-english", device=-1, truncation=True, max_length=512) return vader, bertweet, roberta, roberta_large with st.spinner('Mempersiapkan model AI...'): vader, bertweet, roberta, roberta_large = load_all_models() # ============================================================================== # FUNGSI CLEAN TEXT # ============================================================================== def clean_text(text): text = str(text) # Hapus prefix metadata Twitter/X text = re.sub( r'^.*?·\s*\d+\s*(?:dtk|mnt|jam|s|h|sec|min)\s*(?:Membalas\s+@\w+\s*)?', '', text, flags=re.IGNORECASE ) # Hapus "Tampilkan lebih banyak" (artefak UI Twitter) text = re.sub(r'Tampilkan lebih banyak.*$', '', text, flags=re.IGNORECASE) # Hapus angka trailing engagement (like/retweet count) text = re.sub(r'(\s+\d+)+\s*$', '', text).strip() # Cleaning standar text = text.lower() text = re.sub(r"http\S+", "", text) # hapus URL text = re.sub(r"@\w+", "", text) # hapus @mention text = re.sub(r"#\w+", "", text) # hapus #hashtag text = re.sub(r"[^\w\s]", "", text) # hapus tanda baca text = re.sub(r"\b\d+\b", "", text) # hapus angka sisa text = re.sub(r"\s+", " ", text).strip() # Hapus stopwords tokens = text.split() tokens = [word for word in tokens if word not in stop_words] return " ".join(tokens) # ============================================================================== # THRESHOLD TEXTBLOB # ============================================================================== TEXTBLOB_THRESHOLD = 0.10 def classify_tb(score): if score > TEXTBLOB_THRESHOLD: return 'positive' if score < -TEXTBLOB_THRESHOLD: return 'negative' return 'neutral' def map_roberta(label): return {"LABEL_0": "negative", "LABEL_1": "neutral", "LABEL_2": "positive"}.get(label, "neutral") def map_bertweet(label): return {"pos": "positive", "neu": "neutral", "neg": "negative"}.get(label.lower(), "neutral") def get_daily_label(score): if score > TEXTBLOB_THRESHOLD: return 'Positive' elif score < -TEXTBLOB_THRESHOLD: return 'Negative' else: return 'Neutral' # ============================================================================== # HALAMAN 1 — UJI KALIMAT # ============================================================================== if st.session_state.page == "uji_kalimat": st.markdown('
', unsafe_allow_html=True) col_text, col_img = st.columns([1.1, 1], gap="large") with col_text: st.markdown("""
Website ini bukanlah alat prediksi harga Bitcoin real time, melainkan instrumen untuk melakukan analisis sentimen publik secara batch

Bitcoin Volatility
vs Public Sentiment

Analisis Volatilitas Harga Bitcoin Terhadap Sentimen Publik Pada Platform X Berbasis Python.

Peneliti: Arya Galuh Saputra  ·  H1D022022

""", unsafe_allow_html=True) user_input = st.text_area( "Masukkan Tweet (Bahasa Inggris):", "Great, Bitcoin just fly another 10% today.", height=120 ) st.markdown("
", unsafe_allow_html=True) col_btn1, col_btn2 = st.columns([1.6, 1]) with col_btn1: st.markdown('
', unsafe_allow_html=True) analyze_btn = st.button("Proses Uji Kalimat", use_container_width=True) st.markdown('
', unsafe_allow_html=True) with col_img: st.markdown("
", unsafe_allow_html=True) try: st.image(img_hero, use_container_width=True) except Exception: st.markdown("""
🖼️ Gambar Tidak Ditemukan
Pastikan file bitcoin1.gif ada di direktori
""", unsafe_allow_html=True) st.markdown('
', unsafe_allow_html=True) st.markdown('
', unsafe_allow_html=True) if analyze_btn: scroll_to_target("target-uji-kalimat") col_space_left, col_center_output, col_space_right = st.columns([1, 4, 1]) with col_center_output: st.markdown("""

Output Analisis

Hasil Deteksi Sentimen

""", unsafe_allow_html=True) try: if detect(user_input) != 'en': st.warning("⚠️ Teks sepertinya bukan bahasa Inggris. Hasil prediksi mungkin memiliki bias.") except: pass text = clean_text(user_input) with st.spinner("Mengekstraksi sentimen dengan 5 Model..."): time.sleep(0.5) try: v_compound = vader.polarity_scores(text)['compound'] v_label = "positive" if v_compound > 0.05 else ("negative" if v_compound < -0.05 else "neutral") except: v_compound = 0.0 v_label = "neutral" try: t_label = classify_tb(TextBlob(text).sentiment.polarity) except: t_label = "neutral" try: b_label = map_bertweet(bertweet(text)[0]['label']) except: b_label = "neutral" try: r_label = map_roberta(roberta(text)[0]['label']) except: r_label = "neutral" try: rl_label = roberta_large(text)[0]['label'].lower() except: rl_label = "neutral" def badge_color(label): return {"positive": "#e6fff1", "negative": "#fef1f2", "neutral": "#f1f5f9"}[label] def badge_text_color(label): return {"positive": "#10b981", "negative": "#f43f5e", "neutral": "#64748b"}[label] results = [ ("VADER", v_label, f"compound: {v_compound:.4f}"), ("TextBlob", t_label, f"threshold: ±{TEXTBLOB_THRESHOLD}"), ("BERTweet", b_label, ""), ("RoBERTa Base", r_label, ""), ("RoBERTa Large", rl_label, ""), ] col_a, col_b = st.columns(2) for i, (method, label, detail) in enumerate(results): col = col_a if i % 2 == 0 else col_b bg = badge_color(label) tc = badge_text_color(label) icon = "↗" if label == "positive" else ("↘" if label == "negative" else "→") detail_html = ( '
' + detail + '
' ) if detail else '' if label == 'positive': border_color = '#10b981' elif label == 'negative': border_color = '#f43f5e' else: border_color = '#cbd5e1' html_card = ( '
' '
' '
' + method + '
' '
' + label.capitalize() + '
' + detail_html + '
' '
' + icon + ' ' + label.upper() + '
' '
' ) with col: st.markdown(html_card, unsafe_allow_html=True) with st.expander("🔍 Lihat teks setelah preprocessing"): st.code(text if text.strip() else "(kosong setelah dibersihkan)", language=None) # ============================================================================== # HALAMAN 2 — ANALISIS BATCH # ============================================================================== elif st.session_state.page == "analisis_batch": plt.style.use('default') sns.set_theme(style="whitegrid", rc={ "axes.facecolor": "#FFFFFF", "figure.facecolor": "#FAFAFA", "axes.edgecolor": "#e2e8f0", "text.color": "#0f172a", "xtick.color": "#64748b", "ytick.color": "#64748b", "grid.color": "#f1f5f9", }) st.markdown('
', unsafe_allow_html=True) col_upload, col_img_b = st.columns([1.4, 1], gap="large") with col_upload: st.markdown("""

Analisis Batch Processing

Volatilitas Harga Bitcoin Vs Sentimen Publik
Kolerasi Multi-Metode Analisis Sentimen

Unggah file tweets (.txt) untuk diekstraksi dan dianalisis terhadap volatilitas harga Bitcoin.

""", unsafe_allow_html=True) tweet_files = st.file_uploader( "Pilih file Tweet (.txt)", type=['txt'], accept_multiple_files=True ) with st.expander("Format TXT yang Didukung"): st.code( "username | 2024-03-01 14:00:00\n" "Isi tweet baris pertama di sini\n\n" "username2 | 2024-03-01 15:30:00\n" "Isi tweet baris kedua di sini", language="text" ) st.markdown("
", unsafe_allow_html=True) st.markdown('
', unsafe_allow_html=True) analyze_batch_btn = st.button("Eksekusi Analisis", key="batch_btn", use_container_width=False) st.markdown('
', unsafe_allow_html=True) with col_img_b: st.markdown("
", unsafe_allow_html=True) try: st.image(img_batch, use_container_width=True) except Exception: st.markdown("""
🖼️ Gambar Tidak Ditemukan
Pastikan file bitcoin2.gif ada di direktori
""", unsafe_allow_html=True) st.markdown('
', unsafe_allow_html=True) # ============================================================================== # SECTION TUTORIAL PENGUMPULAN DATA TWEET # ============================================================================== st.markdown("""
Panduan

📥 Cara Mengumpulkan Data Tweet

Sebelum mengunggah file, kumpulkan data tweet dari platform X menggunakan skrip collect.js yang dijalankan langsung di browser. Ikuti langkah-langkah berikut.

""", unsafe_allow_html=True) # ── Tab Tutorial ────────────────────────────────────────────────────────────── tab_langkah, tab_script, tab_format, tab_tips = st.tabs([ "📋 Langkah-Langkah", "💻 Skrip collect.js", "📄 Format File .txt", "💡 Tips & Catatan" ]) with tab_langkah: st.markdown("""
""", unsafe_allow_html=True) langkah_data = [ ("1", "#10b981", "Login ke Platform X", "Buka x.com di browser Chrome/Edge/Firefox. Pastikan sudah login ke akun X Anda. Gunakan akun aktif agar tidak kena pembatasan akses.", "🌐"), ("2", "#10b981", "Cari Keyword 'Bitcoin'", "Di kolom pencarian X, ketik Bitcoin lalu tekan Enter. Pilih tab Latest (Terbaru) — bukan Top — agar hasil terurut cronologis dan lebih representatif untuk analisis harian.", "🔍"), ("3", "#10b981", "Filter Tanggal (Opsional)", "Untuk scraping per hari tertentu, gunakan filter pencarian lanjutan X: until:YYYY-MM-DD since:YYYY-MM-DD. Contoh: Bitcoin since:2026-04-16 until:2026-04-17. Ini memastikan data per file sesuai satu hari.", "📅"), ("4", "#10b981", "Buka Developer Tools", "Tekan F12 (atau klik kanan → Inspect) untuk membuka DevTools browser. Pilih tab Console. Pastikan tidak ada peringatan keamanan — beberapa browser meminta konfirmasi teks sebelum menjalankan skrip.", "🛠️"), ("5", "#10b981", "Jalankan Skrip collect.js", "Copy seluruh isi skrip collect.js dari tab Skrip collect.js di atas, paste ke kolom Console, lalu tekan Enter. Skrip akan mulai men-scrape tweet yang tampil di halaman.", "▶️"), ("6", "#10b981", "Scroll Halaman untuk Load Lebih Banyak Tweet", "Setelah skrip aktif, scroll ke bawah perlahan pada halaman X untuk me-load lebih banyak tweet. Skrip akan otomatis mendeteksi tweet baru yang muncul. Targetkan minimal 200–300 tweet per hari untuk hasil analisis yang valid.", "⬇️"), ("7", "#10b981", "Download File .txt", "Ketik perintah downloadTweets() di Console lalu tekan Enter. File .txt akan otomatis terunduh. Rename file sesuai urutan hari: 1.txt untuk hari pertama, 2.txt untuk hari kedua, dst.", "💾"), ("8", "#10b981", "Ulangi untuk Setiap Hari", "Ulangi langkah 2–7 untuk setiap hari yang ingin dianalisis. Pastikan minimal 30 hari data agar korelasi Pearson memiliki kekuatan statistik yang cukup (r ≥ 0.35 pada n=30).", "🔁"), ("9", "#10b981", "Upload Semua File ke Website", "Setelah semua file siap (1.txt, 2.txt, ..., 30.txt), upload sekaligus ke kolom unggah di bawah ini, lalu klik Eksekusi Analisis.", "🚀"), ] for num, color, title, desc, icon in langkah_data: st.markdown(f"""
{num}
{icon} {title}
{desc}
""", unsafe_allow_html=True) st.markdown("
", unsafe_allow_html=True) with tab_script: st.markdown("""
📌 Cara pakai: Copy seluruh skrip di bawah → Paste di Console browser (F12) saat berada di halaman pencarian X → Tekan Enter → Scroll halaman untuk load tweet → Ketik downloadTweets() → File .txt terunduh otomatis.
""", unsafe_allow_html=True) collect_js = '''// ============================================================ // collect.js — X Tweet Scraper // Jalankan di Console browser saat berada di halaman pencarian X // Keyword yang digunakan: "Bitcoin" (tab: Latest) // ============================================================ (function() { // Menyimpan semua tweet yang sudah dikumpulkan (Set mencegah duplikat) window._collectedTweets = window._collectedTweets || new Set(); window._tweetList = window._tweetList || []; // Fungsi utama: scrape semua tweet yang saat ini tampil di halaman function scrapeTweets() { // Selector untuk artikel tweet di X.com const tweetArticles = document.querySelectorAll('article[data-testid="tweet"]'); let newCount = 0; tweetArticles.forEach(article => { try { // ── Ambil username (handle @...) ────────────────────────── const userEl = article.querySelector('[data-testid="User-Name"]'); const username = userEl ? userEl.innerText.replace(/\n/g, ' ').trim() : 'unknown'; // ── Ambil timestamp ─────────────────────────────────────── const timeEl = article.querySelector('time'); const datetime = timeEl ? timeEl.getAttribute('datetime') // format ISO: 2026-04-16T14:30:00.000Z : new Date().toISOString(); // ── Ambil teks tweet ────────────────────────────────────── const textEl = article.querySelector('[data-testid="tweetText"]'); const tweetText = textEl ? textEl.innerText.trim() : ''; // Skip tweet kosong if (!tweetText) return; // Buat unique key untuk mencegah duplikat const key = username + '|' + datetime + '|' + tweetText.substring(0, 50); if (!window._collectedTweets.has(key)) { window._collectedTweets.add(key); // Format sesuai yang diharapkan source code Python: // "username | datetime" // "isi tweet" const dateFormatted = datetime.replace('T', ' ').replace(/\\.\\d+Z$/, '').replace('Z', ''); window._tweetList.push({ meta: username + ' | ' + dateFormatted, content: tweetText }); newCount++; } } catch (e) { // Skip tweet yang gagal diproses } }); console.log(`[collect.js] +${newCount} tweet baru | Total: ${window._tweetList.length}`); } // ── Auto-scrape setiap 2 detik saat halaman di-scroll ──────────────── if (window._scrapeInterval) { clearInterval(window._scrapeInterval); } window._scrapeInterval = setInterval(scrapeTweets, 2000); // Jalankan sekali langsung saat skrip diload scrapeTweets(); // ── Fungsi download — ketik downloadTweets() di Console ────────────── window.downloadTweets = function(filename) { if (window._tweetList.length === 0) { console.warn('[collect.js] Belum ada tweet terkumpul. Scroll halaman lebih banyak dulu.'); return; } // Susun konten file: setiap tweet dipisah baris kosong const lines = window._tweetList.map(t => t.meta + '\\n' + t.content); const content = lines.join('\\n\\n'); // Tentukan nama file otomatis berdasarkan tanggal tweet pertama if (!filename) { const firstDate = window._tweetList[0].meta.split(' | ')[1]; const dateStr = firstDate ? firstDate.split(' ')[0].replace(/-/g, '') : 'tweets'; filename = dateStr + '_bitcoin.txt'; } // Buat blob dan trigger download const blob = new Blob([content], { type: 'text/plain;charset=utf-8' }); const url = URL.createObjectURL(blob); const a = document.createElement('a'); a.href = url; a.download = filename; a.click(); URL.revokeObjectURL(url); console.log(`[collect.js] ✅ Download: ${filename} (${window._tweetList.length} tweet)`); }; // ── Fungsi reset — ketik resetTweets() untuk mulai hari baru ───────── window.resetTweets = function() { window._collectedTweets = new Set(); window._tweetList = []; console.log('[collect.js] 🔄 Data direset. Siap scraping hari baru.'); }; // ── Fungsi status ───────────────────────────────────────────────────── window.statusTweets = function() { console.log(`[collect.js] 📊 Total tweet: ${window._tweetList.length}`); if (window._tweetList.length > 0) { console.log(' Pertama:', window._tweetList[0].meta); console.log(' Terakhir:', window._tweetList[window._tweetList.length - 1].meta); } }; console.log('[collect.js] ✅ Skrip aktif!'); console.log(' → Scroll halaman X untuk load lebih banyak tweet'); console.log(' → Ketik downloadTweets() untuk download file .txt'); console.log(' → Ketik resetTweets() untuk mulai hari baru'); console.log(' → Ketik statusTweets() untuk cek jumlah tweet'); })();''' st.code(collect_js, language="javascript") st.markdown("""
⚠️ Perhatian: Beberapa browser (terutama Chrome) menampilkan peringatan saat paste skrip ke Console. Jika diminta, ketik allow pasting lalu tekan Enter, kemudian paste ulang skripnya.
""", unsafe_allow_html=True) with tab_format: st.markdown("""
File .txt yang diunggah harus mengikuti format berikut agar dapat dibaca oleh sistem. Setiap tweet terdiri dari 2 baris dan dipisahkan oleh satu baris kosong.
""", unsafe_allow_html=True) col_f1, col_f2 = st.columns(2) with col_f1: st.markdown("**✅ Format yang benar:**") st.code( "username | 2026-04-16 14:30:00\n" "Bitcoin is looking bullish today! Great news for crypto holders.\n\n" "another_user | 2026-04-16 15:45:00\n" "BTC just hit 75k, incredible run. When moon?\n\n" "crypto_analyst | 2026-04-16 16:20:00\n" "Bearish divergence on BTC 4H chart. Be careful traders.", language="text" ) with col_f2: st.markdown("**❌ Format yang salah:**") st.code( "# Jangan ada header CSV\n" "date,user,tweet\n\n" "# Jangan ada spasi ganda antar tweet\n\n\n" "user | 2026-04-16\n" "tweet...\n\n\n" "# Tanggal harus ada jamnya\n" "user | 2026-04-16\n" "tweet tanpa jam...", language="text" ) st.markdown("""

📁 Konvensi Penamaan File

• Satu file = satu hari data
• Nama file: 1.txt (hari ke-1), 2.txt (hari ke-2), dst.
• Sistem mengurutkan file secara alfanumerik sebelum diproses
• Tidak ada batasan jumlah tweet per file
• Encoding: UTF-8 (default output collect.js)
""", unsafe_allow_html=True) with tab_tips: tips_data = [ ("🎯", "Target Minimal Data", "Gunakan minimal 30 hari data untuk hasil korelasi yang bermakna secara statistik. Dengan n=30, nilai r ≥ 0.35 sudah signifikan pada p < 0.05. Makin banyak hari, makin kuat reliabilitas temuan."), ("📊", "Jumlah Tweet per Hari", "Targetkan 200–500 tweet per hari. Terlalu sedikit (< 50 tweet) membuat rata-rata sentimen harian tidak representatif. Scroll halaman X selama 1–2 menit per hari untuk mendapatkan jumlah yang cukup."), ("🔍", "Kata Kunci yang Tepat", "Penelitian ini menggunakan kata kunci tunggal \"Bitcoin\" (tanpa tanda petik di X). Pastikan memilih tab Latest, bukan Top, agar distribusi temporal merata dan tidak bias ke tweet viral."), ("📅", "Konsistensi Periode Waktu", "Gunakan filter tanggal X untuk memastikan setiap file hanya berisi tweet dari satu hari kalender. Contoh: Bitcoin since:2026-04-16 until:2026-04-17. Ini penting agar agregasi harian akurat."), ("🌐", "Bahasa Tweet", "Sistem otomatis memfilter tweet non-Inggris menggunakan langdetect. Dari pengalaman penelitian ini, sekitar 13–14% tweet diskip karena bukan bahasa Inggris. Ini normal dan sudah diperhitungkan."), ("💾", "Reset Antar Hari", "Setelah download file untuk satu hari, selalu ketik resetTweets() di Console sebelum pindah ke hari berikutnya. Ini mencegah tweet dari hari sebelumnya ikut masuk ke file hari berikutnya."), ("⚡", "Performa Browser", "Tutup tab lain yang tidak diperlukan saat scraping untuk mencegah browser melambat. Jika halaman X berhenti load tweet setelah scroll panjang, refresh halaman dan jalankan ulang collect.js (data sebelumnya akan hilang)."), ("🔒", "Batas Scraping X", "Platform X membatasi scraping agresif. Jika halaman tiba-tiba tidak menampilkan tweet baru meski di-scroll, tunggu 5–10 menit sebelum melanjutkan. Alternatif: gunakan akun berbeda atau ganti IP."), ] col_t1, col_t2 = st.columns(2) for i, (icon, title, desc) in enumerate(tips_data): col = col_t1 if i % 2 == 0 else col_t2 with col: st.markdown(f"""
{icon} {title}
{desc}
""", unsafe_allow_html=True) st.markdown("
", unsafe_allow_html=True) st.markdown('
', unsafe_allow_html=True) if tweet_files and analyze_batch_btn: scroll_to_target("target-analisis-batch") col_b_space1, col_b_content, col_b_space2 = st.columns([1, 8, 1]) with col_b_content: st.markdown("""

Hasil Pemrosesan

Dashboard Analisis

""", unsafe_allow_html=True) tweet_files = sorted(tweet_files, key=lambda x: x.name) data = [] with st.status("🔄 Memproses data sentimen...", expanded=True) as status: progress_bar = st.progress(0, text="Mengekstrak sentimen dari data...") total_tweets_uploaded = 0 total_tweets_skipped = 0 for idx, file in enumerate(tweet_files): content = file.getvalue().decode("utf-8").replace("\r\n", "\n").strip() tweets = content.split("\n\n") for tweet in tweets: parts = tweet.strip().split("\n", 1) if len(parts) != 2: continue meta, text_raw = parts try: DetectorFactory.seed = 0 lang = detect(text_raw) if lang != 'en': total_tweets_skipped += 1 continue except: total_tweets_skipped += 1 continue username, date_val = meta.split(" | ") if " | " in meta else ("unknown", "unknown") short_date = date_val[:10] text = clean_text(text_raw) if not text.strip(): total_tweets_skipped += 1 continue try: v_compound = vader.polarity_scores(text)['compound'] vader_label = "positive" if v_compound > 0.05 else ("negative" if v_compound < -0.05 else "neutral") except: v_compound = 0.0 vader_label = "neutral" try: tb_polarity = TextBlob(text).sentiment.polarity tb_label = classify_tb(tb_polarity) except: tb_polarity = 0.0 tb_label = "neutral" try: bertweet_label = map_bertweet(bertweet(text)[0]['label']) except: bertweet_label = "neutral" try: roberta_label = map_roberta(roberta(text)[0]['label']) except: roberta_label = "neutral" try: roberta_large_label = roberta_large(text)[0]['label'].lower() except: roberta_large_label = "neutral" data.append({ "date": short_date, "raw_tweet": text_raw.strip(), "cleaned_tweet": text, "vader": vader_label, "textblob": tb_label, "bertweet": bertweet_label, "roberta": roberta_label, "roberta_large": roberta_large_label, "vader_score": v_compound, "tb_score": tb_polarity, }) total_tweets_uploaded += 1 progress_bar.progress((idx + 1) / len(tweet_files), text=f"Memproses file {idx+1} dari {len(tweet_files)}") status.update(label="✅ Pemrosesan sentimen teks selesai!", state="complete", expanded=False) df = pd.DataFrame(data) if df.empty: st.error("❌ Data kosong. Pastikan format TXT benar dan tweet berbahasa Inggris.") else: col_m1, col_m2, col_m3 = st.columns(3) col_m1.metric("Tweet Diproses", f"{total_tweets_uploaded}", border=True) col_m2.metric("Tweet Diabaikan (Non-EN)", f"{total_tweets_skipped}", border=True) col_m3.metric("Model", "5 Model", border=True) target_dates = sorted(df['date'].unique()) start_unix = int(datetime.strptime(target_dates[0], "%Y-%m-%d").replace(tzinfo=timezone.utc).timestamp()) - 86400 end_unix = int(datetime.strptime(target_dates[-1], "%Y-%m-%d").replace(tzinfo=timezone.utc).timestamp()) + 86400 # ============================================================== # FUNGSI FETCH HARGA BTC # ============================================================== def fetch_via_coingecko(start_ts, end_ts): """Coba CoinGecko dengan retry eksponensial. Return list [[ts_ms, price]] atau raise.""" url = "https://api.coingecko.com/api/v3/coins/bitcoin/market_chart/range" params = {"vs_currency": "usd", "from": start_ts, "to": end_ts} headers = {"accept": "application/json", "User-Agent": "Mozilla/5.0"} wait_times = [5, 15, 30] for attempt, wait in enumerate(wait_times, 1): time.sleep(wait) res = requests.get(url, params=params, headers=headers, timeout=20) if res.status_code == 200: data_j = res.json() if "prices" in data_j: return data_j["prices"] raise ValueError("Key 'prices' tidak ada di respons CoinGecko.") if res.status_code == 429: if attempt < len(wait_times): continue # coba lagi dengan backoff lebih lama raise ConnectionError(f"CoinGecko 429 setelah {attempt} percobaan.") raise ConnectionError(f"CoinGecko error {res.status_code}: {res.text[:200]}") raise ConnectionError("CoinGecko gagal setelah semua percobaan.") def fetch_via_binance(start_ts, end_ts): """ Fallback: Binance Public API — BTCUSDT daily klines. Endpoint bebas API key, limit 1000 candles per request. """ url = "https://api.binance.com/api/v3/klines" prices = [] cur_start = start_ts * 1000 end_ms = end_ts * 1000 while cur_start < end_ms: params = { "symbol": "BTCUSDT", "interval": "1d", "startTime": cur_start, "endTime": end_ms, "limit": 1000, } res = requests.get(url, params=params, timeout=20) if res.status_code != 200: raise ConnectionError(f"Binance error {res.status_code}: {res.text[:200]}") batch = res.json() if not batch: break for candle in batch: open_time = int(candle[0]) # ms close_price = float(candle[4]) # close price prices.append([open_time, close_price]) cur_start = int(batch[-1][0]) + 1 if len(batch) < 1000: break if not prices: raise ValueError("Binance tidak mengembalikan data.") return prices def build_df_price(raw_prices, target_date_list): """Bersihkan raw [[ts_ms, price]] → df_price dengan log_return.""" df_p = pd.DataFrame(raw_prices, columns=["timestamp", "price"]) df_p["date"] = pd.to_datetime(df_p["timestamp"], unit="ms").dt.date df_p = df_p.groupby("date")["price"].mean().reset_index() df_p["pct_change"] = df_p["price"].pct_change() * 100 df_p["log_return"] = np.log(df_p["price"] / df_p["price"].shift(1)) df_p.dropna(inplace=True) df_p = df_p[df_p["date"].isin(pd.to_datetime(target_date_list).date)] return df_p raw_prices = None api_source = None with st.spinner("📡 Mengambil data harga Bitcoin..."): try: raw_prices = fetch_via_coingecko(start_unix, end_unix) api_source = "CoinGecko" except Exception as cg_err: st.warning( f"⚠️ CoinGecko tidak tersedia ({cg_err}). " "Beralih ke **Binance Public API** sebagai fallback..." ) try: raw_prices = fetch_via_binance(start_unix, end_unix) api_source = "Binance" except Exception as bn_err: st.error( f"❌ Kedua sumber data gagal.\n" f"- CoinGecko: {cg_err}\n" f"- Binance: {bn_err}\n\n" "Coba lagi beberapa menit kemudian atau periksa koneksi internet." ) if raw_prices is not None: try: df_price = build_df_price(raw_prices, target_dates) st.info(f"✅ Data harga BTC berhasil diambil dari **{api_source}**.") if df_price.empty: st.warning("⚠️ Data Harga API kosong. Pastikan rentang tanggal di .txt sesuai (yyyy-mm-dd).") else: st.markdown("
", unsafe_allow_html=True) # ── Tabel Data Sentimen ────────────────────────────── st.markdown("🗣️ Data Sentimen") raw_display_cols = ["date","raw_tweet","vader","textblob","bertweet","roberta","roberta_large","vader_score","tb_score"] st.dataframe(df[raw_display_cols], use_container_width=True, hide_index=True) # ================================================== # AGREGASI HARIAN DUAL-MODE # Mode A (Kategorik): konversi {pos:1, neu:0, neg:-1} # Mode B (Numerik): rata-rata vader_score & tb_score # ================================================== sentiment_map = {"positive": 1, "neutral": 0, "negative": -1} df_score = df.copy() models_cat = ["vader","textblob","bertweet","roberta","roberta_large"] for col in models_cat: df_score[col] = df_score[col].map(sentiment_map) # Agregasi kategorik (−1/0/1 mean) df_sentiment_daily = df_score.groupby("date")[models_cat].mean().reset_index() df_sentiment_daily["date"] = pd.to_datetime(df_sentiment_daily["date"]).dt.date # Agregasi numerik VADER compound & TextBlob polarity df_numeric_daily = df.groupby("date")[["vader_score","tb_score"]].mean().reset_index() df_numeric_daily["date"] = pd.to_datetime(df_numeric_daily["date"]).dt.date for col in models_cat: df_sentiment_daily[f"{col}_label"] = df_sentiment_daily[col].apply(get_daily_label) daily_display_cols = ["date"] for col in models_cat: daily_display_cols.extend([col, f"{col}_label"]) # ── Tabel Harga Bitcoin ─────────────────────────── st.markdown("₿ Data Harga & Volatilitas Bitcoin") st.dataframe(df_price[["date","price","pct_change","log_return"]], use_container_width=True, hide_index=True) # Merge data df_merged = pd.merge(df_price, df_sentiment_daily, on="date", how="inner") df_merged = pd.merge(df_merged, df_numeric_daily, on="date", how="inner") # ── Tabel Data Final ────────────────────────────── st.markdown("🗂️ Data Final") final_display_cols = ( ["date","price","pct_change","log_return"] + [c for c in daily_display_cols if c != "date"] + ["vader_score","tb_score"] ) st.dataframe(df_merged[final_display_cols], use_container_width=True, hide_index=True) # Download buttons col_dl1, col_dl2, _ = st.columns([1, 1, 3]) csv_data = df_merged.to_csv(index=False).encode('utf-8') col_dl1.download_button("📥 Unduh CSV", data=csv_data, file_name="bitcoin_volatility_sentiment.csv", mime="text/csv", use_container_width=True) buffer = io.BytesIO() with pd.ExcelWriter(buffer, engine='xlsxwriter') as writer: df_merged.to_excel(writer, index=False) col_dl2.download_button("📥 Unduh Excel", data=buffer.getvalue(), file_name="bitcoin_volatility_sentiment.xlsx", mime="application/vnd.ms-excel", use_container_width=True) st.markdown("
", unsafe_allow_html=True) # ================================================== # UJI KORELASI PEARSON DENGAN LAG # Lag 0 : sentimen hari t vs harga hari t (same-day) # Lag +1: sentimen hari t vs harga hari t+1 # Lag +2: sentimen hari t vs harga hari t+2 # # Kolom korelasi yang diuji: # - vader_score (numerik compound, −1 s.d. 1) # - tb_score (numerik polarity, −1 s.d. 1) # - bertweet (kategorik −1/0/1 mean) # - roberta (kategorik −1/0/1 mean) # - roberta_large(kategorik −1/0/1 mean) # ================================================== st.subheader("🔬 Uji Korelasi Pearson") st.caption( "Menganalisis hubungan statistik antara skor sentimen harian dan " "volatilitas log-return BTC pada lag 0 (hari sama), +1 hari, dan +2 hari. " "VADER & TextBlob menggunakan skor numerik kontinu; model BERT menggunakan " "rata-rata kategorikal (−1/0/1)." ) corr_columns = { "VADER (numerik)": "vader_score", "TextBlob (numerik)": "tb_score", "BERTweet (kategorik)": "bertweet", "RoBERTa Base (kategorik)": "roberta", "RoBERTa Large (kategorik)":"roberta_large", } corr_rows = [] raw_results = [] n_obs = len(df_merged) for label_name, col_key in corr_columns.items(): for lag in [0, 1, 2]: # shift(-lag): harga mundur lag hari ke depan # artinya: sentimen t berkorelasi dengan harga t+lag log_ret_shifted = df_merged["log_return"].shift(-lag) valid_mask = log_ret_shifted.notna() x_vals = df_merged.loc[valid_mask, col_key] y_vals = log_ret_shifted[valid_mask] if len(x_vals) < 4: corr, pval = np.nan, np.nan else: try: corr, pval = pearsonr(x_vals, y_vals) except Exception: corr, pval = np.nan, np.nan arah = "Positif" if (corr is not np.nan and corr > 0) else "Negatif" sig = "✅ Signifikan" if (pval is not np.nan and pval < 0.05) else "Tidak Signifikan" corr_rows.append({ "Metode": label_name, "Lag": f"t+{lag}", "r": f"{corr:.4f}" if not np.isnan(corr) else "N/A", "Arah": arah, "p-value": f"{pval:.4f}" if not np.isnan(pval) else "N/A", "Signifikansi": sig, }) raw_results.append({ "metode": label_name, "col": col_key, "lag": lag, "r": corr if not np.isnan(corr) else 0.0, "p": pval if not np.isnan(pval) else 1.0, }) df_corr_table = pd.DataFrame(corr_rows) # Tandai baris terbaik per metode (|r| terbesar) def highlight_best(row): try: r_abs = abs(float(row["r"])) except: r_abs = 0 sig_ok = "✅" in str(row["Signifikansi"]) if sig_ok: return ["background-color:#e6fff1; font-weight:600"] * len(row) return [""] * len(row) styled = df_corr_table.style.apply(highlight_best, axis=1) st.dataframe(styled, use_container_width=True, hide_index=True) # ── Scatter Plot ────────────────────────────────── st.subheader("🔵 Pola Distribusi Scatter Plot (Lag t+0)") scatter_cols = list(corr_columns.values()) cols_sc = st.columns(3) for idx2, (disp_name, col_key) in enumerate(corr_columns.items()): with cols_sc[idx2 % 3]: fig_s, ax_s = plt.subplots(figsize=(5, 4)) sns.regplot( data=df_merged, x=col_key, y="log_return", ax=ax_s, scatter_kws={"s": 40, "color": "#10b981", "alpha": 0.5}, line_kws={"color": "#0f172a", "linewidth": 2} ) # Tampilkan r dan p pada plot try: r_val, p_val = pearsonr(df_merged[col_key], df_merged["log_return"]) ax_s.set_title(f"{disp_name.split(' ')[0]}\nr={r_val:.3f}, p={p_val:.3f}", fontweight='bold', fontsize=9) except: ax_s.set_title(disp_name.split(' ')[0], fontweight='bold') ax_s.set_xlabel("Sentimen Score") ax_s.set_ylabel("Log Return") plt.tight_layout() st.pyplot(fig_s) # ── Line Chart ──────────────────────────────────── st.subheader("📈 Trend Analisis: Sentiment vs BTC Volatility") fig_line, ax_line = plt.subplots(figsize=(14, 6)) ax_line.plot( df_merged["date"], df_merged["log_return"], label="BTC Log Return", color="#f7931a", linewidth=3 ) colors_line = ["#3B82F6","#10B981","#EC4899","#14B8A6","#6366F1"] for i, (disp_name, col_key) in enumerate(corr_columns.items()): ax_line.plot( df_merged["date"], df_merged[col_key], label=f"Sentimen: {disp_name.split(' ')[0]}", color=colors_line[i], linewidth=1.5, linestyle="--", alpha=0.8 ) ax_line.set_title("Pergerakan Sentimen vs Log Return Bitcoin", fontsize=14, pad=15, fontweight='bold') ax_line.set_xlabel("Tanggal", fontsize=11) ax_line.set_ylabel("Nilai Metrik", fontsize=11) ax_line.legend(loc='upper left', bbox_to_anchor=(1, 1), frameon=True) plt.tight_layout() st.pyplot(fig_line) # ── KESIMPULAN ──────────────────────────────────── st.markdown("
", unsafe_allow_html=True) st.subheader("📝 Kesimpulan") max_idx = df_merged["log_return"].idxmax() min_idx = df_merged["log_return"].idxmin() date_max = df_merged.loc[max_idx, "date"] date_min = df_merged.loc[min_idx, "date"] st.write( f"Puncak lonjakan positif (*max log return*) terjadi pada **{date_max}**, " f"sedangkan penurunan ekstrem terjadi pada **{date_min}**. " f"Dataset mencakup **{n_obs} hari** pengamatan." ) # Cari lag & metode terbaik (|r| terbesar + signifikan) sig_results = [r for r in raw_results if r["p"] < 0.05] all_results = raw_results if sig_results: best = max(sig_results, key=lambda x: abs(x["r"])) arah_text = "berbanding lurus (positif)" if best["r"] > 0 else "berbanding terbalik (negatif)" sig_summary = ", ".join( set(f"{r['metode'].split(' ')[0]} (lag t+{r['lag']})" for r in sig_results) ) st.success(f""" **Hipotesis Diterima (H1):** Ditemukan korelasi linier yang signifikan (*p-value* < 0.05) pada: **{sig_summary}**. Metode & lag dengan korelasi terkuat adalah **{best['metode']} (lag t+{best['lag']})** dengan r = **{best['r']:.4f}**, sifat hubungan **{arah_text}**. """) else: best_overall = max(all_results, key=lambda x: abs(x["r"])) st.warning(f""" **Hipotesis Ditolak (H0 Diterima):** Belum ditemukan bukti empiris korelasi linier yang signifikan pada seluruh metode dan lag yang diuji (*p-value* ≥ 0.05). Korelasi terbesar ditemukan pada **{best_overall['metode']} (lag t+{best_overall['lag']})** dengan r = **{best_overall['r']:.4f}**. Volatilitas harga kemungkinan dipengaruhi faktor teknikal/fundamental di luar sentimen X, atau jumlah observasi belum mencukupi (n = {n_obs}, butuh minimal 30 hari agar r ≥ 0.35 bisa signifikan). """) except Exception as e: st.error(f"⚠️ Terjadi kesalahan saat memproses data harga Bitcoin: {e}") elif analyze_batch_btn and not tweet_files: st.warning("⚠️ Silakan unggah minimal satu file .txt terlebih dahulu.")