from __future__ import annotations import json import sys import os sys.path.insert(0, os.path.join(os.path.dirname(os.path.abspath(__file__)), 'src')) from pathlib import Path import numpy as np import pandas as pd import streamlit as st from human_condition.viz.charts import ( apply_theme, compression_curve, emotion_heatmap, radar_chart, similarity_matrix, source_distribution, summary_stats, topic_barchart, topic_scatter, word_cloud_data, emotion_timeline_chart, ) from human_condition.viz.theme import ( ACCENT_GOLD, ACCENT_CYAN, ACCENT_PINK, ACCENT_PURPLE, BG_CARD, BG_PRIMARY, FONT_FAMILY, PALETTE, PLOTLY_TEMPLATE, TEXT_PRIMARY, ) # ── Paths ────────────────────────────────────────────────────────── DATA_DIR = Path(__file__).parent / "data" WAREHOUSE_DIR = DATA_DIR / "warehouse" PROCESSED_DIR = DATA_DIR / "processed" # ── Page Config ──────────────────────────────────────────────────── st.set_page_config( page_title="The Human Condition", page_icon="\U0001f30d", layout="wide", initial_sidebar_state="expanded", ) # ── Custom CSS ───────────────────────────────────────────────────── st.markdown( f""" """, unsafe_allow_html=True, ) # ── Demo Mode Sample Data ────────────────────────────────────────── def _make_demo_corpus() -> list[dict]: """Generate sample corpus data for demo mode.""" sources = { "quran": [ "In the name of Allah, the Most Gracious, the Most Merciful.", "All praise is due to Allah, Lord of all the worlds.", "Guide us to the straight path, the path of those upon whom You have bestowed favor.", "Read in the name of your Lord who created, created man from a clinging substance.", "And your Lord is most generous, who taught by the pen, taught man that which he knew not.", ], "bible_kjv": [ "In the beginning God created the heaven and the earth.", "And God said, Let there be light: and there was light.", "The Lord is my shepherd; I shall not want.", "For God so loved the world, that he gave his only begotten Son.", "Love is patient, love is kind. It does not envy, it does not boast.", ], "bhagavad_gita": [ "You have the right to work, but never to the fruit of work.", "The soul can neither be cut into pieces nor burnt.", "For the soul there is neither birth nor death at any time.", "Perform your duty equipoised, abandoning all attachment to results.", "The mind is restless and hard to control, but it can be trained by constant practice.", ], "tao_te_ching": [ "The Tao that can be told is not the eternal Tao.", "Be still like a mountain and flow like a great river.", "Knowing others is intelligence; knowing yourself is true wisdom.", "The journey of a thousand miles begins with a single step.", "When I let go of what I am, I become what I might be.", ], "communist_manifesto": [ "A spectre is haunting Europe, the spectre of communism.", "The history of all hitherto existing society is the history of class struggles.", "The proletarians have nothing to lose but their chains.", "Working men of all countries, unite!", "In place of the old bourgeois society we shall have an association in which the free development of each is the condition for the free development of all.", ], "plato_republic": [ "The price good men pay for indifference to public affairs is to be ruled by evil men.", "One of the penalties for refusing to participate in politics is that you end up being governed by your inferiors.", "The direction in which education starts a man will determine his future in life.", "Ignorance, the root and stem of all evil.", "The measure of a man is what he does with power.", ], "reddit_philosophy": [ "Does free will exist if our choices are determined by our brain chemistry?", "I think the hardest part of existentialism is accepting that meaning is subjective.", "The trolley problem isn't really about ethics, it's about how we frame moral dilemmas.", "What if consciousness is just an emergent property of complex information processing?", "Stoicism isn't about suppressing emotions, it's about choosing which ones to act on.", ], } corpus = [] for source, texts in sources.items(): for i, text in enumerate(texts): corpus.append({ "source": source, "title": f"{source.replace('_', ' ').title()} - Passage {i+1}", "text": text, "metadata": {"source_type": "scripture" if source in ("quran", "bible_kjv") else "philosophy"}, }) return corpus def _make_demo_emotions(corpus: list[dict]) -> list[dict]: """Generate sample emotion data for demo mode.""" demo_emotions = { "quran": {"awe": 0.4, "joy": 0.3, "fear": 0.1, "sadness": 0.05, "anger": 0.05, "surprise": 0.05, "love": 0.05}, "bible_kjv": {"awe": 0.35, "joy": 0.3, "love": 0.15, "sadness": 0.08, "fear": 0.05, "anger": 0.02, "surprise": 0.05}, "bhagavad_gita": {"joy": 0.35, "awe": 0.25, "love": 0.15, "sadness": 0.1, "fear": 0.05, "surprise": 0.05, "anger": 0.05}, "tao_te_ching": {"joy": 0.3, "surprise": 0.2, "love": 0.2, "awe": 0.15, "sadness": 0.05, "fear": 0.05, "anger": 0.05}, "communist_manifesto": {"anger": 0.35, "joy": 0.15, "fear": 0.15, "sadness": 0.15, "awe": 0.1, "surprise": 0.05, "love": 0.05}, "plato_republic": {"joy": 0.2, "sadness": 0.2, "anger": 0.15, "awe": 0.15, "fear": 0.1, "surprise": 0.1, "love": 0.1}, "reddit_philosophy": {"sadness": 0.3, "joy": 0.2, "surprise": 0.15, "anger": 0.1, "fear": 0.1, "awe": 0.1, "love": 0.05}, } emotions = [] for doc in corpus: src = doc["source"] scores = demo_emotions.get(src, {"joy": 0.2, "sadness": 0.2, "anger": 0.2, "fear": 0.2, "awe": 0.1, "love": 0.05, "surprise": 0.05}) dominant = max(scores, key=scores.get) emotions.append({ "source": src, "title": doc["title"], "text": doc["text"], "dominant_emotion": dominant, "emotion_scores": scores, }) return emotions def _make_demo_topics(corpus: list[dict]) -> list[dict]: """Generate sample topic data for demo mode.""" source_topics = { "quran": "divine guidance and moral accountability", "bible_kjv": "divine love and covenant relationship", "bhagavad_gita": "duty and detachment from results", "tao_te_ching": "harmony with natural order", "communist_manifesto": "class struggle and collective action", "plato_republic": "justice and the ideal state", "reddit_philosophy": "existential meaning and consciousness", } topics = [] for doc in corpus: topics.append({ "source": doc["source"], "title": doc["title"], "text": doc["text"], "topic_label": source_topics.get(doc["source"], "general"), "topic_id": str(hash(doc["source"]) % 10), }) return topics # ── Data Loading ─────────────────────────────────────────────────── @st.cache_data(ttl=3600) def load_data(): """Load all data from parquet files or demo mode.""" data = { "corpus": [], "emotions": [], "topics": [], "embeddings": None, "is_demo": False, } # Try loading from parquet (data/warehouse/) corpus_path = WAREHOUSE_DIR / "corpus.parquet" emo_path = WAREHOUSE_DIR / "emotions.parquet" topic_path = WAREHOUSE_DIR / "topics.parquet" emb_path = PROCESSED_DIR / "embeddings.npy" # Also check warehouse dir for embeddings emb_path_wh = WAREHOUSE_DIR / "embeddings.npy" has_corpus = corpus_path.exists() has_emo = emo_path.exists() has_topics = topic_path.exists() has_embeddings = emb_path.exists() or emb_path_wh.exists() if has_corpus: df = pd.read_parquet(str(corpus_path)) data["corpus"] = df.to_dict(orient="records") if has_emo: df = pd.read_parquet(str(emo_path)) # Reconstruct emotion_scores from emotion_* columns records = df.to_dict(orient="records") for rec in records: rec["emotion_scores"] = { k.replace("emotion_", ""): v for k, v in rec.items() if k.startswith("emotion_") } data["emotions"] = records if has_topics: df = pd.read_parquet(str(topic_path)) data["topics"] = df.to_dict(orient="records") if has_embeddings: path = str(emb_path) if emb_path.exists() else str(emb_path_wh) data["embeddings"] = np.load(path, allow_pickle=False) # If any data is missing, switch to demo mode if not has_corpus or not has_emo or not has_topics: data["is_demo"] = True data["corpus"] = _make_demo_corpus() data["emotions"] = _make_demo_emotions(data["corpus"]) data["topics"] = _make_demo_topics(data["corpus"]) data["embeddings"] = None return data data = load_data() corpus = data["corpus"] emotions = data["emotions"] topics = data["topics"] embeddings = data["embeddings"] is_demo = data["is_demo"] # ── Demo Mode Banner ─────────────────────────────────────────────── if is_demo: st.info( "\U0001f3ae **Demo Mode** — No pipeline data found. " "Showing sample data from 7 texts. " "Run the full pipeline locally to see real results with 127,000+ documents." ) # ── Sidebar ──────────────────────────────────────────────────────── st.sidebar.markdown( f"