Repository navigation
Expand file tree
/
Copy pathconfig.py
More file actions
113 lines (96 loc) · 4.85 KB
/
Copy pathconfig.py
File metadata and controls
113 lines (96 loc) · 4.85 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
"""
Configuration for the Life Interest Map pipeline.
"""
import os
from pathlib import Path
from dotenv import load_dotenv
load_dotenv()
# ── Paths ──────────────────────────────────────────────────────────────
PROJECT_ROOT = Path(__file__).resolve().parent
DATA_DIR = PROJECT_ROOT / "data"
OUTPUT_DIR = PROJECT_ROOT / "output"
EMBEDDINGS_FILE = DATA_DIR / "embeddings.npz"
THUMBNAILS_FILE = DATA_DIR / "thumbnails.npz"
PROGRESS_FILE = DATA_DIR / "progress.json"
CLUSTER_FILE = DATA_DIR / "clusters.npz"
# ── iCloud ─────────────────────────────────────────────────────────────
ICLOUD_USERNAME = "" # Set via environment variable ICLOUD_USERNAME or edit here
ICLOUD_PASSWORD = "" # Set via environment variable ICLOUD_PASSWORD or edit here
# ── CLIP Model ─────────────────────────────────────────────────────────
CLIP_MODEL_NAME = "ViT-B-32"
CLIP_PRETRAINED = "laion2b_s34b_b79k"
CLIP_BATCH_SIZE = 48 # Tuned for 8GB VRAM
DEVICE = "cuda" # Force GPU
# ── Thumbnails ─────────────────────────────────────────────────────────
THUMBNAIL_SIZE = (64, 64)
THUMBNAIL_QUALITY = 60 # JPEG quality for base64 thumbnails
# ── Video Keyframes ────────────────────────────────────────────────────
MAX_KEYFRAMES_PER_VIDEO = 3
KEYFRAME_INTERVAL_SECONDS = 10 # Extract 1 frame every N seconds
# ── UMAP ───────────────────────────────────────────────────────────────
UMAP_N_NEIGHBORS = 15
UMAP_MIN_DIST = 0.1
UMAP_METRIC = "cosine"
UMAP_N_COMPONENTS_CLUSTER = 50 # For clustering
UMAP_N_COMPONENTS_VIS = 2 # For visualization
# ── HDBSCAN ────────────────────────────────────────────────────────────
HDBSCAN_MIN_CLUSTER_SIZE = 20
HDBSCAN_MIN_SAMPLES = 5
# ── Gemini (LLM Cluster Labeling) ──────────────────────────────────────
GEMINI_MODEL = "gemini-3-flash-preview"
GEMINI_REPRESENTATIVE_IMAGES = 20 # Thumbnails per cluster to send to Gemini
# ── Local LLM (llama-server) ──────────────────────────────────────────
LOCAL_LLM_URL = "http://localhost:8080"
LOCAL_LLM_IMAGES_PER_CLUSTER = 10 # Fewer images to fit in 8GB VRAM
# ── Spending Map ───────────────────────────────────────────────────────
SPENDING_DATA_DIR = DATA_DIR / "spending"
SPENDING_TRANSACTIONS_FILE = SPENDING_DATA_DIR / "transactions.json"
SPENDING_EMBEDDINGS_FILE = SPENDING_DATA_DIR / "embeddings.npz"
SPENDING_CLUSTERS_FILE = SPENDING_DATA_DIR / "clusters.npz"
SPENDING_MODEL_NAME = "all-MiniLM-L6-v2"
SPENDING_BATCH_SIZE = 256
SPENDING_HDBSCAN_MIN_CLUSTER_SIZE = 8
SPENDING_HDBSCAN_MIN_SAMPLES = 3
SPENDING_DEFAULT_CURRENCY = os.environ.get("SPENDING_DEFAULT_CURRENCY", "USD")
# Plaid (optional)
PLAID_CLIENT_ID = os.environ.get("PLAID_CLIENT_ID", "")
PLAID_SECRET = os.environ.get("PLAID_SECRET", "")
PLAID_ENV = os.environ.get("PLAID_ENV", "sandbox")
# ── Cluster Labels ─────────────────────────────────────────────────────
CANDIDATE_LABELS = [
"food and cooking",
"friends hanging out",
"coding and programming",
"memes and humor",
"nature and outdoors",
"gym and fitness",
"travel and vacation",
"selfies and portraits",
"music and concerts",
"gaming",
"work and office",
"screenshots of text and messages",
"night out and parties",
"architecture and buildings",
"pets and animals",
"sports",
"shopping",
"art and design",
"cars and vehicles",
"family",
"beach and water",
"books and reading",
"sunset and sunrise",
"city and urban",
"mountain and hiking",
"coffee and drinks",
"fashion and outfits",
"home and interior",
"sky and clouds",
"flowers and plants",
"documents and receipts",
"maps and navigation",
"social media screenshots",
"app interfaces",
"graphs and charts",
]