-
Notifications
You must be signed in to change notification settings - Fork 7
Expand file tree
/
Copy pathdocker-compose.yml
More file actions
309 lines (298 loc) · 11.4 KB
/
Copy pathdocker-compose.yml
File metadata and controls
309 lines (298 loc) · 11.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
services:
# ── Postgres (primary store — MVCC, real concurrent writers) ───
# The workbook enrichment run depends on concurrent writers; SQLite would
# hit "database is locked". This is the documented Postgres default actually
# wired up. Credentials below match the DATABASE_URL the api/worker use.
postgres:
image: postgres:16-alpine
container_name: lead-data-postgres
environment:
- POSTGRES_USER=${POSTGRES_USER:-yupcha}
- POSTGRES_PASSWORD=${POSTGRES_PASSWORD:-yupcha}
- POSTGRES_DB=${POSTGRES_DB:-yupcha}
volumes:
- pgdata:/var/lib/postgresql/data
ports:
# Host-local only for admin tools; application containers use lead_net.
- "127.0.0.1:${POSTGRES_PORT:-5432}:5432"
healthcheck:
test: ["CMD-SHELL", "pg_isready -U ${POSTGRES_USER:-yupcha} -d ${POSTGRES_DB:-yupcha}"]
interval: 5s
timeout: 5s
retries: 10
restart: unless-stopped
networks:
- lead_net
# ── Schema migration (single owner; must finish before app startup) ──
migrate:
# Every first-party service deliberately shares one immutable application
# image. This prevents an old one-shot migrator from lagging behind a newly
# rebuilt API/worker image during an upgrade.
image: ${YUPCHA_IMAGE:-lead-data-app:local}
build:
context: .
dockerfile: Dockerfile
env_file: .env
environment:
- DATABASE_URL=${DATABASE_URL:-postgresql+psycopg://yupcha:yupcha@postgres:5432/yupcha}
- PYTHONPATH=/app
- YUPCHA_DB_INIT=alembic
- YUPCHA_APP_DB_ROLE=${YUPCHA_APP_DB_ROLE:-yupcha_app}
- YUPCHA_PROVISION_RUNTIME_ROLE=1
- YUPCHA_RUNTIME_DB_USER=${YUPCHA_RUNTIME_DB_USER:-yupcha_runtime}
- YUPCHA_RUNTIME_DB_PASSWORD=${YUPCHA_RUNTIME_DB_PASSWORD:-yupcha_runtime}
command: python -m apps.api.scripts.migrate
depends_on:
postgres:
condition: service_healthy
restart: "no"
networks:
- lead_net
# ── API (FastAPI backend) ──────────────────────────────
api:
image: ${YUPCHA_IMAGE:-lead-data-app:local}
build:
context: .
dockerfile: Dockerfile
container_name: lead-data-api
user: "${APP_UID:-1000}:${APP_GID:-1000}"
env_file: .env
environment:
# Postgres is the source of truth. This overrides any SQLite DB_PATH and
# makes the documented Postgres default take effect. ${VAR:-default} lets
# .env override without editing this file.
- DATABASE_URL=${APP_DATABASE_URL:-postgresql+psycopg://yupcha_runtime:yupcha_runtime@postgres:5432/yupcha}
- PYTHONPATH=/app
- REDIS_URL=redis://redis:6379
# Self-hosted Reacher endpoint (internal only). Inert unless REACHER_ENABLED
# is also set in .env — default OFF means this URL is simply never called.
- REACHER_URL=${REACHER_URL:-http://reacher:8080}
# Job processing is owned by the standalone `worker` service below, which
# claims jobs atomically (FOR UPDATE SKIP LOCKED) and is safe to scale to
# N replicas. Turn OFF the in-API worker so the API only enqueues — this is
# what makes horizontal scaling double-charge-safe.
- RUN_INLINE_WORKER=0
# The one-shot migrate service above owns schema evolution.
- YUPCHA_DB_INIT=skip
volumes:
# data/ still holds the workspaces.db (sqlite meta) + per-workspace leads;
# the primary store is Postgres.
- ./data:/app/data
ports:
# Host-local diagnostics only. LAN traffic goes through nginx on 3010.
- "127.0.0.1:8000:8000"
depends_on:
migrate:
condition: service_completed_successfully
postgres:
condition: service_healthy
redis:
condition: service_healthy
restart: unless-stopped
healthcheck:
test: ["CMD", "curl", "-f", "http://localhost:8000/health"]
interval: 30s
timeout: 10s
retries: 3
networks:
- lead_net
# ── Seed (first-run demo: admin + populated zero-key workbook) ──
# Runs once after Postgres is healthy, creates the demo workspace/workbook
# and enqueues the first enrichment run. Idempotent — safe on every restart.
seed:
image: ${YUPCHA_IMAGE:-lead-data-app:local}
build:
context: .
dockerfile: Dockerfile
container_name: lead-data-seed
user: "${APP_UID:-1000}:${APP_GID:-1000}"
env_file: .env
environment:
- DATABASE_URL=${DATABASE_URL:-postgresql+psycopg://yupcha:yupcha@postgres:5432/yupcha}
- PYTHONPATH=/app
- REDIS_URL=redis://redis:6379
- YUPCHA_DB_INIT=skip
volumes:
- ./data:/app/data
command: python -m apps.api.scripts.seed_demo
depends_on:
postgres:
condition: service_healthy
# Wait for the API after the one-shot migration has completed. The seed
# enqueues work for the standalone worker.
api:
condition: service_healthy
restart: "no"
networks:
- lead_net
# ── Job Worker (durable SQL queue, FOR UPDATE SKIP LOCKED) ─────
# Standalone processor for the `jobs` table. Claims jobs atomically so it is
# safe to scale horizontally: `docker compose up --scale worker=N`. Each
# replica grabs each queued job exactly once (no double-grab / double-charge).
worker:
image: ${YUPCHA_IMAGE:-lead-data-app:local}
build:
context: .
dockerfile: Dockerfile
# NOTE: no fixed container_name so `--scale worker=N` can run multiple.
user: "${APP_UID:-1000}:${APP_GID:-1000}"
env_file: .env
environment:
- DATABASE_URL=${APP_DATABASE_URL:-postgresql+psycopg://yupcha_runtime:yupcha_runtime@postgres:5432/yupcha}
- PYTHONPATH=/app
- REDIS_URL=redis://redis:6379
- REACHER_URL=${REACHER_URL:-http://reacher:8080}
- YUPCHA_DB_INIT=skip
volumes:
- ./data:/app/data
command: python -m apps.api.worker
depends_on:
migrate:
condition: service_completed_successfully
postgres:
condition: service_healthy
redis:
condition: service_healthy
api:
condition: service_healthy
restart: unless-stopped
healthcheck:
# The worker deliberately exposes no HTTP port. Its liveness contract is
# that the PID 1 worker process still exists; queue readiness is reported
# separately through worker/job heartbeats in Postgres.
test: ["CMD", "python", "-c", "import os; os.kill(1, 0)"]
interval: 30s
timeout: 5s
retries: 3
start_period: 10s
networks:
- lead_net
# ── Recurring job scheduler (single reconciliation owner) ─────
scheduler:
image: ${YUPCHA_IMAGE:-lead-data-app:local}
build:
context: .
dockerfile: Dockerfile
user: "${APP_UID:-1000}:${APP_GID:-1000}"
env_file: .env
environment:
- DATABASE_URL=${APP_DATABASE_URL:-postgresql+psycopg://yupcha_runtime:yupcha_runtime@postgres:5432/yupcha}
- PYTHONPATH=/app
- YUPCHA_DB_INIT=skip
- SCHEDULER_RECONCILE_SECONDS=${SCHEDULER_RECONCILE_SECONDS:-60}
volumes:
- ./data:/app/data
command: python -m apps.api.scheduler
depends_on:
migrate:
condition: service_completed_successfully
postgres:
condition: service_healthy
restart: unless-stopped
healthcheck:
test: ["CMD", "python", "-c", "import os; os.kill(1, 0)"]
interval: 30s
timeout: 5s
retries: 3
start_period: 10s
networks:
- lead_net
# ── Frontend asset publisher (refreshes the shared nginx volume) ──
# A named volume is only populated from an image on first creation, which
# otherwise leaves nginx serving an old UI after later image rebuilds. This
# one-shot publisher copies the newly built assets on every deployment.
frontend_assets:
image: ${YUPCHA_IMAGE:-lead-data-app:local}
build:
context: .
dockerfile: Dockerfile
command: sh -c "cp -a /app/apps/web/dist/. /frontend/"
volumes:
- frontend_dist:/frontend
restart: "no"
networks:
- lead_net
# ── Reacher (OPTIONAL self-hosted email verification) ──────────
# Strongest tier of the email cascade: unlimited, $0/email SMTP RCPT probing
# (safe/risky/invalid/unknown + catch-all + disposable + role). DUAL-LICENSED
# AGPL-3.0 OR commercial — we run the AGPL build as a separate networked
# process (unambiguously compliant; no GPL contamination of Yupcha's code).
#
# NOT started by default. Enable with: docker compose --profile reacher up
# Then set REACHER_ENABLED=1 (and a header secret) in .env to wire it into the
# cascade. Without the profile, `docker compose up` is byte-for-byte unchanged.
#
# PORT-25 CAVEAT: Reacher does real outbound :25 RCPT probes. Most clouds block
# egress on :25, so without port-25 access or a SOCKS5 proxy (RCH__PROXY__*)
# Reacher returns mostly `unknown` and the cascade correctly falls back to the
# bundled SMTP probe / HTTP vendors. For hosted SaaS the realistic path is the
# commercial Reacher cloud API (set REACHER_API_KEY, leave this profile off).
#
# NOT published on a host port (internal lead_net only) — never expose it.
reacher:
image: reacherhq/backend:v0.10.0 # pinned (do NOT track `latest`)
container_name: lead-data-reacher
profiles: [reacher]
environment:
- RCH__BACKEND_NAME=yupcha-reacher
- RCH__HTTP_HOST=0.0.0.0
- RCH__SMTP__FROM_EMAIL=${REACHER_FROM_EMAIL:-verify@yupcha.com}
- RCH__SMTP__HELLO_NAME=${REACHER_HELLO_NAME:-verify.yupcha.com}
# Header secret — clients must send `Authorization: Bearer <secret>`.
# Set REACHER_API_KEY in .env to the same value so the cascade authenticates.
- RCH__HEADER_SECRET=${REACHER_API_KEY:-}
# Optional SOCKS5 proxy to escape blocked port-25 egress (operator-supplied):
# - RCH__PROXY__HOST=${REACHER_PROXY_HOST:-}
# - RCH__PROXY__PORT=${REACHER_PROXY_PORT:-}
# - RCH__PROXY__USERNAME=${REACHER_PROXY_USERNAME:-}
# - RCH__PROXY__PASSWORD=${REACHER_PROXY_PASSWORD:-}
restart: unless-stopped
networks:
- lead_net
# ── Redis (live pub/sub; durable jobs live in Postgres) ─────────
redis:
image: redis:7-alpine
container_name: lead-data-redis
ports:
# Redis has no authentication in this self-host profile; never expose it
# to the LAN. Containers reach it through the private lead_net network.
- "127.0.0.1:6379:6379"
volumes:
- redisdata:/data
healthcheck:
test: ["CMD", "redis-cli", "ping"]
interval: 10s
timeout: 5s
retries: 3
restart: unless-stopped
networks:
- lead_net
# ── Nginx (reverse proxy + static files) ───────────────
nginx:
image: nginx:alpine
container_name: lead-data-web
ports:
- "${PORT:-3000}:80"
volumes:
- ./infra/nginx/default.conf:/etc/nginx/conf.d/default.conf:ro
- frontend_dist:/usr/share/nginx/html:ro # Serve built frontend from API build
depends_on:
api:
condition: service_healthy
frontend_assets:
condition: service_completed_successfully
restart: unless-stopped
healthcheck:
test: ["CMD", "wget", "-qO-", "http://127.0.0.1/health"]
interval: 30s
timeout: 5s
retries: 3
networks:
- lead_net
volumes:
pgdata:
redisdata:
frontend_dist:
networks:
lead_net:
driver: bridge