-
Notifications
You must be signed in to change notification settings - Fork 3
Expand file tree
/
Copy pathdocker-compose.yml
More file actions
277 lines (271 loc) · 13.9 KB
/
Copy pathdocker-compose.yml
File metadata and controls
277 lines (271 loc) · 13.9 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
services:
app:
build:
context: .
# Match the container user to the host user so files written into the
# bind mount below are not root-owned. Set UID/GID in .env -- see
# .env.example. Changing these needs a rebuild AND `docker compose down
# -v`, because the anonymous node_modules volume keeps the ownership it
# was first created with.
args:
UID: ${UID:-1000}
GID: ${GID:-1000}
ports:
- '4200:4200'
- '4001:4001'
volumes:
- .:/app
- /app/node_modules
# This ensures the nested tina folder is visible to the watcher
- ./projects/website-angular/tina:/app/projects/website-angular/tina
# localhost inside a container is the container. The Reactome backend
# (Tomcat, and behind it Neo4j and Solr) runs on the host, so reach it
# through the docker bridge gateway. proxy.conf.js reads this and otherwise
# defaults to localhost:8080, which is correct when running on the host.
extra_hosts:
- 'host.docker.internal:host-gateway'
environment:
- NG_CLI_ANALYTICS=false
- NODE_ENV=development
- REACTOME_BACKEND=http://host.docker.internal:8080
# The render service is a sibling container, not loopback. proxy.conf.js
# reads this and otherwise defaults to 127.0.0.1:4310, which is correct
# when both run on the host.
- RENDER_TARGET=http://render:4310
stdin_open: true
tty: true
# Diagram figures for documents -- GIF, PPTX, PDF, PNG, SVG -- rendered by
# driving the app's own render page, so an exported figure cannot drift from
# what a curator sees. Replaces the Java exporters, which reimplement the
# drawing.
render:
build:
context: .
dockerfile: deploy/render-service/Dockerfile
# A render is a browser holding a large canvas, and this box also runs the
# site, Tomcat and Neo4j. If it wedges or runs out of memory, restarting is
# the right response -- there is no state to lose but the cache, which is on
# a volume.
restart: unless-stopped
# Loopback only, never 0.0.0.0. The site's own origin proxies /RenderService
# to this, so a render can only be commissioned through whatever fronts the
# site -- crawlers on the old /ContentService/exporter/* URLs are what
# exhausted Tomcat's heap and took the origin down. Publishing it here rather
# than relying on the compose network because on the dev box the site itself
# runs on the host, not in compose; a deployment where both are containers can
# drop this line and reach it by service name.
ports:
- '127.0.0.1:4310:4310'
environment:
# What it renders. The service name when the site is a container here, the
# host when it is not -- which is the case on the dev box.
- RENDER_BASE=${RENDER_BASE:-http://host.docker.internal:4200}
- RENDER_CACHE=/cache
- RENDER_CONCURRENCY=2
- RENDER_QUEUE=8
volumes:
- render-cache:/cache
# No depends_on: the site it renders may be the app service or may be on the
# host, and starting a second app container would fight the first for :4200.
extra_hosts:
- 'host.docker.internal:host-gateway'
# Chromium's default 64 MB /dev/shm is not enough for a 6000px canvas; a
# renderer that fails only on large diagrams is the usual symptom.
shm_size: '1gb'
# The MCP server (reactome-mcp), which lets an AI assistant read Reactome
# through its own tools rather than through a browser.
#
# Here because until now it had no definition anywhere: it was a `docker run`
# somebody typed once, bind-mounting a git checkout at /srv and running the
# `dist/` inside it. `dist/` is gitignored, so what that served was whatever
# branch happened to be checked out, compiled whenever anyone last ran a
# build -- and a test run rebuilding it from an unmerged branch had already
# put code into the running container that no longer existed on disk.
#
# Built from a git ref instead, so the deployed revision is something someone
# chose. MCP_REF is a branch or tag; the container records the commit it was
# built from in /srv/REVISION.
mcp:
build:
context: .
dockerfile: deploy/mcp/Dockerfile
args:
MCP_REF: ${MCP_REF:-main}
restart: unless-stopped
# Loopback only, and emphatically not 0.0.0.0. Same reason as `render`
# above, and worse per request: a tool call here can submit a real job to
# the Analysis Service, so the only way in is through whatever fronts the
# site, which rate-limits it. The container this replaces set
# MCP_HTTP_HOST=0.0.0.0 safely because it was on a private compose network
# with nothing published; that is not safe under `network_mode: host`, so
# the binding is pinned here rather than inherited.
# Probes a real MCP session, not /health.
#
# /health reports that the process is up, which is a different question from
# whether it can serve anything: a server that binds and then fails every
# session answers it `ok`. That is not hypothetical -- a bad
# MCP_TOOL_GROUPS produced exactly that shape until the server started
# validating before binding, and the general case survives that fix.
#
# So this does what a client does: initialize, require a session id back,
# and then DELETE the session. The delete is not tidiness. Each session gets
# its own server instance and idle ones are only reaped after thirty
# minutes, so a check that opened one every minute and walked away would
# hold about thirty instances for ever, purely to watch. Measured before
# adding it: three initialize calls took the count from 2 to 5 and it stayed
# there; with the delete, three runs left it unchanged.
#
# It also keeps /health's `sessions` honest, which matters because that
# number is the thing anyone would reach for to see whether the server is
# being used.
#
# Node rather than curl or wget, because the image is node:22-slim and has
# neither -- a healthcheck that silently always fails is worse than none,
# since it reports unhealthy for a service that is fine and trains people to
# ignore it.
healthcheck:
test:
[
'CMD',
'node',
'-e',
"const u='http://127.0.0.1:4320/mcp',h={'Content-Type':'application/json','Accept':'application/json, text/event-stream'};fetch(u,{method:'POST',headers:h,body:JSON.stringify({jsonrpc:'2.0',id:1,method:'initialize',params:{protocolVersion:'2024-11-05',capabilities:{},clientInfo:{name:'healthcheck',version:'1'}}})}).then(r=>{const s=r.headers.get('mcp-session-id');if(!r.ok||!s)process.exit(1);return fetch(u,{method:'DELETE',headers:{...h,'mcp-session-id':s}})}).then(()=>process.exit(0)).catch(()=>process.exit(1));",
]
interval: 60s
timeout: 10s
# Generous, because the first session builds a server instance and this
# box is busy. A healthcheck that flaps under load is noise.
retries: 3
start_period: 20s
ports:
- '127.0.0.1:4320:4320'
# Two ways in, for two callers that cannot share one.
#
# nginx runs in host networking, so it reaches the published port above on
# 127.0.0.1. The chatbot is a container on its own private network and
# addresses this by name -- `REACTOME_MCP_URL=http://reactome_mcp:4320` --
# so the alias is what lets it keep doing that. Without it the chatbot would
# need reconfiguring to point at a host gateway, which is a change to
# somebody else's deployment to solve a problem on this side.
#
# The alias is the name of the hand-run container this replaces, on purpose:
# the swap is then invisible to the thing depending on it.
networks:
default: {}
reactome_beta:
aliases:
- reactome_mcp
environment:
# 0.0.0.0 *inside* the container, which the port mapping above then
# exposes on loopback only. A container binding its own loopback would be
# unreachable even from the host.
- MCP_HTTP_HOST=0.0.0.0
- MCP_HTTP_PORT=4320
# Which Reactome it answers from: this deployment's own, never another's.
# An MCP on beta that answered from production would describe a release
# nobody here is running.
- REACTOME_BASE_URL=${MCP_REACTOME_BASE_URL:-https://beta.reactome.org}
# NEO4J_URI and MCP_ALLOW_CYPHER are both deliberately absent, and it
# takes both to register the Cypher tools, the graph-schema resource and
# the instructions that advertise them. Arbitrary graph queries used to
# arrive as a side effect of setting a connection string -- the startup
# schema warm-up wants the same variable -- so an instance could acquire
# them without anyone deciding to. Neither belongs on an instance the
# public can reach.
content-node:
build:
context: .
dockerfile: deploy/content-node/Dockerfile
# It holds two lists in memory and nothing else, so a restart costs one
# rebuild of the caches -- about four seconds -- and never any data. Left
# unsupervised it runs only because somebody started it by hand, and the
# contents page goes blank the first time this box reboots, with nothing to
# connect the outage to a restart hours earlier.
restart: unless-stopped
# Host networking, for as long as the graph is on the host.
#
# Neo4j listens on 127.0.0.1:7687 and should keep doing so. A bridged
# container cannot reach that, and the usual workaround -- binding Neo4j to
# the docker bridge -- widens who can reach the database to fit a container
# in. Sharing the host's namespace keeps the database exactly as reachable
# as it is today and the service on loopback where nginx expects it.
#
# **This is the line to delete when the graph becomes a container**, which
# is the plan: a Neo4j image built per release. Then this joins the compose
# network, `NEO4J_URI` names that service instead of loopback, and it
# publishes 127.0.0.1:4400 like `render` does. `graph.mjs` already reads
# NEO4J_URI from the environment, so that is configuration rather than a
# change here. Recorded in specs/006 D13 so the reason is findable when the
# line looks arbitrary.
network_mode: host
# The credentials file is **mounted, not parsed**. graph.mjs reads it
# literally for a documented reason: a generated secret contains characters
# a shell eats, and sourcing one silently produced an empty password once.
#
# compose's own `env_file` parser does the same thing. Tried first, and it
# read `NEO4J_DATABASE=graph.db` correctly while handing the service
# `NEO4J_PASSWORD=""` -- the service then reported "graph credentials:
# MISSING" and every request 500'd. Exactly the fault the literal reader was
# written to avoid, reintroduced by the layer underneath it.
volumes:
- ${CONTENT_NODE_ENV_FILE:-~/.content-node.env}:/run/secrets/content-node.env:ro
environment:
- CONTENT_NODE_ENV_FILE=/run/secrets/content-node.env
# Runs as the owner of that file, which is 0600 and should stay that way.
# The image's own `node` user is uid 1000 and cannot read it, so the service
# started, reported "graph credentials: MISSING", and answered 500 -- a
# container that is up and useless, which is the state monitoring is worst
# at noticing.
#
# Defaults to this host's operator; override where the file has a different
# owner. `id -u`, `id -g`.
user: '${CONTENT_NODE_UID:-1020}:${CONTENT_NODE_GID:-1001}'
# No depends_on: Neo4j is not in this compose file. If it is down the
# service starts, fails to build its lists, and rebuilds on the next
# request -- which is the behaviour specs/006 D10 asks for.
# The front door. Replaces the hand-created `reactome-nginx` container, which
# mounted its config from /etc/nginx-reactome -- a *copy* of deploy/nginx that
# had to be installed with sudo and could drift from the repository without
# anything saying so. The only reason it was known to match today is that
# somebody diffed it.
#
# Mounting the repository instead makes this the single source: edit
# deploy/nginx, `docker compose exec nginx nginx -t`, reload. No sudo, no
# OS-specific install path, and it works the same on a laptop.
#
# To take over from the hand-created container:
# docker stop reactome-nginx
# docker compose up -d nginx
# To go back: `docker compose stop nginx && docker start reactome-nginx`.
# Both bind the host's ports, so only one can run at a time.
nginx:
build:
context: .
dockerfile: deploy/nginx/Dockerfile
args:
# Which deployment this host is. See deploy/nginx/README.md -- getting
# this wrong on the indexed host would deindex reactome.org.
NGINX_ENV: ${NGINX_ENV:-dev}
restart: unless-stopped
# Host networking, as the container it replaces used: the services it
# proxies -- the site on 4200, content-node on 4400, render on 4310, the
# chatbot on 8000 -- are reached on loopback, and several of them bind there
# deliberately so nothing else can.
network_mode: host
volumes:
# Only the certificates are mounted. The configuration is baked into the
# image above, so a checkout cannot change what this serves.
# The one part that cannot live in the repository. Paths are variables so
# a laptop can point somewhere else, or at an empty directory when it is
# serving plain HTTP and has no certificates at all.
- ${LETSENCRYPT_DIR:-/etc/letsencrypt}:/etc/letsencrypt:ro
- ${CLOUDFLARE_CERT_DIR:-/etc/ssl/cloudflare}:/etc/ssl/cloudflare:ro
networks:
# Created by the chatbot's own stack, not by this one -- declared external so
# compose joins it rather than trying to own it, and never removes it.
reactome_beta:
external: true
volumes:
# Survives a container replacement, which is the point: a cached figure is a
# file read, and rebuilding the cache means paying for every render again.
render-cache: