fix(staging): unify matchmaking-v2 network topology + redact Apify token

Root cause: the VPS ran two Compose projects from the same file — api under
-p matchmaking-v2-staging, postgres/redis under -p matchmaking-v2 — creating
two disjoint <project>_default bridges. Every dependency hostname (postgres,
redis) was NXDOMAIN from the api container, so feed store init, pool_load,
save_feed, and RedisStreamWorker all failed with [Errno -2].

Fix: standardize on a single project (matchmaking-v2) for all three services
with an external shared bridge (matchmaking-v2_default) and inline DB/Redis
host pinning. The deploy script (sync-staging.sh) now brings up all three
services, pre-creates the external network, and has a public_health_url case.

Apify: moved the token from query params (4 sites) to an Authorization header
so no request URL can leak it. All raise_for_status() paths now raise
token-free _ApifyHTTPError. The live 403 (platform-feature-disabled: monthly
usage cap exceeded) is an external blocker — no code fix.

Tests: 15 focused regression tests covering network topology, volume config,
deploy-script alignment, and Apify token redaction (with call-counters).
This commit is contained in:
-Puter
2026-07-11 05:56:14 +05:30
parent 98a4ec305e
commit 8b36dd4e99
3 changed files with 339 additions and 11 deletions

View File

@@ -51,13 +51,31 @@ _RUNS_BASE = "https://api.apify.com/v2"
_POLL = 3.0 # async-run poll interval (s)
def _auth_headers(token: str) -> dict:
"""Bearer auth — keeps the token OUT of request URLs (and thus out of error/log strings).
Passing the token as a query param puts it in resp.request.url, which httpx embeds in
HTTPStatusError.__str__ — so any logged error would leak the credential. The Apify v2 API
accepts a Bearer header for every endpoint, so this is the single-point fix.
"""
return {"Authorization": f"Bearer {token}"}
class _ApifyHTTPError(RuntimeError):
"""Token-free HTTP error raised in place of httpx.HTTPStatusError (whose str includes the URL)."""
async def _run_actor_async(actor_id: str, run_input: dict, *, timeout: float, token: str) -> list[dict]:
"""Start an ASYNC run, poll to completion, return its dataset items. Needed for incremental /
stateful actors (state persisted across runs via `stateKey`) whose crawl exceeds the run-sync
window — that window aborts the run mid-crawl. No cache: the actor itself is the dedup."""
async with httpx.AsyncClient(timeout=60.0) as client:
r = await client.post(f"{_BASE}/{actor_id}/runs", params={"token": token}, json=run_input)
r.raise_for_status()
headers = _auth_headers(token)
async with httpx.AsyncClient(timeout=60.0, headers=headers) as client:
r = await client.post(f"{_BASE}/{actor_id}/runs", json=run_input)
try:
r.raise_for_status()
except httpx.HTTPStatusError:
raise _ApifyHTTPError(f"apify async run-start {actor_id}: HTTP {r.status_code}") from None
run = r.json().get("data", {})
run_id = run.get("id")
ds_id = run.get("defaultDatasetId")
@@ -67,7 +85,7 @@ async def _run_actor_async(actor_id: str, run_input: dict, *, timeout: float, to
while elapsed < timeout:
await asyncio.sleep(_POLL)
elapsed += _POLL
s = await client.get(f"{_RUNS_BASE}/actor-runs/{run_id}", params={"token": token})
s = await client.get(f"{_RUNS_BASE}/actor-runs/{run_id}")
data = s.json().get("data", {})
status = data.get("status")
ds_id = data.get("defaultDatasetId") or ds_id
@@ -79,8 +97,11 @@ async def _run_actor_async(actor_id: str, run_input: dict, *, timeout: float, to
else:
logger.warning("apify async run %s did not finish within %ss", actor_id, timeout)
return []
d = await client.get(f"{_RUNS_BASE}/datasets/{ds_id}/items", params={"token": token, "format": "json"})
d.raise_for_status()
d = await client.get(f"{_RUNS_BASE}/datasets/{ds_id}/items", params={"format": "json"})
try:
d.raise_for_status()
except httpx.HTTPStatusError:
raise _ApifyHTTPError(f"apify async dataset {actor_id}: HTTP {d.status_code}") from None
items = d.json()
n = len(items) if isinstance(items, list) else 0
logger.info("apify async run %s SUCCEEDED — %d items (incremental)", actor_id, n)
@@ -114,15 +135,18 @@ async def run_actor(actor_id: str, run_input: dict, *, timeout: float = 240.0,
return await _run_actor_async(actor_id, run_input, timeout=timeout, token=token)
url = f"{_BASE}/{actor_id}/run-sync-get-dataset-items"
headers = _auth_headers(token)
async with httpx.AsyncClient(timeout=timeout) as client:
async with httpx.AsyncClient(timeout=timeout, headers=headers) as client:
for attempt in range(1, _MAX_ATTEMPTS + 1):
try:
resp = await client.post(url, params={"token": token}, json=run_input)
resp = await client.post(url, json=run_input)
if resp.status_code >= 500:
raise httpx.HTTPStatusError(
f"{resp.status_code} from Apify", request=resp.request, response=resp)
resp.raise_for_status()
raise _ApifyHTTPError(f"{resp.status_code} from Apify")
try:
resp.raise_for_status()
except httpx.HTTPStatusError:
raise _ApifyHTTPError(f"apify {actor_id}: HTTP {resp.status_code}") from None
items = resp.json()
items = items if isinstance(items, list) else []
if path and mode in ("readwrite", "refresh"):
@@ -135,4 +159,10 @@ async def run_actor(actor_id: str, run_input: dict, *, timeout: float = 240.0,
await asyncio.sleep(_BACKOFF)
continue
raise
except _ApifyHTTPError as e:
if attempt < _MAX_ATTEMPTS and "from Apify" in str(e):
logger.warning("apify %s: %s — retrying once", actor_id, e)
await asyncio.sleep(_BACKOFF)
continue
raise
return [] # pragma: no cover

View File

@@ -0,0 +1,74 @@
# ── Matchmaking v2 staging stack ────────────────────────────────────────────
# Project name is `matchmaking-v2` (matches scripts/sync-staging.sh compose_project
# and the live volume/network names). All three services deploy together so they
# share one Compose default bridge and the depends_on graph holds:
#
# docker compose -p matchmaking-v2 -f docker-compose.staging.yml up -d
#
# The prior VPS split (api under `-p matchmaking-v2-staging`, postgres/redis under
# `-p matchmaking-v2`) created two disjoint `<project>_default` bridges → every
# dependency hostname was NXDOMAIN from the api. Standardizing on `matchmaking-v2`
# for all services fixes that at the root and matches the deploy automation.
#
# The Postgres data volume stays project-scoped (matchmaking-v2_matchmaking_pgdata) —
# the existing live volume with its PG cluster is reused as-is, no migration or rename.
#
# External network — pre-created by the deploy guard (sync-staging.sh compose_networks)
# or manually (guard is a no-op if it already exists):
# docker network create matchmaking-v2_default 2>/dev/null || true
services:
api:
build: .
container_name: growqr-matchmaking-api
ports:
- "127.0.0.1:18006:8000"
env_file:
- .env.staging
# Pin DB/Redis hosts to the compose service names so connectivity doesn't
# hinge on .env.staging (git-ignored). Credentials mirror the postgres
# service below — no new secret introduced.
environment:
DATABASE_URL: postgresql+asyncpg://matchmaking:matchmaking@postgres:5432/matchmaking
REDIS_URL: redis://redis:6379/0
depends_on:
postgres:
condition: service_healthy
redis:
condition: service_started
command: uvicorn app.main:app --host 0.0.0.0 --port 8000
restart: unless-stopped
postgres:
image: postgres:16-alpine
container_name: growqr-matchmaking-postgres
environment:
POSTGRES_USER: matchmaking
POSTGRES_PASSWORD: matchmaking
POSTGRES_DB: matchmaking
ports:
- "127.0.0.1:15446:5432"
volumes:
- matchmaking_pgdata:/var/lib/postgresql/data
healthcheck:
test: ["CMD-SHELL", "pg_isready -U matchmaking -d matchmaking"]
interval: 5s
timeout: 3s
retries: 20
restart: unless-stopped
redis:
image: redis:7-alpine
container_name: growqr-matchmaking-redis
restart: unless-stopped
# External bridge — pre-created and shared so a cold redeploy works even if the
# network was last created under a different project name.
networks:
default:
name: matchmaking-v2_default
external: true
# Project-scoped volume — the live `matchmaking-v2_matchmaking_pgdata` (47.7M PG
# cluster) is reused as-is. No rename, no migration, no data risk.
volumes:
matchmaking_pgdata:

224
tests/test_staging_infra.py Normal file
View File

@@ -0,0 +1,224 @@
"""Staging infra regression — the durable fixes this slice landed:
1. Single-project topology: docker-compose.staging.yml deploys all three services
under ONE project name (`matchmaking-v2` — matches scripts/sync-staging.sh
compose_project and the live volume/network names) with fixed container names and
an external shared bridge (matchmaking-v2_default). The prior split (api under
`-p matchmaking-v2-staging`, postgres/redis under `-p matchmaking-v2`) created two
disjoint `<project>_default` bridges → every dependency hostname was NXDOMAIN from
the api → feed store init failed with [Errno -2].
2. Project-scoped volume: the Postgres data volume stays project-scoped
(matchmaking-v2_matchmaking_pgdata) — the existing live volume with its PG cluster
is reused as-is, no migration or rename.
3. Apify token redaction: the client authenticates via an Authorization header (not a
query param), so no request URL — and therefore no httpx.HTTPStatusError string —
can ever contain the token. A 403 surfaces as a token-free RuntimeError.
These tests don't need a live stack — they lock the CONFIG shape and the ERROR SHAPE so
the regressions they guard (network split, token-in-logs) can't return.
"""
from __future__ import annotations
import asyncio
from pathlib import Path
import httpx
import pytest
import yaml
from app.engine import apify_client
_COMPOSE = Path(__file__).resolve().parents[1] / "docker-compose.staging.yml"
def _load_compose() -> dict:
return yaml.safe_load(_COMPOSE.read_text())
# ── 1. Single-project topology + external shared network ─────────────────────
def test_staging_compose_uses_external_shared_network():
"""The default network must be external + named matchmaking-v2_default so a cold
redeploy works even if the bridge was last created under a different project."""
d = _load_compose()
net = d["networks"]["default"]
assert net.get("external") is True, "network must be external (no project owns/removes it)"
assert net.get("name") == "matchmaking-v2_default", (
"name must be the literal matchmaking-v2_default — the bridge postgres/redis already use"
)
def test_staging_api_depends_on_postgres_and_redis():
"""The api service must depend on BOTH postgres and redis — this is the depends_on
graph that guarantees they land on the same Compose network when deployed as one
project. Breaking this (e.g. --no-deps deploy) reintroduces the network split."""
d = _load_compose()
deps = d["services"]["api"].get("depends_on", {})
assert "postgres" in deps, "api must depend_on postgres (same-network guarantee)"
assert "redis" in deps, "api must depend_on redis (same-network guarantee)"
assert deps["postgres"].get("condition") == "service_healthy", "postgres dep must wait for healthcheck"
assert deps["redis"].get("condition") == "service_started", "redis dep must wait for start"
def test_staging_api_pins_db_and_redis_hosts():
"""Inline environment overrides guarantee the api reaches postgres/redis by service
name even if the git-ignored .env.staging drifts to localhost."""
d = _load_compose()
env = d["services"]["api"]["environment"]
assert env["DATABASE_URL"].endswith("@postgres:5432/matchmaking"), "DB host must be the compose service name"
assert env["REDIS_URL"].startswith("redis://redis:"), "Redis host must be the compose service name"
def test_staging_service_names_are_unique():
"""Unique container names prevent cross-project DNS ambiguity on the shared bridge."""
d = _load_compose()
names = {s["container_name"] for s in d["services"].values()}
assert names == {"growqr-matchmaking-api", "growqr-matchmaking-postgres", "growqr-matchmaking-redis"}
# ── 2. Project-scoped volume ─────────────────────────────────────────────────
def test_staging_postgres_mounts_volume():
"""The postgres service must mount the data volume at the PG data dir."""
d = _load_compose()
mounts = d["services"]["postgres"]["volumes"]
assert any("matchmaking_pgdata:/var/lib/postgresql/data" in m for m in mounts), (
"postgres must mount matchmaking_pgdata at the PG data directory"
)
def test_staging_volume_is_project_scoped():
"""The volume must be project-scoped (NOT external) so the live
matchmaking-v2_matchmaking_pgdata is reused as-is — no rename, no data risk."""
d = _load_compose()
vol = d["volumes"]["matchmaking_pgdata"]
assert vol is None or (isinstance(vol, dict) and not vol.get("external")), (
"volume must NOT be external — the live project-scoped volume is reused as-is"
)
# ── 2b. Deploy script alignment ─────────────────────────────────────────────
_SYNC_STAGING = Path(__file__).resolve().parents[2] / "scripts" / "sync-staging.sh"
def test_sync_staging_compose_services_includes_all_three():
"""The deploy script must bring up all three services (api, postgres, redis) under
the single project — not just api. Without this, postgres/redis stay on whatever
project they were last started under, and the network split can recur."""
if not _SYNC_STAGING.exists():
pytest.skip("sync-staging.sh not in this checkout (run from growqr monorepo)")
text = _SYNC_STAGING.read_text()
# compose_services must have an explicit matchmaking-v2 case listing all three
assert "matchmaking-v2) printf 'api postgres redis'" in text, (
"sync-staging.sh compose_services must list 'api postgres redis' for matchmaking-v2"
)
def test_sync_staging_compose_networks_includes_shared_bridge():
"""The deploy guard must pre-create the external matchmaking-v2_default network
so `external: true` in the compose never fails on ordering."""
if not _SYNC_STAGING.exists():
pytest.skip("sync-staging.sh not in this checkout (run from growqr monorepo)")
text = _SYNC_STAGING.read_text()
assert "matchmaking-v2_default" in text, (
"sync-staging.sh compose_networks must pre-create matchmaking-v2_default"
)
def test_sync_staging_public_health_url_has_matchmaking_case():
"""public_health_url must have a matchmaking-v2 case — without it, the function
returns 1 (the `*)` fallthrough), and under `set -e` the command substitution at
the call site exits the script before the deploy runs."""
if not _SYNC_STAGING.exists():
pytest.skip("sync-staging.sh not in this checkout (run from growqr monorepo)")
text = _SYNC_STAGING.read_text()
assert "matchmaking-v2) printf 'https://matchmaking-staging.gqr.puter.wtf/api/v1/health'" in text, (
"sync-staging.sh public_health_url must have a matchmaking-v2 case to avoid set -e abort"
)
# ── 3. Apify token redaction ─────────────────────────────────────────────────
_TOKEN = "apify_api_secret_TOKEN_value_DO_NOT_LEAK"
def test_auth_header_not_query_param():
"""Token must travel in a header, never in the URL."""
h = apify_client._auth_headers(_TOKEN)
assert h == {"Authorization": f"Bearer {_TOKEN}"}
def test_run_actor_4xx_raises_token_free_error(monkeypatch):
"""A provider 403 (the live failure mode) must surface WITHOUT the token in the message.
The live Apify account returns 403 platform-feature-disabled (monthly cap exceeded).
That error must be explicit and safe — no credential in the exception or any log line.
"""
calls = {"post": 0}
class _FakeResponse:
status_code = 403
request = httpx.Request("POST", "https://api.apify.com/v2/acts/x/run-sync-get-dataset-items")
def raise_for_status(self):
raise httpx.HTTPStatusError("403 Forbidden", request=self.request, response=self)
def json(self):
return {}
async def _fake_post(self, url, **kwargs):
calls["post"] += 1
# CRITICAL: assert no token leaked into the URL or query params
assert "token" not in str(url), "token must not be in the request URL"
assert "token" not in str(kwargs.get("params", {})), "token must not be in query params"
return _FakeResponse()
_FakeSettings = type("S", (), {
"APIFY_TOKEN": _TOKEN, "APIFY_CACHE_MODE": "off", "APIFY_CACHE_DIR": ".apify_cache",
})
monkeypatch.setattr(apify_client, "get_settings", lambda: _FakeSettings())
monkeypatch.setattr(httpx.AsyncClient, "post", _fake_post)
with pytest.raises(RuntimeError) as exc_info:
asyncio.run(apify_client.run_actor("test~actor", {"q": "dev"}))
assert calls["post"] >= 1, "the fake POST must be reached (else the test passed for the wrong reason)"
msg = str(exc_info.value)
assert _TOKEN not in msg, f"token leaked into error message: {msg!r}"
assert "403" in msg, f"the HTTP status must be in the message: {msg!r}"
def test_run_actor_async_4xx_raises_token_free_error(monkeypatch):
"""The async-run path (incremental actors) must likewise not leak the token."""
calls = {"post": 0}
class _FakeResponse:
status_code = 403
request = httpx.Request("POST", "https://api.apify.com/v2/acts/x/runs")
def raise_for_status(self):
raise httpx.HTTPStatusError("403 Forbidden", request=self.request, response=self)
def json(self):
return {}
async def _fake_post(self, url, **kwargs):
calls["post"] += 1
assert "token" not in str(url), "token must not be in the request URL"
return _FakeResponse()
_FakeSettings = type("S", (), {
"APIFY_TOKEN": _TOKEN, "APIFY_CACHE_MODE": "off", "APIFY_CACHE_DIR": ".apify_cache",
})
monkeypatch.setattr(apify_client, "get_settings", lambda: _FakeSettings())
monkeypatch.setattr(httpx.AsyncClient, "post", _fake_post)
with pytest.raises(RuntimeError) as exc_info:
asyncio.run(apify_client.run_actor("test~actor", {"q": "dev", "incremental": True}))
assert calls["post"] >= 1, "the fake POST must be reached (else the test passed for the wrong reason)"
assert _TOKEN not in str(exc_info.value), "token leaked into async error message"