diff --git a/Dockerfile.test b/Dockerfile.test index 9f8dcb1..3fabcf0 100644 --- a/Dockerfile.test +++ b/Dockerfile.test @@ -25,7 +25,7 @@ COPY tests ./tests # docker-compose.yml.j2 is here because the compose-generation test renders the # REAL template — a stub would assert nothing about what actually ships. COPY requirements-api.txt requirements-mcpunifier.txt docker-compose.yml.example docker-compose.yml.j2 run.sh ./ -COPY scripts/config_helper.py scripts/start.bat ./scripts/ +COPY scripts/config_helper.py scripts/start.bat scripts/check_health.py scripts/healthcheck.sh ./scripts/ ENV PYTHONPATH=/app ENV PYTHONDONTWRITEBYTECODE=1 diff --git a/docs/backtesting.md b/docs/backtesting.md index 35d8255..be046ef 100644 --- a/docs/backtesting.md +++ b/docs/backtesting.md @@ -179,7 +179,12 @@ Only one tester runs at a time per API process (serialized by an internal lock); additional submissions queue. If the API restarts while a job is queued or running, startup recovery marks -that orphaned job as `failed` with `API restarted before completion`. +that orphaned job as `failed` with `API restarted before completion`. Recovery +runs in a background thread so a large job history can never delay the API from +serving, and only inspects jobs touched within the last `BACKTEST_SWEEP_LOOKBACK` +(default `24h`). Completed/failed job state and staging dirs are pruned once +older than `BACKTEST_JOB_RETENTION` (default `30d`), so job status/report/log +URLs for pruned jobs return 404 once past that window. ### Asset sources diff --git a/mt5api/backtest/handler.py b/mt5api/backtest/handler.py index 5d714a6..b29c1ef 100644 --- a/mt5api/backtest/handler.py +++ b/mt5api/backtest/handler.py @@ -24,6 +24,8 @@ import time import uuid +import psutil + from flask import Response, abort, jsonify, request, send_file from mt5api.backtest import cache_parser, ini_builder, jobs, optimization_parser, set_builder @@ -44,6 +46,15 @@ from mt5api.logger import log RUN_LOCK = threading.Lock() + +# How long to keep looking for a replacement terminal process after the one we +# launched exits 0 without a report. Covers the ~7s LiveUpdate gap with margin. +RELAUNCH_GRACE_SECONDS = 30 +# Only a process that died fast enough to be an update restart is worth waiting +# on; a genuine failure after a real run must still fail immediately. +RELAUNCH_MAX_EXIT_SECONDS = 60 +# The report is written just before the terminal exits. +REPORT_SETTLE_SECONDS = 2 DIAGNOSTIC_TAIL_CHARS = 4000 DEFAULT_TOP_PASSES = 50 MAX_TOP_PASSES = 500 @@ -120,7 +131,11 @@ def _read_submission(upload, asset_name, asset_subdir, field, *, required, requi def _parse_ini(text): - parser = configparser.ConfigParser() + # RawConfigParser, not ConfigParser: MT5 INI values are literals and a bare + # '%' is ordinary text in them (EA comments, percentage inputs). Interpolation + # rejects those outright — ConfigParser raises InterpolationSyntaxError on + # "Risk 2% per trade" — which fails the submission before the test ever runs. + parser = configparser.RawConfigParser() parser.optionxform = str parser.read_string(text) if "Tester" not in parser: @@ -265,6 +280,86 @@ def _tail_terminal_log(lines=20): return "\n".join(tail_lines[-lines:]) +def _terminal_process_alive(): + """True while a terminal64.exe belonging to THIS terminal directory runs. + + Matched by directory rather than by the PID we spawned on purpose: the + point of this check is to see the process MT5 started to *replace* the one + we launched, which we never get a handle on. + """ + for proc in psutil.process_iter(["name", "exe"]): + try: + if (proc.info.get("name") or "").lower() != "terminal64.exe": + continue + exe = proc.info.get("exe") or "" + if exe and TERMINAL_DIR.lower() in exe.lower(): + return True + except (psutil.NoSuchProcess, psutil.AccessDenied): + continue + return False + + +def _report_candidates(job): + """Every path that counts as this run's output. + + Mode-3 optimizations write `.symbols.xml` instead of the .htm the INI + asks for, so a mode-3 run has two acceptable outputs and checking only the + .htm would make a finished job look empty. + """ + paths = [job["reportPath"]] + if job.get("optimizationType") == 3: + paths.append(f"{os.path.splitext(job['reportPath'])[0]}.symbols.xml") + return paths + + +def _any_report_exists(paths): + return any(os.path.exists(path) for path in paths) + + +def _await_self_relaunch(job_id, report_paths, deadline): + """Wait out an MT5 self-relaunch, returning True if the report then lands. + + When a LiveUpdate is pending, MT5 applies it by having the process we + launched spawn the updater and exit 0 within a few seconds, then relaunches + itself to run the actual test. `subprocess.run` returns for that first, + short-lived process, so the report legitimately does not exist yet and the + job used to be failed as "Report not generated" while the backtest it was + reporting on went on to finish perfectly well minutes later. + + Seen across the build 6090 rollout: terminal exits at T+3s, the replacement + starts at T+9s, and the test finishes at T+5m34s having written a valid + report. One sacrificial job per terminal, on every MT5 build push. + """ + # The updater runs from AppData, not from TERMINAL_DIR, so there is a gap + # with no matching process at all between the exit and the relaunch — + # ~7s when observed. Poll the whole grace window before concluding that + # nothing is coming back. + grace_end = min(time.time() + RELAUNCH_GRACE_SECONDS, deadline) + while time.time() < grace_end: + if _terminal_process_alive(): + break + if _any_report_exists(report_paths): + return True + time.sleep(1) + else: + return _any_report_exists(report_paths) + + log.info( + "backtest terminal relaunched itself (pending LiveUpdate) broker=%s " + "account=%s job=%s — waiting for the replacement run", + BROKER, ACCOUNT, job_id, + ) + while time.time() < deadline: + if not _terminal_process_alive(): + break + time.sleep(2) + + # MT5 writes the report and then exits, so the file can still be settling + # in the moment the process disappears. + time.sleep(REPORT_SETTLE_SECONDS) + return _any_report_exists(report_paths) + + def _parse_top_passes(raw_value): raw_value = (raw_value or "").strip() if not raw_value: @@ -507,6 +602,14 @@ def _execute_job(job_id): ) return + # A clean exit with no report, fast enough to be an update restart rather + # than a real run, means MT5 probably relaunched itself — wait for the + # replacement rather than failing a backtest that is still going to finish. + report_paths = _report_candidates(job) + if duration < RELAUNCH_MAX_EXIT_SECONDS and not _any_report_exists(report_paths): + _await_self_relaunch(job_id, report_paths, start_time + job["timeoutSeconds"]) + duration = round(time.time() - start_time, 3) + if not os.path.exists(job["reportPath"]) and job.get("optimizationType") == 3: # MT5 writes mode-3 optimization output to .symbols.xml. symbols_report_path = f"{os.path.splitext(job['reportPath'])[0]}.symbols.xml" diff --git a/mt5api/backtest/ini_builder.py b/mt5api/backtest/ini_builder.py index 40457ea..e1055c5 100644 --- a/mt5api/backtest/ini_builder.py +++ b/mt5api/backtest/ini_builder.py @@ -11,7 +11,7 @@ from __future__ import annotations import io -from configparser import ConfigParser +from configparser import RawConfigParser from datetime import date, datetime, timedelta, timezone # MT5 Strategy Tester modelling modes. @@ -217,7 +217,12 @@ def build_ini(params: dict) -> str: raise ValueError("forwardMode must be 0..4") visual = int(bool(params.get("visual", 0))) - parser = ConfigParser() + # RawConfigParser, not ConfigParser: the values below are MT5 INI literals + # and a bare '%' is ordinary text in them — symbol, currency, report name + # and the uploaded filenames all reach here unfiltered. Interpolation + # rejects those on assignment ("invalid interpolation syntax"), surfacing + # as a 400 that looks like a validation error. Mirrors handler._parse_ini. + parser = RawConfigParser() parser.optionxform = str # preserve key casing — MT5 is case-sensitive. parser["Common"] = { diff --git a/mt5api/backtest/jobs.py b/mt5api/backtest/jobs.py index e730e46..12a3468 100644 --- a/mt5api/backtest/jobs.py +++ b/mt5api/backtest/jobs.py @@ -3,17 +3,26 @@ Each job has one JSON file at logs/backtest-jobs/.json. The in-memory dict (BACKTEST_JOBS) is a write-through cache guarded by JOB_LOCK. State files survive API restarts; sweep_orphans() marks any in-flight job as failed at -startup so callers do not poll forever. +startup so callers do not poll forever, and prune_old_jobs() retires terminal +(completed/failed) jobs older than the retention window so the directory cannot +grow without bound. Every backtest API on a VM shares this directory, so both +passes only inspect recently-touched files and are safe to run concurrently. """ from __future__ import annotations import json import os import re +import shutil import threading +import time from datetime import datetime, timezone -from mt5api.config import BACKTEST_JOB_DIR +from mt5api.config import ( + BACKTEST_JOB_DIR, + BACKTEST_JOB_RETENTION_SECONDS, + BACKTEST_SWEEP_LOOKBACK_SECONDS, +) from mt5api.logger import log JOB_LOCK = threading.Lock() @@ -23,6 +32,18 @@ TERMINAL_STATUSES = frozenset({"completed", "failed"}) ACTIVE_STATUSES = frozenset({"queued", "running"}) +# Only state files touched within the last SWEEP_LOOKBACK_SECONDS can be an +# active run that needs sweeping; anything older was long since terminal. +SWEEP_LOOKBACK_SECONDS = BACKTEST_SWEEP_LOOKBACK_SECONDS +# prune_old_jobs() retires terminal jobs older than JOB_RETENTION_SECONDS. +JOB_RETENTION_SECONDS = BACKTEST_JOB_RETENTION_SECONDS +# Don't bother listing a directory unless it has grown this big. +PRUNE_MIN_ENTRIES = 1000 +# All APIs on a VM share the job dir, so gate pruning with a marker file: only +# the first process to pass the gate scans the whole directory per interval. +PRUNE_INTERVAL_SECONDS = 24 * 3600 +PRUNE_MARKER_NAME = ".job-prune-marker" + def now_iso() -> str: return datetime.now(timezone.utc).replace(microsecond=0, tzinfo=None).isoformat() + "Z" @@ -33,14 +54,22 @@ def _state_path(job_id: str) -> str: return os.path.join(BACKTEST_JOB_DIR, f"{job_id}.json") -def _write(job: dict) -> None: - path = _state_path(job["jobId"]) +def _atomic_write(path: str, job: dict) -> None: + """Write job state atomically (tmp + rename). + + A concurrent reader — including a sweep running in another API process that + shares this directory — must never observe a truncated file. + """ tmp = f"{path}.tmp" with open(tmp, "w", encoding="utf-8") as handle: json.dump(job, handle, indent=2, sort_keys=True) os.replace(tmp, path) +def _write(job: dict) -> None: + _atomic_write(_state_path(job["jobId"]), job) + + def store_job(job: dict) -> None: with JOB_LOCK: BACKTEST_JOBS[job["jobId"]] = job @@ -120,23 +149,44 @@ def public_payload(job: dict) -> dict: return payload -def sweep_orphans() -> int: +def sweep_orphans(lookback_seconds: int | None = None) -> int: """Mark any queued/running jobs on disk as failed. - Called at API startup. Returns the number of jobs swept. + Called at API startup. Only state files touched within the last + ``lookback_seconds`` are considered: a live job rewrites its file on every + state transition and a run is bounded by its timeout, so a file untouched + longer than the window cannot be an active run. Sweeping the whole history + parsed tens of thousands of files on every boot — and because every backtest + API on a VM shares the same job directory, that full scan repeated once per + process. Returns the number of jobs swept. """ + if lookback_seconds is None: + lookback_seconds = SWEEP_LOOKBACK_SECONDS if not os.path.isdir(BACKTEST_JOB_DIR): return 0 + cutoff = time.time() - lookback_seconds + recent = [] + try: + with os.scandir(BACKTEST_JOB_DIR) as entries: + for entry in entries: + if not entry.name.endswith(".json"): + continue + try: + if entry.stat(follow_symlinks=False).st_mtime < cutoff: + continue + except OSError: + continue + recent.append(entry.name) + except OSError: + return 0 swept = 0 - for entry in sorted(os.listdir(BACKTEST_JOB_DIR)): - if not entry.endswith(".json"): - continue - path = os.path.join(BACKTEST_JOB_DIR, entry) + for name in sorted(recent): + path = os.path.join(BACKTEST_JOB_DIR, name) try: with open(path, "r", encoding="utf-8") as handle: job = json.load(handle) except (OSError, json.JSONDecodeError) as exc: - log.warning("backtest sweep: cannot read %s: %s", entry, exc) + log.warning("backtest sweep: cannot read %s: %s", name, exc) continue if job.get("status") not in ACTIVE_STATUSES: continue @@ -144,16 +194,94 @@ def sweep_orphans() -> int: job["error"] = "API restarted before completion" job["finishedAt"] = now_iso() try: - with open(path, "w", encoding="utf-8") as handle: - json.dump(job, handle, indent=2, sort_keys=True) + _atomic_write(path, job) except OSError as exc: - log.warning("backtest sweep: cannot write %s: %s", entry, exc) + log.warning("backtest sweep: cannot write %s: %s", name, exc) continue swept += 1 - log.info("backtest sweep: marked %s as failed (was %s)", job.get("jobId"), entry) + log.info("backtest sweep: marked %s as failed (was %s)", job.get("jobId"), name) return swept +def _remove_job_state(json_path: str, job_id: str | None) -> None: + targets = [json_path] + if job_id: + targets.append(os.path.join(BACKTEST_JOB_DIR, job_id)) + for target in targets: + try: + if os.path.isdir(target) and not os.path.islink(target): + shutil.rmtree(target) + else: + os.remove(target) + except OSError as exc: + log.warning("backtest prune: cannot remove %s: %s", target, exc) + + +def _touch(path: str) -> None: + try: + with open(path, "w", encoding="utf-8") as handle: + handle.write(str(int(time.time()))) + except OSError as exc: + log.warning("backtest prune: cannot write marker %s: %s", path, exc) + + +def prune_old_jobs(max_age_seconds: int | None = None) -> int: + """Retire terminal (completed/failed) jobs older than the retention window. + + Removes the state file and the per-job staging dir (staged .ex5/.set, + normalized.ini, tester.ini, run.log). Nothing reads old staging: the report + artifact lives in the terminal's own Reports dir, /backtest//log reads + run.log only as a best-effort fallback, and the backend archives both report + and log into its own data dir right after each job. Bounds the shared job + dir so startup stays cheap as jobs accumulate. Returns the number pruned. + """ + if max_age_seconds is None: + max_age_seconds = JOB_RETENTION_SECONDS + if not os.path.isdir(BACKTEST_JOB_DIR): + return 0 + marker = os.path.join(BACKTEST_JOB_DIR, PRUNE_MARKER_NAME) + try: + if time.time() - os.path.getmtime(marker) < PRUNE_INTERVAL_SECONDS: + return 0 + except OSError: + pass + try: + with os.scandir(BACKTEST_JOB_DIR) as entries: + listed = list(entries) + except OSError: + return 0 + if len(listed) < PRUNE_MIN_ENTRIES: + return 0 + cutoff = time.time() - max_age_seconds + pruned = 0 + for entry in listed: + if entry.name == PRUNE_MARKER_NAME or not entry.name.endswith(".json"): + continue + try: + if entry.stat(follow_symlinks=False).st_mtime >= cutoff: + continue + except OSError: + continue + try: + with open(entry.path, "r", encoding="utf-8") as handle: + job = json.load(handle) + except (OSError, json.JSONDecodeError) as exc: + log.warning("backtest prune: cannot read %s: %s", entry.name, exc) + continue + if job.get("status") in ACTIVE_STATUSES: + continue + _remove_job_state(entry.path, job.get("jobId")) + pruned += 1 + _touch(marker) + if pruned: + log.info( + "backtest prune: retired %d terminal job(s) older than %ds", + pruned, + max_age_seconds, + ) + return pruned + + # ── Report summary parser ─────────────────────────────────────────── # MT5 HTML report layout varies by build; treat every field as best-effort # and return None for anything we cannot parse. diff --git a/mt5api/config.py b/mt5api/config.py index 4370c90..24ad880 100644 --- a/mt5api/config.py +++ b/mt5api/config.py @@ -67,6 +67,7 @@ def _parse_args(): _DURATION_RE = re.compile( r"^\s*(?P[+-])?\s*" + r"(?:(?P\d+(?:\.\d+)?)\s*d)?\s*" r"(?:(?P\d+(?:\.\d+)?)\s*h)?\s*" r"(?:(?P\d+(?:\.\d+)?)\s*m)?\s*" r"(?:(?P\d+(?:\.\d+)?)\s*s)?\s*$", @@ -75,7 +76,7 @@ def _parse_args(): def parse_duration_to_seconds(value): - """Parse '3h', '3h30m', '-2h', '90m', '0' into integer seconds. + """Parse '3d', '3h', '3h30m', '-2h', '90m', '0' into integer seconds. Bare numbers (e.g. '3' or '3.5' or 3) are interpreted as HOURS for convenience — most brokers run on whole-hour offsets. @@ -93,15 +94,16 @@ def parse_duration_to_seconds(value): except ValueError: pass m = _DURATION_RE.match(s) - if not m or not (m.group("h") or m.group("m") or m.group("s")): + if not m or not (m.group("d") or m.group("h") or m.group("m") or m.group("s")): raise ValueError( f"Invalid duration: {value!r}. " - "Use '3h', '3h30m', '-2h', '90m', or a bare number (hours)." + "Use '3d', '3h', '3h30m', '-2h', '90m', or a bare number (hours)." ) + d = float(m.group("d") or 0) h = float(m.group("h") or 0) minutes = float(m.group("m") or 0) secs = float(m.group("s") or 0) - total = h * 3600 + minutes * 60 + secs + total = d * 86400 + h * 3600 + minutes * 60 + secs if m.group("sign") == "-": total = -total return int(round(total)) @@ -207,6 +209,22 @@ def load_terminal_config(): ) BACKTEST_TIMEOUT = BACKTEST_TIMEOUT_RAW BACKTEST_TIMEOUT_SECONDS = parse_duration_to_seconds(BACKTEST_TIMEOUT) + +# Startup backtest-cleanup windows. sweep_orphans() only inspects state files +# touched within the lookback: a live job rewrites its file on every state +# transition and a run is bounded by its timeout, so anything older than the +# window is dead history and needs no sweep. prune_old_jobs() retires terminal +# (completed/failed) jobs older than the retention window so the shared +# backtest-jobs dir — every backtest API on a VM points at it — cannot grow +# without bound and make every boot's directory scan slower. +_SWEEP_LOOKBACK_ENV = os.environ.get("BACKTEST_SWEEP_LOOKBACK") +BACKTEST_SWEEP_LOOKBACK_SECONDS = parse_duration_to_seconds( + _SWEEP_LOOKBACK_ENV if _SWEEP_LOOKBACK_ENV not in (None, "") else "24h" +) +_JOB_RETENTION_ENV = os.environ.get("BACKTEST_JOB_RETENTION") +BACKTEST_JOB_RETENTION_SECONDS = parse_duration_to_seconds( + _JOB_RETENTION_ENV if _JOB_RETENTION_ENV not in (None, "") else "30d" +) _MODE_RAW = (_args.mode or _terminal_config.get("mode") or os.environ.get("MT5_MODE") or "live") MODE = str(_MODE_RAW).strip().lower() or "live" if MODE not in ("live", "backtest"): diff --git a/mt5api/main.py b/mt5api/main.py index 2bfb1a3..91372ac 100644 --- a/mt5api/main.py +++ b/mt5api/main.py @@ -126,6 +126,27 @@ def _background_init(): time.sleep(RETRY_INTERVAL) +def _run_backtest_startup_cleanup(): + """Sweep orphaned in-flight jobs and retire old completed/failed jobs. + + Runs in a background daemon thread so a large job history can never delay + the API from accepting connections. Every backtest API on a VM shares the + same backtest-jobs directory (the fast VM holds tens of thousands of files), + so scanning it synchronously at boot added minutes of outage. Best-effort + and idempotent: it is safe for every backtest API to run it in parallel + against the shared directory. + """ + try: + swept = backtest_jobs.sweep_orphans() + if swept: + log.warning("Backtest sweep marked %d orphaned job(s) as failed.", swept) + pruned = backtest_jobs.prune_old_jobs() + if pruned: + log.info("Backtest retention retired %d old job(s).", pruned) + except Exception: + log.exception("Backtest startup cleanup failed.") + + def _handle_signal(sig, _frame): log.critical("Received signal %d — exiting.", sig) sys.exit(sig) @@ -170,12 +191,14 @@ def main(): start_monitor() - swept = backtest_jobs.sweep_orphans() - if swept: - log.warning("Backtest sweep marked %d orphaned job(s) as failed.", swept) - _start_mcp_session_manager() + threading.Thread( + target=_run_backtest_startup_cleanup, + name="backtest-cleanup", + daemon=True, + ).start() + log.info( "HTTP API listening on %s:%d (waitress, threads=%d, conn_limit=%d, max_queue_depth=%d)", HOST, PORT, WSGI_THREADS, WSGI_CONNECTION_LIMIT, MAX_QUEUE_DEPTH, diff --git a/scripts/check_health.py b/scripts/check_health.py index 0e09ff5..fae1272 100644 --- a/scripts/check_health.py +++ b/scripts/check_health.py @@ -6,6 +6,7 @@ import datetime import os import socket +import sys try: import yaml @@ -13,6 +14,19 @@ print(' ERROR: pyyaml not installed') raise SystemExit(1) +# Reuse config_helper's per-VM filter rather than re-deriving it. It sits in +# this same directory and guards its main() behind __name__, so importing it +# is side-effect free. +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) +try: + from config_helper import _in_group, _vm_group_filter +except ImportError: + def _vm_group_filter(): + return None + + def _in_group(terminal, allowed): + return True + SHARED = r'C:\Users\Docker\Desktop\Shared' CONFIG = os.path.join(SHARED, 'config', 'config.yaml') FULL_LOG = os.path.join(SHARED, 'logs', 'full.log') @@ -28,6 +42,15 @@ now = datetime.datetime.now().strftime('%Y-%m-%d %H:%M:%S') dead = [] +# Only this VM's terminals. Without the filter each VM probes every port in +# config.yaml, including the ones another VM hosts, and reports them DOWN — so +# the container sits permanently "unhealthy" and a genuine failure is +# indistinguishable from the usual noise. Observed 2026-08-04: the fast VM was +# unhealthy with a 25,272-long failing streak, listing only bulk-VM ports, +# while all 12 of its own terminals were serving normally. +allowed = _vm_group_filter() +terminals = [t for t in terminals if _in_group(t, allowed)] + for t in terminals: port = t['port'] broker = t['broker'] diff --git a/scripts/healthcheck.sh b/scripts/healthcheck.sh index 0382e31..a2c48ca 100755 --- a/scripts/healthcheck.sh +++ b/scripts/healthcheck.sh @@ -19,6 +19,9 @@ set -u readonly CONFIG=/shared/config/config.yaml +# Per-VM terminal list, bind-mounted by docker-compose from +# data/vm-group-.txt. Absent on single-VM installs, which means no filter. +readonly VM_GROUP=/shared/config/vm-group.txt readonly DNSMASQ_LEASES=/var/lib/misc/dnsmasq.leases readonly PROBE_PATH=/ping readonly PROBE_TIMEOUT_SECONDS=3 @@ -30,10 +33,55 @@ readonly FALLBACK_HOSTS='127.0.0.1 localhost' exit 1 } -# YAML schema: only terminals[] entries have a `port:` key, so a plain grep -# on indented `port: ` lines is sufficient and avoids needing python -# / a yaml parser inside the alpine-based dockurr/windows container. -PORTS=$(grep -E '^[[:space:]]+port:[[:space:]]*[0-9]+' "$CONFIG" | grep -oE '[0-9]+$') +# Walk terminals[] keeping only the ones this VM actually hosts. The old plain +# grep for `port:` took every port in config.yaml, so on a multi-VM install each +# VM probed the other VM's terminals, found them down, and reported unhealthy +# forever — the fast VM sat at a 25,272-long failing streak listing only +# bulk-VM ports while all 12 of its own terminals were serving. A healthcheck +# that is always red says nothing, and hides the outage it exists to catch. +# +# awk, not python: the dockurr/windows container is alpine and has no python. +# The group file is read inside BEGIN rather than as a second input file +# because the usual FNR==NR idiom misreads the config as the group list when +# that file is empty, which would emit zero ports and fail every single-VM +# install. No group file => no filter, same ports as before. +PORTS=$(awk -v groupfile="$VM_GROUP" ' + function clean(s) { + gsub(/^[[:space:]]+|[[:space:]]+$/, "", s); gsub(/^"|"$/, "", s); return s + } + function value(line, v) { + v = line; sub(/^[^:]*:/, "", v); sub(/,[[:space:]]*$/, "", v); return clean(v) + } + function emit() { + if (port == "") return + if (!have_group) { print port; return } + if ((broker SUBSEP account SUBSEP (instance == "" ? "default" : instance)) in allowed) + print port + } + function reset() { broker = ""; account = ""; instance = ""; port = "" } + BEGIN { + if (groupfile != "") { + while ((getline line < groupfile) > 0) { + sub(/#.*/, "", line) + n = split(line, f, /[[:space:]]+/) + cnt = 0 + for (i = 1; i <= n; i++) if (f[i] != "") { cnt++; g[cnt] = f[i] } + if (cnt >= 2) { + allowed[g[1] SUBSEP g[2] SUBSEP (cnt >= 3 ? g[3] : "default")] = 1 + have_group = 1 + } + } + close(groupfile) + } + } + /^[[:space:]]*-[[:space:]]*"?broker"?[[:space:]]*:/ { + emit(); reset(); broker = value($0); next + } + /^[[:space:]]*"?account"?[[:space:]]*:/ { account = value($0); next } + /^[[:space:]]*"?instance"?[[:space:]]*:/ { instance = value($0); next } + /^[[:space:]]*"?port"?[[:space:]]*:/ { port = value($0); next } + END { emit() } +' "$CONFIG") [ -n "$PORTS" ] || { echo "no ports parsed from $CONFIG" exit 1 @@ -78,5 +126,7 @@ if [ -n "$dead" ]; then exit 1 fi -echo "ok all ports up: $PORTS (vm_ip=$VM_IP)" +# Unquoted on purpose: collapses the newline-separated list onto one line, so +# the verdict `docker inspect` shows stays readable. +echo "ok all ports up:" $PORTS "(vm_ip=$VM_IP)" exit 0 diff --git a/tests/test_backtest_ini_builder.py b/tests/test_backtest_ini_builder.py index 3fb3a69..7f3f7f8 100644 --- a/tests/test_backtest_ini_builder.py +++ b/tests/test_backtest_ini_builder.py @@ -10,7 +10,9 @@ def _parse(text): - parser = configparser.ConfigParser() + # RawConfigParser to match how the builder writes and how MT5 reads: values + # are literals, so a '%' in one must not be treated as an interpolation. + parser = configparser.RawConfigParser() parser.optionxform = str parser.read_string(text) return parser @@ -202,3 +204,22 @@ def test_set_must_end_with_set(): def test_negative_last_days_rejected(): with pytest.raises(ValueError, match="positive"): ini_builder.build_ini({"symbol": "X", "timeframe": "M15", "expert": "EA.ex5", "lastDays": -1}) + + +# --- INI values are literals, not interpolation templates --------------------- + + +def test_percent_in_report_name_is_literal_text(): + """A '%' is ordinary text in an MT5 INI value, and report names come + straight from the caller. configparser's default interpolation treats it as + a template escape and refuses the assignment, failing INI generation with + what surfaces as a 400 validation error.""" + text = ini_builder.build_ini(_base(reportName="drawdown 5% run")) + + assert _parse(text)["Tester"]["Report"] == "Reports\\drawdown 5% run.htm" + + +def test_percent_in_symbol_is_literal_text(): + text = ini_builder.build_ini(_base(symbol="EUR%USD")) + + assert _parse(text)["Tester"]["Symbol"] == "EUR%USD" diff --git a/tests/test_backtest_jobs.py b/tests/test_backtest_jobs.py index 89d6fd6..44a62d3 100644 --- a/tests/test_backtest_jobs.py +++ b/tests/test_backtest_jobs.py @@ -1,8 +1,9 @@ -"""Unit tests for mt5api.backtest.jobs sweep + summary parser.""" +"""Unit tests for mt5api.backtest.jobs sweep + retention + summary parser.""" from __future__ import annotations import json import os +import time import pytest @@ -50,11 +51,76 @@ def test_sweep_handles_corrupt_files(tmp_jobs_dir): assert jobs.sweep_orphans() == 1 +def test_sweep_skips_stale_files(tmp_jobs_dir): + _write(tmp_jobs_dir, "fresh", "running") + _write(tmp_jobs_dir, "stale", "running") + old = time.time() - jobs.SWEEP_LOOKBACK_SECONDS - 3600 + os.utime(tmp_jobs_dir / "stale.json", (old, old)) + assert jobs.sweep_orphans() == 1 + assert json.loads((tmp_jobs_dir / "fresh.json").read_text())["status"] == "failed" + assert json.loads((tmp_jobs_dir / "stale.json").read_text())["status"] == "running" + + +def test_sweep_honors_custom_lookback(tmp_jobs_dir): + _write(tmp_jobs_dir, "oldish", "running") + old = time.time() - 7200 + os.utime(tmp_jobs_dir / "oldish.json", (old, old)) + assert jobs.sweep_orphans(lookback_seconds=3600) == 0 + assert jobs.sweep_orphans(lookback_seconds=10800) == 1 + + def test_sweep_no_directory(monkeypatch, tmp_path): monkeypatch.setattr(jobs, "BACKTEST_JOB_DIR", str(tmp_path / "missing")) assert jobs.sweep_orphans() == 0 +def test_prune_removes_old_terminal_jobs_and_staging(tmp_jobs_dir, monkeypatch): + monkeypatch.setattr(jobs, "PRUNE_MIN_ENTRIES", 1) + _write(tmp_jobs_dir, "olddone", "completed") + _write(tmp_jobs_dir, "oldfail", "failed") + _write(tmp_jobs_dir, "newdone", "completed") + _write(tmp_jobs_dir, "activenow", "running") + (tmp_jobs_dir / "olddone").mkdir() + (tmp_jobs_dir / "newdone").mkdir() + old = time.time() - 10 * 24 * 3600 + for name in ("olddone.json", "oldfail.json"): + os.utime(tmp_jobs_dir / name, (old, old)) + assert jobs.prune_old_jobs(max_age_seconds=24 * 3600) == 2 + assert not (tmp_jobs_dir / "olddone.json").exists() + assert not (tmp_jobs_dir / "olddone").exists() + assert not (tmp_jobs_dir / "oldfail.json").exists() + assert (tmp_jobs_dir / "newdone.json").exists() + assert (tmp_jobs_dir / "newdone").exists() + assert (tmp_jobs_dir / "activenow.json").exists() + + +def test_prune_keeps_active_jobs_even_when_old(tmp_jobs_dir, monkeypatch): + monkeypatch.setattr(jobs, "PRUNE_MIN_ENTRIES", 1) + _write(tmp_jobs_dir, "stalequeued", "queued") + old = time.time() - 10 * 24 * 3600 + os.utime(tmp_jobs_dir / "stalequeued.json", (old, old)) + assert jobs.prune_old_jobs(max_age_seconds=24 * 3600) == 0 + assert (tmp_jobs_dir / "stalequeued.json").exists() + + +def test_prune_skips_small_dir(tmp_jobs_dir): + _write(tmp_jobs_dir, "olddone", "completed") + old = time.time() - 10 * 24 * 3600 + os.utime(tmp_jobs_dir / "olddone.json", (old, old)) + assert jobs.prune_old_jobs() == 0 + assert (tmp_jobs_dir / "olddone.json").exists() + + +def test_prune_honors_marker_interval(tmp_jobs_dir, monkeypatch): + monkeypatch.setattr(jobs, "PRUNE_MIN_ENTRIES", 1) + _write(tmp_jobs_dir, "olddone", "completed") + old = time.time() - 10 * 24 * 3600 + os.utime(tmp_jobs_dir / "olddone.json", (old, old)) + (tmp_jobs_dir / jobs.PRUNE_MARKER_NAME).write_text(str(int(time.time()))) + assert jobs.prune_old_jobs() == 0 + assert (tmp_jobs_dir / "olddone.json").exists() + + def test_summary_parser_returns_all_keys_for_empty_html(): summary = jobs.parse_report_summary("") expected_keys = { diff --git a/tests/test_backtest_relaunch.py b/tests/test_backtest_relaunch.py new file mode 100644 index 0000000..aded6b2 --- /dev/null +++ b/tests/test_backtest_relaunch.py @@ -0,0 +1,252 @@ +"""MT5 self-relaunch (pending LiveUpdate) handling in the backtest runner. + +When MT5 has an update queued, the process we launch spawns the updater and +exits 0 within a few seconds, then MT5 relaunches itself and runs the actual +test. `subprocess.run` returns for the first, short-lived process, so the +report does not exist yet — and the job used to be failed as "Report not +generated" while the backtest went on to finish normally minutes later. + +These cover both directions: the update restart must be waited out, and a +genuine no-report failure must still fail without inheriting that wait. +""" +from __future__ import annotations + +import io +import time +from types import SimpleNamespace + +import pytest + +from mt5api.backtest import handler, jobs +from mt5api.server import app + + +def _far_future(): + return time.time() + 300 + + +@pytest.fixture +def terminal(monkeypatch, tmp_path): + terminal_dir = tmp_path / "terminal" + terminal_dir.mkdir() + monkeypatch.setattr(handler, "TERMINAL_DIR", str(terminal_dir)) + monkeypatch.setattr(handler, "RELAUNCH_GRACE_SECONDS", 0.2) + monkeypatch.setattr(handler, "REPORT_SETTLE_SECONDS", 0) + monkeypatch.setattr( + handler, "time", SimpleNamespace(time=time.time, sleep=lambda _s: None) + ) + return terminal_dir + + +def _scripted_terminal(monkeypatch, states, *, writes_report_at_exit=None): + """Drive _terminal_process_alive through a fixed sequence of states. + + The last state repeats once exhausted. If `writes_report_at_exit` is given, + the file appears the first time the sequence reports the process gone after + having been alive — MT5 writes the report just before exiting. + """ + remaining = list(states) + seen_alive = {"value": False} + + def _alive(): + state = remaining.pop(0) if len(remaining) > 1 else remaining[0] + if state: + seen_alive["value"] = True + elif seen_alive["value"] and writes_report_at_exit is not None: + writes_report_at_exit.write_text("report") + return state + + monkeypatch.setattr(handler, "_terminal_process_alive", _alive) + + +def test_waits_out_the_relaunch_and_finds_the_late_report(terminal, monkeypatch): + """The build-6090 case: the launched process exits with no report, nothing is + running during the updater gap, a replacement appears, runs the real test, + and writes the report on its way out.""" + report = terminal / "report.htm" + _scripted_terminal( + monkeypatch, + [False, False, True, True, False], + writes_report_at_exit=report, + ) + + assert handler._await_self_relaunch("job1", [str(report)], _far_future()) is True + assert report.exists() + + +def test_no_relaunch_and_no_report_still_fails(terminal, monkeypatch): + """A genuine failure must not be rescued: nothing comes back within the grace + window, so the report stays missing and the caller goes on to fail the job.""" + report = terminal / "report.htm" + _scripted_terminal(monkeypatch, [False]) + + assert handler._await_self_relaunch("job1", [str(report)], _far_future()) is False + + +def test_report_appearing_during_the_grace_window_is_accepted(terminal, monkeypatch): + """The report can land in the gap itself, with the process already gone.""" + report = terminal / "report.htm" + report.write_text("report") + _scripted_terminal(monkeypatch, [False]) + + assert handler._await_self_relaunch("job1", [str(report)], _far_future()) is True + + +def test_deadline_stops_the_wait_even_while_the_terminal_runs(terminal, monkeypatch): + """The replacement is still going when the job's timeout budget runs out — + the wait ends rather than running past the deadline.""" + report = terminal / "report.htm" + _scripted_terminal(monkeypatch, [True]) + + assert handler._await_self_relaunch("job1", [str(report)], time.time() + 0.2) is False + + +# --- wiring: the short-exit gate in _execute_job ------------------------------ + +_INI = """[Tester] +Expert=MyEA +Symbol=EURUSD +Period=H1 +FromDate=2024.01.01 +ToDate=2024.02.01 +""" + + +class _NoopThread: + def __init__(self, *args, **kwargs): + pass + + def start(self): + pass + + +@pytest.fixture +def client(monkeypatch, tmp_path): + terminal_dir = tmp_path / "terminal" + terminal_dir.mkdir() + terminal_path = terminal_dir / "terminal64.exe" + terminal_path.write_text("stub") + assets_dir = tmp_path / "assets" + (assets_dir / "experts").mkdir(parents=True) + (assets_dir / "sets").mkdir(parents=True) + log_dir = tmp_path / "logs" + log_dir.mkdir() + + monkeypatch.setattr(handler, "TERMINAL_DIR", str(terminal_dir)) + monkeypatch.setattr(handler, "TERMINAL_PATH", str(terminal_path)) + monkeypatch.setattr(handler, "ASSETS_DIR", str(assets_dir)) + monkeypatch.setattr(handler, "LOG_DIR", str(log_dir)) + monkeypatch.setattr(handler, "BACKTEST_JOB_DIR", str(log_dir / "backtest-jobs")) + monkeypatch.setattr(handler, "BROKER", "testbroker") + monkeypatch.setattr(handler, "ACCOUNT", "testacct") + monkeypatch.setattr(jobs, "BACKTEST_JOB_DIR", str(log_dir / "backtest-jobs")) + jobs.BACKTEST_JOBS.clear() + monkeypatch.setattr(handler, "_load_account_config", lambda: { + "login": 1, "password": "p", "server": "S", + }) + monkeypatch.setattr(handler.threading, "Thread", _NoopThread) + monkeypatch.setattr("mt5api.server.API_TOKEN", "") + app.config["TESTING"] = True + return app.test_client() + + +def _submit(c): + resp = c.post( + "/backtest", + data={ + "ini": (io.BytesIO(_INI.encode()), "tester.ini"), + "expert": (io.BytesIO(b"MZstub"), "MyEA.ex5"), + }, + content_type="multipart/form-data", + ) + assert resp.status_code == 202, resp.get_data(as_text=True) + return resp.get_json()["jobId"] + + +def _run_with_exit_after(monkeypatch, elapsed_seconds): + """Stub terminal64.exe as a clean exit that took `elapsed_seconds`.""" + real_time = time.time + offset = {"value": 0.0} + + def _clock(): + return real_time() + offset["value"] + + def _fake_run(*args, **kwargs): + offset["value"] = elapsed_seconds + return type("Result", (), {"returncode": 0})() + + monkeypatch.setattr( + handler, "time", SimpleNamespace(time=_clock, sleep=lambda _s: None) + ) + monkeypatch.setattr(handler.subprocess, "run", _fake_run) + + waited = [] + monkeypatch.setattr( + handler, "_await_self_relaunch", + lambda job_id, reports, deadline: waited.append(job_id) or False, + ) + return waited + + +def test_fast_clean_exit_without_report_waits_for_a_relaunch(client, monkeypatch): + job_id = _submit(client) + waited = _run_with_exit_after(monkeypatch, 3) + + handler._execute_job(job_id) + + assert waited == [job_id], "a fast clean exit with no report must wait for a relaunch" + assert jobs.load_job(job_id)["status"] == "failed" + + +def test_mode3_symbols_xml_counts_as_the_report_and_skips_the_wait(tmp_path): + """Mode-3 optimizations write `.symbols.xml` rather than the .htm the + INI asks for. Judging only the .htm would make every finished mode-3 run + look empty and pay the full relaunch grace window before succeeding.""" + report = tmp_path / "report.htm" + symbols = tmp_path / "report.symbols.xml" + symbols.write_text("") + + job = {"reportPath": str(report), "optimizationType": 3} + candidates = handler._report_candidates(job) + + assert str(symbols) in candidates + assert handler._any_report_exists(candidates) is True + + plain = {"reportPath": str(report), "optimizationType": 0} + assert handler._any_report_exists(handler._report_candidates(plain)) is False + + +def test_slow_clean_exit_without_report_fails_without_waiting(client, monkeypatch): + """A process that ran long enough to have been a real backtest is not an + update restart, so a missing report there fails immediately.""" + job_id = _submit(client) + waited = _run_with_exit_after(monkeypatch, handler.RELAUNCH_MAX_EXIT_SECONDS + 5) + + handler._execute_job(job_id) + + assert waited == [], "a long run that produced no report must fail without waiting" + assert jobs.load_job(job_id)["status"] == "failed" + + +# --- INI parsing: literals, not interpolation templates ----------------------- + + +def test_percent_in_an_ini_value_is_literal_text(): + """MT5 INI values are literals. A bare '%' is ordinary text in them — EA + comments and percentage inputs carry it routinely — but configparser's + default interpolation treats it as a template escape and refuses to parse, + failing the submission before the test ever runs.""" + parsed = handler._parse_ini( + "[Tester]\nExpert=MyEA\n[Common]\nComment=Risk 2% per trade\n" + ) + + assert parsed["Common"]["Comment"] == "Risk 2% per trade" + + +def test_percent_paren_in_an_ini_value_is_not_a_substitution(): + """The shape interpolation would actually try to expand.""" + parsed = handler._parse_ini( + "[Tester]\nExpert=MyEA\nReport=Reports\\%(name)s.htm\n" + ) + + assert parsed["Tester"]["Report"] == "Reports\\%(name)s.htm" diff --git a/tests/test_duration.py b/tests/test_duration.py index f583279..729060d 100644 --- a/tests/test_duration.py +++ b/tests/test_duration.py @@ -26,6 +26,12 @@ ("1h2m3s", 3723), ("-2h", -7200), ("-1h30m", -5400), + # Day-suffixed forms (retention windows) + ("1d", 86400), + ("2d", 172800), + ("1d12h", 129600), + ("-3d", -259200), + ("0d", 0), # Whitespace tolerated (" 3h ", 10800), ]) diff --git a/tests/test_terminal_instances.py b/tests/test_terminal_instances.py index 06d4237..c1ebbfc 100644 --- a/tests/test_terminal_instances.py +++ b/tests/test_terminal_instances.py @@ -111,4 +111,43 @@ def test_config_helper_nginx_conf_includes_instance_routes(duplicate_terminals_c assert "location /darwinex/live/a/" in content assert "location /darwinex/live/b/" in content assert f"location /ictrading/demo/{helper.DEFAULT_INSTANCE}/" in content - assert "location /ictrading/demo/" in content \ No newline at end of file + assert "location /ictrading/demo/" in content + + +def test_check_health_probes_only_this_vms_terminals(): + """The per-VM healthcheck must apply the same group filter as the launcher. + + Probing every port in config.yaml means each VM reports the other VM's + terminals DOWN and the container is permanently unhealthy, which makes a + real failure indistinguishable from the standing noise. On 2026-08-04 the + fast VM had a 25,272-long failing streak listing only bulk-VM ports while + all 12 of its own terminals were serving. + """ + src = (Path(__file__).resolve().parents[1] / "scripts" / "check_health.py").read_text( + encoding="utf-8" + ) + + assert "_vm_group_filter" in src, "check_health.py no longer applies the per-VM filter" + assert "_in_group" in src + # The filter has to be applied to the list that gets probed, not merely + # imported and forgotten. + assert "_in_group(t, allowed)" in src + + +def test_container_healthcheck_probes_only_this_vms_ports(): + """The container healthcheck must scope its probes to this VM's terminals. + + Docker surfaces this script's verdict as the container health status. Taking + every port in config.yaml means each VM probes the other VM's terminals and + reports unhealthy forever, so the status carries no signal and a real outage + looks exactly like the standing noise. + """ + src = (Path(__file__).resolve().parents[1] / "scripts" / "healthcheck.sh").read_text( + encoding="utf-8" + ) + + assert "VM_GROUP=/shared/config/vm-group.txt" in src + assert 'groupfile="$VM_GROUP"' in src, "the group file is no longer passed to the filter" + # No group file must mean no filter, not zero ports — otherwise every + # single-VM install fails its healthcheck. + assert 'if (!have_group) { print port; return }' in src