diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 72136e9..e98b161 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -21,7 +21,7 @@ jobs: with: python-version: ${{ matrix.python }} cache: pip - - run: python -m pip install -e '.[dev,viz]' + - run: python -m pip install -e '.[dev]' - if: matrix.backend == 'numba-cuda' # The device backend imports CUDA runtime metadata even for CPU-only tests. # Only the runtime is needed here; simulation does not require a GPU/driver. diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index aaf74bf..3f40900 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -1,16 +1,18 @@ # Contributing -Install `.[dev,viz]`; run `pytest`, `ruff check .`, and `ruff format --check .`. +Install `.[dev]`; run `pytest`, `ruff check .`, and `ruff format --check .`. See [tests/README.md](tests/README.md) for CPU, simulator, and hardware commands. Keep boundaries clear: - `src/gpu_backtest/core/` contains generic computation only. It must not import - examples, test references, benchmark code, CLI, workflows, or deployment tools. -- `workflows/` uses core primitives for splitting/analysis/output charts. + examples, test references, benchmark code, CLI, or deployment tools. - Example trading rules/data stay under `examples/` and helper code under `tools/`. - CLI modules are thin adapters; they do not own numerical or cloud logic. +Keep output raw: no built-in scoring, ranking, statistical inference, splitting or charts. +Preserve four grouped array values and their parameter alignment. + New example strategies need a device-function spec and hand-calculated case. Use synthetic data. Keep credentials, private plugins/datasets, and research output out of Git. New tracked files require a reviewed update to the explicit allowlist diff --git a/README.md b/README.md index b1cfd1d..4837558 100644 --- a/README.md +++ b/README.md @@ -2,10 +2,12 @@ ## 10× faster on our public billion-pair benchmark -**CPU: 7 min 44.60 s → RTX 4090: 44.95 s.** Same RSI grid: +**CPU: 7 min 44.60 s → RTX 4090: 44.95 s.** The same public RSI grid: **1,000,000,000 pairs × 1,024 bars**, compared with an **eight-thread compiled -Numba CPU baseline**. Measured speedup **10.34×**, saving about seven minutes per -sweep. Cloud setup is additional; [method and raw evidence](docs/benchmarks.md). +Numba CPU baseline**. Measured speedup **10.34×** on v0.4, saving about seven +minutes per sweep. Cloud setup is additional; [method and raw evidence](docs/benchmarks.md). +v0.6 preserves the GPU reductions and saves raw results instead of ranked analysis. +The historical timing includes v0.4's output processing; it is not a new v0.6 timing. - **Your algorithm:** load a strategy module/object; private rules can remain private. - **Large grids:** deterministic GPU reductions without materializing the full return matrix. @@ -15,28 +17,26 @@ sweep. Cloud setup is additional; [method and raw evidence](docs/benchmarks.md). ```text src/gpu_backtest/ - core/ GPU engine, kernels, grids, indicators, statistics, output - workflows/ Generic split/common analysis and charts + core/ GPU engine, kernels, grids, indicators, raw output cli/ Thin command-line adapters examples/gpu_backtest_examples/ - rsi/ Example strategy + config + generated CSV - data.py Example/benchmark synthetic data generator + rsi/ Educational strategy + config + generated CSV + data.py Synthetic data generator tools/gpu_backtest_tools/ runpod/ Optional API / SSH / bundle / lifecycle helper - benchmarks/ Performance runner and compiled CPU baseline - checks/ Hardware smoke checks and CPU test reference + benchmarks/ Performance runner and compiled CPU comparison + checks/ Hardware smoke checks and CPU numeric reference tests/ - cpu/ CPU contracts, examples, workflows, helper tests + cpu/ Contracts, examples, CLI, helpers and CPU reference gpu/ Isolated CUDA simulation and real-GPU tests docs/ Strategy contract, RunPod usage, benchmark method benchmarks/results/ Historical measurements and validation evidence ``` -**Start with `core/engine.py`** for the GPU run. Numerical kernels are in +**Start with `core/engine.py`** for a GPU run. CUDA kernels are in `core/kernels.py`; trading rules are supplied by a plugin. Core imports no example, -CPU comparison engine, benchmark, or RunPod code. CPU preprocessing of market data -and indicator tables is part of the GPU pipeline; the separate CPU backtest baseline -is only a benchmark/reference tool. +CPU backtest comparison, benchmark, or RunPod code. Market validation and indicator +preparation use the CPU before GPU execution. ## Try the RSI example without a GPU @@ -45,17 +45,13 @@ Python 3.11–3.13, from a checkout: ```bash python3 -m venv .venv source .venv/bin/activate -python -m pip install -e '.[dev,viz]' +python -m pip install -e '.[dev]' NUMBA_ENABLE_CUDASIM=1 gpu-backtest run \ --config examples/gpu_backtest_examples/rsi/config.json --out-prefix runs/rsi -gpu-backtest charts --input runs/rsi_top_entry.csv runs/rsi_top_exit.csv \ - --output runs/rsi.html ``` -The simulator is for tiny examples/tests. Open `runs/rsi.html` in a browser; -its Vega libraries load from a public CDN. The [RSI example](examples/gpu_backtest_examples/rsi/README.md) -is educational and uses generated data. It is packaged separately as -`gpu_backtest_examples.rsi.strategy`, not as an engine builtin. +The CUDA simulator is for tiny examples/tests. The [RSI example](examples/gpu_backtest_examples/rsi/README.md) +uses generated data and is packaged separately as `gpu_backtest_examples.rsi.strategy`. ## Use your own strategy @@ -69,43 +65,64 @@ results = run(example, "market.csv", "runs/custom", buy=0.0015, sell=0.0015) ``` Or use `gpu-backtest run --strategy my_strategies.example --input market.csv ---out-prefix runs/custom`. Plugins implement a complete Numba CUDA trading loop; -see [the contract](docs/strategy.md). Entry/exit parameters are separate Cartesian -axes (up to four dimensions each), with up to four precomputed tables. Shared-parameter -diagonal-only sweeps are not supported. The API returns grouped sums/squares and -optional ranked artifacts, not a trade ledger, equity curve, Sharpe, or drawdown series. +--out-prefix runs/custom`. Plugins implement their complete Numba CUDA trading loop; +see [the contract](docs/strategy.md). Entry/exit parameters form separate Cartesian +axes, up to four dimensions each, with up to four precomputed indicator tables. +Shared-parameter diagonal-only sweeps are not supported. -## Optional workflows and helpers +## Raw output -| Command | Owner / purpose | +A run returns four float64 arrays and, when an output prefix is supplied, writes: + +- `_results.npz`: `entry_sum`, `entry_sumsq`, `exit_sum`, `exit_sumsq`. +- `_manifest.json`: ordered parameter dimensions, counts, fees, array schema, + result-file SHA-256, and reduction consistency checks. + +Entry arrays aggregate each entry parameter set across **all exits**; exit arrays +aggregate each exit set across **all entries**. Array indices follow Cartesian +parameter order, with the last dimension varying fastest. Individual returns and +their squares round to float32 before float64 accumulation. No full pair matrix, +trade ledger or equity curve is stored. + +```python +import numpy as np + +with np.load("runs/rsi_results.npz", allow_pickle=False) as results: + entry_sum = results["entry_sum"] + exit_sum = results["exit_sum"] +``` + +Apply your own analysis downstream. v0.6 removes built-in scoring, ranking, +common-parameter analysis, date-split pipelines and charts. The `top_n` option and +old ranked CSV/schema-1 artifacts are removed; numerical return arrays remain the +same. Old config keys are rejected rather than silently ignored. + +[RTX 4090 validation](benchmarks/results/rtx4090_raw_output_parity_20261005.json): +v0.5/v0.6 returned arrays matched byte-for-byte on small, 16,777,216-pair and +1,000,000,000-pair grids. Clean-wheel results matched the default RunPod bundle, +all five physical-GPU tests passed, and owned-pod deletion was confirmed. + +## RunPod and validation tools + +| Command | Purpose | |---|---| -| `pipeline` | `workflows/`: split, run each segment, intersect ranked parameters | -| `split`, `common`, `charts` | `workflows/`: standalone data/result analysis | -| `runpod` | `tools/runpod/`: lease one GPU, upload selected files, download, delete | -| `benchmark` | `tools/benchmarks/`: reproduce the public CPU/GPU measurements | -| `gpu-check` | `tools/checks/`: small numeric checks on actual hardware | +| `run` | One GPU backtest with raw results | +| `runpod` | Lease one GPU, upload selected files, run, download, delete | +| `benchmark` | CPU/GPU performance and numeric comparison | +| `gpu-check` | Small numerical checks on actual hardware | ```bash gpu-backtest runpod --config examples/gpu_backtest_examples/rsi/config.json \ - --ssh-key ~/.ssh/runpod_ed25519 --output-dir runs/runpod-rsi --charts + --ssh-key ~/.ssh/runpod_ed25519 --output-dir runs/runpod-rsi ``` -RunPod requires your account/API key and registered SSH key. It is an optional -execution helper; [setup, manual GPU route, and cleanup](docs/runpod.md). -Commands keep their existing names. The public `from gpu_backtest import run` API -and old `rsi_meanrev` shorthand remain usable; direct internal imports moved under -`core/`, `workflows/`, or `gpu_backtest_tools` in v0.5. - -## Performance and development - -| Same billion-pair RSI job | CPU, eight threads | RTX 4090 | Saved per sweep | -|---|---|---|---| -| 1,000,000,000 pairs × 1,024 bars, through output | 7 min 44.60 s | 44.95 s | 6 min 59.65 s; 10.34× faster | +RunPod requires your account/API key and registered SSH key. It defaults to one +full-input run; [setup, manual GPU route, and cleanup](docs/runpod.md). +The public `from gpu_backtest import run` API and old `rsi_meanrev` shorthand +remain usable. Direct internal imports moved under `core/` or +`gpu_backtest_tools` in v0.5. -This is one measured run per side on the published setup, not a universal speed -claim. Small CPU jobs may not justify cloud startup. Statistics describe parameter -combination groups, not independent market samples or a forecast of profits. -See [benchmark details](docs/benchmarks.md) for scope and raw data. +## Development ```bash python -m pytest diff --git a/benchmarks/results/rtx4090_raw_output_parity_20261005.json b/benchmarks/results/rtx4090_raw_output_parity_20261005.json new file mode 100644 index 0000000..1228883 --- /dev/null +++ b/benchmarks/results/rtx4090_raw_output_parity_20261005.json @@ -0,0 +1,358 @@ +{ + "schema_version": 1, + "recorded_at_utc": "2026-10-05T15:48:45.133035+00:00", + "baseline_version": "0.5.0", + "candidate_version": "0.6.0", + "baseline_commit": "6bf72d725ce95ac7b34789ef2b811f6b0b08b9d3", + "gpu": { + "name": "NVIDIA GeForce RTX 4090", + "compute_capability": [ + 8, + 9 + ], + "driver_query": "NVIDIA GeForce RTX 4090, 570.172.08" + }, + "python": "3.11.10", + "packages": { + "gpu-backtest-engine": "0.6.0", + "numba": "0.65.1", + "numba-cuda": "0.30.4", + "numpy": "2.4.6", + "pandas": "3.0.6", + "llvmlite": "0.47.0", + "cuda-bindings": "12.9.9" + }, + "simulation": false, + "kernel_file_sha256": "3c6563b01e4489c424cd69ba7b3dcc83c00107444f5a313268a56661e9d2e3c8", + "all_arrays_byte_identical": true, + "default_runpod_bundle_matches_clean_wheel": true, + "current_npz_matches_api_arrays": true, + "current_manifest_schema_and_hash_checked": true, + "physical_gpu_tests_passed": 5, + "comparisons": [ + { + "name": "example_fee", + "bars": 128, + "pair_count": 16, + "data_sha256": "d66e7b48d7daea048db45f2c8b2d39855cd0dc4cc11169cf6370d7a100d14ab3", + "entry_dims": [ + [ + "p_e", + 2, + 4, + 2, + false + ], + [ + "buy_lvl", + 20, + 30, + 10, + false + ] + ], + "exit_dims": [ + [ + "p_x", + 2, + 4, + 2, + false + ], + [ + "sell_lvl", + 70, + 80, + 10, + false + ] + ], + "buy": 0.0015, + "sell": 0.0015, + "threads_per_block": 128, + "arrays": { + "entry_sum": { + "shape": [ + 4 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "1f88270eabd98af76559aa7acfeef52f90072e9a75db235e8f40a0695aba9726" + }, + "entry_sumsq": { + "shape": [ + 4 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "0b269272a61c098f3f26eebe8faff26b65921f1ec16a081f8485e1c4a71a27cd" + }, + "exit_sum": { + "shape": [ + 4 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "4405df6a7f94385e99c3b5af3a3092383775725ea4f236428f23fe3f025a00c2" + }, + "exit_sumsq": { + "shape": [ + 4 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "f13dde05c1c32ebc64f34af57103375444c0270c6d7d5e348ca48e9d83a87bcb" + } + } + }, + { + "name": "example_zero_fee", + "bars": 128, + "pair_count": 16, + "data_sha256": "d66e7b48d7daea048db45f2c8b2d39855cd0dc4cc11169cf6370d7a100d14ab3", + "entry_dims": [ + [ + "p_e", + 2, + 4, + 2, + false + ], + [ + "buy_lvl", + 20, + 30, + 10, + false + ] + ], + "exit_dims": [ + [ + "p_x", + 2, + 4, + 2, + false + ], + [ + "sell_lvl", + 70, + 80, + 10, + false + ] + ], + "buy": 0.0, + "sell": 0.0, + "threads_per_block": 128, + "arrays": { + "entry_sum": { + "shape": [ + 4 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "6078ae66029d60ab20d93f42063a941f8c9ca27c8852330fa6b33cccf0edf65a" + }, + "entry_sumsq": { + "shape": [ + 4 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "b803e529bdf9931585219ac2ccc683e311912e7a4952688e1fc7b04e87ed3f3d" + }, + "exit_sum": { + "shape": [ + 4 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "a440f4605201f2f136891826f1fce613d240641e8a36c130d1807910d3beb9e2" + }, + "exit_sumsq": { + "shape": [ + 4 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "62fa5f5c079499543709b4e28a1ab8603efb64face4f251d5176e2805652ab91" + } + } + }, + { + "name": "comparison", + "bars": 1024, + "pair_count": 16777216, + "data_sha256": "e57584c2def67be06de09b52ee72463d0626f20de30d2aace06f735d823909e2", + "entry_dims": [ + [ + "p_e", + 2, + 65, + 1, + false + ], + [ + "buy_lvl", + 20.0, + 51.5, + 0.5, + true + ] + ], + "exit_dims": [ + [ + "p_x", + 2, + 65, + 1, + false + ], + [ + "sell_lvl", + 49.0, + 80.5, + 0.5, + true + ] + ], + "buy": 0.0015, + "sell": 0.0015, + "threads_per_block": 128, + "arrays": { + "entry_sum": { + "shape": [ + 4096 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "fe4529ddae6bf486be07bafce26ffe2c0c394529ea0ceb5906d3b4e4ec0acd5f" + }, + "entry_sumsq": { + "shape": [ + 4096 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "476faf29b1eed0b64114f4d55ab872870d51ea266cbc97e745beff6647d26e01" + }, + "exit_sum": { + "shape": [ + 4096 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "4ec9f5f536a99225cf381a565511b31b58b396f6af0a810c305a06743548335e" + }, + "exit_sumsq": { + "shape": [ + 4096 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "f5ee99c688a5a4fec3cc3efdfa6f2474853aee934dd6652d6eb5963c810ab454" + } + } + }, + { + "name": "billion", + "bars": 1024, + "pair_count": 1000000000, + "data_sha256": "e57584c2def67be06de09b52ee72463d0626f20de30d2aace06f735d823909e2", + "entry_dims": [ + [ + "p_e", + 2, + 401, + 1, + false + ], + [ + "buy_lvl", + 10, + 59, + 1, + false + ] + ], + "exit_dims": [ + [ + "p_x", + 2, + 1001, + 1, + false + ], + [ + "sell_lvl", + 50, + 99, + 1, + false + ] + ], + "buy": 0.0015, + "sell": 0.0015, + "threads_per_block": 128, + "arrays": { + "entry_sum": { + "shape": [ + 20000 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "88a3fe69d130efeda5e1d51ac99230982368c74ce4ec55a0c7a34c72aab10e06" + }, + "entry_sumsq": { + "shape": [ + 20000 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "ef9cafc4c5ed4dc56261f261491dd872aff0f1d68ef6a145132f1d83c09ea4fe" + }, + "exit_sum": { + "shape": [ + 50000 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "492d4bccfd3f5d909a6fe6c8b41375a83321b865d4e91ee318893a5204d64114" + }, + "exit_sumsq": { + "shape": [ + 50000 + ], + "dtype": "float64", + "byte_identical": true, + "max_absolute_difference": 0.0, + "sha256": "10a22ec01cf8d5c80bd609c45321d4a08218409e3fb8271d78ea4397c4853dcb" + } + } + } + ], + "scope": "Numerical parity, not a speed benchmark. Old API results independently saved; candidate artifacts checked against returned arrays. Ranked output was intentionally removed.", + "cleanup": "deleted; absence confirmed through RunPod API", + "active_pods_after_validation": 0, + "candidate_source_commit": "99f0a88f27a375c389794893f9d6e4c334baf5b6", + "tested_wheel_sha256": { + "0.5.0": "682fa9dc682c808009edacc3f5fb6c82bac51a4cde62feb60afb8e5108d52eb5", + "0.6.0": "90e2a1646f49d885aba8a50d7de5ffc78f86d6618e4f21970053546a9ac3ebc8" + } +} diff --git a/docs/benchmarks.md b/docs/benchmarks.md index a84329c..0e5e8f4 100644 --- a/docs/benchmarks.md +++ b/docs/benchmarks.md @@ -1,6 +1,6 @@ # Public CPU/GPU benchmark: minutes saved on a billion-pair sweep -Measured on 2026-10-03 with the public RSI example and deterministic generated +Historical v0.4 measurement on 2026-10-03 with the public RSI example and deterministic generated OHLCV data. The headline compares **the same full billion-pair job on CPU and GPU**. The [full raw report](../benchmarks/results/rtx4090_rsi_matched_billion_20261003.json) contains exact timings, both parameter grids, hardware/software, and input SHA-256. @@ -83,7 +83,7 @@ host with eight CPU threads, 16,777,216 pairs × 1,024 bars took CPU median 6.41 vs GPU median 1.654 s, or 3.88×. Its GPU-only billion run took 43.341 s; that older record did not run the full billion grid on CPU. -The new matched run also measured that smaller grid: CPU median 8.286 s vs GPU +The matched v0.4 run also measured that smaller grid: CPU median 8.286 s vs GPU median 1.717 s, or 4.83×. The difference between hosts illustrates why results must state hardware, thread count, workload, and timing scope. @@ -94,7 +94,12 @@ input/indicator preparation, GPU transfers/allocation, compilation, result copie statistics and CSV writing are excluded. CPU output array allocation is included. All four grouped arrays matched exactly in those smaller comparisons. -## Reproduce the full comparison +## Run the current comparison + +v0.6 keeps the same grids, numerical kernels and CPU/GPU checks, but writes raw +NPZ/schema-2 manifests instead of statistics and ranked CSVs. The historical JSON +reports above are unchanged. New reports describe their actual raw-output timing; +current runs are not an exact reproduction of the old output-processing workload. On a compatible GPU machine: @@ -115,3 +120,21 @@ The RunPod route now includes the full CPU sweep. It can take several minutes, rents one paid GPU pod, downloads results, and deletes its pod. Hardware, host load, thread count, strategy, grid shape, bar count, and cold-start overhead affect whether GPU use is worthwhile. No universal speedup or trading-return claim is made. + + +## v0.6 raw-output validation + +On 2026-10-05, one RTX 4090 RunPod ran the released v0.5 wheel and reviewed v0.6 +wheel against the same inputs, dimensions, fees and block size. Every float64 array +matched byte-for-byte across four workloads: tiny RSI with/without fees, 16,777,216 +pairs, and 1,000,000,000 pairs (both large grids used 1,024 bars). The entire CUDA +kernel file was identical. Candidate NPZ values matched API returns, ordered +parameter manifests and file hashes were checked, and default RunPod bundle output +matched the clean wheel. Five physical-GPU tests passed, including independent +single-combination profit, commission and loss anchors. + +[Validation evidence](../benchmarks/results/rtx4090_raw_output_parity_20261005.json) +records grids, data/array/kernel hashes, versions, hardware and zero differences. +This validates numerical parity, not a new speedup claim. The result files are +intentionally NPZ/schema-2 rather than ranked CSV/schema-1. Download completed, +the owned pod was deleted, and the API confirmed zero active pods. diff --git a/docs/runpod.md b/docs/runpod.md index 834240d..f19af94 100644 --- a/docs/runpod.md +++ b/docs/runpod.md @@ -1,165 +1,110 @@ -# Run on a rented RunPod GPU +# RunPod GPU helper -RunPod supplies the GPU computer; the same engine runs on that computer as on a -local NVIDIA workstation. The launcher runs on your laptop, leases one GPU, -uploads selected files, runs the job, downloads results, and deletes the pod. -You do not need a local NVIDIA GPU. +Use a local NVIDIA machine or lease one RunPod GPU. RunPod is an optional execution +helper, separate from the numerical engine. It runs a backtest and downloads raw +results; it does not analyze or rank them. -## Account and SSH setup +## Prerequisites -You need a funded RunPod account, an API key allowed to create/read/delete Pods, -and an SSH key whose public half is registered under RunPod Credentials. -See [RunPod SSH setup](https://docs.runpod.io/pods/configuration/use-ssh). +- Python 3.11–3.13 locally, an OpenSSH client, and this package installed. +- Your RunPod account with credits and API access. +- A registered SSH public key; keep its private key on your own machine. +- API key supplied through `RUNPOD_API_KEY`, or the local RunPod CLI configuration + at `~/.runpod/config.toml`. Never put credentials in the repository/config JSON. -If you do not already have a suitable key, create one: +The helper reads credentials locally. API credentials are not sent over SSH or +included in upload archives. Your private strategy stays outside this repository; +running it on a rented GPU necessarily uploads the selected strategy source and data. -```bash -ssh-keygen -t ed25519 -f ~/.ssh/runpod_ed25519 -cat ~/.ssh/runpod_ed25519.pub -``` - -Register the public key with RunPod; keep the private key on your laptop. The -launcher uses OpenSSH batch mode, so unlock encrypted keys in your SSH agent first. -New host keys use `accept-new`, with a separate known-hosts file for each run. - -Provide the API key through `RUNPOD_API_KEY`, or an `apikey` field in your local -`~/.runpod/config.toml`. The key is excluded from the upload, SSH process environment, -remote job, and state file. Server error bodies/auth headers are not printed. - -## Automated example - -From an installed checkout: +## Run one backtest ```bash gpu-backtest runpod --config examples/gpu_backtest_examples/rsi/config.json \ - --ssh-key ~/.ssh/runpod_ed25519 --output-dir runs/runpod-rsi --charts + --ssh-key ~/.ssh/runpod_ed25519 --output-dir runs/runpod-rsi ``` -The default `pipeline` mode evaluates the full period and its two halves and -produces common-parameter results. Add `--mode run` for one full-period sweep. -To run only the built-in numeric hardware checks: +Default mode is `run`, on one RTX 4090 secure-cloud pod. A hardware smoke check +runs first, followed by one backtest over the complete configured input CSV. +The helper then downloads `output/` and attempts deletion on success, errors, +Ctrl-C or termination. Check `runpod_state.json` for `cleanup: deleted`. -```bash -gpu-backtest runpod --mode check --ssh-key ~/.ssh/runpod_ed25519 \ - --output-dir runs/runpod-check -``` +The output directory must be new or empty. Downloaded files are under `results/`: -The output directory must be new or empty. A dry run reads no API credentials, -requires no SSH key, and rents no GPU: +- `result_results.npz`: four raw grouped float64 arrays. +- `result_manifest.json`: parameters, fees, counts, output schema/hash and checks. +- `gpu_check.json`, `gpu.csv`, `requirements-resolved.txt`: environment evidence. -```bash -gpu-backtest runpod --config examples/gpu_backtest_examples/rsi/config.json \ - --output-dir runs/runpod-preview --dry-run -``` +`remote.log` contains installation/job logs. Local `upload.tar.gz` and lifecycle +state are retained for inspection. If deletion fails, the helper reports its owned +pod ID; delete that pod in your RunPod console. `--keep-pod` explicitly retains a +paid pod and must be used deliberately. -It writes `upload.tar.gz` and lists the exact files it contains. Inspect this -bundle before using custom plugins. Its contents go to your rented machine; -they are not published to GitHub. - -To reproduce the [published CPU/GPU and billion-pair benchmark](benchmarks.md): +## Settings and dry run ```bash -gpu-backtest runpod --mode benchmark --ssh-key ~/.ssh/runpod_ed25519 \ - --output-dir runs/runpod-benchmark --max-seconds 1800 +gpu-backtest runpod --config examples/gpu_backtest_examples/rsi/config.json \ + --output-dir runs/runpod-preview --dry-run ``` -This mode needs no config/market file: it generates public synthetic data, requests -at least eight vCPUs, measures an eight-thread compiled CPU baseline and the GPU, -then runs the entire billion-pair RSI grid on **both** CPU and GPU. Allow several -minutes for the CPU measurement. It uses one paid GPU pod and the normal cleanup -behavior. Timing can vary on another host. - -## Execution and output - -The launcher validates the selected CSV, normalizes config paths, and packages -an explicit engine source-file list plus that CSV and any selected Python plugin -directory. It does not upload an entire repository. It creates one pod, records -its ID, waits for SSH, installs a fresh environment, and forces real CUDA execution. +Dry run validates and builds the selected upload without reading API credentials +or creating a pod. Supported modes are `run`, `check`, and `benchmark`. +`check` needs no config; `benchmark` uses the packaged public RSI/synthetic workload. -Each job first checks an explicit reduction matrix, repeat determinism, and -hand-calculated RSI returns on hardware. It then runs the configured job, optionally -creates charts, downloads results, and attempts pod deletion on completion, error, -timeout, or normal cancellation. Only the pod created for this launch is deleted. +The default hourly-rate cap is $1/hour and the startup/job/download deadline is +1,800 seconds. Use `--max-hourly-rate` and `--max-seconds` to change them. The rate +cap is checked after the provider returns the quote; an unacceptable quote causes +owned-pod deletion. This is not a provider-side total-spend cap. -`remote.log` records installation/job progress. `runpod_state.json` records the -pod ID, status, and cleanup outcome. Downloaded output is under `results/`, including -`gpu_check.json`, `gpu.csv`, exact resolved Python requirements, and the selected -job's artifacts. Downloaded tar links and path traversal are rejected. +The default image is +`runpod/pytorch:2.4.0-py3.11-cuda12.4.1-devel-ubuntu22.04`. The remote recipe installs +the bundled `tools/gpu_backtest_tools/runpod/requirements.txt` environment, disables +CUDA simulation, and uses the image's CUDA toolkit. Image overrides must provide +Python 3.11–3.13, OpenSSH, NVIDIA access, and a compatible development toolkit. -## Environment and limits +## Your strategy -Default image: - -```text -runpod/pytorch:2.4.0-py3.11-cuda12.4.1-devel-ubuntu22.04 +```bash +gpu-backtest runpod --config my-job.json --plugin-dir ./my-plugins \ + --ssh-key ~/.ssh/runpod_ed25519 --output-dir runs/custom-gpu ``` -This existing official image supplies Python 3.11, SSH, and CUDA development -libraries. The engine does not use its PyTorch installation. The launcher creates -an isolated venv and installs the engine with the pinned -[validated environment](../tools/gpu_backtest_tools/runpod/requirements.txt), using the -image's CUDA toolkit. Optional charts use Altair 6.3.0. Overrides must supply Python 3.11–3.13, -compatible CUDA development libraries/driver, SSH, tar, and `nvidia-smi`. - -The default is one RTX 4090 on Secure Cloud. Override explicitly with `--gpu`, -`--cloud`, or `--image`; there is no silent GPU/cloud fallback. The -[RunPod Pods API](https://docs.runpod.io/api-reference/pods/POST/pods) handles -availability and allocation. +`--plugin-dir` is a Python import root. For `my_strategies.example`, it should +contain `my_strategies/example.py` and package initialization as needed. Only +selected `.py` files are uploaded; Git, environments and caches are excluded. +Plugin symlinks are rejected. The helper creates an isolated virtual environment +and installs only the public package and bundled requirements. Extra plugin dependencies +are not installed automatically; use the manual route and install them into that +virtual environment, or adapt the remote installation recipe. -`--max-hourly-rate` defaults to 1.00 USD/hour. The reported rate is checked **after -allocation**; an over-limit or unknown-rate pod is immediately deleted, even with -`--keep-pod`. A brief charge may be incurred before rejection. `--max-seconds` -defaults to 1800 and limits local startup/SSH/job waits after allocation; API -cleanup requests have their own bounded timeouts. +The config accepts only `strategy`, `input`, `buy`, `sell`, `entry_dims`, +`exit_dims`, `expected_interval`, and `threads_per_block`. Relative input paths +resolve against the config directory. Removed analysis fields, `pipeline` mode +and `--charts` are rejected before allocation. -`--keep-pod` deliberately retains a debug machine and its ongoing charges. Normal -interruption triggers cleanup, but a killed launcher, lost machine/network, or -unavailable API can prevent deletion. Check the state file and RunPod Console in -that case. Ambiguous creation is never blindly retried: the launcher reconciles -the unique launch name with your pod list or reports that name for manual inspection. +## Manual GPU route -## Private strategies - -Name the dotted module in your JSON config, for example `my_strategies.example`, -and select the directory containing that package: +Create your own GPU pod, register your SSH key, and connect using the provider's +SSH command. Upload only the public package plus the strategy/data you intend to run. +On the remote machine: ```bash -gpu-backtest runpod --config /path/to/job.json \ - --plugin-dir /path/to/strategy-project/src \ - --ssh-key ~/.ssh/runpod_ed25519 --output-dir runs/custom +python3 -m venv env +source env/bin/activate +python -m pip install -e . -r tools/gpu_backtest_tools/runpod/requirements.txt +NUMBA_ENABLE_CUDASIM=0 gpu-backtest gpu-check --output runs/gpu_check.json +NUMBA_ENABLE_CUDASIM=0 gpu-backtest run \ + --config examples/gpu_backtest_examples/rsi/config.json --out-prefix runs/rsi ``` -Only `.py` files are copied from the selected directory. Git, venv, and cache -directories and non-Python files are excluded; source symlinks are rejected. -Put the strategy package directly beneath the selected directory. Plugins can -use Numba, NumPy, pandas, the engine, and the standard library. Extra dependencies, -non-Python package resources, namespace-only strategy entry packages, and private -package-index authentication are not installed automatically. Use an appropriate -manual route for these cases. The launcher's isolated venv does not inherit global -Python packages preinstalled in a custom image. Plugins are trusted code. +Download the NPZ/manifest files and delete your manually created pod through +RunPod. The automatic helper cannot manage a pod created outside its lifecycle. -## Manual route - -Create a Pod with the image above, exposing TCP port 22, then connect using the -SSH command shown by RunPod. Run these commands **inside the Pod**: +## Benchmark ```bash -git clone https://github.com/howardc38/gpu-backtest-engine.git -cd gpu-backtest-engine -python3 -m venv .venv -source .venv/bin/activate -python -m pip install -e '.[dev,viz,cuda]' 'cuda-bindings>=12.9.1,<13' -NUMBA_ENABLE_CUDASIM=0 gpu-backtest gpu-check --output runs/gpu_check.json -NUMBA_ENABLE_CUDASIM=0 gpu-backtest pipeline \ - --config examples/gpu_backtest_examples/rsi/config.json --output-dir runs/example -gpu-backtest charts --input runs/example/*_top_entry.csv runs/example/*_top_exit.csv \ - --output runs/example/charts.html +gpu-backtest runpod --mode benchmark --ssh-key ~/.ssh/runpod_ed25519 \ + --output-dir runs/runpod-benchmark --max-seconds 1800 ``` -Download `runs/` before deleting the Pod. Manual pods are managed by you; the -launcher deletes only pods it creates. Small hardware checks establish GPU -compilation and explicit numeric cases, not large-grid speed or every strategy's -correctness. - -See [the hardware validation record](../benchmarks/results/runpod_validation_20261003.md) for the tested image, -driver, package versions, numeric results, and successful cleanup. +This runs the full billion-pair GPU and eight-thread CPU workloads and may take +several minutes. See [measurement scope and historical evidence](benchmarks.md). diff --git a/docs/strategy.md b/docs/strategy.md index 8b4150c..8eb6bd2 100644 --- a/docs/strategy.md +++ b/docs/strategy.md @@ -4,7 +4,8 @@ Pass a module/object to `gpu_backtest.run`, or an installed dotted module name such as `my_strategies.example`. The separate example module is `gpu_backtest_examples.rsi.strategy`. The old `rsi_meanrev` shorthand is translated only at the public API/CLI edge; the core imports no example strategy. -The package neither copies nor uploads an external plugin. +A local engine run imports your installed plugin without copying it. The optional +[RunPod helper](runpod.md) uploads selected strategy source and input data to the rented GPU. ## Required declarations @@ -19,7 +20,7 @@ Each dimension is `(name, low, high, step, is_float)`. Endpoints are inclusive; `high` must lie on the grid. Names must be unique across both sides. Each side has one to four dimensions. Overrides may change ranges, but must preserve the declared names and order. Table windows must be positive integer dimensions. -Ranked output requires at least two combinations on each side. +A single combination on either side is supported, including raw file output. The last dimension varies fastest. `ep` and `xp` are four-element float64 arrays; only the slots corresponding to declared dimensions are defined. Integer @@ -99,3 +100,26 @@ not establish that the intended rules are correct. See [the separate RSI example](../examples/gpu_backtest_examples/rsi/README.md) for a complete plugin, runnable config/data, and its execution/fee conventions. + + +## Raw results + +`run()` returns `entry_sum`, `entry_sumsq`, `exit_sum`, `exit_sumsq`, each float64. +With an output prefix it saves the same arrays in `_results.npz`, plus a +schema-2 `_manifest.json`. Entry arrays have `entry_count` elements; exit +arrays have `exit_count` elements. Each side aggregates across all combinations +of the opposite side; these are grouped results, not a full pair-return matrix. +Each per-pair return and square rounds to float32 before float64 accumulation. + +Array order follows declared dimension order; the last dimension varies fastest. +The manifest stores dimensions as `[name, low, high, step, is_float]`, fees, counts, +reduction checks, array dtypes/shapes and the NPZ SHA-256. Decode index `i` with +`gpu_backtest.core.grid.decode_values(dims, i)` using the corresponding dimensions. + +Output contains no scoring, ranking, probabilities or intervals. `top_n` and date +split settings are no longer accepted. Read NPZ with `numpy.load(..., allow_pickle=False)` +and apply your own downstream analysis. + +The manifest marks completed output. A failed publication invalidates that marker; +ignore an NPZ without its matching manifest. Check the recorded SHA-256 before +consuming a persisted result pair. diff --git a/examples/gpu_backtest_examples/rsi/README.md b/examples/gpu_backtest_examples/rsi/README.md index 1ca0d1d..0f7447b 100644 --- a/examples/gpu_backtest_examples/rsi/README.md +++ b/examples/gpu_backtest_examples/rsi/README.md @@ -39,3 +39,8 @@ For the seven-bar hand-calculated fixture, the zero-fee trade buys 125 units at 80 and sells them at 120: return 50%. At 0.15% per side, buy fee is 15 and sell fee is 22.5; final equity is 14,962.5, giving 49.625%. Tests also cover a losing round trip, no-entry conditions, and ignored final-bar pending orders. + + +The run writes `runs/rsi_results.npz` and `runs/rsi_manifest.json`. The four raw +arrays retain Cartesian parameter order; the manifest records dimensions and fees. +Apply your own result analysis outside the engine. diff --git a/examples/gpu_backtest_examples/rsi/config.json b/examples/gpu_backtest_examples/rsi/config.json index 44b12c4..badb37c 100644 --- a/examples/gpu_backtest_examples/rsi/config.json +++ b/examples/gpu_backtest_examples/rsi/config.json @@ -35,7 +35,5 @@ false ] ], - "top_n": 4, - "expected_interval": "1D", - "num_splits": 3 + "expected_interval": "1D" } diff --git a/pyproject.toml b/pyproject.toml index ac5c488..e24f70c 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,8 +4,8 @@ build-backend = "setuptools.build_meta" [project] name = "gpu-backtest-engine" -version = "0.5.0" -description = "Pluggable GPU parameter sweeps with deterministic entry/exit effect-size analysis" +version = "0.6.0" +description = "Pluggable GPU parameter sweeps with deterministic raw grouped results" readme = "README.md" requires-python = ">=3.11,<3.14" license = "MIT" @@ -15,7 +15,6 @@ dependencies = ["numba>=0.65,<0.66", "numpy>=2.3,<2.5", "pandas>=2.2,<4"] [project.optional-dependencies] dev = ["pytest>=8,<10", "build>=1,<2", "ruff>=0.14,<1"] -viz = ["altair>=6,<7"] cuda = ["numba-cuda>=0.30,<0.31; platform_system != 'Darwin'"] [project.scripts] diff --git a/scripts/check_release.py b/scripts/check_release.py index eac3dbe..48f6f7f 100644 --- a/scripts/check_release.py +++ b/scripts/check_release.py @@ -13,6 +13,7 @@ "LICENSE", "MANIFEST.in", "README.md", + "benchmarks/results/rtx4090_raw_output_parity_20261005.json", "benchmarks/results/rtx4090_rsi_20261003.json", "benchmarks/results/rtx4090_rsi_matched_billion_20261003.json", "benchmarks/results/runpod_validation_20261003.md", @@ -32,7 +33,6 @@ "src/gpu_backtest/__init__.py", "src/gpu_backtest/__main__.py", "src/gpu_backtest/cli/__init__.py", - "src/gpu_backtest/cli/analysis.py", "src/gpu_backtest/cli/backtest.py", "src/gpu_backtest/cli/options.py", "src/gpu_backtest/cli/tools.py", @@ -43,24 +43,15 @@ "src/gpu_backtest/core/indicators.py", "src/gpu_backtest/core/kernels.py", "src/gpu_backtest/core/output.py", - "src/gpu_backtest/core/statistics.py", "src/gpu_backtest/core/strategy.py", - "src/gpu_backtest/workflows/__init__.py", - "src/gpu_backtest/workflows/charts.py", - "src/gpu_backtest/workflows/common.py", - "src/gpu_backtest/workflows/pipeline.py", - "src/gpu_backtest/workflows/splits.py", "tests/README.md", "tests/cpu/test_architecture.py", "tests/cpu/test_benchmark.py", "tests/cpu/test_cli.py", - "tests/cpu/test_common_contract.py", "tests/cpu/test_engine_contract.py", "tests/cpu/test_examples.py", "tests/cpu/test_gpu_check.py", "tests/cpu/test_runpod.py", - "tests/cpu/test_split_contract.py", - "tests/cpu/test_statistics.py", "tests/cpu/test_strategy.py", "tests/cpu/test_tables.py", "tests/gpu/fixtures.py", diff --git a/src/gpu_backtest/__init__.py b/src/gpu_backtest/__init__.py index 026173e..d0f29a9 100644 --- a/src/gpu_backtest/__init__.py +++ b/src/gpu_backtest/__init__.py @@ -1,6 +1,6 @@ -"""Pluggable GPU parameter sweeps and grouped-return analysis.""" +"""Pluggable GPU parameter sweeps with raw grouped results.""" -__version__ = "0.5.0" +__version__ = "0.6.0" def resolve_strategy_name(value): diff --git a/src/gpu_backtest/cli/__init__.py b/src/gpu_backtest/cli/__init__.py index ca8aa57..85f9ab8 100644 --- a/src/gpu_backtest/cli/__init__.py +++ b/src/gpu_backtest/cli/__init__.py @@ -2,13 +2,13 @@ import argparse -from . import analysis, backtest, tools +from . import backtest, tools def main(argv=None): parser = argparse.ArgumentParser(prog="gpu-backtest") subparsers = parser.add_subparsers(dest="command", required=True) - for group in (backtest, analysis, tools): + for group in (backtest, tools): group.register(subparsers) args = parser.parse_args(argv) try: diff --git a/src/gpu_backtest/cli/analysis.py b/src/gpu_backtest/cli/analysis.py deleted file mode 100644 index 26e1cfc..0000000 --- a/src/gpu_backtest/cli/analysis.py +++ /dev/null @@ -1,49 +0,0 @@ -"""Analysis commands consume generic input/results rather than strategy code.""" - - -def register(subparsers): - split = subparsers.add_parser("split") - split.add_argument("--input", required=True) - split.add_argument("--start-date") - split.add_argument("--end-date") - split.add_argument("--num-splits", type=int, default=3) - split.set_defaults(handler=execute) - common = subparsers.add_parser("common") - common.add_argument("--input", nargs="+", required=True) - common.add_argument("--keys", required=True) - common.add_argument("--labels") - common.add_argument("--output", required=True) - common.set_defaults(handler=execute) - charts = subparsers.add_parser("charts") - charts.add_argument("--input", nargs="+", required=True) - charts.add_argument("--output", required=True) - charts.set_defaults(handler=execute) - - -def execute(args): - if args.command == "split": - from gpu_backtest.workflows.splits import _read_input, split_csv_by_date_range - - data, column = _read_input(args.input, "Time") - split_csv_by_date_range( - args.input, - args.start_date or data[column].min().strftime("%d-%m-%Y"), - args.end_date or data[column].max().strftime("%d-%m-%Y"), - args.num_splits, - column, - ) - elif args.command == "common": - from gpu_backtest.workflows.common import process_data - - process_data( - { - "input_files": args.input, - "key_columns": [p.strip() for p in args.keys.split(",") if p.strip()], - "file_labels": args.labels.split(",") if args.labels else None, - "output_file": args.output, - } - ) - else: - from gpu_backtest.workflows.charts import write_charts - - print(write_charts(args.input, args.output)) diff --git a/src/gpu_backtest/cli/backtest.py b/src/gpu_backtest/cli/backtest.py index 7019e40..6effe38 100644 --- a/src/gpu_backtest/cli/backtest.py +++ b/src/gpu_backtest/cli/backtest.py @@ -1,42 +1,23 @@ -"""Run commands: the GPU engine and generic split/common pipeline.""" +"""Command-line adapter for one GPU backtest run.""" from .options import job_options def register(subparsers): - for name in ("run", "pipeline"): - command = subparsers.add_parser(name) - command.add_argument("--config") - command.add_argument("--strategy", help="Installed strategy module name") - command.add_argument("--input") - command.add_argument("--buy", type=float) - command.add_argument("--sell", type=float) - command.add_argument("--top-n", dest="top_n", type=int) - command.add_argument("--threads-per-block", dest="threads_per_block", type=int) - command.add_argument("--expected-interval", dest="expected_interval") - command.set_defaults(handler=execute) - if name == "run": - command.add_argument("--out-prefix", required=True) - else: - command.add_argument("--output-dir", required=True) + command = subparsers.add_parser("run", help="Run a strategy and save raw GPU results") + command.add_argument("--config") + command.add_argument("--strategy", help="Installed strategy module name") + command.add_argument("--input") + command.add_argument("--buy", type=float) + command.add_argument("--sell", type=float) + command.add_argument("--threads-per-block", dest="threads_per_block", type=int) + command.add_argument("--expected-interval", dest="expected_interval") + command.add_argument("--out-prefix", required=True) + command.set_defaults(handler=execute) def execute(args): - strategy, source, options, config = job_options(args) - if args.command == "run": - from gpu_backtest.core.engine import run + from gpu_backtest.core.engine import run - run(strategy, source, args.out_prefix, **options) - else: - from gpu_backtest.workflows.pipeline import run_pipeline - - run_pipeline( - strategy, - source, - args.output_dir, - num_splits=config.get("num_splits", 3), - start_date=config.get("start_date"), - end_date=config.get("end_date"), - ranges=config.get("ranges"), - **options, - ) + strategy, source, options, _ = job_options(args) + run(strategy, source, args.out_prefix, **options) diff --git a/src/gpu_backtest/cli/options.py b/src/gpu_backtest/cli/options.py index 6d2ac86..a83c0b7 100644 --- a/src/gpu_backtest/cli/options.py +++ b/src/gpu_backtest/cli/options.py @@ -5,6 +5,17 @@ from gpu_backtest import resolve_strategy_name +CONFIG_KEYS = { + "strategy", + "input", + "buy", + "sell", + "entry_dims", + "exit_dims", + "threads_per_block", + "expected_interval", +} + def job_options(args): config = {} @@ -13,6 +24,9 @@ def job_options(args): config = json.loads(config_path.read_text()) if not isinstance(config, dict): raise ValueError("Config must be a JSON object") + unknown = set(config) - CONFIG_KEYS + if unknown: + raise ValueError(f"Unknown engine config keys: {sorted(unknown)}") if config.get("input"): config["input"] = str(config_path.parent / config["input"]) strategy = args.strategy if args.strategy is not None else config.get("strategy") @@ -23,7 +37,6 @@ def job_options(args): for name, default in ( ("buy", 0.0015), ("sell", 0.0015), - ("top_n", 100), ("threads_per_block", 128), ("expected_interval", None), ): diff --git a/src/gpu_backtest/cli/tools.py b/src/gpu_backtest/cli/tools.py index 2c9c9d4..a738d8a 100644 --- a/src/gpu_backtest/cli/tools.py +++ b/src/gpu_backtest/cli/tools.py @@ -22,16 +22,13 @@ def register(subparsers): runpod.add_argument("--output-dir", required=True) runpod.add_argument("--ssh-key") runpod.add_argument("--plugin-dir") - runpod.add_argument( - "--mode", choices=("run", "pipeline", "check", "benchmark"), default="pipeline" - ) + runpod.add_argument("--mode", choices=("run", "check", "benchmark"), default="run") runpod.add_argument("--image", default=DEFAULT_IMAGE) runpod.add_argument("--gpu", default="NVIDIA GeForce RTX 4090") runpod.add_argument("--cloud", choices=("SECURE", "COMMUNITY"), default="SECURE") runpod.add_argument("--max-seconds", type=int, default=1800) runpod.add_argument("--max-hourly-rate", type=float, default=1.0) runpod.add_argument("--keep-pod", action="store_true") - runpod.add_argument("--charts", action="store_true") runpod.add_argument("--dry-run", action="store_true") runpod.set_defaults(handler=execute) @@ -68,6 +65,5 @@ def execute(args): max_seconds=args.max_seconds, max_hourly_rate=args.max_hourly_rate, keep_pod=args.keep_pod, - charts=args.charts, dry_run=args.dry_run, ) diff --git a/src/gpu_backtest/core/engine.py b/src/gpu_backtest/core/engine.py index d69f387..3a7e5e8 100644 --- a/src/gpu_backtest/core/engine.py +++ b/src/gpu_backtest/core/engine.py @@ -9,7 +9,7 @@ from .grid import dim_arrays, space_count from .indicators import prepare_tables from .kernels import build_kernels -from .output import validate_reduction_outputs, write_top_csv +from .output import validate_reduction_outputs, write_results from .strategy import MAX_DIMS, load_strategy, validate_strategy @@ -19,7 +19,6 @@ def run( out_prefix, buy=0.0015, sell=0.0015, - top_n=10000, threads_per_block=128, entry_dims=None, exit_dims=None, @@ -33,8 +32,6 @@ def run( raise ValueError(f"Entry and exit dimensions must each contain 1 to {MAX_DIMS} dimensions") if not np.isfinite([buy, sell]).all() or buy < 0 or sell < 0: raise ValueError("Commissions must be finite and non-negative") - if isinstance(top_n, bool) or not isinstance(top_n, (int, np.integer)) or top_n < 1: - raise ValueError("top_n must be a positive integer") if ( isinstance(threads_per_block, bool) or not isinstance(threads_per_block, (int, np.integer)) @@ -43,10 +40,6 @@ def run( raise ValueError("threads_per_block must be an integer between 1 and 1024") validate_strategy(strategy, e_dims, x_dims) entry_count, exit_count = (space_count(e_dims), space_count(x_dims)) - if out_prefix and (entry_count < 2 or exit_count < 2): - raise ValueError( - "Effect-size isolation requires at least two entry and two exit combinations" - ) if verbose: print( f"[{strategy.NAME}] entry={entry_count} × exit={exit_count} = {entry_count * exit_count:,} combos" @@ -121,7 +114,7 @@ class _S: print(f"grand total check: diff={reduction_checks['sum']['difference']:.2e}") out = {"entry_sum": es, "entry_sumsq": esq, "exit_sum": xs, "exit_sumsq": xsq} if out_prefix: - write_top_csv( + write_results( out_prefix, strategy.NAME, input_csv, @@ -131,8 +124,9 @@ class _S: esq, xs, xsq, - top_n, reduction_checks, verbose, + buy=buy, + sell=sell, ) return out diff --git a/src/gpu_backtest/core/output.py b/src/gpu_backtest/core/output.py index efdedfe..379390f 100644 --- a/src/gpu_backtest/core/output.py +++ b/src/gpu_backtest/core/output.py @@ -1,16 +1,15 @@ -"""Validate grouped outputs and write ranked CSVs and their manifest.""" +"""Validate GPU reductions and persist their raw arrays without ranking or analysis.""" +import hashlib import json import os from pathlib import Path import numpy as np -import pandas as pd -from . import statistics as cs -from .grid import decode_values +from .grid import space_count -ROUND_TRIP_FLOAT_FORMAT = "%.17g" +RESULT_KEYS = ("entry_sum", "entry_sumsq", "exit_sum", "exit_sumsq") def validate_reduction_outputs(entry_sum, entry_sumsq, exit_sum, exit_sumsq): @@ -22,8 +21,11 @@ def validate_reduction_outputs(entry_sum, entry_sumsq, exit_sum, exit_sumsq): "exit_sumsq": np.asarray(exit_sumsq, dtype=np.float64), } for name, values in arrays.items(): - if values.size == 0 or not np.isfinite(values).all(): - raise RuntimeError(f"{name} is empty or contains non-finite values") + if values.ndim != 1 or values.size == 0 or not np.isfinite(values).all(): + raise RuntimeError(f"{name} must be a nonempty finite one-dimensional array") + for side in ("entry", "exit"): + if arrays[f"{side}_sum"].shape != arrays[f"{side}_sumsq"].shape: + raise RuntimeError(f"{side} sum/sumsq shapes must match") if (arrays["entry_sumsq"] < 0).any() or (arrays["exit_sumsq"] < 0).any(): raise RuntimeError("Reduction sum-of-squares values must be non-negative") checks = {} @@ -49,11 +51,14 @@ def validate_reduction_outputs(entry_sum, entry_sumsq, exit_sum, exit_sumsq): return checks -def _format_parameter(value): - return np.format_float_positional(float(value), trim="-") +def _dimensions(dims): + return [ + [name, *[float(v) if is_float else int(v) for v in (low, high, step)], is_float] + for name, low, high, step, is_float in dims + ] -def write_top_csv( +def write_results( prefix, strategy_name, input_csv, @@ -63,63 +68,69 @@ def write_top_csv( esq, xs, xsq, - top_n, reduction_checks, verbose, + *, + buy, + sell, + engine="gpu_backtest.core.engine", ): - output_records = {} - for side, dims, s, sq, n1 in ( - ("entry", e_dims, es, esq, len(xs)), - ("exit", x_dims, xs, xsq, len(es)), - ): - st = cs.stats_from_groups(s, sq, n1, s.sum(), sq.sum(), len(s) * n1) - cols = { - d[0]: np.array([decode_values(dims, i)[j] for i in range(len(s))]) - for j, d in enumerate(dims) - } - df = pd.DataFrame({**cols, **st}) - if not np.isfinite(df["effect_size"].to_numpy(dtype=np.float64)).all(): - raise RuntimeError(f"{side} effect_size contains non-finite values") - key_columns = [d[0] for d in dims] - df = df.sort_values( - ["effect_size", *key_columns], - ascending=[False, *[True] * len(key_columns)], - kind="mergesort", - ).head(min(top_n, len(s))) - serialized = df.copy() - for dim in dims: - if dim[4]: - serialized[dim[0]] = serialized[dim[0]].map(_format_parameter) - path = f"{prefix}_top_{side}.csv" - Path(path).parent.mkdir(parents=True, exist_ok=True) - temp_path = f"{path}.tmp" - serialized.to_csv(temp_path, index=False, float_format=ROUND_TRIP_FLOAT_FORMAT) - os.replace(temp_path, path) - output_records[side] = { - "file": Path(path).name, - "rows": int(len(df)), - "key_columns": key_columns, - "columns": list(serialized.columns), - "sort": ["effect_size descending", "parameter columns ascending"], - } - if verbose: - print(f" -> {path} ({len(df)} rows)") + """Save the four float64 arrays in parameter-index order, plus their schema.""" + arrays = dict(zip(RESULT_KEYS, (es, esq, xs, xsq))) + entries, exits = space_count(e_dims), space_count(x_dims) + for key, value in arrays.items(): + count = entries if key.startswith("entry_") else exits + if value.dtype != np.float64 or value.shape != (count,): + raise RuntimeError(f"{key} must have float64 dtype and shape ({count},)") + path = Path(f"{prefix}_results.npz") + manifest_path = Path(f"{prefix}_manifest.json") + path.parent.mkdir(parents=True, exist_ok=True) + temp_path = path.with_name(path.name + ".tmp") + temp_manifest = manifest_path.with_name(manifest_path.name + ".tmp") manifest = { - "schema_version": 1, - "engine": "gpu_backtest.engine", + "schema_version": 2, + "engine": engine, "strategy": strategy_name, "source_file": Path(input_csv).name, - "entry_count": int(len(es)), - "exit_count": int(len(xs)), - "pair_count": int(len(es) * len(xs)), - "top_n": int(top_n), - "float_serialization": "IEEE-754 float64 round-trip (17 significant digits)", + "entry_dims": _dimensions(e_dims), + "exit_dims": _dimensions(x_dims), + "entry_count": entries, + "exit_count": exits, + "pair_count": entries * exits, + "buy": float(buy), + "sell": float(sell), + "parameter_order": "Cartesian index order; last dimension varies fastest", + "reduction_semantics": { + "entry": "Each entry group sums returns across all exit combinations", + "exit": "Each exit group sums returns across all entry combinations", + "return_dtype": "float32", + "square_dtype": "float32", + "accumulator_dtype": "float64", + }, "reduction_checks": reduction_checks, - "outputs": output_records, + "outputs": { + "file": path.name, + "format": "npz", + "arrays": { + key: {"dtype": str(a.dtype), "shape": list(a.shape)} for key, a in arrays.items() + }, + }, } - manifest_path = f"{prefix}_top_manifest.json" - temp_manifest = f"{manifest_path}.tmp" - Path(temp_manifest).write_text(json.dumps(manifest, indent=2) + "\n", encoding="utf-8") - os.replace(temp_manifest, manifest_path) + try: + with temp_path.open("wb") as handle: + np.savez(handle, **arrays) + with temp_path.open("rb") as handle: + manifest["outputs"]["sha256"] = hashlib.file_digest(handle, "sha256").hexdigest() + temp_manifest.write_text(json.dumps(manifest, indent=2) + "\n", encoding="utf-8") + # The manifest marks a completed pair. Invalidate it before replacing data + # so a failed publication cannot leave new arrays with stale parameters. + manifest_path.unlink(missing_ok=True) + os.replace(temp_path, path) + os.replace(temp_manifest, manifest_path) + finally: + temp_path.unlink(missing_ok=True) + temp_manifest.unlink(missing_ok=True) if verbose: + print(f" -> {path} (raw grouped arrays)") print(f" -> {manifest_path}") + return manifest_path diff --git a/src/gpu_backtest/core/statistics.py b/src/gpu_backtest/core/statistics.py deleted file mode 100644 index d949f57..0000000 --- a/src/gpu_backtest/core/statistics.py +++ /dev/null @@ -1,78 +0,0 @@ -"""Generic grouped-return statistics. Observations are parameter combinations.""" - -import math - -import numpy as np - -_STAT_COLS = [ - "sample_size_g1", - "sample_size_g2", - "group1_mean", - "group2_mean", - "diff_mean", - "posterior_prob_superior", - "effect_size", - "diff_ci_lower", - "diff_ci_upper", - "group2_mean_plus_diff_ci_lower", - "group2_mean_plus_diff_ci_upper", -] - - -def _norm_cdf(z): - """Evaluate the normal CDF without SciPy.""" - vec = np.vectorize(lambda v: 0.5 * (1.0 + math.erf(v / math.sqrt(2.0)))) - return vec(z) - - -def stats_from_groups(sum1, sumsq1, n1, total_sum, total_sumsq, total_count): - """Compare every parameter group with all other groups using pooled variance.""" - sum1 = np.asarray(sum1, dtype=np.float64) - sumsq1 = np.asarray(sumsq1, dtype=np.float64) - n1 = float(n1) - if sum1.ndim != 1 or sumsq1.shape != sum1.shape or sum1.size < 2: - raise ValueError( - "sum1 and sumsq1 must be matching one-dimensional arrays with at least two groups" - ) - if not np.isfinite(sum1).all() or not np.isfinite(sumsq1).all(): - raise ValueError("Group sums must be finite") - if not np.isfinite([n1, total_sum, total_sumsq, total_count]).all(): - raise ValueError("Group counts and grand totals must be finite") - if not n1.is_integer() or float(total_count) != float(sum1.size) * n1: - raise ValueError("total_count must equal number_of_groups * observations_per_group") - if n1 <= 1 or total_count - n1 <= 1: - raise ValueError( - "Each effect-size comparison requires at least two observations in both groups" - ) - if (sumsq1 < 0).any() or total_sumsq < 0: - raise ValueError("Sum-of-squares values must be non-negative") - mean1 = sum1 / n1 - var1 = (sumsq1 - sum1**2 / n1) / (n1 - 1) - sum2 = total_sum - sum1 - n2 = total_count - n1 - sumsq2 = total_sumsq - sumsq1 - mean2 = sum2 / n2 - var2 = (sumsq2 - sum2**2 / n2) / (n2 - 1) - diff_mean = mean1 - mean2 - std_error = np.sqrt(np.maximum(var1 / n1 + var2 / n2, 0.0)) - z = np.divide(diff_mean, std_error, out=np.zeros_like(diff_mean), where=std_error > 0) - posterior = _norm_cdf(z) - sp2 = ((n1 - 1) * var1 + (n2 - 1) * var2) / (n1 + n2 - 2) - effect_size = np.divide( - diff_mean, np.sqrt(np.where(sp2 > 0, sp2, 1.0)), out=np.zeros_like(diff_mean), where=sp2 > 0 - ) - diff_ci_lower = diff_mean - 1.96 * std_error - diff_ci_upper = diff_mean + 1.96 * std_error - return dict( - sample_size_g1=np.full_like(sum1, n1), - sample_size_g2=np.full_like(sum1, n2), - group1_mean=mean1, - group2_mean=mean2, - diff_mean=diff_mean, - posterior_prob_superior=posterior, - effect_size=effect_size, - diff_ci_lower=diff_ci_lower, - diff_ci_upper=diff_ci_upper, - group2_mean_plus_diff_ci_lower=mean2 + diff_ci_lower, - group2_mean_plus_diff_ci_upper=mean2 + diff_ci_upper, - ) diff --git a/src/gpu_backtest/core/strategy.py b/src/gpu_backtest/core/strategy.py index 4becefd..e7b09a0 100644 --- a/src/gpu_backtest/core/strategy.py +++ b/src/gpu_backtest/core/strategy.py @@ -4,8 +4,6 @@ import math from numbers import Real -from .statistics import _STAT_COLS - MAX_DIMS = 4 MAX_TABLES = 4 MAX_INDEX = 2**63 - 1 @@ -39,8 +37,8 @@ def _validate_dims(dims, side): if not isinstance(dim, (list, tuple)) or len(dim) != 5: raise ValueError("Each dimension must be (name, low, high, step, is_float)") name, low, high, step, is_float = dim - if not isinstance(name, str) or not name.isidentifier() or name in _STAT_COLS: - raise ValueError(f"Invalid or reserved parameter name: {name!r}") + if not isinstance(name, str) or not name.isidentifier(): + raise ValueError(f"Invalid parameter name: {name!r}") if type(is_float) is not bool: raise ValueError(f"{name}: is_float must be a boolean") if any(isinstance(v, bool) or not isinstance(v, Real) for v in (low, high, step)): diff --git a/src/gpu_backtest/workflows/__init__.py b/src/gpu_backtest/workflows/__init__.py deleted file mode 100644 index e32e536..0000000 --- a/src/gpu_backtest/workflows/__init__.py +++ /dev/null @@ -1 +0,0 @@ -"""Generic analysis workflows built on the core.""" diff --git a/src/gpu_backtest/workflows/charts.py b/src/gpu_backtest/workflows/charts.py deleted file mode 100644 index 28fa509..0000000 --- a/src/gpu_backtest/workflows/charts.py +++ /dev/null @@ -1,44 +0,0 @@ -"""Render generic parameter/effect-size plots as a directly openable HTML file.""" - -from pathlib import Path - -import numpy as np -import pandas as pd - -from gpu_backtest.core.statistics import _STAT_COLS - - -def write_charts(input_files, output_file): - try: - import altair as alt - except ImportError as exc: - raise ValueError("Charts require: pip install 'gpu-backtest-engine[viz]'") from exc - charts = [] - for filename in input_files: - data = pd.read_csv(filename) - if "effect_size" not in data or data.empty: - raise ValueError(f"{filename}: requires non-empty top data with effect_size") - parameters = [column for column in data if column not in _STAT_COLS] - if not parameters: - raise ValueError(f"{filename}: no parameter columns") - if not np.isfinite(data[[*parameters, "effect_size"]].to_numpy(dtype=float)).all(): - raise ValueError(f"{filename}: parameter/effect-size values must be finite") - for parameter in parameters: - charts.append( - alt.Chart(data) - .mark_circle(size=40, opacity=0.6) - .encode( - x=alt.X(f"{parameter}:Q", title=parameter.replace("_", " ").title()), - y=alt.Y("effect_size:Q", title="Effect size"), - tooltip=[*parameters, "effect_size"], - ) - .properties(width=700, height=240, title=f"{Path(filename).name}: {parameter}") - .interactive(name=f"zoom_{len(charts)}") - ) - if not charts: - raise ValueError("At least one top CSV is required") - output = Path(output_file) - output.parent.mkdir(parents=True, exist_ok=True) - with alt.data_transformers.enable("default", max_rows=None): - alt.vconcat(*charts).save(str(output)) - return output diff --git a/src/gpu_backtest/workflows/common.py b/src/gpu_backtest/workflows/common.py deleted file mode 100644 index d3f9bf8..0000000 --- a/src/gpu_backtest/workflows/common.py +++ /dev/null @@ -1,159 +0,0 @@ -"""Find parameter combinations present in every ranked top artifact.""" - -import csv -import os -from decimal import Decimal, InvalidOperation, getcontext -from pathlib import Path - -getcontext().prec = 50 -VALUE_COLUMN = "effect_size" - - -def extract_label_from_filename(filename): - base_name = Path(filename).stem - parts = base_name.split("_") - for i, part in enumerate(parts): - if part == "to" and 0 < i < len(parts) - 1: - before_years = [ - segment - for segment in parts[i - 1].split("-") - if segment.isdigit() and len(segment) == 4 - ] - after_years = [ - segment - for segment in parts[i + 1].split("-") - if segment.isdigit() and len(segment) == 4 - ] - if before_years and after_years: - return f"{before_years[0]}_{after_years[0]}" - for i, part in enumerate(parts[:-1]): - if part.isdigit() and len(part) == 4: - following = parts[i + 1] - if following.isdigit() and len(following) == 4: - return f"{part}_{following}" - return base_name - - -def _decimal(raw_value, context): - try: - value = Decimal(str(raw_value).strip()) - except (InvalidOperation, ValueError) as exc: - raise ValueError(f"{context}: expected a decimal number, got {raw_value!r}") from exc - if not value.is_finite(): - raise ValueError(f"{context}: value must be finite") - return value - - -def _format_decimal(value): - if value == 0: - return "0" - return format(value.normalize(), "f") - - -def _read_top_artifact(filename, key_columns, value_column=VALUE_COLUMN): - path = Path(filename) - if not path.is_file(): - raise FileNotFoundError(f"Input file not found: {filename}") - with path.open("r", newline="", encoding="utf-8") as handle: - reader = csv.DictReader(handle) - fieldnames = reader.fieldnames or [] - if len(fieldnames) != len(set(fieldnames)): - raise ValueError(f"{filename}: duplicate CSV column names") - required = [*key_columns, value_column] - missing = [column for column in required if column not in fieldnames] - if missing: - raise ValueError(f"{filename}: missing required columns: {missing}") - data = {} - previous_effect = None - for row_number, row in enumerate(reader, start=2): - key = tuple( - ( - _decimal(row[column], f"{filename}:{row_number}:{column}") - for column in key_columns - ) - ) - effect = _decimal(row[value_column], f"{filename}:{row_number}:{value_column}") - if key in data: - raise ValueError(f"{filename}:{row_number}: duplicate parameter combination {key}") - if previous_effect is not None and effect > previous_effect: - raise ValueError( - f"{filename}:{row_number}: {value_column} is not sorted descending" - ) - previous_effect = effect - data[key] = effect - if not data: - raise ValueError(f"{filename}: no data rows") - return data - - -def _resolve_labels(input_files, configured_labels=None): - labels = configured_labels or [extract_label_from_filename(path) for path in input_files] - if len(labels) != len(input_files): - raise ValueError("The number of labels must match the number of input files") - labels = [label.strip() for label in labels] - if any((not label for label in labels)): - raise ValueError("Labels must not be empty") - if len(labels) != len(set(labels)): - raise ValueError(f"Labels must be unique: {labels}") - return labels - - -def process_data(config): - input_files = config["input_files"] - key_columns = config["key_columns"] - output_file = Path(config["output_file"]) - output_file.unlink(missing_ok=True) - if len(input_files) < 2: - raise ValueError("At least two input files are required") - if not key_columns or len(key_columns) != len(set(key_columns)): - raise ValueError("key_columns must contain unique column names") - labels = _resolve_labels(input_files, config.get("file_labels")) - all_data = [_read_top_artifact(path, key_columns) for path in input_files] - common_keys = set(all_data[0]) - for data in all_data[1:]: - common_keys.intersection_update(data) - if not common_keys: - pairwise = [] - for i in range(len(all_data)): - for j in range(i + 1, len(all_data)): - pairwise.append( - f"{labels[i]} & {labels[j]}={len(set(all_data[i]) & set(all_data[j]))}" - ) - raise ValueError( - "No common combinations found; pairwise intersections: " + ", ".join(pairwise) - ) - result = [] - for key in common_keys: - effects = [data[key] for data in all_data] - average = sum(effects, Decimal(0)) / Decimal(len(effects)) - result.append({"key": key, "effects": effects, "average": average, "minimum": min(effects)}) - result.sort(key=lambda item: (-item["average"], *item["key"])) - effect_columns = [f"{VALUE_COLUMN}_{label}" for label in labels] - fieldnames = [*key_columns, *effect_columns, f"avg_{VALUE_COLUMN}", f"min_{VALUE_COLUMN}"] - output_file.parent.mkdir(parents=True, exist_ok=True) - temp_file = output_file.with_name(output_file.name + ".tmp") - try: - with temp_file.open("w", newline="", encoding="utf-8") as handle: - writer = csv.DictWriter(handle, fieldnames=fieldnames, lineterminator="\n") - writer.writeheader() - for item in result: - row = { - column: _format_decimal(value) - for column, value in zip(key_columns, item["key"]) - } - row.update( - { - column: _format_decimal(value) - for column, value in zip(effect_columns, item["effects"]) - } - ) - row[f"avg_{VALUE_COLUMN}"] = _format_decimal(item["average"]) - row[f"min_{VALUE_COLUMN}"] = _format_decimal(item["minimum"]) - writer.writerow(row) - os.replace(temp_file, output_file) - finally: - if temp_file.exists(): - temp_file.unlink() - print(f"Common combinations: {len(result)}") - print(f"Results saved to: {output_file}") - return result diff --git a/src/gpu_backtest/workflows/pipeline.py b/src/gpu_backtest/workflows/pipeline.py deleted file mode 100644 index fecb458..0000000 --- a/src/gpu_backtest/workflows/pipeline.py +++ /dev/null @@ -1,65 +0,0 @@ -"""Run independently configured date splits and intersect their ranked parameters.""" - -import json -import shutil -from pathlib import Path - -from gpu_backtest.core.engine import run -from gpu_backtest.core.strategy import load_strategy - -from .common import process_data -from .splits import _read_input, split_csv_by_custom_ranges, split_csv_by_date_range - - -def run_pipeline( - strategy, - input_csv, - output_dir, - *, - num_splits=3, - start_date=None, - end_date=None, - ranges=None, - **run_options, -): - output_dir = Path(output_dir) - output_dir.mkdir(parents=True, exist_ok=True) - data_dir = output_dir / "data" - data_dir.mkdir(exist_ok=True) - staged = data_dir / Path(input_csv).name - if Path(input_csv).resolve() == staged.resolve(): - raise ValueError("Pipeline input must be outside its output data directory") - shutil.copyfile(input_csv, staged) - if ranges: - manifest_path = split_csv_by_custom_ranges(staged, ranges, "Time") - else: - data, _ = _read_input(staged, "Time") - start_date = start_date or data["Time"].min().strftime("%d-%m-%Y") - end_date = end_date or data["Time"].max().strftime("%d-%m-%Y") - manifest_path = split_csv_by_date_range(staged, start_date, end_date, num_splits, "Time") - manifest = json.loads(manifest_path.read_text()) - if len(manifest["splits"]) < 2: - raise ValueError("Common analysis requires at least two splits") - shutil.copyfile(manifest_path, output_dir / "split_manifest.json") - strategy_module = load_strategy(strategy) - entry_dims = run_options.get("entry_dims") - exit_dims = run_options.get("exit_dims") - entry_dims = strategy_module.ENTRY_DIMS if entry_dims is None else entry_dims - exit_dims = strategy_module.EXIT_DIMS if exit_dims is None else exit_dims - paths = {"entry": [], "exit": []} - for record in manifest["splits"]: - source = manifest_path.parent / record["file"] - prefix = output_dir / source.stem - run(strategy_module, source, str(prefix), **run_options) - for side in paths: - paths[side].append(f"{prefix}_top_{side}.csv") - for side, dims in (("entry", entry_dims), ("exit", exit_dims)): - process_data( - { - "input_files": paths[side], - "key_columns": [d[0] for d in dims], - "file_labels": [s["label"] for s in manifest["splits"]], - "output_file": output_dir / f"common_{side}.csv", - } - ) - return output_dir diff --git a/src/gpu_backtest/workflows/splits.py b/src/gpu_backtest/workflows/splits.py deleted file mode 100644 index ec0c20f..0000000 --- a/src/gpu_backtest/workflows/splits.py +++ /dev/null @@ -1,184 +0,0 @@ -"""Split OHLCV CSVs and record boundaries and content hashes.""" - -import hashlib -import json -import os -from pathlib import Path - -import pandas as pd - -SPLIT_CONTRACT_VERSION = 2 - - -def _sha256(path): - digest = hashlib.sha256() - with open(path, "rb") as handle: - for chunk in iter(lambda: handle.read(1024 * 1024), b""): - digest.update(chunk) - return digest.hexdigest() - - -def detect_date_column(df): - candidates = {"time", "date", "datetime", "timestamp"} - for col in df.columns: - if col.lower() in candidates: - return col - raise ValueError("No date column found") - - -def _read_input(input_file, date_column=None): - df = pd.read_csv(input_file) - if date_column is None: - date_column = detect_date_column(df) - if date_column not in df.columns: - raise ValueError(f"Date column '{date_column}' not found") - if df.empty: - raise ValueError(f"Input CSV is empty: {input_file}") - try: - df[date_column] = pd.to_datetime(df[date_column], errors="raise") - except Exception as exc: - raise ValueError(f"Column '{date_column}' contains invalid timestamps") from exc - if df[date_column].isna().any(): - raise ValueError(f"Column '{date_column}' contains missing timestamps") - if df[date_column].duplicated().any(): - raise ValueError(f"Column '{date_column}' contains duplicate timestamps") - if not df[date_column].is_monotonic_increasing: - raise ValueError(f"Column '{date_column}' must be strictly increasing") - return (df, date_column) - - -def _bound(value, tz, *, end_exclusive=False): - result = pd.to_datetime(value, format="%d-%m-%Y") - if tz is not None: - result = result.tz_localize(tz) - if end_exclusive: - result += pd.Timedelta(days=1) - return result - - -def _output_dir(input_file): - parent = Path(input_file).resolve().parent - output = parent if parent.name == "output" else parent / "output" - output.mkdir(parents=True, exist_ok=True) - return output - - -def _iso(value): - return None if value is None else value.isoformat() - - -def _write_splits(input_file, date_column, source_rows, filtered_df, specs, mode): - output_dir = _output_dir(input_file) - base_name = Path(input_file).stem - records = [] - for index, (label, split_start, split_end, split_df) in enumerate(specs, start=1): - if split_df.empty: - raise ValueError(f"Split {index} ('{label}') is empty") - start_str = split_start.strftime("%d-%m-%Y") - end_str = (split_end - pd.Timedelta(nanoseconds=1)).strftime("%d-%m-%Y") - filename = f"{base_name}_split{index}_{start_str}_to_{end_str}.csv" - output_file = output_dir / filename - split_df.to_csv(output_file, index=False) - records.append( - { - "index": index, - "label": label, - "file": filename, - "rows": int(len(split_df)), - "first_timestamp": _iso(split_df[date_column].iloc[0]), - "last_timestamp": _iso(split_df[date_column].iloc[-1]), - "requested_start": _iso(split_start), - "requested_end_exclusive": _iso(split_end), - "sha256": _sha256(output_file), - } - ) - print(f"Saved split {index}: {output_file} ({len(split_df)} rows)") - manifest = { - "schema_version": 2, - "split_contract_version": SPLIT_CONTRACT_VERSION, - "mode": mode, - "source_file": Path(input_file).name, - "source_sha256": _sha256(input_file), - "date_column": date_column, - "source_rows": int(source_rows), - "filtered_rows": int(len(filtered_df)), - "filtered_first_timestamp": _iso(filtered_df[date_column].iloc[0]), - "filtered_last_timestamp": _iso(filtered_df[date_column].iloc[-1]), - "splits": records, - } - manifest_path = output_dir / "split_manifest.json" - temp_path = manifest_path.with_suffix(".json.tmp") - temp_path.write_text(json.dumps(manifest, indent=2) + "\n", encoding="utf-8") - os.replace(temp_path, manifest_path) - print(f"Split manifest: {manifest_path}") - return manifest_path - - -def split_csv_by_date_range(input_file, start_date, end_date, num_splits, date_column=None): - """Split by date. Three splits mean full range, first half, and second half.""" - if num_splits < 1: - raise ValueError("num_splits must be at least 1") - print(f"Reading CSV file: {input_file}") - df, date_column = _read_input(input_file, date_column) - tz = df[date_column].dt.tz - start = _bound(start_date, tz) - end_exclusive = _bound(end_date, tz, end_exclusive=True) - if start >= end_exclusive: - raise ValueError("start_date must not be after end_date") - filtered_df = df[(df[date_column] >= start) & (df[date_column] < end_exclusive)].copy() - if filtered_df.empty: - raise ValueError(f"No data found between {start_date} and {end_date}") - print(f"Found {len(filtered_df)} rows between {start_date} and {end_date}") - if num_splits == 3: - midpoint = start + (end_exclusive - start) / 2 - first = filtered_df[filtered_df[date_column] <= midpoint] - second = filtered_df[filtered_df[date_column] > midpoint] - first_ids = set(first.index) - second_ids = set(second.index) - full_ids = set(filtered_df.index) - if first_ids & second_ids or first_ids | second_ids != full_ids: - raise RuntimeError( - "Split contract failed: half ranges must be disjoint and cover the full range" - ) - specs = [ - ("full", start, end_exclusive, filtered_df), - ("first_half", start, midpoint + pd.Timedelta(nanoseconds=1), first), - ("second_half", midpoint + pd.Timedelta(nanoseconds=1), end_exclusive, second), - ] - return _write_splits( - input_file, date_column, len(df), filtered_df, specs, "full_and_halves" - ) - boundaries = [start + (end_exclusive - start) * i / num_splits for i in range(num_splits + 1)] - specs = [] - covered = set() - for i in range(num_splits): - split_start, split_end = (boundaries[i], boundaries[i + 1]) - split_df = filtered_df[ - (filtered_df[date_column] >= split_start) & (filtered_df[date_column] < split_end) - ] - ids = set(split_df.index) - if covered & ids: - raise RuntimeError("Split contract failed: generated ranges overlap") - covered |= ids - specs.append((f"part_{i + 1}", split_start, split_end, split_df)) - if covered != set(filtered_df.index): - raise RuntimeError("Split contract failed: generated ranges do not cover the filtered data") - return _write_splits(input_file, date_column, len(df), filtered_df, specs, "partition") - - -def split_csv_by_custom_ranges(input_file, ranges, date_column=None): - """Write explicit DD-MM-YYYY ranges; overlap is allowed by definition.""" - df, date_column = _read_input(input_file, date_column) - tz = df[date_column].dt.tz - specs = [] - included = set() - for i, (start_text, end_text) in enumerate(ranges, start=1): - start = _bound(start_text, tz) - end_exclusive = _bound(end_text, tz, end_exclusive=True) - if start >= end_exclusive: - raise ValueError(f"Custom range {i} has start after end") - split_df = df[(df[date_column] >= start) & (df[date_column] < end_exclusive)] - included |= set(split_df.index) - specs.append((f"range_{i}", start, end_exclusive, split_df)) - filtered_df = df.loc[sorted(included)] - return _write_splits(input_file, date_column, len(df), filtered_df, specs, "custom_ranges") diff --git a/tests/README.md b/tests/README.md index 03b0465..2caaac2 100644 --- a/tests/README.md +++ b/tests/README.md @@ -2,9 +2,9 @@ | Directory | Runs where | What it checks | |---|---|---| -| `cpu/` | CPU only | Core contracts, indicators/statistics, CLI/workflows, example reference, mocked RunPod lifecycle, benchmark CPU baseline, import boundaries | +| `cpu/` | CPU only | Core contracts, indicators, raw artifacts, CLI, example reference, mocked RunPod lifecycle, benchmark CPU baseline, import boundaries | | `gpu/test_simulator.py` + `simulator_cases.py` | CPU CUDA simulator in a child process | Actual GPU kernels against explicit matrix/RSI answers and CPU reference | -| `gpu/test_hardware.py` | Real NVIDIA GPU | Kernel reductions and RSI reference comparison; marked `gpu` | +| `gpu/test_hardware.py` | Real NVIDIA GPU | Kernel reductions, RSI reference and single-combination raw output; marked `gpu` | ```bash python -m pytest # CPU + isolated simulation; no rented pod @@ -35,3 +35,9 @@ For changes to numerical logic, run both simulation and real-GPU checks. A pure module/layout refactor must preserve the kernel body and compare previous/current outputs. Installing the wheel and checking its RunPod export catches packaging errors that editable-checkout tests can miss. + + +Raw output tests reload NPZ with pickle disabled, assert exact float64 array values, +ordered dimensions, fees, schema-2 counts and file SHA-256. Single-combination output +is anchored against hand-calculated profitable, fee-paying and losing trades. +Tests also reject removed analysis commands/config fields before RunPod allocation. diff --git a/tests/cpu/test_benchmark.py b/tests/cpu/test_benchmark.py index 5b26428..50adf38 100644 --- a/tests/cpu/test_benchmark.py +++ b/tests/cpu/test_benchmark.py @@ -72,7 +72,7 @@ def test_generated_data_reproduces_public_example(tmp_path): assert generated.read_bytes() == original.read_bytes() -def test_cpu_baseline_writes_complete_ranked_artifacts(tmp_path): +def test_cpu_baseline_writes_raw_artifacts(tmp_path): import json source = generate_csv(tmp_path / "bars.csv", 32) @@ -85,11 +85,14 @@ def test_cpu_baseline_writes_complete_ranked_artifacts(tmp_path): finally: numba.set_num_threads(old) assert set(result) == set(benchmark.RESULT_KEYS) - manifest = json.loads((tmp_path / "cpu_top_manifest.json").read_text()) + manifest = json.loads((tmp_path / "cpu_manifest.json").read_text()) assert manifest["engine"] == "gpu_backtest.benchmark_cpu" assert manifest["pair_count"] == 16 - assert (tmp_path / "cpu_top_entry.csv").is_file() - assert (tmp_path / "cpu_top_exit.csv").is_file() + with np.load(tmp_path / "cpu_results.npz", allow_pickle=False) as saved: + for key in benchmark.RESULT_KEYS: + assert saved[key].tobytes() == result[key].tobytes() + assert manifest["schema_version"] == 2 + assert not list(tmp_path.glob("*_top*")) def test_full_billion_cpu_flag_requires_billion_profile(monkeypatch, tmp_path): diff --git a/tests/cpu/test_cli.py b/tests/cpu/test_cli.py index cf365e6..2732b11 100644 --- a/tests/cpu/test_cli.py +++ b/tests/cpu/test_cli.py @@ -4,7 +4,7 @@ import sys from pathlib import Path -import pandas as pd +import numpy as np import pytest ROOT = Path(__file__).resolve().parents[2] @@ -36,8 +36,11 @@ def test_config_paths_resolve_outside_checkout(tmp_path): cwd=tmp_path, ) assert result.returncode == 0, result.stdout + result.stderr - manifest = json.loads((tmp_path / "rsi_top_manifest.json").read_text()) + manifest = json.loads((tmp_path / "rsi_manifest.json").read_text()) assert manifest["pair_count"] == 16 and manifest["strategy"] == "rsi_meanrev" + with np.load(tmp_path / "rsi_results.npz", allow_pickle=False) as saved: + assert saved["entry_sum"].shape == (4,) + assert saved["entry_sum"].dtype == np.float64 def test_strategy_is_always_explicit(tmp_path): @@ -49,41 +52,19 @@ def test_strategy_is_always_explicit(tmp_path): tmp_path / "x", ) assert result.returncode == 2 and "explicitly" in result.stderr - assert not (tmp_path / "x_top_entry.csv").exists() + assert not (tmp_path / "x_results.npz").exists() -def test_complete_public_pipeline(tmp_path): - output = tmp_path / "pipeline" - result = invoke( - "pipeline", - "--config", - ROOT / "examples/gpu_backtest_examples/rsi/config.json", - "--output-dir", - output, - simulation=True, - ) - assert result.returncode == 0, result.stdout + result.stderr - manifest = json.loads((output / "split_manifest.json").read_text()) - assert [record["label"] for record in manifest["splits"]] == [ - "full", - "first_half", - "second_half", - ] - assert len(list(output.glob("*_top_manifest.json"))) == 3 - for side in ("entry", "exit"): - common = pd.read_csv(output / f"common_{side}.csv") - assert len(common) == 4 - assert {"avg_effect_size", "min_effect_size", "effect_size_full"}.issubset(common.columns) +@pytest.mark.parametrize("command", ["split", "common", "charts", "pipeline"]) +def test_analysis_commands_are_removed(command): + result = invoke(command) + assert result.returncode == 2 and "invalid choice" in result.stderr -def test_optional_charts_detect_arbitrary_parameter_names(tmp_path): - pytest.importorskip("altair") - source = tmp_path / "sample_top.csv" - pd.DataFrame({"custom_threshold": [1, 2], "effect_size": [0.1, 0.2]}).to_csv( - source, index=False - ) - output = tmp_path / "charts.html" - result = invoke("charts", "--input", source, "--output", output) - assert result.returncode == 0, result.stderr - html = output.read_text() - assert "custom_threshold" in html and "vegaEmbed" in html +@pytest.mark.parametrize("key", ["top_n", "num_splits", "ranges", "start_date", "end_date"]) +def test_removed_config_options_are_rejected(tmp_path, key): + config = tmp_path / "job.json" + config.write_text(json.dumps({"strategy": "unused", "input": "unused.csv", key: 3})) + result = invoke("run", "--config", config, "--out-prefix", tmp_path / "x") + assert result.returncode == 2 and "Unknown engine config keys" in result.stderr + assert not (tmp_path / "x_results.npz").exists() diff --git a/tests/cpu/test_common_contract.py b/tests/cpu/test_common_contract.py deleted file mode 100644 index f9bca16..0000000 --- a/tests/cpu/test_common_contract.py +++ /dev/null @@ -1,112 +0,0 @@ -import csv -import os -import subprocess -import sys - -import pandas as pd -import pytest - - -def _write_top(path, rows): - with open(path, "w", newline="") as handle: - writer = csv.DictWriter(handle, fieldnames=["parameter", "effect_size"]) - writer.writeheader() - writer.writerows(rows) - - -def _run(inputs, output, labels="full,first_half,second_half", env=None): - return subprocess.run( - [ - sys.executable, - "-m", - "gpu_backtest", - "common", - "--input", - *map(str, inputs), - "--keys", - "parameter", - "--labels", - labels, - "--output", - str(output), - ], - capture_output=True, - text=True, - env=env, - timeout=60, - ) - - -def test_common_uses_full_precision_and_writes_audit_columns(tmp_path): - files = [tmp_path / f"top_{i}.csv" for i in range(3)] - _write_top( - files[0], - [ - {"parameter": "1", "effect_size": "0.10000000000000006"}, - {"parameter": "2", "effect_size": "0.1"}, - ], - ) - for path in files[1:]: - _write_top( - path, - [ - {"parameter": "2", "effect_size": "0.10000000000000005"}, - {"parameter": "1", "effect_size": "0.1"}, - ], - ) - output = tmp_path / "common.csv" - output.write_text("stale output\n") - result = _run(files, output) - assert result.returncode == 0, result.stderr - data = pd.read_csv(output, dtype=str) - assert list(data["parameter"]) == ["2", "1"] - assert list(data.columns) == [ - "parameter", - "effect_size_full", - "effect_size_first_half", - "effect_size_second_half", - "avg_effect_size", - "min_effect_size", - ] - assert data.loc[0, "avg_effect_size"] != data.loc[1, "avg_effect_size"] - - -def test_common_ties_are_byte_deterministic_across_hash_seeds(tmp_path): - files = [tmp_path / f"top_{i}.csv" for i in range(3)] - rows = [ - {"parameter": "10", "effect_size": "0.5"}, - {"parameter": "2", "effect_size": "0.5"}, - {"parameter": "1", "effect_size": "0.5"}, - ] - for path in files: - _write_top(path, rows) - outputs = [] - for seed in ("1", "999"): - output = tmp_path / f"common_{seed}.csv" - env = {**os.environ, "PYTHONHASHSEED": seed} - result = _run(files, output, env=env) - assert result.returncode == 0, result.stderr - outputs.append(output.read_bytes()) - assert outputs[0] == outputs[1] - assert list(pd.read_csv(tmp_path / "common_1.csv")["parameter"]) == [1, 2, 10] - - -@pytest.mark.parametrize("case", ["missing", "no_common", "duplicate"]) -def test_common_fails_closed_without_valid_output(tmp_path, case): - files = [tmp_path / f"top_{i}.csv" for i in range(3)] - for path in files: - _write_top(path, [{"parameter": "1", "effect_size": "0.5"}]) - if case == "missing": - files[1] = tmp_path / "missing.csv" - elif case == "no_common": - _write_top(files[2], [{"parameter": "2", "effect_size": "0.5"}]) - else: - _write_top( - files[1], - [{"parameter": "1", "effect_size": "0.5"}, {"parameter": "1", "effect_size": "0.4"}], - ) - output = tmp_path / "common.csv" - output.write_text("stale output\n") - result = _run(files, output) - assert result.returncode != 0 - assert not output.exists() diff --git a/tests/cpu/test_engine_contract.py b/tests/cpu/test_engine_contract.py index 78275f3..caf3a40 100644 --- a/tests/cpu/test_engine_contract.py +++ b/tests/cpu/test_engine_contract.py @@ -58,36 +58,103 @@ def test_reduction_contract_checks_sum_and_sumsq(): artifacts.validate_reduction_outputs([3.0, 7.0], [5.0, 25.0], [4.0, 7.0], [10.0, 20.0]) -def test_top_artifact_preserves_float64_effects_and_writes_manifest(tmp_path): +def test_raw_artifact_preserves_float64_arrays_and_parameter_order(tmp_path): + import hashlib + matrix = np.array([[1.125, 2.25], [3.5, 8.75]], dtype=np.float32) - entry_sum = matrix.sum(axis=1, dtype=np.float64) - entry_sumsq = (matrix * matrix).sum(axis=1, dtype=np.float64) - exit_sum = matrix.sum(axis=0, dtype=np.float64) - exit_sumsq = (matrix * matrix).sum(axis=0, dtype=np.float64) - checks = artifacts.validate_reduction_outputs(entry_sum, entry_sumsq, exit_sum, exit_sumsq) + arrays = ( + matrix.sum(axis=1, dtype=np.float64), + (matrix * matrix).sum(axis=1, dtype=np.float64), + matrix.sum(axis=0, dtype=np.float64), + (matrix * matrix).sum(axis=0, dtype=np.float64), + ) + checks = artifacts.validate_reduction_outputs(*arrays) prefix = str(tmp_path / "result") - entry_dims = [("entry_parameter", 0.1, 0.2, 0.1, True)] + entry_dims = [("effect_size", 0.1, 0.2, 0.1, True)] exit_dims = [("exit_parameter", 1, 2, 1, False)] - artifacts.write_top_csv( + artifacts.write_results( prefix, "test_strategy", "source.csv", entry_dims, exit_dims, - entry_sum, - entry_sumsq, - exit_sum, - exit_sumsq, - 2, + *arrays, checks, False, + buy=0.0015, + sell=0.002, ) - expected = artifacts.cs.stats_from_groups( - entry_sum, entry_sumsq, 2, entry_sum.sum(), entry_sumsq.sum(), 4 - )["effect_size"] - output = pd.read_csv(f"{prefix}_top_entry.csv") - np.testing.assert_array_equal(output["effect_size"].to_numpy(), np.sort(expected)[::-1]) - assert "0.20000000000000001" not in open(f"{prefix}_top_entry.csv").read() - manifest = json.load(open(f"{prefix}_top_manifest.json")) + path = tmp_path / "result_results.npz" + with np.load(path, allow_pickle=False) as saved: + assert set(saved.files) == set(artifacts.RESULT_KEYS) + for key, expected in zip(artifacts.RESULT_KEYS, arrays): + assert saved[key].dtype == np.float64 + assert saved[key].tobytes() == expected.tobytes() + manifest = json.loads((tmp_path / "result_manifest.json").read_text()) + assert manifest["schema_version"] == 2 + assert manifest["entry_dims"] == [["effect_size", 0.1, 0.2, 0.1, True]] + assert manifest["exit_dims"] == [["exit_parameter", 1, 2, 1, False]] assert manifest["pair_count"] == 4 - assert manifest["outputs"]["entry"]["rows"] == 2 + assert manifest["buy"] == 0.0015 and manifest["sell"] == 0.002 + assert manifest["outputs"]["sha256"] == hashlib.sha256(path.read_bytes()).hexdigest() + assert not list(tmp_path.glob("*.tmp")) + assert not list(tmp_path.glob("*_top*")) + + +@pytest.mark.parametrize( + "arrays", + [ + ([[1.0]], [1.0], [1.0], [1.0]), + ([1.0, 2.0], [1.0], [3.0], [1.0]), + ], +) +def test_reduction_validation_rejects_invalid_shapes(arrays): + with pytest.raises(RuntimeError, match="one-dimensional|shapes must match"): + artifacts.validate_reduction_outputs(*arrays) + + +def test_raw_writer_failure_does_not_publish_partial_results(tmp_path, monkeypatch): + def fail_save(*args, **kwargs): + raise OSError("Injected disk error") + + monkeypatch.setattr(artifacts.np, "savez", fail_save) + dims = [("p", 1, 1, 1, False)] + arrays = [np.array([1.0])] * 4 + with pytest.raises(OSError, match="Injected disk error"): + artifacts.write_results( + tmp_path / "result", + "test", + "source.csv", + dims, + dims, + *arrays, + {}, + False, + buy=0.0, + sell=0.0, + ) + assert not list(tmp_path.iterdir()) + + +def test_failed_publication_invalidates_previous_manifest(tmp_path, monkeypatch): + dims = [("p", 1, 1, 1, False)] + prefix = tmp_path / "result" + arrays = [np.array([1.0])] * 4 + artifacts.write_results( + prefix, "test", "source.csv", dims, dims, *arrays, {}, False, buy=0.0, sell=0.0 + ) + original = artifacts.os.replace + + def fail_manifest(source, destination): + if str(destination).endswith("_manifest.json"): + raise OSError("Injected manifest publication failure") + return original(source, destination) + + monkeypatch.setattr(artifacts.os, "replace", fail_manifest) + new_arrays = [np.array([2.0])] * 4 + with pytest.raises(OSError, match="manifest publication failure"): + artifacts.write_results( + prefix, "test", "source.csv", dims, dims, *new_arrays, {}, False, buy=0.1, sell=0.1 + ) + assert not (tmp_path / "result_manifest.json").exists() + assert not list(tmp_path.glob("*.tmp")) diff --git a/tests/cpu/test_runpod.py b/tests/cpu/test_runpod.py index 4f17c95..aad81d1 100644 --- a/tests/cpu/test_runpod.py +++ b/tests/cpu/test_runpod.py @@ -183,7 +183,6 @@ def forbidden(): dry_run=True, client=client, ) - assert client.created == [] assert not (tmp_path / "output/runpod_state.json").exists() @@ -340,3 +339,34 @@ def test_api_key_is_removed_from_ssh_environment(tmp_path, monkeypatch): def test_invalid_server_ssh_metadata_is_rejected(pod): with pytest.raises(runpod.RunPodError, match="invalid public SSH"): runpod.endpoint(pod) + + +def test_default_runpod_bundle_runs_raw_backtest(tmp_path): + archive = tmp_path / "run.tar.gz" + runpod.build_bundle( + archive, mode="run", config_path=ROOT / "examples/gpu_backtest_examples/rsi/config.json" + ) + with tarfile.open(archive) as bundle: + command = bundle.extractfile("job.sh").read().decode() + assert "gpu_backtest run --config job.json --out-prefix output/result" in command + assert "pipeline" not in command and "charts" not in command and "altair" not in command + assert not any( + "workflows/" in name or name.endswith("statistics.py") for name in bundle.getnames() + ) + + +@pytest.mark.parametrize("key", ["top_n", "num_splits", "ranges", "start_date", "end_date"]) +def test_runpod_rejects_removed_analysis_keys_before_allocation(tmp_path, key): + config = json.loads((ROOT / "examples/gpu_backtest_examples/rsi/config.json").read_text()) + config["input"] = str(ROOT / "examples/gpu_backtest_examples/rsi/synthetic.csv") + config[key] = 3 + path = tmp_path / "job.json" + path.write_text(json.dumps(config)) + + class NoAllocation: + def create(self, body): + pytest.fail("Invalid config must fail before allocation") + + client = NoAllocation() + with pytest.raises(ValueError, match="documented engine config keys"): + runpod.launch(config_path=path, output_dir=tmp_path / "out", client=client) diff --git a/tests/cpu/test_split_contract.py b/tests/cpu/test_split_contract.py deleted file mode 100644 index 2372e01..0000000 --- a/tests/cpu/test_split_contract.py +++ /dev/null @@ -1,60 +0,0 @@ -import json - -import pandas as pd -import pytest - -from gpu_backtest.workflows import splits as splitter - - -def _market_frame(periods=10): - time = pd.date_range("2021-01-01", periods=periods, freq="12h", tz="Asia/Hong_Kong") - close = pd.Series(range(periods), dtype=float) + 100.0 - return pd.DataFrame( - { - "Time": time, - "Open": close, - "High": close + 2.0, - "Low": close - 2.0, - "Close": close + 1.0, - "Volume": 1000.0, - } - ) - - -def test_full_and_halves_are_disjoint_and_cover_full_range(tmp_path): - source = tmp_path / "sample.csv" - _market_frame().to_csv(source, index=False) - manifest_path = splitter.split_csv_by_date_range( - str(source), "01-01-2021", "05-01-2021", 3, "Time" - ) - manifest = json.loads(manifest_path.read_text()) - assert manifest["mode"] == "full_and_halves" - assert manifest["schema_version"] == 2 - assert manifest["split_contract_version"] == splitter.SPLIT_CONTRACT_VERSION - assert len(manifest["source_sha256"]) == 64 - assert all((len(record["sha256"]) == 64 for record in manifest["splits"])) - assert [record["label"] for record in manifest["splits"]] == [ - "full", - "first_half", - "second_half", - ] - split_dir = manifest_path.parent - full, first, second = [pd.read_csv(split_dir / record["file"]) for record in manifest["splits"]] - assert len(first) + len(second) == len(full) == manifest["filtered_rows"] - assert set(first["Time"]).isdisjoint(second["Time"]) - assert set(first["Time"]) | set(second["Time"]) == set(full["Time"]) - - -@pytest.mark.parametrize( - "mutation,error", [("duplicate", "duplicate timestamps"), ("unordered", "strictly increasing")] -) -def test_split_rejects_invalid_timestamp_order(tmp_path, mutation, error): - data = _market_frame() - if mutation == "duplicate": - data.loc[4, "Time"] = data.loc[3, "Time"] - else: - data.loc[[3, 4], "Time"] = data.loc[[4, 3], "Time"].to_numpy() - source = tmp_path / "invalid.csv" - data.to_csv(source, index=False) - with pytest.raises(ValueError, match=error): - splitter.split_csv_by_date_range(str(source), "01-01-2021", "05-01-2021", 3, "Time") diff --git a/tests/cpu/test_statistics.py b/tests/cpu/test_statistics.py deleted file mode 100644 index 1cfcf3b..0000000 --- a/tests/cpu/test_statistics.py +++ /dev/null @@ -1,91 +0,0 @@ -import warnings - -import numpy as np - -from gpu_backtest.core import statistics as cs - - -def _effect_size_scalar(sum1, sumsq1, n1, tot_sum, tot_sumsq, tot_count): - mean1 = sum1 / n1 - var1 = (sumsq1 - sum1**2 / n1) / (n1 - 1) - sum2 = tot_sum - sum1 - n2 = tot_count - n1 - sumsq2 = tot_sumsq - sumsq1 - mean2 = sum2 / n2 - var2 = (sumsq2 - sum2**2 / n2) / (n2 - 1) - diff = mean1 - mean2 - sp2 = ((n1 - 1) * var1 + (n2 - 1) * var2) / (n1 + n2 - 2) - return diff / np.sqrt(sp2) if sp2 > 0 else 0.0 - - -def test_stats_from_groups_matches_scalar_formula(): - rng = np.random.default_rng(7) - n_groups, n1 = (50, 30) - data = rng.normal(3.0, 2.0, (n_groups, n1)) - sums = data.sum(axis=1) - sumsqs = (data**2).sum(axis=1) - tot_sum, tot_sumsq, tot_count = (sums.sum(), sumsqs.sum(), n_groups * n1) - st = cs.stats_from_groups(sums, sumsqs, n1, tot_sum, tot_sumsq, tot_count) - want = np.array( - [ - _effect_size_scalar(sums[i], sumsqs[i], n1, tot_sum, tot_sumsq, tot_count) - for i in range(n_groups) - ] - ) - np.testing.assert_allclose(st["effect_size"], want, rtol=0, atol=1e-12) - - -def _cohen_effect(group, rest): - diff = group.mean() - rest.mean() - pooled = ((len(group) - 1) * group.var(ddof=1) + (len(rest) - 1) * rest.var(ddof=1)) / ( - len(group) + len(rest) - 2 - ) - return diff / np.sqrt(pooled) - - -def test_effect_size_isolates_each_entry_and_exit_against_all_counterparts(): - returns = np.array([[1.0, 2.0, 4.0, 8.0], [3.0, 5.0, 7.0, 9.0], [6.0, 10.0, 11.0, 12.0]]) - total_sum = returns.sum() - total_sumsq = (returns**2).sum() - total_count = returns.size - entry_stats = cs.stats_from_groups( - returns.sum(axis=1), - (returns**2).sum(axis=1), - returns.shape[1], - total_sum, - total_sumsq, - total_count, - ) - expected_entry = np.array( - [ - _cohen_effect(returns[e, :], np.delete(returns, e, axis=0).ravel()) - for e in range(returns.shape[0]) - ] - ) - np.testing.assert_allclose(entry_stats["effect_size"], expected_entry, rtol=0, atol=1e-12) - exit_stats = cs.stats_from_groups( - returns.sum(axis=0), - (returns**2).sum(axis=0), - returns.shape[0], - total_sum, - total_sumsq, - total_count, - ) - expected_exit = np.array( - [ - _cohen_effect(returns[:, x], np.delete(returns, x, axis=1).ravel()) - for x in range(returns.shape[1]) - ] - ) - np.testing.assert_allclose(exit_stats["effect_size"], expected_exit, rtol=0, atol=1e-12) - - -def test_stats_degenerate_data_no_warning(): - n_groups, n1 = (4, 10) - data = np.full((n_groups, n1), 0.1) - sums = data.sum(axis=1) - sumsqs = (data**2).sum(axis=1) - with warnings.catch_warnings(): - warnings.simplefilter("error") - st = cs.stats_from_groups(sums, sumsqs, n1, sums.sum(), sumsqs.sum(), n_groups * n1) - np.testing.assert_allclose(st["effect_size"], 0.0, atol=1e-09) diff --git a/tests/cpu/test_strategy.py b/tests/cpu/test_strategy.py index 18e48ce..fdf97cb 100644 --- a/tests/cpu/test_strategy.py +++ b/tests/cpu/test_strategy.py @@ -34,7 +34,7 @@ def test_builtin_and_module_object_loading(): ([("entry", 2, 1, 1, False)], "low <= high"), ([("entry", 1, 2, 0.5, False)], "integer bounds"), ([("entry", 0.0, 1.0, 0.3, True)], "lie on"), - ([("effect_size", 1, 2, 1, False)], "reserved"), + ([("bad name", 1, 2, 1, False)], "Invalid parameter name"), ([("entry", 1, float("inf"), 1, False)], "finite"), ([("entry", 1, 2, 1, "i")], "boolean"), ([("entry", 1, 2, 1, False)] * 5, "1 to 4"), @@ -84,3 +84,7 @@ def test_parameter_decode_preserves_sub_micro_precision(): def test_invalid_import_names(value): with pytest.raises(ValueError, match="module name"): load_strategy(value) + + +def test_parameter_names_are_independent_of_removed_analysis_columns(): + validate_strategy(specification(ENTRY_DIMS=[("effect_size", 1, 2, 1, False)])) diff --git a/tests/gpu/fixtures.py b/tests/gpu/fixtures.py index 6c5988e..eba0b08 100644 --- a/tests/gpu/fixtures.py +++ b/tests/gpu/fixtures.py @@ -49,7 +49,7 @@ def check_matrix(tmp_path, monkeypatch): source = tmp_path / "bars.csv" write_bars(source, rsi_bars()) prefix = str(tmp_path / "result") - first = run(plugin, source, prefix, top_n=3, threads_per_block=32, verbose=False) + first = run(plugin, source, prefix, threads_per_block=32, verbose=False) expected = { "entry_sum": [39.0, 69.0], "exit_sum": [34.0, 36.0, 38.0], @@ -58,21 +58,47 @@ def check_matrix(tmp_path, monkeypatch): } for key, values in expected.items(): np.testing.assert_array_equal(first[key], values) - artifacts = [tmp_path / f"result_top_{side}.csv" for side in ("entry", "exit")] - artifacts.append(tmp_path / "result_top_manifest.json") + artifacts = [tmp_path / "result_results.npz", tmp_path / "result_manifest.json"] before = [path.read_bytes() for path in artifacts] - second = run( - load_strategy(plugin), source, prefix, top_n=3, threads_per_block=32, verbose=False - ) + second = run(load_strategy(plugin), source, prefix, threads_per_block=32, verbose=False) for key in first: - np.testing.assert_array_equal(first[key], second[key]) + assert first[key].tobytes() == second[key].tobytes() assert [path.read_bytes() for path in artifacts] == before - manifest = json.loads(artifacts[-1].read_text()) + with np.load(artifacts[0], allow_pickle=False) as saved: + for key, values in expected.items(): + np.testing.assert_array_equal(saved[key], values) + assert saved[key].dtype == np.float64 + manifest = json.loads(artifacts[1].read_text()) + assert manifest["schema_version"] == 2 assert manifest["strategy"] == "matrix" and manifest["pair_count"] == 6 - for side, count, order in (("entry", 2, [2, 1]), ("exit", 3, [3, 2, 1])): - data = pd.read_csv(tmp_path / f"result_top_{side}.csv") - assert len(data) == count - assert data[side].tolist() == order + assert manifest["entry_dims"] == [["entry", 1, 2, 1, False]] + assert manifest["exit_dims"] == [["exit", 1, 3, 1, False]] + assert not list(tmp_path.glob("*_top*")) + + +def check_single_output(tmp_path, exit_open, fee, expected_sum, expected_square): + source = tmp_path / "single.csv" + write_bars(source, rsi_bars(exit_open)) + result = run( + "gpu_backtest_examples.rsi.strategy", + source, + str(tmp_path / "single"), + buy=fee, + sell=fee, + entry_dims=[("p_e", 2, 2, 1, False), ("buy_lvl", 30, 30, 1, False)], + exit_dims=[("p_x", 2, 2, 1, False), ("sell_lvl", 70, 70, 1, False)], + verbose=False, + ) + with np.load(tmp_path / "single_results.npz", allow_pickle=False) as saved: + for side in ("entry", "exit"): + np.testing.assert_allclose(saved[f"{side}_sum"], [expected_sum], rtol=0, atol=0.0001) + np.testing.assert_allclose( + saved[f"{side}_sumsq"], [expected_square], rtol=0, atol=0.0001 + ) + assert saved[f"{side}_sum"].tobytes() == result[f"{side}_sum"].tobytes() + manifest = json.loads((tmp_path / "single_manifest.json").read_text()) + assert manifest["entry_count"] == manifest["exit_count"] == manifest["pair_count"] == 1 + assert manifest["buy"] == manifest["sell"] == fee def check_rsi_reference(tmp_path): diff --git a/tests/gpu/simulator_cases.py b/tests/gpu/simulator_cases.py index afc9f7c..26669ef 100644 --- a/tests/gpu/simulator_cases.py +++ b/tests/gpu/simulator_cases.py @@ -1,7 +1,7 @@ """Explicitly collected only in an isolated CUDA-simulator process.""" import pytest -from fixtures import check_matrix, check_rsi_reference, rsi_bars, single_rsi +from fixtures import check_matrix, check_rsi_reference, check_single_output, rsi_bars, single_rsi def test_known_matrix_external_module_and_determinism(tmp_path, monkeypatch): @@ -55,3 +55,15 @@ def test_open_position_is_marked_at_final_open_without_sell_fee(tmp_path): ] actual = single_rsi(tmp_path / "bars.csv", bars, buy=0.0015, sell=0.5) assert actual == pytest.approx(-25.15, abs=0.0001) + + +@pytest.mark.parametrize( + "exit_open,fee,total,square", + [ + (120, 0.0, 50.0, 2500.0), + (120, 0.0015, 49.625, 2462.640625), + (60, 0.0, -25.0, 625.0), + ], +) +def test_single_combination_raw_output(tmp_path, exit_open, fee, total, square): + check_single_output(tmp_path, exit_open, fee, total, square) diff --git a/tests/gpu/test_hardware.py b/tests/gpu/test_hardware.py index 2b53095..5bb6c76 100644 --- a/tests/gpu/test_hardware.py +++ b/tests/gpu/test_hardware.py @@ -1,5 +1,5 @@ import pytest -from fixtures import check_matrix, check_rsi_reference +from fixtures import check_matrix, check_rsi_reference, check_single_output from numba import config, cuda pytestmark = pytest.mark.gpu @@ -19,3 +19,15 @@ def test_real_gpu_known_matrix(tmp_path, monkeypatch): def test_real_gpu_rsi_reference(tmp_path): check_rsi_reference(tmp_path) + + +@pytest.mark.parametrize( + "exit_open,fee,total,square", + [ + (120, 0.0, 50.0, 2500.0), + (120, 0.0015, 49.625, 2462.640625), + (60, 0.0, -25.0, 625.0), + ], +) +def test_single_combination_raw_output(tmp_path, exit_open, fee, total, square): + check_single_output(tmp_path, exit_open, fee, total, square) diff --git a/tools/gpu_backtest_tools/benchmarks/runner.py b/tools/gpu_backtest_tools/benchmarks/runner.py index 50a44f6..091d956 100644 --- a/tools/gpu_backtest_tools/benchmarks/runner.py +++ b/tools/gpu_backtest_tools/benchmarks/runner.py @@ -24,7 +24,7 @@ from gpu_backtest.core.indicators import prepare_tables from gpu_backtest.core.kernels import build_kernels from gpu_backtest.core.output import validate_reduction_outputs -from gpu_backtest.core.output import write_top_csv as _write_top_csv +from gpu_backtest.core.output import write_results as _write_results from gpu_backtest.core.strategy import validate_strategy from .cpu import build_cpu_reducer @@ -141,7 +141,7 @@ def _cpu_info(threads): def run_cpu_baseline(source, prefix, entry_dims, exit_dims, reducer=None): - """CSV through ranked output using the compiled CPU two-pass reference.""" + """CSV through raw NPZ/manifest output using the compiled CPU two-pass reference.""" data = validate_market_data(source, expected_interval="1D") arrays = [ data[c].to_numpy(dtype=np.float32) for c in ("Close", "Open", "High", "Low", "Volume") @@ -151,21 +151,19 @@ def run_cpu_baseline(source, prefix, entry_dims, exit_dims, reducer=None): results = reducer(*args) checks = validate_reduction_outputs(*results) if prefix: - _write_top_csv( + _write_results( str(prefix), "rsi_meanrev", str(source), entry_dims, exit_dims, *results, - 100, checks, False, + buy=0.0015, + sell=0.0015, + engine="gpu_backtest.benchmark_cpu", ) - manifest = Path(f"{prefix}_top_manifest.json") - content = json.loads(manifest.read_text()) - content["engine"] = "gpu_backtest.benchmark_cpu" - manifest.write_text(json.dumps(content, indent=2) + "\n") return dict(zip(RESULT_KEYS, results)) @@ -267,7 +265,7 @@ def benchmark(output, *, bars=1024, cpu_threads=8, repeats=3, billion=False, bil "gpu_compile_and_warmup_seconds": gpu_warmup, "max_absolute_differences": differences, }, - "timing_scope": "Comparison: warmed two-pass reductions only. Shared input/table preparation, GPU transfers, compilation, result copying, statistics and CSV writing are excluded. CPU output allocation is included.", + "timing_scope": "Comparison: warmed two-pass reductions only. Shared input/table preparation, GPU transfers, compilation, result copying, raw NPZ/manifest writing are excluded. CPU output allocation is included.", } output.write_text(json.dumps(report, indent=2) + "\n") print( @@ -287,7 +285,6 @@ def benchmark(output, *, bars=1024, cpu_threads=8, repeats=3, billion=False, bil str(output.parent / "billion"), entry_dims=e_dims, exit_dims=x_dims, - top_n=100, verbose=True, ) elapsed = time.perf_counter() - started @@ -302,7 +299,7 @@ def benchmark(output, *, bars=1024, cpu_threads=8, repeats=3, billion=False, bil "reduction_array_bytes": sum((a.nbytes for a in large.values())), "hypothetical_float32_matrix_bytes": entries * exits * 4, "nonzero_entry_groups": int(np.count_nonzero(large["entry_sum"])), - "timing_scope": "Normal engine run including CSV validation, indicator tables, allocation/transfers, kernel construction/JIT, two GPU passes, copies, statistics and top CSV/manifest output. Excludes pod provisioning and dependency installation.", + "timing_scope": "Normal engine run including CSV validation, indicator tables, allocation/transfers, kernel construction/JIT, two GPU passes, copies, raw NPZ/manifest output. Excludes pod provisioning and dependency installation.", } output.write_text(json.dumps(report, indent=2) + "\n") print(f"Billion-pair normal engine run completed in {elapsed:.3f}s.", flush=True) @@ -325,7 +322,7 @@ def benchmark(output, *, bars=1024, cpu_threads=8, repeats=3, billion=False, bil end_to_end_speedup=cpu_elapsed / elapsed, elapsed_seconds_saved=cpu_elapsed - elapsed, cpu_gpu_max_absolute_differences=differences, - cpu_timing_scope="Compiled CPU reference through CSV validation, indicator preparation, both passes, allocations, checks, statistics and CSV/manifest output. Reuses the reducer already compiled for the smaller comparison. No CUDA simulation. One full measured run, not an extrapolation.", + cpu_timing_scope="Compiled CPU reference through CSV validation, indicator preparation, both passes, allocations, checks, raw NPZ/manifest output. Reuses the reducer already compiled for the smaller comparison. No CUDA simulation. One full measured run, not an extrapolation.", ) output.write_text(json.dumps(report, indent=2) + "\n") print( diff --git a/tools/gpu_backtest_tools/runpod/bundle.py b/tools/gpu_backtest_tools/runpod/bundle.py index bd36eae..54bb82f 100644 --- a/tools/gpu_backtest_tools/runpod/bundle.py +++ b/tools/gpu_backtest_tools/runpod/bundle.py @@ -10,13 +10,13 @@ from pathlib import Path from gpu_backtest import __version__, resolve_strategy_name +from gpu_backtest.cli.options import CONFIG_KEYS PACKAGE_FILES = { "gpu_backtest": [ "__init__.py", "__main__.py", "cli/__init__.py", - "cli/analysis.py", "cli/backtest.py", "cli/options.py", "cli/tools.py", @@ -27,13 +27,7 @@ "core/indicators.py", "core/kernels.py", "core/output.py", - "core/statistics.py", "core/strategy.py", - "workflows/__init__.py", - "workflows/charts.py", - "workflows/common.py", - "workflows/pipeline.py", - "workflows/splits.py", ], "gpu_backtest_examples": [ "__init__.py", @@ -58,31 +52,16 @@ "runpod/requirements.txt", ], } -CONFIG_KEYS = { - "num_splits", - "exit_dims", - "input", - "buy", - "strategy", - "top_n", - "end_date", - "entry_dims", - "start_date", - "expected_interval", - "threads_per_block", - "ranges", - "sell", -} IGNORED_PARTS = {"__pycache__", ".git", ".ruff_cache", "venv", ".venv", ".pytest_cache"} -def build_bundle(archive, *, mode, config_path=None, plugin_dir=None, charts=False): - if mode not in ("run", "pipeline", "check", "benchmark"): - raise ValueError("RunPod mode must be run, pipeline, check, or benchmark") +def build_bundle(archive, *, mode, config_path=None, plugin_dir=None): + if mode not in ("run", "check", "benchmark"): + raise ValueError("RunPod mode must be run, check, or benchmark") config = {} - if mode in ("run", "pipeline"): + if mode == "run": if not config_path: - raise ValueError("RunPod run/pipeline requires --config") + raise ValueError("RunPod run requires --config") config_path = Path(config_path).resolve() config = json.loads(config_path.read_text()) if not isinstance(config, dict) or set(config) - CONFIG_KEYS: @@ -135,7 +114,7 @@ def build_bundle(archive, *, mode, config_path=None, plugin_dir=None, charts=Fal if str(entry) == "LICENSE" or str(entry).endswith("/licenses/LICENSE"): shutil.copyfile(distribution.locate_file(entry), engine / "LICENSE") break - if mode in ("run", "pipeline"): + if mode == "run": (staging / "input").mkdir() shutil.copyfile(source, staging / "input/market.csv") config["input"] = "input/market.csv" @@ -157,19 +136,16 @@ def build_bundle(archive, *, mode, config_path=None, plugin_dir=None, charts=Fal copied += 1 if not copied: raise ValueError("Plugin directory contains no Python files") - if mode in ("run", "pipeline"): + if mode == "run": config["strategy"] = resolve_strategy_name(config["strategy"]) (staging / "job.json").write_text(json.dumps(config, indent=2) + "\n") - if ( - mode in ("run", "pipeline") - and config["strategy"] != "gpu_backtest_examples.rsi.strategy" - ): + if mode == "run" and config["strategy"] != "gpu_backtest_examples.rsi.strategy": relative = Path(*config["strategy"].split(".")) if not (staging / "plugins" / relative.with_suffix(".py")).is_file() and ( not (staging / "plugins" / relative / "__init__.py").is_file() ): raise ValueError("External strategy must be present in --plugin-dir") - (staging / "job.sh").write_text(remote_script(mode, charts=charts)) + (staging / "job.sh").write_text(remote_script(mode)) files = [] for path in sorted(staging.rglob("*")): if path.is_file(): @@ -187,7 +163,7 @@ def build_bundle(archive, *, mode, config_path=None, plugin_dir=None, charts=Fal return files -def remote_script(mode, *, charts=False): +def remote_script(mode): script = """#!/bin/bash set -euo pipefail export NUMBA_ENABLE_CUDASIM=0 @@ -203,13 +179,7 @@ def remote_script(mode, *, charts=False): script += ( "env/bin/python -m gpu_backtest run --config job.json --out-prefix output/result\n" ) - elif mode == "pipeline": - script += "env/bin/python -m gpu_backtest pipeline --config job.json --output-dir output/pipeline\n" elif mode == "benchmark": script += "env/bin/python -m gpu_backtest benchmark --output output/benchmark.json --billion --billion-cpu\n" - if charts and mode in ("run", "pipeline"): - folder = "output/pipeline" if mode == "pipeline" else "output" - script += "env/bin/python -m pip install 'altair==6.3.0'\n" - script += f"env/bin/python -m gpu_backtest charts --input {folder}/*_top_entry.csv {folder}/*_top_exit.csv --output output/charts.html\n" script += "env/bin/python -m pip freeze > output/requirements-resolved.txt\n" return script diff --git a/tools/gpu_backtest_tools/runpod/launcher.py b/tools/gpu_backtest_tools/runpod/launcher.py index f42cf12..855dd07 100644 --- a/tools/gpu_backtest_tools/runpod/launcher.py +++ b/tools/gpu_backtest_tools/runpod/launcher.py @@ -81,7 +81,7 @@ def launch( config_path=None, output_dir, ssh_key=None, - mode="pipeline", + mode="run", plugin_dir=None, image=DEFAULT_IMAGE, gpu="NVIDIA GeForce RTX 4090", @@ -89,7 +89,6 @@ def launch( max_seconds=1800, max_hourly_rate=1.0, keep_pod=False, - charts=False, dry_run=False, client=None, transport_factory=SSHTransport, @@ -109,7 +108,6 @@ def launch( mode=mode, config_path=config_path, plugin_dir=plugin_dir, - charts=charts, ) if dry_run: print(