diff --git a/.gitignore b/.gitignore index 9443e36..db6dbb3 100644 --- a/.gitignore +++ b/.gitignore @@ -4,6 +4,7 @@ __pycache__/ .pytest_cache/ .ruff_cache/ .coverage +coverage.json htmlcov/ dist/ build/ diff --git a/README.md b/README.md index 1fda9fd..53958e6 100644 --- a/README.md +++ b/README.md @@ -4,7 +4,7 @@ [![PyPI](https://img.shields.io/pypi/v/ml4t-models)](https://pypi.org/project/ml4t-models/) [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT) -Finance-native model implementations for latent-factor estimation, stochastic discount factor learning, direct asset prediction, and end-to-end portfolio learning. +Finance-specific models for asset pricing, prediction, and portfolio learning. Documentation: [ml4trading.io/docs/models](https://www.ml4trading.io/docs/models/) @@ -69,96 +69,42 @@ Documentation tools are contributor dependencies. From a source checkout, run ## Quick Start -### 1. Latent-Factor Forecast Pipeline +The base package can produce a first forecast on a small synthetic panel without credentials or +an accelerator. Run this complete example with `pip install ml4t-models`: ```python import numpy as np - from ml4t.models import ( BetaLambdaMapper, - CrossSectionBatch, ExpandingMeanFactorForecaster, - IPCAConfig, - IPCAModel, LatentFactorForecastPipeline, + PCAConfig, + PCAModel, + PersistentPanelBatch, ) -batch = CrossSectionBatch( - characteristics=np.random.randn(24, 200, 12), - returns=np.random.randn(24, 200), - timestamps=tuple(range(24)), +asset_ids = tuple(f"asset_{i}" for i in range(6)) +train = PersistentPanelBatch( + returns=np.random.default_rng(1).normal(scale=0.02, size=(12, 6)), + timestamps=tuple(f"2024-{month:02d}" for month in range(1, 13)), + asset_ids=asset_ids, ) - +future = PersistentPanelBatch(timestamps=("2025-01", "2025-02"), asset_ids=asset_ids) pipeline = LatentFactorForecastPipeline( - model=IPCAModel(IPCAConfig(n_factors=3)), + model=PCAModel(PCAConfig(n_factors=2)), forecaster=ExpandingMeanFactorForecaster(), mapper=BetaLambdaMapper(), ) -pipeline.fit(batch) -prediction = pipeline.predict(batch) - -print(prediction.asset_forecast.expected_returns.shape) -# (24, 200) -``` - -### 2. Weight-Native Stochastic Discount Factor - -```python -import numpy as np - -from ml4t.models import ( - CrossSectionBatch, - StochasticDiscountFactorConfig, - StochasticDiscountFactorModel, -) - -batch = CrossSectionBatch( - characteristics=np.random.randn(36, 300, 16), - returns=np.random.randn(36, 300), - context_features=np.random.randn(36, 8), - timestamps=tuple(range(36)), -) - -model = StochasticDiscountFactorModel( - StochasticDiscountFactorConfig(checkpoint_epochs=(256, 512, 768, 1024)) -) -model.fit(batch) -state = model.extract(batch, checkpoint=1280) - -print(state.asset_weights.shape) -# (36, 300) +pipeline.fit(train) +forecast = pipeline.predict(future).asset_forecast.expected_returns +assert forecast.shape == (2, 6) and np.isfinite(forecast).all() +print(forecast.shape) # (2, 6) ``` -### 3. End-to-End Portfolio Learning - -```python -import numpy as np - -from ml4t.models import LSTMPortfolioConfig, LSTMPortfolioModel, PortfolioSequenceBatch - -batch = PortfolioSequenceBatch( - features=np.random.randn(8, 63, 20, 10), - returns=np.random.randn(8, 63, 20), - timestamps=tuple(range(63)), - asset_ids=tuple(f"asset_{i}" for i in range(20)), -) - -model = LSTMPortfolioModel(LSTMPortfolioConfig(max_iters=20, checkpoint_every=5)) -model.fit(batch) -weights = model.predict(batch, checkpoint=20) - -print(weights.weights.shape) -# (8, 63, 20) -``` - -### 4. Hand Off Predictions To The Rest Of ML4T - -```python -from ml4t.models import predictions_frame_from_asset_forecast, write_backtest_frames - -frame = predictions_frame_from_asset_forecast(prediction.asset_forecast) -write_backtest_frames("artifacts/run_001", predictions=frame) -``` +The forecast uses the training factor history and preserves the future dates and asset order. +It does not imply trading performance. The [Quickstart](docs/getting-started/quickstart.md) +explains the result, and the [Book Guide](docs/book-guide/index.md) links to pinned teaching files. +The `deep` extra is needed for neural models; the `integration` extra adds Polars and Specs support. ## Model Families diff --git a/docs/api/index.md b/docs/api/index.md index c09209d..56d7056 100644 --- a/docs/api/index.md +++ b/docs/api/index.md @@ -74,7 +74,7 @@ The package root re-exports the main model classes, configs, batches, results, a ## Stability The [API Stability](../reference/api-stability.md) page defines the public -surface intended to remain stable through the `0.1` beta series. +surface for the `0.1` stable line. ## Integration @@ -92,3 +92,14 @@ surface intended to remain stable through the `0.1` beta series. | `ml4t.models.stochastic_discount_factor` | weight-native SDF estimation and return projections | | `ml4t.models.asset_prediction` | direct asset-level predictors | | `ml4t.models.portfolio` | end-to-end portfolio learners | + +## Portfolio model signatures + +The portfolio models are imported lazily from `ml4t.models`. Their class signatures are rendered +here explicitly so all three supported allocators are covered by the generated reference. + +::: ml4t.models.portfolio.linear.LinearFeaturePortfolioModel + +::: ml4t.models.portfolio.lstm.LSTMPortfolioModel + +::: ml4t.models.portfolio.deep_portfolio.DeepPortfolioModel diff --git a/docs/book-guide/index.md b/docs/book-guide/index.md index c4f548a..ab32a85 100644 --- a/docs/book-guide/index.md +++ b/docs/book-guide/index.md @@ -1,132 +1,52 @@ # Book Guide -`ml4t-models` is the library form of the model families developed manually in the book notebooks. +The public companion repository at revision +[`d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb`](https://github.com/stefan-jansen/machine-learning-for-trading/tree/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb) +provides the teaching files below. Every linked path was checked against that revision's Git tree. +These notebooks explain methods and often use their own data and dependencies. None of the Chapter 14 +teaching notebooks below imports `ml4t.models`; run the library's small examples for its API. -![From The Factor Zoo To A Library Taxonomy](../images/figure_14_1_factor_zoo_to_discipline.jpeg) +## Latent factors and factor forecasts -The goal is not to hide the teaching implementation. The goal is to: +| Public book file | What it does | Related library task | +|---|---|---| +| [IPCA notebook](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/14_latent_factors/04_ipca.ipynb) | Manually teaches characteristic-dependent betas and factor forecasts. | [Fit a latent-factor pipeline](../user-guide/latent-factor-pipelines.md) with `IPCAModel` and a separate forecaster. | +| [Risk-premium PCA notebook](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/14_latent_factors/05_rp_pca.ipynb) | Manually teaches pricing-aware factor extraction. | [Choose a latent-factor model](../user-guide/latent-factor-models.md) with `RPPCAModel`. | +| [Conditional autoencoder notebook](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/14_latent_factors/06_conditional_autoencoder.ipynb) | Manually builds a neural conditional factor model. | [Choose a latent-factor model](../user-guide/latent-factor-models.md) with `CAEModel`; the library requires `deep`. | +| [Case-study insights notebook](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/14_latent_factors/09_case_study_insights.ipynb) | Illustrates analysis of stored case-study results, not a first API example. | [Hand results downstream](../user-guide/integration.md). | -- show the architecture and mathematics clearly in the chapter notebooks -- use the library for repeatable case-study execution and downstream integration +The book's [case-study library bridge](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/case_studies/utils/latent_factors/library_bridge.py) +*does* import `ml4t.models` for PCA, IPCA, CAE, SDF, and SAE runs. It is case-study integration +code with data and registry prerequisites, not a standalone quickstart. -## Chapter Mapping +## SDF and direct prediction -### Chapter 14: Latent Factors +| Public book file | What it does | Related library task | +|---|---|---| +| [Adversarial SDF notebook](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/14_latent_factors/07_stochastic_discount_factor.ipynb) | Manually teaches phase-aware SDF training. | [Estimate SDF weights](../user-guide/stochastic-discount-factor.md) with `StochasticDiscountFactorModel`. | +| [Supervised autoencoder notebook](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/14_latent_factors/08_supervised_autoencoder.ipynb) | Manually teaches a direct supervised predictor. | [Predict asset signals](../user-guide/direct-asset-prediction.md) with `SAEModel`. | -The latent-factor chapter corresponds most directly to: +These neural notebooks need PyTorch and book data. Their full training runs are longer than the +small CPU examples in this site's task guides. The book's targets and splits may also differ from +the synthetic examples; results are not directly comparable. -- `PCAModel` -- `RPPCAModel` -- `IPCAModel` -- `CAEModel` -- `StochasticDiscountFactorModel` -- `SAEModel` as supervised autoencoder direct prediction +## Portfolio learning -The key conceptual transition from the notebooks to the library is: +| Public book file | What it does | Related library task | +|---|---|---| +| [Deep portfolio optimization](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/17_portfolio_construction/11_dl_portfolio_allocation.ipynb) | Illustrates a related neural allocation workflow, without calling this library. | [Learn portfolio weights](../user-guide/portfolio-learning.md). | +| [VLSTM portfolio](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/17_portfolio_construction/12_vlstm_portfolio.ipynb) | Manually teaches variable selection and sequence allocation. | [Learn portfolio weights](../user-guide/portfolio-learning.md) with `LSTMPortfolioModel`. | +| [DeePM regime robustness](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/17_portfolio_construction/13_deepm_regime_robust.ipynb) | Manually teaches a related DeePM architecture; it is not an exact library implementation. | [Learn portfolio weights](../user-guide/portfolio-learning.md) with `DeepPortfolioModel`. | -- notebook exposition may derive the math and architecture step by step -- library code enforces the clean separation between: - - structural extraction - - factor forecasting - - asset mapping +Portfolio notebooks use book-specific data and longer neural training. Start with the library's +[linear CPU example](https://github.com/ml4t/models/blob/main/examples/portfolio_learning.py) +for an observable result. -That separation matters most for `IPCAModel` and `CAEModel`. In the teaching notebooks, it -is helpful to show the full architecture and fitted-return logic step by step. In the -library, the corresponding production object is the two-step pipeline: +## Data and downstream boundaries -```text -structural estimator -> factor-premium forecaster -> asset mapper -``` +The library's [data contracts](../user-guide/data-contracts.md) preserve timestamps and asset +identity. [Integration](../user-guide/integration.md) converts predictions and weights to frames for +`ml4t-diagnostic` and `ml4t-backtest`. The Chapter 14 case-study insights notebook illustrates +analysis after model runs, but it does not replace those packages' guides or APIs. -### Chapter 17: Portfolio Construction - -The end-to-end allocation family corresponds to: - -- `LinearFeaturePortfolioModel` -- `LSTMPortfolioModel` -- `DeepPortfolioModel` - -These models are designed to connect naturally to: - -- Chapter 18 cost modeling -- Chapter 19 risk controls -- Chapter 20 strategy analysis - -## Why The Library Split Matters - -The book often needs to compare multiple modeling ideas side by side: - -- latent-factor models -- no-arbitrage SDF models -- direct signal models -- end-to-end allocation models - -The library turns those into explicit families instead of treating them as one generic “deep learning model.” - -## Case Studies - -The case studies are intended to act as: - -- integration tests -- realistic pressure tests for the API -- examples of how to hand model outputs into `ml4t-backtest` and `ml4t-diagnostic` - -They should not define the public API by accident. - -## Compatibility Status - -The `0.1.0` stable line is validated against the Chapter 14 teaching flow and the shared case-study -latent-factor bridge. - -| Book surface | Validation status | -|---|---| -| `14_latent_factors/04_ipca.ipynb` | full notebook execution passed | -| `14_latent_factors/05_rp_pca.ipynb` | full notebook execution passed | -| `14_latent_factors/06_conditional_autoencoder.ipynb` | full notebook execution passed | -| `14_latent_factors/07_stochastic_discount_factor.ipynb` | full notebook execution passed | -| `14_latent_factors/08_supervised_autoencoder.ipynb` | Papermill smoke execution passed; full production training is long-running | -| `14_latent_factors/09_case_study_insights.ipynb` | full notebook execution passed | -| `case_studies.utils.latent_factors.library_bridge` | synthetic PCA, IPCA, CAE, SAE, and SDF bridge smoke checks passed | - -The teaching notebooks keep hand-built implementations where that improves exposition. The -case-study path uses `ml4t-models` through the shared latent-factor bridge so the same -contracts are exercised in walk-forward validation and registry-backed analysis. - -## Case-Study Validation - -The beta gate also checks that each case-study family can execute its model-specific -latent-factor notebooks through the shared bridge. - -| Case study | Validation status | -|---|---| -| ETF returns | PCA, IPCA, SDF, and SAE cached executions passed; CAE passed in cached three-fold validation mode | -| US firm characteristics | IPCA, CAE, SDF, and SAE passed in cached three-fold validation mode | -| S&P 500 option analytics | PCA, IPCA, CAE, SDF, and SAE passed in cached three-fold validation mode | - -The ETF CAE notebook also reached the cached full-fold execution path and loaded the -registry-backed model outputs before the notebook kernel exited while processing the large -cached result set. The three-fold validation run exercises the same library bridge, -checkpoint handling, prediction schema, and registry persistence path with a bounded -runtime footprint. - -## Evaluation Boundary - -Case-study IC reporting is delegated to `ml4t-diagnostic`: - -- fold-level scoring calls `ml4t.diagnostic.metrics.cross_sectional_ic` -- pooled model-analysis summaries call `cross_sectional_ic` and `cross_sectional_ic_series` -- model outputs are converted into `PredictionsFrame`, `SignalsFrame`, `WeightsFrame`, and - `ml4t-backtest` handoff payloads by library adapters - -`ml4t-models` remains responsible for fitting and output contracts. Statistical diagnostics -and execution simulation remain owned by `ml4t-diagnostic` and `ml4t-backtest`. - -## Recommended Reading Order - -If you are moving from the book notebooks to the library: - -1. [Data Contracts](../user-guide/data-contracts.md) -2. [Latent-Factor Pipelines](../user-guide/latent-factor-pipelines.md) -3. [Stochastic Discount Factor](../user-guide/stochastic-discount-factor.md) -4. [Portfolio Learning](../user-guide/portfolio-learning.md) -5. [Integration](../user-guide/integration.md) +The [Quickstart](../getting-started/quickstart.md) is the first runnable library workflow. diff --git a/docs/getting-started/quickstart.md b/docs/getting-started/quickstart.md index cec4691..2020c24 100644 --- a/docs/getting-started/quickstart.md +++ b/docs/getting-started/quickstart.md @@ -1,183 +1,74 @@ -# Quickstart +# Quickstart: forecast returns from a small panel -This quickstart shows the three main workflows in the library: +This CPU workflow fits PCA to 12 months of synthetic returns, forecasts factor premia from the +training history, and maps them to six assets at two future dates. It uses the base installation: +`pip install ml4t-models`. No data download, credentials, or accelerator is needed. -1. latent-factor forecasting -2. stochastic discount factor extraction -3. end-to-end portfolio learning - -The same workflows are available as executable smoke examples in the repository's -`examples/` directory. - -## 1. Latent-Factor Forecasting - -The latent-factor path is intentionally three-stage: - -1. fit a structural model -2. forecast factor premia -3. map those forecasts back to assets +Save the following as `first_forecast.py` and run `python first_forecast.py`: ```python import numpy as np from ml4t.models import ( BetaLambdaMapper, - CrossSectionBatch, ExpandingMeanFactorForecaster, - IPCAConfig, - IPCAModel, LatentFactorForecastPipeline, + PCAConfig, + PCAModel, + PersistentPanelBatch, ) -batch = CrossSectionBatch( - characteristics=np.random.randn(36, 250, 12), - returns=np.random.randn(36, 250), - timestamps=tuple(range(36)), +rng = np.random.default_rng(1) +asset_ids = tuple(f"asset_{i}" for i in range(6)) +train = PersistentPanelBatch( + returns=rng.normal(scale=0.02, size=(12, 6)), + timestamps=tuple(f"2024-{month:02d}" for month in range(1, 13)), + asset_ids=asset_ids, +) +future = PersistentPanelBatch( + timestamps=("2025-01", "2025-02"), + asset_ids=asset_ids, ) pipeline = LatentFactorForecastPipeline( - model=IPCAModel(IPCAConfig(n_factors=3)), + model=PCAModel(PCAConfig(n_factors=2)), forecaster=ExpandingMeanFactorForecaster(), mapper=BetaLambdaMapper(), ) - -fit_result = pipeline.fit(batch) -lf_prediction = pipeline.predict(batch) - -print(fit_result.structural_fit.converged) -print(lf_prediction.state.asset_betas.shape) # (36, 250, 3) -print(lf_prediction.factor_forecast.factor_premia.shape) # (36, 3) -print(lf_prediction.asset_forecast.expected_returns.shape) +fit = pipeline.fit(train) +prediction = pipeline.predict(future) +forecast = prediction.asset_forecast.expected_returns + +assert fit.structural_fit.converged +assert forecast.shape == (2, 6) +assert np.isfinite(forecast).all() +print(f"forecast shape: {forecast.shape}") ``` -### Why This Matters - -This separation matches the actual finance workflow: - -- `IPCAModel` estimates conditional exposures and factor history -- `ExpandingMeanFactorForecaster` forecasts factor premia from that history -- `BetaLambdaMapper` computes asset-level expected returns - -The same pipeline can be used with `PCAModel`, `RPPCAModel`, and `CAEModel`. - -## 2. Weight-Native Stochastic Discount Factor - -The stochastic discount factor family is different. It does not expose a `beta × lambda` forecast path as the native object. - -```python -import numpy as np +Expected output: -from ml4t.models import ( - CrossSectionBatch, - StochasticDiscountFactorConfig, - StochasticDiscountFactorModel, -) - -batch = CrossSectionBatch( - characteristics=np.random.randn(48, 300, 16), - returns=np.random.randn(48, 300), - context_features=np.random.randn(48, 8), - timestamps=tuple(range(48)), -) - -config = StochasticDiscountFactorConfig( - checkpoint_epochs=(256, 512, 768, 1024), - default_checkpoint=("conditional", 1024), -) -model = StochasticDiscountFactorModel(config) -fit_summary = model.fit(batch) -state = model.extract(batch) - -print(fit_summary.best_epoch) -print(state.asset_weights.shape) # (48, 300) -print(state.sdf_values.shape) # (48,) +```text +forecast shape: (2, 6) ``` -Use this family when you want: +Each row is a future timestamp and each column is an asset in `asset_ids` order. These are model +forecasts from synthetic data, not evidence of trading performance. The forecaster uses the fitted +factor history; the future batch supplies dates and persistent asset identity without future returns. -- no-arbitrage training -- weight-native outputs -- phase-aware checkpointed estimation +The [latent-factor task guide](../user-guide/latent-factor-pipelines.md) explains the stages and +input requirements. See the exact +[`LatentFactorForecastPipeline` API](../api/index.md#pipelines) and the +[runnable repository example](https://github.com/ml4t/models/blob/main/examples/latent_factor_pipeline.py). -## 3. Direct Asset Prediction With SAE - -`SAEModel` is treated as a direct predictor in this library. - -```python -import numpy as np - -from ml4t.models import CrossSectionBatch, SAEConfig, SAEModel - -batch = CrossSectionBatch( - characteristics=np.random.randn(24, 200, 20), - returns=np.random.randn(24, 200), - timestamps=tuple(range(24)), -) - -model = SAEModel(SAEConfig(n_epochs=20, checkpoint_interval=5)) -fit_summary = model.fit(batch, validation_batch=batch) -signals = model.predict(batch) - -print(fit_summary.best_epoch) -print(signals.signal_values.shape) -``` - -## 4. End-To-End Portfolio Learning - -Portfolio models learn weights directly. - -```python -import numpy as np - -from ml4t.models import LSTMPortfolioConfig, LSTMPortfolioModel, PortfolioSequenceBatch - -batch = PortfolioSequenceBatch( - features=np.random.randn(8, 63, 30, 10), - returns=np.random.randn(8, 63, 30), - timestamps=tuple(range(63)), - asset_ids=tuple(f"asset_{i}" for i in range(30)), -) - -model = LSTMPortfolioModel( - LSTMPortfolioConfig(max_iters=20, checkpoint_every=5, default_checkpoint=20) -) -model.fit(batch, validation_batch=batch) -portfolio_prediction = model.predict(batch) - -print(portfolio_prediction.weights.shape) -print(portfolio_prediction.checkpoint_step) -``` - -## 5. Export Frames For Backtesting And Diagnostics - -```python -from ml4t.models import ( - backtest_inputs_from_asset_forecast, - predictions_frame_from_asset_forecast, - write_backtest_frames, -) - -frame = predictions_frame_from_asset_forecast(forecast=lf_prediction.asset_forecast) -written = write_backtest_frames("artifacts/run_001", predictions=frame) - -print(written["predictions"]) -``` - -With the integration extra installed, you can also build a `DataFeed` handoff payload: - -```python -inputs = backtest_inputs_from_asset_forecast( - lf_prediction.asset_forecast, - prices_path="prices.parquet", - timestamp_col="timestamp", - entity_col="asset", - close_col="close", -) -``` +## Other supported tasks -## Next Steps +| Task | Guide | Runnable example | Requirement | +|---|---|---|---| +| RP-PCA, IPCA, and CAE factor forecasts | [Latent-factor models](../user-guide/latent-factor-models.md) | [Bounded variant examples](https://github.com/ml4t/models/blob/main/examples/latent_factor_variants.py) | `deep` for CAE; production neural training can be long | +| SDF weights and optional return mapping | [SDF estimation](../user-guide/stochastic-discount-factor.md) | [CPU smoke example](https://github.com/ml4t/models/blob/main/examples/stochastic_discount_factor.py) | `ml4t-models[deep]`; production training can be long | +| Direct asset signals | [SAE prediction](../user-guide/direct-asset-prediction.md) | [CPU smoke example](https://github.com/ml4t/models/blob/main/examples/direct_asset_prediction.py) | `ml4t-models[deep]`; production training can be long | +| Portfolio weights | [Portfolio learning](../user-guide/portfolio-learning.md) | [Linear CPU example](https://github.com/ml4t/models/blob/main/examples/portfolio_learning.py) and [neural smoke example](https://github.com/ml4t/models/blob/main/examples/portfolio_neural.py) | Base install for the linear model; `deep` for LSTM and DeepPortfolio | +| Long-frame inputs and downstream frames | [Data contracts](../user-guide/data-contracts.md) and [Integration](../user-guide/integration.md) | [Adapter example](https://github.com/ml4t/models/blob/main/examples/integration_handoff.py) | `integration` extra for Parquet or Specs objects | -- [Data Contracts](../user-guide/data-contracts.md) -- [Latent-Factor Pipelines](../user-guide/latent-factor-pipelines.md) -- [Portfolio Learning](../user-guide/portfolio-learning.md) -- [Integration](../user-guide/integration.md) +The [Book Guide](../book-guide/index.md) links to teaching notebooks and identifies where they +implement the methods manually. The [API Reference](../api/index.md) provides signatures and options. diff --git a/docs/index.md b/docs/index.md index 0531bd6..534da8d 100644 --- a/docs/index.md +++ b/docs/index.md @@ -1,6 +1,6 @@ # ML4T Models -Build finance-native latent-factor, stochastic discount factor, direct signal, and portfolio-learning models without collapsing everything into one generic trainer. +Finance-specific models for asset pricing, prediction, and portfolio learning. `ml4t.models` is the modeling layer in the ML4T stack. It packages model families that matter in empirical asset pricing and portfolio construction while keeping the contracts explicit: @@ -8,7 +8,10 @@ Build finance-native latent-factor, stochastic discount factor, direct signal, a - what object it estimates - what must still happen before you have an implementable forecast or tradable weight vector -If you are new to the library, start with the [Quickstart](getting-started/quickstart.md). If you are coming from *Machine Learning for Trading*, the [Book Guide](book-guide/index.md) maps the chapter implementations to the production API. +Start with the [bounded CPU Quickstart](getting-started/quickstart.md) for a complete forecast and +expected result. Use the [task guides](user-guide/index.md) for your own inputs, +the [API Reference](api/index.md) for exact signatures, and the [Book Guide](book-guide/index.md) +for verified public teaching notebooks.
@@ -62,37 +65,10 @@ Many finance models look similar at the tensor level but behave very differently The library reflects those differences instead of hiding them behind one catch-all `fit/predict` story. -## Quick Example - -```python -import numpy as np - -from ml4t.models import ( - BetaLambdaMapper, - CrossSectionBatch, - ExpandingMeanFactorForecaster, - IPCAConfig, - IPCAModel, - LatentFactorForecastPipeline, -) - -batch = CrossSectionBatch( - characteristics=np.random.randn(24, 150, 10), - returns=np.random.randn(24, 150), - timestamps=tuple(range(24)), -) - -pipeline = LatentFactorForecastPipeline( - model=IPCAModel(IPCAConfig(n_factors=3)), - forecaster=ExpandingMeanFactorForecaster(), - mapper=BetaLambdaMapper(), -) -pipeline.fit(batch) -prediction = pipeline.predict(batch) - -print(prediction.state.asset_betas.shape) -print(prediction.asset_forecast.expected_returns.shape) -``` +## First result + +The [Quickstart](getting-started/quickstart.md) runs PCA on a small synthetic panel and checks a +finite forecast with shape `(2, 6)`. It needs only the base package and NumPy. ## Three Core Contracts diff --git a/docs/overrides/assets/stylesheets/ml4t-docs-theme.css b/docs/overrides/assets/stylesheets/ml4t-docs-theme.css index 01dca7d..1267438 100644 --- a/docs/overrides/assets/stylesheets/ml4t-docs-theme.css +++ b/docs/overrides/assets/stylesheets/ml4t-docs-theme.css @@ -385,6 +385,18 @@ body { /* ── Responsive ────────────────────────────────────── */ @media screen and (max-width: 76.1875em) { + .md-header { + border-bottom: 1px solid var(--ml4t-silver-muted); + color: var(--ml4t-navy); + } + + .md-header__inner { + display: flex; + height: 2.4rem; + padding: 0 0.8rem; + overflow: visible; + } + .md-content { max-width: 100%; } diff --git a/docs/reference/api-stability.md b/docs/reference/api-stability.md index af20a96..e4fbc61 100644 --- a/docs/reference/api-stability.md +++ b/docs/reference/api-stability.md @@ -1,7 +1,6 @@ # API Stability -This page defines the public surface that is intended to be stable for the -`0.1` beta series. +This page defines the public surface for the `0.1` stable line. ## Stable Import Surface @@ -20,7 +19,7 @@ and integration helpers used throughout the documentation. ## Stable Families -The beta API is organized around four model families: +The public API is organized around four model families: | Family | Stable models | |---|---| @@ -67,6 +66,6 @@ contracts before introducing a new public type. ## Deferred Surface -The beta release does not freeze internals under `ml4t.models._internal`. +Stable releases do not freeze internals under `ml4t.models._internal`. Functions and modules prefixed with `_` are implementation details and may change between `0.1` releases. diff --git a/docs/user-guide/data-contracts.md b/docs/user-guide/data-contracts.md index a9c4f0a..e07543b 100644 --- a/docs/user-guide/data-contracts.md +++ b/docs/user-guide/data-contracts.md @@ -1,5 +1,16 @@ # Data Contracts +## Verify identities before fitting + +Run the [long-frame adapter example](https://github.com/ml4t/models/blob/main/examples/integration_handoff.py) +with the base install. It checks a two-date, two-asset `PersistentPanelBatch` and preserves +`("A", "B")` as the asset order. With your own frame, check that each date and asset has at most +one row, inspect missing values, and confirm that returns are aligned with the features used to +predict them. See the [typed contracts](../api/index.md#typed-contracts) and +[integration API](../api/index.md#integration) for exact options. The +[Book Guide](../book-guide/index.md) maps method notebooks; it has no exact notebook for these +batch constructors. + The library uses three primary batch contracts because the underlying finance problems are not all the same. ## PersistentPanelBatch diff --git a/docs/user-guide/direct-asset-prediction.md b/docs/user-guide/direct-asset-prediction.md index a3e1f1b..2d280c8 100644 --- a/docs/user-guide/direct-asset-prediction.md +++ b/docs/user-guide/direct-asset-prediction.md @@ -1,5 +1,18 @@ # Direct Asset Prediction +## Run and verify + +Install `ml4t-models[deep]` and run the +[small CPU SAE example](https://github.com/ml4t/models/blob/main/examples/direct_asset_prediction.py). +It checks that fitting converges and that the signal matrix has one value per date and asset. +The two-epoch run verifies the contract, not predictive skill. Keep validation and test dates +separate for a real study. The [API reference](../api/index.md) describes `SAEConfig`, +`SAEModel`, and `AssetSignalResult`. + +The [supervised autoencoder notebook](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/14_latent_factors/08_supervised_autoencoder.ipynb) +teaches a manual implementation with book data and longer PyTorch training; it does not call +`SAEModel`. + This family covers models that predict asset-level signals directly rather than first estimating latent structure. diff --git a/docs/user-guide/index.md b/docs/user-guide/index.md index 24f775a..5d4581e 100644 --- a/docs/user-guide/index.md +++ b/docs/user-guide/index.md @@ -92,6 +92,15 @@ These models optimize allocation decisions directly rather than first estimating - Evaluation belongs in `ml4t-diagnostic`, not in this library. - Execution belongs in `ml4t-backtest`, not in this library. +## Supported and unsupported boundaries + +The four model families above are supported public workflows. No other model family is designated +experimental in the public API. Modules under `ml4t.models._internal` can change between releases; +use the [public API reference](../api/index.md) for supported imports. This package does not fetch +market data, compute diagnostic scores, or simulate trades. The [Integration guide](integration.md) +describes the frames it can hand to those other steps. Neural models require the `deep` extra; +their small CPU examples verify contracts, not trained investment performance. + ## A Good Reading Strategy If you want the economic logic first: diff --git a/docs/user-guide/integration.md b/docs/user-guide/integration.md index aef4fd9..5b393b7 100644 --- a/docs/user-guide/integration.md +++ b/docs/user-guide/integration.md @@ -1,5 +1,19 @@ # Integration +## Run and verify a handoff + +The [CPU adapter example](https://github.com/ml4t/models/blob/main/examples/integration_handoff.py) +creates a two-date panel and converts a future `AssetForecastResult` into two prediction rows. +Check `timestamp`, `asset`, and `prediction_value` before passing the frame downstream. +Parquet writing and Specs objects need `ml4t-models[integration]`; `ml4t-backtest` and +`ml4t-diagnostic` are separate packages. A frame conversion alone does not run a backtest or +compute IC. Exact signatures are in the [integration API](../api/index.md#integration). + +The book's [case-study library bridge](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/case_studies/utils/latent_factors/library_bridge.py) +calls `ml4t.models` with case-study data and registry requirements. The +[case-study insights notebook](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/14_latent_factors/09_case_study_insights.ipynb) +illustrates related analysis of stored results; it is not an API example for these adapters. + `ml4t-models` integrates with the rest of the ML4T stack at boundaries. It does not try to absorb execution or evaluation logic. ## Boundary Design diff --git a/docs/user-guide/latent-factor-models.md b/docs/user-guide/latent-factor-models.md index d70be6c..fd9d999 100644 --- a/docs/user-guide/latent-factor-models.md +++ b/docs/user-guide/latent-factor-models.md @@ -1,5 +1,17 @@ # Latent-Factor Models +For a bounded first result, run the [PCA quickstart](../getting-started/quickstart.md) and verify +the forecast shape and finite values. The [variant example](https://github.com/ml4t/models/blob/main/examples/latent_factor_variants.py) +also checks one-date forecasts from RP-PCA, IPCA, and CAE. `PCAModel` and `RPPCAModel` need persistent asset identities; +`IPCAModel` and `CAEModel` need dated characteristics. CAE requires `ml4t-models[deep]` and its +full training cost depends on epochs and data size. The [API reference](../api/index.md) lists +the released classes and configs. + +The book [RP-PCA](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/14_latent_factors/05_rp_pca.ipynb) +and [CAE](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/14_latent_factors/06_conditional_autoencoder.ipynb) +notebooks teach their methods manually; they do not call these model classes. See the +[Book Guide](../book-guide/index.md) for the verified revision and prerequisites. + This guide covers the structural latent-factor family: - `PCAModel` diff --git a/docs/user-guide/latent-factor-pipelines.md b/docs/user-guide/latent-factor-pipelines.md index 67b2081..e0c0319 100644 --- a/docs/user-guide/latent-factor-pipelines.md +++ b/docs/user-guide/latent-factor-pipelines.md @@ -1,5 +1,19 @@ # Latent-Factor Pipelines +## Run and verify a forecast + +Start with the [complete PCA workflow](../getting-started/quickstart.md). It fits on 12 observed +periods, predicts at two future dates, and asserts a finite `(2, 6)` asset forecast. Keep the +same asset IDs and order in the future `PersistentPanelBatch`. Use a dated +`CrossSectionBatch` with characteristics for IPCA or CAE, and separate training observations +from the dates being forecast. Fit the structural model and forecaster on training data before +mapping future exposures; fitting on evaluation returns leaks future information. + +The [pipeline API](../api/index.md#pipelines) gives exact signatures and config options. The +[IPCA teaching notebook](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/14_latent_factors/04_ipca.ipynb) +implements the method manually and does not call this library. Its data and runtime differ from +the synthetic CPU example. + The core latent-factor abstraction in `ml4t-models` is: ```text diff --git a/docs/user-guide/portfolio-learning.md b/docs/user-guide/portfolio-learning.md index ada55b8..23aa145 100644 --- a/docs/user-guide/portfolio-learning.md +++ b/docs/user-guide/portfolio-learning.md @@ -1,5 +1,24 @@ # Portfolio Learning +## Run and verify weights + +Run the [linear CPU example](https://github.com/ml4t/models/blob/main/examples/portfolio_learning.py) +with the base install. It verifies `(windows, periods, assets)` output and that postprocessed +gross exposure stays at or below `0.8`. The sequence batch must retain the same asset IDs and +time order as the features and returns. LSTM and DeepPortfolio need `ml4t-models[deep]`; their +production training may need an accelerator and considerably more time. A small CPU smoke run +can check shapes and constraints, but not investment performance. Consult the +[API reference](../api/index.md) for each config and result contract. + +The [neural smoke example](https://github.com/ml4t/models/blob/main/examples/portfolio_neural.py) +checks bounded LSTM and DeepPortfolio fits on CPU with two optimization steps. Use a separate +validation period and more training for any research comparison. + +The book's [VLSTM](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/17_portfolio_construction/12_vlstm_portfolio.ipynb) +and [DeePM](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/17_portfolio_construction/13_deepm_regime_robust.ipynb) +notebooks teach related manual implementations. They do not call this library and are not exact +equivalents of these classes. + Portfolio models in `ml4t-models` learn weights directly. They do not first estimate expected returns and then call a separate optimizer unless you diff --git a/docs/user-guide/stochastic-discount-factor.md b/docs/user-guide/stochastic-discount-factor.md index 13197ea..a65a181 100644 --- a/docs/user-guide/stochastic-discount-factor.md +++ b/docs/user-guide/stochastic-discount-factor.md @@ -1,5 +1,19 @@ # Stochastic Discount Factor +## Run and verify + +Install `ml4t-models[deep]`, then run the +[bounded CPU SDF example](https://github.com/ml4t/models/blob/main/examples/stochastic_discount_factor.py). +It uses six dated cross-sections with observed returns, checks an `(6, 5)` weight matrix, and +checks the optional return projection. It uses only a few training epochs as a contract smoke +check; that is not enough for a research result. Use a separate validation period and explicit +checkpoints for longer training. The [API reference](../api/index.md) has exact configuration and +state signatures. + +The [book's SDF notebook](https://github.com/stefan-jansen/machine-learning-for-trading/blob/d2edec54b1c7a6a9d7a97d8129eb05db4491e1eb/14_latent_factors/07_stochastic_discount_factor.ipynb) +teaches the adversarial method manually and needs book data and longer PyTorch training. It does +not call `StochasticDiscountFactorModel`. + `StochasticDiscountFactorModel` is a separate model family because the native object is not a latent factor with a premium forecast. The native object is a weight-based pricing-kernel proxy. diff --git a/examples/integration_handoff.py b/examples/integration_handoff.py new file mode 100644 index 0000000..d964c9b --- /dev/null +++ b/examples/integration_handoff.py @@ -0,0 +1,28 @@ +from __future__ import annotations + +import numpy as np + +from ml4t.models import ( + AssetForecastResult, + persistent_panel_batch_from_long_frame, + predictions_frame_from_asset_forecast, +) + +frame = { + "timestamp": np.array(["2024-01", "2024-01", "2024-02", "2024-02"]), + "asset": np.array(["A", "B", "A", "B"]), + "return": np.array([0.01, 0.02, -0.01, 0.03]), +} +batch = persistent_panel_batch_from_long_frame(frame, return_col="return") +assert batch.returns is not None and batch.returns.shape == (2, 2) +assert batch.asset_ids == ("A", "B") + +forecast = AssetForecastResult( + expected_returns=np.array([[0.02, 0.01]]), + timestamps=("2024-03",), + asset_ids=batch.asset_ids, +) +predictions = predictions_frame_from_asset_forecast(forecast) +assert predictions.columns == ("timestamp", "asset", "prediction_value") +assert len(predictions.rows) == 2 +print(f"panel shape: {batch.returns.shape}; prediction rows: {len(predictions.rows)}") diff --git a/examples/latent_factor_variants.py b/examples/latent_factor_variants.py new file mode 100644 index 0000000..6ac423a --- /dev/null +++ b/examples/latent_factor_variants.py @@ -0,0 +1,78 @@ +from __future__ import annotations + +import numpy as np + +from ml4t.models import ( + BetaLambdaMapper, + CAEConfig, + CAEModel, + CrossSectionBatch, + ExpandingMeanFactorForecaster, + IPCAConfig, + IPCAModel, + LatentFactorForecastPipeline, + PersistentPanelBatch, + RPPCAConfig, + RPPCAModel, +) + +rng = np.random.default_rng(13) +asset_ids = tuple(f"A{i}" for i in range(7)) +returns = rng.normal(scale=0.02, size=(8, 7)) +panel = PersistentPanelBatch( + returns=returns, + timestamps=tuple(f"2024-{i:02d}" for i in range(1, 9)), + asset_ids=asset_ids, +) +future_panel = PersistentPanelBatch(timestamps=("2024-09",), asset_ids=asset_ids) + +rp_pipeline = LatentFactorForecastPipeline( + model=RPPCAModel(RPPCAConfig(n_factors=1, gamma=1.0)), + forecaster=ExpandingMeanFactorForecaster(), + mapper=BetaLambdaMapper(), +) +rp_pipeline.fit(panel) +rp_forecast = rp_pipeline.predict(future_panel).asset_forecast.expected_returns +assert rp_forecast.shape == (1, 7) and np.isfinite(rp_forecast).all() + +characteristics = rng.normal(size=(8, 7, 3)) +returns = 0.03 * characteristics[..., 0] - 0.02 * characteristics[..., 1] +returns += 0.01 * rng.normal(size=returns.shape) +cross_sections = CrossSectionBatch( + characteristics=characteristics, + returns=returns, + timestamps=panel.timestamps, + asset_ids=asset_ids, +) +future_cross_section = CrossSectionBatch( + characteristics=rng.normal(size=(1, 7, 3)), + timestamps=("2024-09",), + asset_ids=asset_ids, +) + +for label, model in ( + ("IPCA", IPCAModel(IPCAConfig(n_factors=1, max_iter=30))), + ( + "CAE", + CAEModel( + CAEConfig( + n_factors=1, + hidden_units=(4,), + n_epochs=4, + checkpoint_interval=2, + batch_size=8, + ) + ), + ), +): + pipeline = LatentFactorForecastPipeline( + model=model, + forecaster=ExpandingMeanFactorForecaster(), + mapper=BetaLambdaMapper(), + ) + pipeline.fit(cross_sections) + forecast = pipeline.predict(future_cross_section).asset_forecast.expected_returns + assert forecast.shape == (1, 7) and np.isfinite(forecast).all() + print(f"{label} forecast shape: {forecast.shape}") + +print(f"RP-PCA forecast shape: {rp_forecast.shape}") diff --git a/examples/portfolio_neural.py b/examples/portfolio_neural.py new file mode 100644 index 0000000..d217aa5 --- /dev/null +++ b/examples/portfolio_neural.py @@ -0,0 +1,59 @@ +from __future__ import annotations + +import numpy as np + +from ml4t.models import ( + DeepPortfolioConfig, + DeepPortfolioModel, + LSTMPortfolioConfig, + LSTMPortfolioModel, + PortfolioSequenceBatch, +) + +rng = np.random.default_rng(11) +features = rng.normal(size=(3, 4, 3, 4)) +returns = 0.03 * features[..., 0] - 0.01 * features[..., 1] +batch = PortfolioSequenceBatch( + features=features, + returns=returns, + vol_scale=np.ones((3, 4, 3)), + mask=np.ones((3, 4, 3), dtype=bool), + asset_ids=("A", "B", "C"), +) + +common = { + "dropout": 0.0, + "batch_size": 2, + "max_iters": 2, + "eval_every": 1, + "checkpoint_every": 1, + "default_checkpoint": 2, + "seed": 7, + "device": "cpu", +} +for label, model in ( + ( + "LSTM", + LSTMPortfolioModel(LSTMPortfolioConfig(hidden_size=8, n_layers=1, **common)), + ), + ( + "DeepPortfolio", + DeepPortfolioModel( + DeepPortfolioConfig( + d_model=8, + n_heads=1, + lstm_layers=1, + temporal_mha_layers=1, + cross_attention_heads=1, + macro_gnn_heads=1, + **common, + ) + ), + ), +): + fit = model.fit(batch, validation_batch=batch) + weights = model.predict(batch) + assert fit.converged + assert weights.weights.shape == (3, 4, 3) + assert np.isfinite(weights.weights).all() + print(f"{label} weights shape: {weights.weights.shape}") diff --git a/tests/test_examples.py b/tests/test_examples.py index 64e4c1d..839f7a6 100644 --- a/tests/test_examples.py +++ b/tests/test_examples.py @@ -1,25 +1,19 @@ from __future__ import annotations -import ast import re import runpy from pathlib import Path import pytest -from ml4t.models import StochasticDiscountFactorConfig - EXAMPLES = ( "latent_factor_pipeline.py", + "latent_factor_variants.py", "stochastic_discount_factor.py", "direct_asset_prediction.py", "portfolio_learning.py", -) - -DOCUMENTS = ( - "README.md", - "docs/getting-started/quickstart.md", - "docs/user-guide/stochastic-discount-factor.md", + "portfolio_neural.py", + "integration_handoff.py", ) @@ -28,31 +22,10 @@ def test_examples_execute(example: str) -> None: runpy.run_path(str(Path(__file__).parents[1] / "examples" / example)) -@pytest.mark.parametrize( - "document", - DOCUMENTS, -) -def test_documented_sdf_checkpoint_configs_are_valid(document: str) -> None: +def test_documented_quickstart_executes(capsys: pytest.CaptureFixture[str]) -> None: root = Path(__file__).parents[1] - content = (root / document).read_text(encoding="utf-8") - code_blocks = re.findall(r"```python\r?\n(.*?)```", content, re.DOTALL) - checked = 0 - for code_block in code_blocks: - tree = ast.parse(code_block) - for call in (node for node in ast.walk(tree) if isinstance(node, ast.Call)): - if ( - not isinstance(call.func, ast.Name) - or call.func.id != "StochasticDiscountFactorConfig" - ): - continue - keywords = { - keyword.arg: ast.literal_eval(keyword.value) - for keyword in call.keywords - if keyword.arg - in {"checkpoint_epochs", "default_checkpoint", "n_epochs_unc", "n_epochs_cond"} - } - if not keywords: - continue - StochasticDiscountFactorConfig(**keywords) - checked += 1 - assert checked > 0 + content = (root / "docs/getting-started/quickstart.md").read_text(encoding="utf-8") + source = re.search(r"```python\r?\n(.*?)```", content, re.DOTALL) + assert source is not None + exec(compile(source.group(1), "quickstart.md", "exec"), {"__name__": "__main__"}) + assert capsys.readouterr().out == "forecast shape: (2, 6)\n" diff --git a/tests/test_release_workflow.py b/tests/test_release_workflow.py index f7e97f4..89ee054 100644 --- a/tests/test_release_workflow.py +++ b/tests/test_release_workflow.py @@ -149,8 +149,6 @@ def run(command: list[str], *, check: bool) -> subprocess.CompletedProcess[bytes def test_readme_quick_start_is_an_executable_installed_package_contract() -> None: readme = ROOT / "README.md" - source = readme_smoke.extract_quick_start(readme.read_text(encoding="utf-8")) - assert "IPCAModel" in source readme_smoke.run(readme, __version__)