From 736b70fb6600c1baa6fbfb32690f2d0f8db135c2 Mon Sep 17 00:00:00 2001 From: Stefan Jansen Date: Sat, 10 Oct 2026 04:50:49 -0400 Subject: [PATCH 1/5] docs: enforce published link integrity --- .github/workflows/docs.yml | 4 +- .github/workflows/release.yml | 1 + mkdocs.yml | 2 +- scripts/verify_documentation_identity.py | 129 +++++++++++++++++++++++ scripts/verify_readme_links.py | 90 ++++++++++++++++ tests/test_release_pipeline.py | 65 +++++++++++- 6 files changed, 288 insertions(+), 3 deletions(-) create mode 100644 scripts/verify_readme_links.py diff --git a/.github/workflows/docs.yml b/.github/workflows/docs.yml index ed57e27e..c10b665b 100644 --- a/.github/workflows/docs.yml +++ b/.github/workflows/docs.yml @@ -43,7 +43,9 @@ jobs: echo "ML4T_DOCS_VERSION=$(uv run python -c 'from ml4t.data import __version__; print(__version__)')" >> "$GITHUB_ENV" echo "ML4T_DOCS_COMMIT=${GITHUB_SHA}" >> "$GITHUB_ENV" - name: Build documentation strictly - run: uv run mkdocs build --strict + run: | + uv run python scripts/verify_readme_links.py + uv run mkdocs build --strict - name: Verify rendered identity run: >- uv run python scripts/verify_documentation_identity.py diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index e3a45bb7..349410b8 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -113,6 +113,7 @@ jobs: ML4T_DOCS_COMMIT: ${{ needs.preflight.outputs.commit }} ML4T_DOCS_VERSION: ${{ needs.preflight.outputs.version }} run: | + uv run python scripts/verify_readme_links.py uv run mkdocs build --strict uv run python scripts/verify_documentation_identity.py \ --site site \ diff --git a/mkdocs.yml b/mkdocs.yml index c722079b..19821e0a 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -38,7 +38,7 @@ theme: repo: fontawesome/brands/github logo: assets/logo.svg - favicon: assets/favicon.ico + favicon: assets/images/favicon.png # Custom CSS for Picasso palette extra_css: diff --git a/scripts/verify_documentation_identity.py b/scripts/verify_documentation_identity.py index 51221223..825214bc 100644 --- a/scripts/verify_documentation_identity.py +++ b/scripts/verify_documentation_identity.py @@ -5,6 +5,7 @@ import argparse import time import urllib.error +import urllib.parse import urllib.request from html.parser import HTMLParser from pathlib import Path @@ -27,6 +28,28 @@ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None self.values[name] = content +class _ReferenceParser(HTMLParser): + def __init__(self) -> None: + super().__init__() + self.references: list[str] = [] + self.anchors: set[str] = set() + + def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: + values = dict(attrs) + anchor = values.get("id") or (values.get("name") if tag == "a" else None) + if anchor: + self.anchors.add(anchor) + attribute = { + "a": "href", + "img": "src", + "link": "href", + "script": "src", + }.get(tag) + reference = values.get(attribute) if attribute else None + if reference: + self.references.append(reference) + + def identity_failures( html: str, *, @@ -66,6 +89,109 @@ def _read_url(url: str) -> str: return response.read().decode("utf-8") +def _local_destination( + site: Path, + page: Path, + reference: str, + site_path: str, +) -> tuple[Path, str] | None: + parsed = urllib.parse.urlsplit(reference) + if parsed.path.startswith("/"): + if not parsed.path.startswith(site_path): + return None + destination = site / urllib.parse.unquote(parsed.path.removeprefix(site_path)) + else: + destination = page.parent / urllib.parse.unquote(parsed.path) + if not parsed.path or parsed.path.endswith("/") or destination.is_dir(): + destination /= "index.html" + return destination.resolve(), urllib.parse.unquote(parsed.fragment) + + +def site_link_failures(site: Path, site_path: str = "/docs/data/") -> list[str]: + """Return broken internal links, assets, and fragments in a rendered site.""" + failures: list[str] = [] + site_root = site.resolve() + anchor_cache: dict[Path, set[str]] = {} + for page in _site_pages(site): + parser = _ReferenceParser() + parser.feed(page.read_text(encoding="utf-8")) + anchor_cache[page.resolve()] = parser.anchors + for reference in parser.references: + parsed = urllib.parse.urlsplit(reference) + if parsed.scheme or parsed.netloc or reference.startswith(("mailto:", "tel:", "data:")): + continue + resolved = _local_destination(site, page, reference, site_path) + if resolved is None: + continue + destination, fragment = resolved + if destination != site_root and site_root not in destination.parents: + failures.append(f"{page}: {reference!r} escapes the rendered site") + continue + if not destination.exists(): + failures.append(f"{page}: {reference!r} does not resolve") + continue + if fragment and destination.suffix == ".html": + anchors = anchor_cache.get(destination) + if anchors is None: + target_parser = _ReferenceParser() + target_parser.feed(destination.read_text(encoding="utf-8")) + anchors = target_parser.anchors + anchor_cache[destination] = anchors + if fragment not in anchors: + failures.append(f"{page}: {reference!r} has no matching anchor") + return failures + + +def page_reference_urls(html: str, *, page_url: str, site_root: str) -> list[str]: + """Return distinct same-site documentation references from one deployed page.""" + parser = _ReferenceParser() + parser.feed(html) + root = urllib.parse.urlsplit(site_root) + references: list[str] = [] + for reference in parser.references: + absolute = urllib.parse.urljoin(page_url, reference) + parsed = urllib.parse.urlsplit(absolute) + if parsed.scheme not in {"http", "https"} or parsed.netloc != root.netloc: + continue + if not parsed.path.startswith(root.path): + continue + normalized = urllib.parse.urlunsplit( + (parsed.scheme, parsed.netloc, parsed.path, parsed.query, "") + ) + if normalized not in references: + references.append(normalized) + return references + + +def _probe_url(url: str) -> None: + request = urllib.request.Request(url, headers={"User-Agent": "ml4t-release-verifier"}) + with urllib.request.urlopen(request, timeout=30) as response: # noqa: S310 + if response.status != 200: + raise ValueError(f"HTTP {response.status}") + response.read(1) + + +def deployed_link_failures(urls: list[str]) -> list[str]: + """Return unavailable navigation, asset, and internal-link targets.""" + site_root = urls[0] + references: list[str] = [] + failures: list[str] = [] + for url in urls: + try: + html = _read_url(url) + for reference in page_reference_urls(html, page_url=url, site_root=site_root): + if reference not in references: + references.append(reference) + except (OSError, UnicodeDecodeError, urllib.error.URLError, ValueError) as error: + failures.append(f"{url}: {error}") + for reference in references: + try: + _probe_url(reference) + except (OSError, urllib.error.URLError, ValueError) as error: + failures.append(f"{reference}: {error}") + return failures + + def deployed_identity_failures( urls: list[str], *, @@ -135,6 +261,7 @@ def main() -> int: source=str(page), ) ) + failures.extend(site_link_failures(args.site)) except (OSError, UnicodeDecodeError, urllib.error.URLError, ValueError) as error: failures.append(str(error)) if args.url is not None: @@ -148,6 +275,8 @@ def main() -> int: delay=args.delay, ) ) + if not failures: + failures.extend(deployed_link_failures(args.url)) for failure in failures: print(failure) diff --git a/scripts/verify_readme_links.py b/scripts/verify_readme_links.py new file mode 100644 index 00000000..d2c5f843 --- /dev/null +++ b/scripts/verify_readme_links.py @@ -0,0 +1,90 @@ +"""Verify every local and external link published in README.md.""" + +from __future__ import annotations + +import argparse +import re +import time +import urllib.error +import urllib.request +from collections.abc import Callable +from pathlib import Path +from urllib.parse import unquote, urlsplit + +MARKDOWN_LINK = re.compile(r"!?\[[^\]]*\]\((?P<[^>]+>|[^\s)]+)") +REMOTE_SCHEMES = {"http", "https"} +IGNORED_SCHEMES = {"mailto", "tel"} + + +def extract_markdown_targets(markdown: str) -> list[str]: + """Return distinct Markdown link and image targets in source order.""" + targets: list[str] = [] + for match in MARKDOWN_LINK.finditer(markdown): + target = match.group("target").removeprefix("<").removesuffix(">") + if target not in targets: + targets.append(target) + return targets + + +def _probe_url(url: str, attempts: int = 3, delay: float = 1.0) -> None: + request = urllib.request.Request( + url, + headers={ + "Accept": "text/html,application/xhtml+xml,image/*,*/*;q=0.8", + "User-Agent": "ml4t-documentation-link-verifier", + }, + ) + error: Exception | None = None + for attempt in range(attempts): + try: + with urllib.request.urlopen(request, timeout=30) as response: # noqa: S310 + if response.status >= 400: + raise ValueError(f"HTTP {response.status}") + response.read(1) + return + except (OSError, urllib.error.URLError, ValueError) as caught: + error = caught + if attempt + 1 < attempts: + time.sleep(delay) + raise ValueError(str(error)) + + +def readme_link_failures( + readme: Path, + probe_url: Callable[[str], None] = _probe_url, +) -> list[str]: + """Return missing local targets and unavailable external URLs.""" + failures: list[str] = [] + for target in extract_markdown_targets(readme.read_text(encoding="utf-8")): + parsed = urlsplit(target) + if parsed.scheme in REMOTE_SCHEMES: + try: + probe_url(target) + except (OSError, urllib.error.URLError, ValueError) as error: + failures.append(f"{target}: {error}") + continue + if parsed.scheme in IGNORED_SCHEMES or target.startswith("#"): + continue + if parsed.scheme: + failures.append(f"{target}: unsupported link scheme {parsed.scheme!r}") + continue + destination = readme.parent / unquote(parsed.path) + if not destination.exists(): + failures.append(f"{target}: local target does not exist") + return failures + + +def main() -> int: + """Validate the configured README and print one line per failure.""" + parser = argparse.ArgumentParser() + parser.add_argument("readme", type=Path, nargs="?", default=Path("README.md")) + args = parser.parse_args() + + failures = readme_link_failures(args.readme) + for failure in failures: + print(failure) + return int(bool(failures)) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/tests/test_release_pipeline.py b/tests/test_release_pipeline.py index c301b095..053f133a 100644 --- a/tests/test_release_pipeline.py +++ b/tests/test_release_pipeline.py @@ -24,8 +24,13 @@ metadata_failures, ) from scripts.run_readme_quickstart import extract_quick_start -from scripts.verify_documentation_identity import identity_failures +from scripts.verify_documentation_identity import ( + identity_failures, + page_reference_urls, + site_link_failures, +) from scripts.verify_published_release import published_release_failures +from scripts.verify_readme_links import extract_markdown_targets, readme_link_failures COMMIT = "a" * 40 TREE = "b" * 40 @@ -179,6 +184,64 @@ def test_documentation_identity_and_quickstart_are_executable_contracts() -> Non assert extract_quick_start("## Quick start\n\n```python\nvalue = 1\n```\n") == "value = 1\n" +def test_rendered_documentation_rejects_broken_links_assets_and_fragments(tmp_path: Path) -> None: + site = tmp_path / "site" + (site / "guide").mkdir(parents=True) + (site / "assets").mkdir() + (site / "assets/app.js").write_text("", encoding="utf-8") + (site / "guide/index.html").write_text('

Guide

', encoding="utf-8") + (site / "index.html").write_text( + 'Guide' + '', + encoding="utf-8", + ) + + failures = site_link_failures(site) + + assert any("no matching anchor" in failure for failure in failures) + assert any("does not resolve" in failure for failure in failures) + + +def test_deployed_documentation_discovers_only_same_route_references() -> None: + html = ( + 'Guide' + 'Other libraryExternal' + ) + + assert page_reference_urls( + html, + page_url="https://www.ml4trading.io/docs/data/", + site_root="https://www.ml4trading.io/docs/data/", + ) == [ + "https://www.ml4trading.io/docs/data/guide/", + "https://www.ml4trading.io/docs/data/assets/logo.svg", + ] + + +def test_readme_link_check_rejects_missing_local_and_unavailable_remote_targets( + tmp_path: Path, +) -> None: + readme = tmp_path / "README.md" + (tmp_path / "LICENSE").write_text("MIT\n", encoding="utf-8") + readme.write_text( + "[license](LICENSE) [missing](missing.md) [remote](https://example.invalid/)\n", + encoding="utf-8", + ) + + failures = readme_link_failures( + readme, + probe_url=lambda url: (_ for _ in ()).throw(ValueError(f"unavailable: {url}")), + ) + + assert extract_markdown_targets(readme.read_text(encoding="utf-8")) == [ + "LICENSE", + "missing.md", + "https://example.invalid/", + ] + assert any("local target does not exist" in failure for failure in failures) + assert any("unavailable" in failure for failure in failures) + + def test_external_workflow_actions_use_full_commit_pins() -> None: action = re.compile(r"^\s*uses:\s*(?!\./)([^\s#]+)@([^\s#]+)", re.MULTILINE) failures = [] From b07427e62b1e49f5d004f34df06473477dd0dfe0 Mon Sep 17 00:00:00 2001 From: Stefan Jansen Date: Sat, 10 Oct 2026 04:54:18 -0400 Subject: [PATCH 2/5] docs: replace stale contributor guidance --- docs/contributing/architecture.md | 577 ++-------------- docs/contributing/creating-a-provider.md | 839 ++++------------------- docs/contributing/index.md | 87 +-- docs/contributing/testing.md | 163 ++--- 4 files changed, 291 insertions(+), 1375 deletions(-) diff --git a/docs/contributing/architecture.md b/docs/contributing/architecture.md index 91a3df96..e92cf3b3 100644 --- a/docs/contributing/architecture.md +++ b/docs/contributing/architecture.md @@ -1,530 +1,103 @@ -# Extending ML4T Data - Architecture Guide +# Architecture -This document explains ML4T Data's architecture, design patterns, and extension points for contributors. +`ml4t-data` separates external data access from validation, persistence, and update orchestration. +The separation lets provider adapters vary without changing the canonical data passed to downstream +ML4T libraries. -## Table of Contents +## Package boundaries -- [Architecture Overview](#architecture-overview) -- [Core Abstractions](#core-abstractions) -- [Design Patterns](#design-patterns) -- [Extension Points](#extension-points) -- [Best Practices](#best-practices) +| Area | Responsibility | +|---|---| +| `providers/` | External-service adapters, provider protocols, registry metadata, retry, rate-limit, and HTTP behavior | +| `validation/` and `anomaly/` | Schema, invariant, quality, and anomaly checks | +| `storage/` | Storage protocols, Parquet layouts, metadata, atomic publication, and migrations | +| `futures/` | Futures acquisition, calendars, rolls, and continuous-contract construction | +| `assets/`, `calendar/`, and `sessions/` | Asset identity and trading-session semantics | +| `core/` and `config/` | Shared exceptions, data models, and configuration resolution | +| `data_manager.py`, `update_manager.py`, and `managers/` | Coordination of provider reads, storage writes, and updates | +| `cli/` | Command-line entry points over the same public services | -## Architecture Overview +The package uses a PEP 420 namespace. `src/ml4t/` intentionally has no `__init__.py`. -### High-Level Structure +## Provider contracts -``` -ML4T Data Architecture -├── Providers (Data Acquisition) -│ ├── BaseProvider (Template Method) -│ ├── Rate Limiting (Global) -│ ├── Circuit Breakers -│ └── Retry Logic -│ -├── Storage (Data Persistence) -│ ├── Hive-Partitioned Parquet -│ ├── Metadata Tracking -│ └── Incremental Updates -│ -├── Validation (Data Quality) -│ ├── Schema Validation -│ ├── OHLCV Invariants -│ └── Anomaly Detection -│ -└── Interfaces - ├── Python API - └── CLI -``` +`providers/protocols.py` defines structural interfaces for OHLCV, factor, event, and asynchronous +providers. Code that only needs an interface should depend on the protocol rather than a concrete +adapter. -### Design Philosophy +Most OHLCV adapters subclass `BaseProvider`. Its `fetch_ohlcv()` template owns the shared request +lifecycle: -1. **Separation of Concerns** - Each layer has single responsibility -2. **Template Method Pattern** - Common workflow, provider-specific implementation -3. **Type Safety** - Public interface type hints checked with ty -4. **Performance First** - Polars, not pandas -5. **Production Ready** - Error handling, retry logic, monitoring +1. validate the symbol and date inputs; +2. acquire the provider rate limit; +3. call the adapter's fetch and transformation method; +4. execute the call through the circuit breaker; and +5. normalize and validate the returned frame. -## Core Abstractions +An adapter implements either `_fetch_and_transform_data()` or the pair `_fetch_raw_data()` and +`_transform_data()`. It does not override `fetch_ohlcv()` merely to duplicate rate limiting, +retries, logging, or schema validation. -### 1. BaseProvider +The synchronous request helpers in `providers/base.py` map transport failures and HTTP status codes +to the shared exception hierarchy. Native asynchronous providers use the async session mixin and +must retain the same public error and cleanup behavior. -**Location**: `src/ml4t-data/providers/base.py` +## Canonical OHLCV data -The `BaseProvider` class is the foundation of all data providers. It implements the **Template Method** pattern. +The validation mixin requires these columns in order: -#### Key Features +| Column | Normalized type | Constraint | +|---|---|---| +| `timestamp` | timezone-aware `Datetime` in UTC | No duplicate timestamp and symbol pair | +| `symbol` | `String` | One normalized symbol per single-symbol request | +| `open`, `high`, `low`, `close` | `Float64` | Finite values and valid OHLC relationships | +| `volume` | `Float64` | Finite and non-negative | -```python -class BaseProvider(ABC): - """Base class for all data providers. +Provider-specific columns may follow the canonical columns. A zero-column empty frame is normalized +to the canonical empty schema. Other malformed responses raise `DataValidationError` rather than +passing incomplete data downstream. - Provides: - - Rate limiting (global, per-provider) - - Circuit breaker pattern - - Retry logic with exponential backoff - - HTTP session management - - Structured logging - """ +## Provider discovery - # Class variables (override in subclass) - DEFAULT_RATE_LIMIT: ClassVar[tuple[int, float]] = (60, 60.0) - CIRCUIT_BREAKER_CONFIG: ClassVar[dict[str, Any]] = {...} +`providers/registry.py` is the static source for provider names, import targets, capabilities, +credentials, configuration requirements, and optional extras. It supports discovery without +constructing providers or contacting external services. - # Abstract methods (must implement) - @abstractmethod - def name(self) -> str: - """Return provider name (lowercase).""" +The registry and public imports serve different purposes: - @abstractmethod - def _fetch_raw_data(self, symbol, start, end, frequency) -> Any: - """Fetch raw data from API.""" +- registry metadata answers whether an adapter is advertised and configured; +- `providers/__init__.py` exposes supported provider classes; and +- provider pages document data scope, credentials, optional dependencies, and service limits. - @abstractmethod - def _transform_data(self, raw_data, symbol) -> pl.DataFrame: - """Transform to standard DataFrame.""" -``` +A new adapter normally updates all three surfaces. -#### Template Method +## Storage and publication -The `fetch_ohlcv()` method defines the workflow: +Storage implementations satisfy the protocols in `storage/protocols.py`. Layout and metadata code +define how logical dataset keys map to files and records. Publication must keep data and metadata +consistent under concurrent writes and must reject unsafe paths and symlink escapes. -```python -@circuit_breaker(...) -@retry(...) -def fetch_ohlcv(self, symbol, start, end, frequency="daily"): - """Template method - don't override! +Callers should use the storage protocols or manager layer instead of depending on a backend's +private files. Migrations belong in `storage/migration.py` and require recovery tests for partially +completed work. - Workflow: - 1. Log request - 2. Call _fetch_raw_data() (provider-specific) - 3. Call _transform_data() (provider-specific) - 4. Validate output - 5. Return standardized DataFrame - """ - raw_data = self._fetch_raw_data(symbol, start, end, frequency) - df = self._transform_data(raw_data, symbol) - return df -``` +## Configuration and errors -**Key Insight**: Subclasses implement `_fetch_raw_data()` and `_transform_data()`, but never override `fetch_ohlcv()`. This ensures consistent error handling, logging, and retries across all providers. +Configuration helpers resolve public settings such as `ML4T_DATA_PATH`. Provider credentials and +service-specific settings remain scoped to the adapter that consumes them. Imports must stay usable +without credentials and without unrelated optional dependencies. -### 2. Rate Limiting +Public failures use the exception hierarchy in `core/exceptions.py`. Adapters preserve useful +provider context and exception chaining while classifying authentication, rate-limit, missing-data, +validation, and transient network failures consistently. -**Location**: `src/ml4t-data/utils/global_rate_limit.py` +## Extension points -Rate limiting is **global** (not per-instance) to prevent parallel requests from violating API limits. +- Add an OHLCV source by following [Creating a provider](creating-a-provider.md). +- Implement a protocol directly when `BaseProvider` supplies behavior that does not apply to the + data source. +- Add a storage backend behind the storage protocols and contract tests. +- Add validation without embedding provider-specific acquisition logic in the validation layer. -```python -from ml4t.data.utils.global_rate_limit import global_rate_limit_manager - -# Get rate limiter for provider (shared across instances) -rate_limiter = global_rate_limit_manager.get_rate_limiter( - provider_name="tiingo", - max_calls=1000, - period=86400.0, # 1 day in seconds -) - -# In _fetch_raw_data(): -self.rate_limiter.acquire(blocking=True) # Blocks until token available -response = self.session.get(url) -``` - -**Why Global?** -- Multiple provider instances share the same API quota -- Prevents race conditions in parallel execution -- Respects daily/monthly limits across application - -### 3. Circuit Breaker - -**Location**: `src/ml4t-data/providers/base.py:CircuitBreaker` - -Implements the Circuit Breaker pattern to prevent cascading failures. - -**States:** -- **CLOSED** - Normal operation -- **OPEN** - Too many failures, block all requests -- **HALF_OPEN** - Test if service recovered - -```python -class CircuitBreaker: - """Circuit breaker with exponential backoff. - - Configuration: - - failure_threshold: Number of failures before opening (default: 5) - - reset_timeout: Seconds before attempting reset (default: 300) - - expected_exception: Exceptions that trigger circuit breaker - """ -``` - -**Usage:** -```python -@circuit_breaker( - failure_threshold=3, - reset_timeout=600.0, - expected_exception=NetworkError -) -def fetch_ohlcv(...): - ... -``` - -### 4. Exception Hierarchy - -**Location**: `src/ml4t-data/core/exceptions.py` - -All exceptions inherit from `ProviderError`: - -``` -ProviderError (base) -├── AuthenticationError (401, invalid API key) -├── RateLimitError (429, too many requests) -├── DataNotAvailableError (404, no data for symbol) -├── DataValidationError (invalid parameters or data) -├── NetworkError (connection issues, timeouts) -└── CircuitBreakerOpenError (circuit breaker tripped) -``` - -**Best Practice:** -```python -# Good - specific exception -if response.status_code == 429: - raise RateLimitError( - provider="tiingo", - retry_after=60.0, - message="Rate limit exceeded" - ) - -# Bad - generic exception -if error: - raise Exception("Error occurred") # ❌ Don't do this -``` - -### 5. Standard DataFrame Schema - -All providers must return DataFrames with this schema: - -```python -{ - "timestamp": pl.Datetime, # UTC datetime - "symbol": pl.String, # Uppercase symbol - "open": pl.Float64, # Opening price - "high": pl.Float64, # High price - "low": pl.Float64, # Low price - "close": pl.Float64, # Closing price (adjusted) - "volume": pl.Float64, # Trading volume -} -``` - -**Optional columns:** -- `adj_open`, `adj_high`, `adj_low`, `adj_close` - Unadjusted prices -- `dividend` - Dividend amount -- `split_factor` - Stock split factor - -## Design Patterns - -### 1. Template Method Pattern - -**Used in**: `BaseProvider.fetch_ohlcv()` - -**Purpose**: Define skeleton algorithm, let subclasses fill in steps - -**Example:** -```python -# BaseProvider (template) -def fetch_ohlcv(self, ...): # DON'T override - raw_data = self._fetch_raw_data(...) # Override this - df = self._transform_data(...) # Override this - return df - -# TiingoProvider (concrete) -def _fetch_raw_data(self, ...): - # Tiingo-specific fetching - return tiingo_data - -def _transform_data(self, ...): - # Tiingo-specific transformation - return dataframe -``` - -### 2. Strategy Pattern - -**Used in**: Provider selection - -**Purpose**: Swap providers at runtime - -**Example:** -```python -# Different strategies for different use cases -providers = { - "crypto": CoinGeckoProvider(), - "stocks": TiingoProvider(api_key="..."), - "futures": DataBentoProvider(api_key="..."), -} - -# Select strategy at runtime -provider = providers[asset_class] -data = provider.fetch_ohlcv(symbol, start, end) -``` - -### 3. Factory Pattern - -**Used in**: Storage backend selection - -**Purpose**: Create objects without specifying exact class - -**Example:** -```python -def create_storage(storage_type: str, config: StorageConfig): - if storage_type == "hive": - return HiveStorage(config) - elif storage_type == "flat": - return FlatStorage(config) - else: - raise ValueError(f"Unknown storage type: {storage_type}") -``` - -### 4. Decorator Pattern - -**Used in**: `@circuit_breaker`, `@retry` - -**Purpose**: Add behavior to methods without modifying them - -**Example:** -```python -@circuit_breaker(failure_threshold=5, reset_timeout=300.0) -@retry(stop=stop_after_attempt(3), wait=wait_exponential(...)) -def fetch_ohlcv(self, ...): - # Method gets automatic retry and circuit breaker - ... -``` - -## Extension Points - -### 1. Adding a New Provider - -**See**: [Creating a Provider](creating-a-provider.md) - -**Steps:** -1. Inherit from `BaseProvider` -2. Implement `name()`, `_fetch_raw_data()`, `_transform_data()` -3. Configure rate limiting and circuit breaker -4. Add integration tests -5. Register in `__init__.py` - -**Example:** -```python -class NewProvider(BaseProvider): - DEFAULT_RATE_LIMIT = (10, 60.0) # 10/min - - def name(self) -> str: - return "newprovider" - - def _fetch_raw_data(self, symbol, start, end, frequency): - # Your fetching logic - return raw_data - - def _transform_data(self, raw_data, symbol): - # Your transformation logic - return polars_dataframe -``` - -### 2. Adding a New Storage Backend - -**Location**: `src/ml4t-data/storage/` - -**Interface**: `StorageProtocol` (in `protocols.py`) - -**Required methods:** -- `write(df, symbol, provider)` - Store data -- `read(symbol, start, end, provider)` - Retrieve data -- `list_symbols(provider)` - List available symbols -- `get_metadata(symbol, provider)` - Get metadata - -**Example:** -```python -class S3Storage: - """Store data in AWS S3.""" - - def write(self, df, symbol, provider): - # Upload to S3 - ... - - def read(self, symbol, start, end, provider): - # Download from S3 - ... -``` - -### 3. Adding a New Validator - -**Location**: `src/ml4t-data/validation/` - -**Interface**: `BaseValidator` - -**Example:** -```python -class CustomValidator(BaseValidator): - """Custom validation rules.""" - - def validate(self, df: pl.DataFrame) -> ValidationResult: - errors = [] - - # Check custom rules - if (df["close"] < 0).any(): - errors.append("Negative prices found") - - return ValidationResult( - valid=len(errors) == 0, - errors=errors, - ) -``` - -### 4. Adding a New CLI Command - -**Location**: `src/ml4t-data/cli.py` - -**Framework**: Click - -**Example:** -```python -@cli.command() -@click.argument("symbol") -@click.option("--provider", default="tiingo") -def analyze(symbol, provider): - """Analyze symbol data quality.""" - # Your command logic - ... -``` - -## Best Practices - -### 1. Error Handling - -**Always use ml4t-data exceptions:** -```python -from ml4t.data.core.exceptions import ( - AuthenticationError, - DataNotAvailableError, - RateLimitError, -) - -# Good -if response.status_code == 404: - raise DataNotAvailableError( - provider="myProvider", - symbol=symbol, - start=start, - end=end, - ) - -# Bad -if response.status_code == 404: - raise Exception("Not found") # ❌ -``` - -### 2. Logging - -**Use structured logging:** -```python -self.logger.info( - "Fetching data", - symbol=symbol, - start=start, - end=end, - provider=self.name(), -) - -# Not this: -print(f"Fetching {symbol} from {start} to {end}") # ❌ -``` - -### 3. Type Hints - -**Use type hints everywhere:** -```python -def _transform_data( - self, - raw_data: list[dict[str, Any]], # ✅ - symbol: str, -) -> pl.DataFrame: # ✅ - ... - -# Not this: -def _transform_data(self, raw_data, symbol): # ❌ - ... -``` - -### 4. Testing - -**Test real API calls:** -```python -@pytest.mark.integration -def test_fetch_real_data(provider): - """Test with actual API call.""" - df = provider.fetch_ohlcv("AAPL", "2024-01-01", "2024-01-31") - assert len(df) > 0 - assert all(df["high"] >= df["low"]) -``` - -### 5. Documentation - -**Document everything:** -```python -def _fetch_raw_data(self, symbol, start, end, frequency): - """Fetch raw data from API. - - This method makes HTTP requests to the provider's API - and returns the raw response with minimal processing. - - Args: - symbol: Stock symbol (e.g., "AAPL") - start: Start date in YYYY-MM-DD format - end: End date in YYYY-MM-DD format - frequency: Data frequency (daily, weekly, monthly) - - Returns: - Raw API response (dict or str depending on format) - - Raises: - RateLimitError: If rate limit exceeded - DataNotAvailableError: If no data found - NetworkError: If request fails - """ -``` - -## Architecture Decisions - -### Why Polars Instead of Pandas? - -Providers and storage use the same columnar frame type, schema expressions, and native Parquet -integration. This avoids conversion at the provider-storage boundary. Pandas remains a core -dependency for calendar and vendor interoperability. - -### Why Global Rate Limiting? - -**Problem**: Multiple instances of same provider share API quota -**Solution**: Global rate limiter ensures compliance -**Trade-off**: Slightly more complex, but much safer - -### Why Template Method Pattern? - -**Problem**: Ensure consistent error handling across providers -**Solution**: Template method in base class -**Benefit**: Add circuit breaker/retry once, applies everywhere - -### Why Circuit Breaker? - -**Problem**: Cascading failures when API goes down -**Solution**: Fail fast, recover gracefully -**Benefit**: Better user experience, prevents wasted API calls - -## Contributing Guidelines - -1. **Read first**: [Contribution guide](index.md) -2. **Use templates**: `provider_template/` directory -3. **Follow patterns**: Study existing providers -4. **Test thoroughly**: Integration tests required -5. **Document well**: Code should explain itself - ---- - -**Questions?** Open a GitHub Discussion or issue! +The generated [API reference](../api/index.md) is authoritative for signatures. Source-level +orientation and required verification commands are in the repository's nested `AGENTS.md` files. diff --git a/docs/contributing/creating-a-provider.md b/docs/contributing/creating-a-provider.md index b10d51fd..a2588817 100644 --- a/docs/contributing/creating-a-provider.md +++ b/docs/contributing/creating-a-provider.md @@ -1,775 +1,194 @@ -# Creating a New Provider +# Create an OHLCV provider -This guide walks you through creating a new data provider for ML4T Data. We'll use a fictional "Stooq" provider as an example. - -## Table of Contents - -- [Overview](#overview) -- [Prerequisites](#prerequisites) -- [Step-by-Step Guide](#step-by-step-guide) -- [Testing Your Provider](#testing-your-provider) -- [Documentation](#documentation) -- [Checklist](#checklist) - -## Overview - -A ML4T Data provider is a class that inherits from `BaseProvider` and implements methods to: -1. Fetch raw data from an API -2. Transform it into standardized Polars DataFrames -3. Handle errors, rate limiting, and retries - -**Time estimate:** 2-4 hours for a basic provider +This guide adds a synchronous OHLCV adapter that uses the shared provider lifecycle. Read +[Architecture](architecture.md) first if the source does not return symbol-based OHLCV data. Factor, +event, file-backed, and native asynchronous sources may need a different protocol or an existing +adapter as their model. ## Prerequisites -- Familiarity with Python async/await -- Understanding of the target API -- API key from the data provider (if required) -- Development environment set up (see the [contribution guide](index.md)) +Before writing code, record the source's: -## Step-by-Step Guide +- authentication and optional-dependency requirements; +- supported instruments, frequencies, and history; +- start and end date semantics; +- request quotas and paid-tier boundaries; +- pagination and response schema; +- timestamp timezone; and +- status codes or response fields for authentication, rate limits, missing symbols, and transient + failures. -### Step 1: Research the API +Do not infer these contracts from an SDK method name. Link the provider documentation from the new +provider page and test the observed boundary behavior. -**Before writing code**, understand the API: +## Implement the adapter -```python -# Questions to answer: -# 1. Authentication: API key? OAuth? Bearer token? -# 2. Base URL: https://api.example.com -# 3. Endpoints: /v1/ohlcv, /v1/quote, etc. -# 4. Rate limits: requests per minute/day -# 5. Data format: JSON? CSV? XML? -# 6. Date format: YYYY-MM-DD? Unix timestamp? -# 7. Response structure: nested? flat? -# 8. Error codes: 429 for rate limit? 401 for auth? -``` +Create `src/ml4t/data/providers/.py`. For a JSON service, the two-step form keeps acquisition +and transformation separate: -**Example** - Stooq API research: ```python -# Stooq API (fictional example): -# Base URL: https://stooq.com/q/d/l/ -# Authentication: None (public API) -# Format: CSV -# Parameters: -# s = symbol (e.g., AAPL.US) -# d1 = start date (YYYYMMDD) -# d2 = end date (YYYYMMDD) -# i = frequency (d=daily, w=weekly, m=monthly) -# Rate limit: 10 requests/minute -# Response: CSV with Date,Open,High,Low,Close,Volume -``` - -### Step 2: Create Provider File - -Create `src/ml4t-data/providers/stooq.py`: - -```python -"""Stooq data provider. - -Stooq provides free market data for international exchanges. - -API Documentation: https://stooq.com/ -Features: -- No API key required -- 50+ international exchanges -- Daily/weekly/monthly OHLCV data -- CSV format - -Rate Limits: -- 10 requests per minute -- No daily limit +from __future__ import annotations -Example: - >>> from ml4t.data.providers.stooq import StooqProvider - >>> provider = StooqProvider() - >>> data = provider.fetch_ohlcv("AAPL.US", "2024-01-01", "2024-01-31") - >>> provider.close() -""" - -import os -from datetime import datetime -from typing import Any, ClassVar, Optional +from datetime import UTC, datetime +from typing import Any, ClassVar import polars as pl -import structlog -from ml4t.data.core.exceptions import ( - DataNotAvailableError, - DataValidationError, - NetworkError, -) +from ml4t.data.core.exceptions import DataValidationError from ml4t.data.providers.base import BaseProvider +from ml4t.data.providers.protocols import ProviderCapabilities -logger = structlog.get_logger() - - -class StooqProvider(BaseProvider): - """Stooq data provider for international equities. - - Attributes: - base_url: API base URL - """ - - # Rate limit: 10 requests/minute - DEFAULT_RATE_LIMIT: ClassVar[tuple[int, float]] = (10, 60.0) - - CIRCUIT_BREAKER_CONFIG: ClassVar[dict[str, Any]] = { - "failure_threshold": 3, - "reset_timeout": 300.0, # 5 minutes - } - - # Map frequency strings to Stooq codes - FREQUENCY_MAP: ClassVar[dict[str, str]] = { - "daily": "d", - "1d": "d", - "day": "d", - "weekly": "w", - "1w": "w", - "week": "w", - "monthly": "m", - "1M": "m", - "month": "m", - } - def __init__( - self, - rate_limit: Optional[tuple[int, float]] = None, - session_config: Optional[dict[str, Any]] = None, - circuit_breaker_config: Optional[dict[str, Any]] = None, - ): - """Initialize Stooq provider. - - Args: - rate_limit: Custom rate limiting override - session_config: HTTP session configuration - circuit_breaker_config: Circuit breaker configuration - """ - self.base_url = "https://stooq.com/q/d/l/" - - super().__init__( - rate_limit=rate_limit, - session_config=session_config, - circuit_breaker_config=circuit_breaker_config, - ) +class ExampleProvider(BaseProvider): + DEFAULT_RATE_LIMIT: ClassVar[tuple[int, float]] = (30, 60.0) - self.logger.info("Initialized Stooq provider") + def __init__(self, api_key: str, **session_config: Any) -> None: + self.api_key = api_key + self.base_url = "https://api.example.test/v1" + super().__init__(session_config=session_config) + @property def name(self) -> str: - """Return provider name.""" - return "stooq" + return "example" + + def capabilities(self) -> ProviderCapabilities: + return ProviderCapabilities(requires_api_key=True, rate_limit=self.DEFAULT_RATE_LIMIT) def _fetch_raw_data( self, symbol: str, start: str, end: str, - frequency: str = "daily", - ) -> str: - """Fetch raw CSV data from Stooq API. - - Args: - symbol: Stock symbol with exchange (e.g., "AAPL.US") - start: Start date (YYYY-MM-DD) - end: End date (YYYY-MM-DD) - frequency: Data frequency (daily, weekly, monthly) - - Returns: - Raw CSV string - - Raises: - DataValidationError: If frequency is invalid - NetworkError: If request fails - """ - # 1. Validate frequency - freq_code = self.FREQUENCY_MAP.get(frequency.lower()) - if not freq_code: - raise DataValidationError( - provider="stooq", - message=f"Unsupported frequency '{frequency}'. " - f"Supported: {list(self.FREQUENCY_MAP.keys())}", - field="frequency", - value=frequency, - ) - - # 2. Convert dates to Stooq format (YYYYMMDD) - start_formatted = start.replace("-", "") - end_formatted = end.replace("-", "") - - # 3. Build URL with parameters - params = { - "s": symbol.upper(), - "d1": start_formatted, - "d2": end_formatted, - "i": freq_code, - } - - try: - # 4. Apply rate limiting - self.rate_limiter.acquire(blocking=True) - - # 5. Make HTTP request - response = self.session.get(self.base_url, params=params) - - # 6. Check for errors - if response.status_code != 200: - raise NetworkError( - provider="stooq", - message=f"HTTP {response.status_code}: {response.text}", - ) - - # 7. Return raw data - csv_data = response.text - - # 8. Basic validation - if not csv_data or len(csv_data) < 50: - raise DataNotAvailableError( - provider="stooq", - symbol=symbol, - start=start, - end=end, - frequency=frequency, - ) - - return csv_data - - except (NetworkError, DataNotAvailableError): - raise - except Exception as err: - raise NetworkError( - provider="stooq", - message=f"Request failed: {self.base_url}", - ) from err + frequency: str, + ) -> list[dict[str, object]]: + payload = self._request_json( + f"{self.base_url}/bars", + resource=symbol, + params={ + "symbol": symbol, + "start": start, + "end": end, + "frequency": frequency, + "api_key": self.api_key, + }, + ) + records = payload.get("data") + if not isinstance(records, list): + raise DataValidationError(self.name, "Response does not contain a data list") + return records def _transform_data( self, - raw_data: str, + raw_data: list[dict[str, object]], symbol: str, ) -> pl.DataFrame: - """Transform CSV data to Polars DataFrame. - - Args: - raw_data: Raw CSV string - symbol: Symbol for logging - - Returns: - Polars DataFrame with OHLCV data - - Raises: - DataValidationError: If transformation fails - """ - try: - # 1. Parse CSV into Polars DataFrame - df = pl.read_csv(raw_data.encode()) - - # 2. Rename columns to standard format - # Stooq CSV: Date,Open,High,Low,Close,Volume - df = df.rename({ - "Date": "date", - "Open": "open", - "High": "high", - "Low": "low", - "Close": "close", - "Volume": "volume", - }) - - # 3. Convert date string to datetime - df = df.with_columns( - pl.col("date").str.to_date("%Y%m%d").cast(pl.Datetime).alias("timestamp") - ) - - # 4. Drop original date column - df = df.drop("date") - - # 5. Convert numeric columns to float - for col in ["open", "high", "low", "close", "volume"]: - if col in df.columns: - df = df.with_columns(pl.col(col).cast(pl.Float64)) - - # 6. Add symbol column - df = df.with_columns(pl.lit(symbol.upper()).alias("symbol")) - - # 7. Sort by timestamp - df = df.sort("timestamp") - - # 8. Select final columns in standard order - df = df.select([ - "timestamp", - "symbol", - "open", - "high", - "low", - "close", - "volume", - ]) - - return df - - except Exception as err: - raise DataValidationError( - provider="stooq", - message=f"Failed to transform data for {symbol}", - ) from err - - def close(self) -> None: - """Close HTTP client.""" - if hasattr(self, "session"): - self.session.close() - self.logger.debug("Closed Stooq API client") -``` - -**Key points:** -1. **Module docstring** - Explain what the provider does -2. **Inherit from BaseProvider** - Gets rate limiting, circuit breaker, etc. -3. **Class variables** - `DEFAULT_RATE_LIMIT`, `CIRCUIT_BREAKER_CONFIG`, mappings -4. **`__init__`** - Set up provider-specific config, call `super().__init__()` -5. **`name()`** - Return lowercase provider name -6. **`_fetch_raw_data()`** - Get data from API, minimal processing -7. **`_transform_data()`** - Convert to standard DataFrame format -8. **`close()`** - Clean up resources - -### Step 3: Handle Edge Cases - -Common edge cases to handle: - -```python -def _fetch_raw_data(self, symbol, start, end, frequency="daily"): - # ... existing code ... - - # Edge case: Empty response - if not csv_data or csv_data.strip() == "": - raise DataNotAvailableError( - provider="stooq", - symbol=symbol, - message="API returned empty response" - ) - - # Edge case: Error message in response - if "error" in csv_data.lower(): - raise ProviderError( - provider="stooq", - message=f"API error: {csv_data[:100]}" - ) - - # Edge case: Invalid symbol (returns HTML instead of CSV) - if csv_data.startswith(" dict: - """Update symbol with new data. - - Args: - symbol: Symbol to update - start_time: Start date (YYYY-MM-DD), defaults to 90 days ago - end_time: End date (YYYY-MM-DD), defaults to today - frequency: Data frequency - incremental: If True, only fetch new data since last update - dry_run: If True, don't actually store data - - Returns: - Result dictionary with status and metrics - """ - from datetime import timedelta - - # Default date range - if not end_time: - end_time = datetime.now().strftime("%Y-%m-%d") - if not start_time: - start_time = (datetime.now() - timedelta(days=90)).strftime("%Y-%m-%d") - - try: - # Fetch data - df = self.provider.fetch_ohlcv(symbol, start_time, end_time, frequency) - - if df.is_empty(): - return { - "success": True, - "symbol": symbol, - "records_fetched": 0, - "message": "No data available", - } - - # Store data if not dry run - if not dry_run: - self.storage.write(df, symbol, self.provider_name) - - return { - "success": True, - "symbol": symbol, - "records_fetched": len(df), - "start_date": start_time, - "end_date": end_time, - } +The returned frame must satisfy the canonical schema described in [Architecture](architecture.md). +Do not silently invent volume, timezone, symbol, or date-boundary semantics. - except Exception as err: - self.logger.error(f"Failed to update {symbol}: {err}") - return { - "success": False, - "symbol": symbol, - "error": str(err), - } -``` - -### Step 5: Register Provider - -Add to `src/ml4t-data/providers/__init__.py`: +Always close providers in application code or use their context-manager support: ```python -# In the imports section -try: - from ml4t.data.providers.stooq import StooqProvider, StooqUpdater -except ImportError: - StooqProvider = None # type: ignore - StooqUpdater = None # type: ignore - -# In __all__ list -__all__ = [ - # ... existing providers ... - "StooqProvider", - "StooqUpdater", -] - -# In docstring -""" -Available Providers: - - StooqProvider: International equities (free, no API key) - - ... -""" +with ExampleProvider(api_key="...") as provider: + frame = provider.fetch_ohlcv("ABC", "2026-01-01", "2026-01-31", "daily") ``` -## Testing Your Provider - -### Step 6: Create Integration Tests - -Create `tests/integration/test_stooq.py`: - -```python -"""Integration tests for Stooq provider (real API calls). - -These tests verify the Stooq provider works correctly with actual API calls. - -Requirements: - - No API key needed (public API) - - Rate limit: 10 requests/minute - -Test Coverage: - - Stock daily OHLCV data (AAPL.US) - - Multiple frequencies (daily, weekly, monthly) - - International exchanges - - Error handling -""" - -import os -from datetime import datetime, timedelta - -import polars as pl -import pytest - -from ml4t.data.core.exceptions import DataNotAvailableError -from ml4t.data.providers.stooq import StooqProvider - - -@pytest.fixture -def provider(): - """Create Stooq provider.""" - provider = StooqProvider() - yield provider - provider.close() +## Register the provider +Add one `_spec()` entry to `src/ml4t/data/providers/registry.py`. Record: -class TestStooqProvider: - """Test Stooq provider with real API calls.""" - - def test_provider_initialization(self): - """Test provider can be initialized.""" - provider = StooqProvider() - assert provider.name() == "stooq" - provider.close() - - def test_fetch_ohlcv_daily(self, provider): - """Test fetching daily stock data with real API call.""" - end_date = datetime.now().strftime("%Y-%m-%d") - start_date = (datetime.now() - timedelta(days=30)).strftime("%Y-%m-%d") - - df = provider.fetch_ohlcv( - symbol="AAPL.US", - start=start_date, - end=end_date, - frequency="daily", - ) +- the stable provider name, module, and class; +- factual capabilities; +- required or optional credential environment variables; +- required configuration fields; +- the optional dependency extra, if one exists; and +- whether the manager-compatible OHLCV path applies. - # Verify data structure - assert isinstance(df, pl.DataFrame) - assert not df.is_empty() +Expose the supported class from `src/ml4t/data/providers/__init__.py`. Keep optional integrations +importable without installing unrelated extras. - # Check required columns - required_cols = ["timestamp", "symbol", "open", "high", "low", "close", "volume"] - assert all(col in df.columns for col in required_cols) +## Write deterministic contract tests - # Verify data types - assert df["timestamp"].dtype == pl.Datetime - assert df["symbol"].dtype == pl.String - assert df["open"].dtype == pl.Float64 +Unit tests must exercise the public `fetch_ohlcv()` path with a transport fixture or local response +fixture. A useful provider test proves that: - # Verify OHLCV relationships - assert (df["high"] >= df["low"]).all() - assert (df["high"] >= df["open"]).all() +- request parameters preserve inclusive or exclusive date semantics; +- timestamps are interpreted in the source timezone and normalized to UTC; +- pagination cannot silently truncate the requested range; +- malformed responses raise the expected shared exception; +- HTTP authentication, missing-symbol, rate-limit, and transient errors are classified correctly; +- the frame satisfies canonical schema, ordering, symbol, and OHLC invariants; and +- the HTTP client is closed. - print(f"✅ Fetched {len(df)} rows of AAPL.US daily data") - - def test_invalid_symbol(self, provider): - """Test error handling for invalid symbol.""" - end_date = datetime.now().strftime("%Y-%m-%d") - start_date = (datetime.now() - timedelta(days=30)).strftime("%Y-%m-%d") - - with pytest.raises(DataNotAvailableError): - provider.fetch_ohlcv( - symbol="INVALID_XYZ.US", - start=start_date, - end=end_date, - frequency="daily", - ) - - print("✅ Invalid symbol correctly raises DataNotAvailableError") -``` - -### Step 7: Run Tests - -```bash -# Run integration tests -pytest tests/integration/test_stooq.py -v -s - -# Check coverage -pytest tests/integration/test_stooq.py --cov=src/ml4t-data/providers/stooq -``` - -## Documentation - -### Step 8: Create Example Script - -Create `examples/stooq_example.py`: +`httpx.MockTransport` can drive the real request and transformation path without network access: ```python -"""Example usage of Stooq provider for international equities.""" - -import sys -from datetime import datetime, timedelta -from pathlib import Path - -# Add src to path -sys.path.insert(0, str(Path(__file__).parent.parent / "src")) - -from ml4t.data.providers import StooqProvider, StooqUpdater -from ml4t.data.storage.backend import StorageConfig -from ml4t.data.storage.hive import HiveStorage - - -def example_basic_usage(): - """Example 1: Basic OHLCV fetching.""" - print("\n" + "=" * 60) - print(" Example 1: Basic OHLCV Data Fetching") - print("=" * 60 + "\n") - - provider = StooqProvider() - - try: - end = datetime.now().strftime("%Y-%m-%d") - start = (datetime.now() - timedelta(days=30)).strftime("%Y-%m-%d") - - # Fetch AAPL data - print(f"Fetching AAPL.US data from {start} to {end}...\n") - df = provider.fetch_ohlcv("AAPL.US", start, end, frequency="daily") - - print(f"Fetched {len(df)} records:") - print(df.head()) - print(f"\nPrice range: ${df['close'].min():.2f} - ${df['close'].max():.2f}") +import httpx - finally: - provider.close() +def test_example_provider_preserves_end_date() -> None: + def respond(request: httpx.Request) -> httpx.Response: + assert request.url.params["end"] == "2026-01-31" + return httpx.Response(200, json={"data": []}) -def example_international_exchanges(): - """Example 2: Fetching from multiple international exchanges.""" - print("\n" + "=" * 60) - print(" Example 2: International Exchanges") - print("=" * 60 + "\n") + with ExampleProvider( + api_key="test", + transport=httpx.MockTransport(respond), + ) as provider: + result = provider.fetch_ohlcv("ABC", "2026-01-01", "2026-01-31", "daily") - provider = StooqProvider() - - stocks = [ - ("AAPL.US", "Apple (US)"), - ("VOD.UK", "Vodafone (London)"), - ("BMW.DE", "BMW (Germany)"), - ] - - try: - end = datetime.now().strftime("%Y-%m-%d") - start = (datetime.now() - timedelta(days=30)).strftime("%Y-%m-%d") - - for symbol, name in stocks: - print(f"\n📊 {name} ({symbol}):") - df = provider.fetch_ohlcv(symbol, start, end) - print(f" Records: {len(df)}") - print(f" Last close: {df['close'][-1]:.2f}") - - finally: - provider.close() - - -if __name__ == "__main__": - example_basic_usage() - example_international_exchanges() + assert result.is_empty() ``` -### Step 9: Update README - -Add the provider to the provider reference: - -````markdown -## Phase X Providers +Adapt the fixture to the provider's real response shape. Tests that contact the live service use the +`integration` marker and, when applicable, `requires_api_key`, `paid_tier`, `slow`, or `expensive`. +Add a narrowly bounded job to `.github/workflows/provider-contracts.yml` when ongoing live evidence +is necessary. -| Provider | Asset Classes | API Key | Rate Limit | Best For | -|----------|--------------|---------|------------|----------| -| **Stooq** | International Equities | No | 10/min | Free international data | - -### Stooq +Run the focused offline tests first: ```bash -# No API key needed! +uv run pytest tests/test__provider.py -q -ra +uv run ruff check src/ml4t/data/providers/.py tests/test__provider.py +uv run ty check ``` -```python -from ml4t.data.providers import StooqProvider - -provider = StooqProvider() -data = provider.fetch_ohlcv("AAPL.US", "2024-01-01", "2024-01-31") -``` - -**Why Stooq:** -- Free international equities -- 50+ exchanges worldwide -- No registration required -- Simple CSV format -```` - -## Checklist - -Before submitting your provider, verify: - -### Code Quality -- [ ] Inherits from `BaseProvider` -- [ ] Implements `name()`, `_fetch_raw_data()`, `_transform_data()` -- [ ] Has rate limiting configured -- [ ] Has circuit breaker configured -- [ ] Uses ml4t-data exception classes -- [ ] Has type hints on all public methods -- [ ] Has Google-style docstrings -- [ ] Passes `ruff` linting -- [ ] Passes `uv run ty check` - -### Testing -- [ ] Integration tests created -- [ ] Tests cover happy path -- [ ] Tests cover error cases -- [ ] Tests respect rate limits -- [ ] 80%+ code coverage -- [ ] All tests passing - -### Documentation -- [ ] Module docstring complete -- [ ] Class docstring complete -- [ ] Method docstrings complete -- [ ] Example script created -- [ ] README updated -- [ ] Provider registered in `__init__.py` - -### Optional but Recommended -- [ ] Updater class implemented -- [ ] Updater tests created -- [ ] Multiple frequencies supported -- [ ] Custom parameters documented - -## Common Pitfalls - -❌ **Forgetting to call `super().__init__()`** -```python -def __init__(self): - # Missing super().__init__() - rate limiting won't work! - pass -``` +## Document the service boundary -✅ **Always call super()** -```python -def __init__(self): - super().__init__(rate_limit=..., ...) -``` - -❌ **Not handling rate limits** -```python -def _fetch_raw_data(self, ...): - # Makes request without rate limiting - will get banned! - response = self.session.get(url) -``` - -✅ **Use rate limiter** -```python -def _fetch_raw_data(self, ...): - self.rate_limiter.acquire(blocking=True) - response = self.session.get(url) -``` - -❌ **Generic exceptions** -```python -if error: - raise Exception("Something went wrong") -``` - -✅ **Specific exceptions** -```python -if response.status_code == 429: - raise RateLimitError(provider="stooq", retry_after=60.0) -``` +Add `docs/providers/.md` and include it in `mkdocs.yml`. State the source, supported data, +credentials, optional extra, quota or paid-service boundary, symbol and date semantics, and one +supported example. Update the appropriate asset-class source page without duplicating the provider +reference. -## Getting Help +Do not add a provider to the README unless it changes the package-level installation or integration +boundary. The provider index and registry are the complete provider inventory. -- Check existing providers for patterns: `src/ml4t-data/providers/` -- Use the template: `provider_template/` -- Read the [architecture guide](architecture.md) for implementation details -- Ask in GitHub Discussions +## Verify the contribution ---- +Before opening the pull request, run the full gates in the root `AGENTS.md`. The strict MkDocs build +must resolve the new page and navigation entry, and the package build must keep the base import +usable without the provider's optional dependency or credentials. -**Ready to contribute?** Follow this guide, and submit a PR! 🚀 +The pull request should link its owning issue and identify any compatibility or release-note impact. diff --git a/docs/contributing/index.md b/docs/contributing/index.md index c8878d37..7fdb84eb 100644 --- a/docs/contributing/index.md +++ b/docs/contributing/index.md @@ -1,70 +1,47 @@ # Contributing -Welcome to ML4T Data! We appreciate your interest in contributing. +Contributions should preserve the public provider, storage, and validation contracts while keeping +the default test suite deterministic and offline. -## Ways to Contribute +## Set up the repository -
- -- :material-source-pull:{ .lg .middle } __Create a Provider__ - - --- - - Add support for a new data source. - - [:octicons-arrow-right-24: Provider Guide](creating-a-provider.md) - -- :material-bug:{ .lg .middle } __Fix Bugs__ - - --- - - Help improve reliability and fix issues. - - [:octicons-arrow-right-24: Testing Guide](testing.md) - -- :material-file-document:{ .lg .middle } __Improve Docs__ - - --- - - Enhance documentation and examples. - - [:octicons-arrow-right-24: Architecture](architecture.md) - -
- -## Quick Start +Install `uv`, clone the canonical repository, and create the complete locked environment: ```bash -# Clone the repository -git clone https://github.com/stefan-jansen/ml4t-data.git -cd ml4t-data - -# Install with dev dependencies -uv sync --all-extras +git clone https://github.com/ml4t/data.git +cd data +uv sync --locked --all-extras --all-groups +uv run pre-commit install +``` -# Install pre-commit hooks -pre-commit install +Run the offline test lane once to confirm the checkout: -# Run tests -pytest +```bash +uv run pytest tests -q -ra ``` -## Code Style +Provider tests that contact live services are excluded by default. See the +[testing guide](testing.md) before running a credentialed or paid-tier test. + +## Choose the relevant guide -- **Formatter**: ruff (100 char line length) -- **Type checking**: ty -- **Docstrings**: Google style -- **Tests**: pytest with 80%+ coverage +- [Creating a provider](creating-a-provider.md) covers the provider contract, registry metadata, + deterministic tests, and documentation required for a new adapter. +- [Architecture](architecture.md) explains the boundaries among providers, validation, storage, + configuration, and orchestration. +- [Testing](testing.md) lists the offline, resource-warning, focused, and live-provider lanes. -## Pull Request Process +The root [`AGENTS.md`](https://github.com/ml4t/data/blob/main/AGENTS.md) records the repository map, +change rules, and complete verification commands. More specific `AGENTS.md` files apply under the +provider, storage, and futures source directories. -1. Fork the repository -2. Create a feature branch -3. Make your changes with tests -4. Run `pre-commit run --all-files` -5. Submit a pull request +## Pull requests -## Getting Help +1. Open or reference one issue that defines the problem and acceptance criteria. +2. Make the smallest coherent change and add a test that fails for the behavior being corrected. +3. Run focused checks while editing, then the repository gates from `AGENTS.md`. +4. Complete the pull request template, including compatibility and release implications. -- [GitHub Issues](https://github.com/stefan-jansen/ml4t-data/issues) -- [Discussions](https://github.com/stefan-jansen/ml4t-data/discussions) +Use [private vulnerability reporting](https://github.com/ml4t/data/security/advisories/new) for a +suspected security issue. Use the [issue tracker](https://github.com/ml4t/data/issues) for other +bugs, documentation problems, and feature requests. diff --git a/docs/contributing/testing.md b/docs/contributing/testing.md index 22e3791d..a56bad7f 100644 --- a/docs/contributing/testing.md +++ b/docs/contributing/testing.md @@ -1,146 +1,93 @@ -# Testing Guide +# Testing -## Quick Start +The default pytest configuration runs the deterministic offline lane. Live API, paid-tier, +credentialed, expensive, and slow tests require an explicit marker or a dedicated workflow. -### Run Tests (Parallel by Default) -```bash -# Default: runs tests in parallel using all CPU cores (FAST!) -pytest - -# Verbose output -pytest -v +## Install the test environment -# Disable parallel execution (slower, but better for debugging) -pytest -n 0 +Use the lock file and install all provider extras and development groups: -# Use specific number of workers -pytest -n 4 +```bash +uv sync --locked --all-extras --all-groups ``` -**Performance**: Tests run in parallel by default using `pytest-xdist`: -- Parallel execution is optional and should be measured on the target machine -- **Sequential** (`-n 0`): ~15 minutes for full suite +## Run the default lane -### Run Integration Tests ```bash -# Run ALL tests including integration (if you have API keys) -pytest -m "" - -# Run only integration tests -pytest -m integration - -# Run specific provider integration tests -pytest tests/integration/test_coingecko.py -m "" +uv run pytest tests -q -ra ``` -## Test Categories +`pyproject.toml` excludes `slow`, `paid_tier`, `integration`, and `requires_api_key` tests from this +command. It also enables strict marker validation, so an undeclared marker fails collection. -### Unit Tests (Fast) -- **Location**: Mostly in `tests/` root, some in `tests/unit/` -- **Duration**: <1 minute for all unit tests -- **Dependencies**: Core dependencies only -- **Markers**: None (run by default) +Run the separate resource-leak lane before submitting a change that opens files, HTTP clients, or +other managed resources: -### Integration Tests (Slow) -- **Location**: `tests/integration/` -- **Duration**: ~15 minutes -- **Dependencies**: Real APIs, may require API keys -- **Markers**: `@pytest.mark.integration` +```bash +uv run pytest tests -q -ra -W error::ResourceWarning +``` -## API Keys for Integration Tests +## Run focused tests -Integration tests for provider APIs require API keys: +Use a file, node ID, or expression while developing: ```bash -# Free tier (no key required) -export COINGECKO_API_KEY="" # Optional - -# Requires API keys -export MASSIVE_API_KEY="your_key_here" -export TWELVE_DATA_API_KEY="your_key_here" -export CRYPTOCOMPARE_API_KEY="your_key_here" -export FINNHUB_API_KEY="your_key_here" -export TIINGO_API_KEY="your_key_here" -export EODHD_API_KEY="your_key_here" -export ALPHA_VANTAGE_API_KEY="your_key_here" -export OANDA_API_KEY="your_key_here" -export DATABENTO_API_KEY="your_key_here" +uv run pytest tests/test_storage_paths.py -q -ra +uv run pytest tests/test_storage_paths.py::test_legacy_env_warns -q -ra +uv run pytest tests -q -ra -k yahoo +uv run pytest tests -q -ra -x ``` -## Test Markers - -| Marker | Description | Skip by Default? | -|--------|-------------|------------------| -| `integration` | Real API calls, slow tests | ✅ Yes | -| `slow` | Tests taking >10 seconds | ✅ Yes | -| `requires_api_key` | Needs specific API key | ⚠️ If key missing | -| `expensive` | High API costs | ⚠️ In CI only | +The suite runs sequentially by default. Add `-n auto` only when the focused tests are known to be +safe under parallel execution. -## Common Test Commands - -```bash -# Run specific test file -pytest tests/test_core_models.py +## Test categories -# Run specific test -pytest tests/test_core_models.py::test_function_name +| Marker | Contract | Default lane | +|---|---|---:| +| `integration` | Contacts an external service or exercises a cross-system boundary | Excluded | +| `requires_api_key` | Requires one or more provider credentials | Excluded | +| `paid_tier` | Can consume a metered or paid provider allowance | Excluded | +| `slow` | Unsuitable for the routine offline lane | Excluded | +| `expensive` | Has a material resource or provider cost | Not excluded automatically | +| `network_guard_probe` | Verifies that the offline network guard blocks network access | Included | -# Run tests matching pattern -pytest -k "yahoo" +Mark a test according to what it does, not according to the directory containing it. A provider +test using `httpx.MockTransport`, a fixture, or a local file belongs in the default lane. -# Stop after first failure -pytest -x +## Run live provider contracts -# Show print statements -pytest -s +The `Provider Contract` workflow runs one selected provider with only that provider's credentials. +Use it for repository-level evidence. For local diagnosis, configure the named environment variable +and run only the intended node: -# Generate coverage report -pytest --cov=ml4t-data --cov-report=html -open htmlcov/index.html +```bash +uv run pytest tests/integration/test_coingecko.py::TestCoinGeckoProvider::test_fetch_ohlcv_btc \ + -m integration -q -ra ``` -## CI/CD +Do not run every integration test against a live account. Several providers impose quotas, require +subscriptions, or charge for requests. -GitHub Actions runs: -- **PR checks**: Unit tests only (fast) -- **Main branch**: Unit + integration tests (with API keys) +## Optional dependencies -## Troubleshooting +The complete development environment includes every provider extra. To reproduce a minimal optional +dependency boundary, create an isolated environment with the relevant extra, for example: -### "ModuleNotFoundError: No module named 'databento'" ```bash -pip install -e ".[databento]" +uv sync --locked --extra databento --group test ``` -### Tests taking too long -```bash -# Make sure you're running unit tests only -pytest -m "not integration and not slow" -``` +Import-time behavior must not require unrelated extras or credentials. -### Integration tests skipped -```bash -# Check if API keys are set -env | grep API_KEY +## Coverage and failure diagnosis -# Run with specific marker -pytest -m integration -``` +Generate a local report with the package import path: -### Parallel execution issues (debugging) ```bash -# Disable parallel execution for clearer error messages -pytest -n 0 - -# Use less workers to reduce resource contention -pytest -n 4 - -# Run single test file -pytest tests/test_specific_file.py -n 0 +uv run pytest tests -q --cov=ml4t.data --cov-report=term-missing ``` -### Test output not showing (parallel mode) -```bash -# Use -s with sequential execution to see print statements -pytest -n 0 -s -``` +When a test fails only in the full suite, rerun it sequentially and preserve the order-dependent +reproduction. Do not remove the resource-warning lane, weaken markers, or replace a live contract +with a mock merely to obtain a green result. From cd05d835f721fddb787f0723fa6e0d01842c85a7 Mon Sep 17 00:00:00 2001 From: Stefan Jansen Date: Sat, 10 Oct 2026 04:58:23 -0400 Subject: [PATCH 3/5] fix: retry deployed documentation probes --- scripts/verify_documentation_identity.py | 19 ++++++++++++++----- 1 file changed, 14 insertions(+), 5 deletions(-) diff --git a/scripts/verify_documentation_identity.py b/scripts/verify_documentation_identity.py index 825214bc..3d78e7cc 100644 --- a/scripts/verify_documentation_identity.py +++ b/scripts/verify_documentation_identity.py @@ -163,12 +163,21 @@ def page_reference_urls(html: str, *, page_url: str, site_root: str) -> list[str return references -def _probe_url(url: str) -> None: +def _probe_url(url: str, attempts: int = 3, delay: float = 1.0) -> None: request = urllib.request.Request(url, headers={"User-Agent": "ml4t-release-verifier"}) - with urllib.request.urlopen(request, timeout=30) as response: # noqa: S310 - if response.status != 200: - raise ValueError(f"HTTP {response.status}") - response.read(1) + error: Exception | None = None + for attempt in range(attempts): + try: + with urllib.request.urlopen(request, timeout=30) as response: # noqa: S310 + if response.status != 200: + raise ValueError(f"HTTP {response.status}") + response.read(1) + return + except (OSError, urllib.error.URLError, ValueError) as caught: + error = caught + if attempt + 1 < attempts: + time.sleep(delay) + raise ValueError(str(error)) def deployed_link_failures(urls: list[str]) -> list[str]: From 5262d065fa0f3f3674865931fac66462d0075b21 Mon Sep 17 00:00:00 2001 From: Stefan Jansen Date: Sat, 10 Oct 2026 05:14:15 -0400 Subject: [PATCH 4/5] docs: remove stale legacy guidance --- docs/ASYNC_STORAGE.md | 285 ------------ docs/FEATURES.md | 363 ---------------- docs/INSTALLATION.md | 304 ------------- docs/INTEGRATION_TESTING.md | 241 ----------- docs/SESSION_MANAGEMENT.md | 528 ----------------------- docs/crypto_providers.md | 161 ------- docs/export_guide.md | 282 ------------ docs/providers/crypto.md | 2 +- docs/providers/cryptocompare.md | 11 +- docs/providers/databento.md | 2 +- docs/providers/databento_reference.md | 161 ------- docs/providers/eodhd.md | 3 +- docs/providers/index.md | 6 +- docs/providers/mock.md | 68 --- docs/providers/polymarket.md | 2 +- docs/providers/synthetic.md | 1 - docs/user-guide/exporting.md | 100 +++++ docs/user-guide/incremental-updates.md | 377 ++++------------ docs/user-guide/sessions.md | 122 ++++++ docs/yahoo_provider.md | 144 ------- mkdocs.yml | 15 +- scripts/verify_documentation_identity.py | 91 ++++ tests/test_release_pipeline.py | 25 ++ 23 files changed, 438 insertions(+), 2856 deletions(-) delete mode 100644 docs/ASYNC_STORAGE.md delete mode 100644 docs/FEATURES.md delete mode 100644 docs/INSTALLATION.md delete mode 100644 docs/INTEGRATION_TESTING.md delete mode 100644 docs/SESSION_MANAGEMENT.md delete mode 100644 docs/crypto_providers.md delete mode 100644 docs/export_guide.md delete mode 100644 docs/providers/databento_reference.md delete mode 100644 docs/providers/mock.md create mode 100644 docs/user-guide/exporting.md create mode 100644 docs/user-guide/sessions.md delete mode 100644 docs/yahoo_provider.md diff --git a/docs/ASYNC_STORAGE.md b/docs/ASYNC_STORAGE.md deleted file mode 100644 index 1ced3efd..00000000 --- a/docs/ASYNC_STORAGE.md +++ /dev/null @@ -1,285 +0,0 @@ -# Async Storage Backend Guide - -ML4T Data includes an async storage backend that provides non-blocking I/O for workloads that -coordinate storage with other asynchronous operations. - -## Overview - -The async storage backend is built on top of `aiofiles` and provides the same interface as the synchronous storage backend, but with async/await support for better performance in high-concurrency scenarios. - -## Features - -- **Non-blocking I/O** - Uses `aiofiles` for async file operations -- **Shared Lock Coordination** - Prevents race conditions with coordinated locking -- **Migration Utilities** - Easy conversion between sync and async storage -- **Atomic Publication** - Immutable generations become visible through an atomic commit pointer -- **Exponential Backoff** - Automatic retry with intelligent backoff -- **Path Security** - Built-in protection against path traversal attacks - -## Installation - -```bash -# Install async storage dependencies -pip install -e ".[async]" -``` - -## Basic Usage - -### Async Storage Backend - -```python -import asyncio -from pathlib import Path -from ml4t.data.storage.async_filesystem import AsyncFileSystemBackend -from ml4t.data.core.models import DataObject, Metadata -import polars as pl -from datetime import datetime - -async def main(): - # Create async backend - storage = AsyncFileSystemBackend( - data_root=Path("./data"), - lock_timeout=30.0, - max_retries=3 - ) - - # Create sample data - df = pl.DataFrame({ - "timestamp": [datetime(2024, 1, 1), datetime(2024, 1, 2)], - "open": [100.0, 101.0], - "high": [105.0, 106.0], - "low": [99.0, 100.0], - "close": [104.0, 105.0], - "volume": [1000000, 1100000] - }) - - metadata = Metadata( - provider="test", - symbol="AAPL", - asset_class="equities", - frequency="daily", - schema_version="1.0" - ) - - data = DataObject(data=df, metadata=metadata) - - # Write data - key = await storage.write(data) - print(f"Data written with key: {key}") - - # Read data - loaded_data = await storage.read(key) - print(f"Loaded data for {loaded_data.metadata.symbol}") - - # Check existence - exists = await storage.exists(key) - print(f"Data exists: {exists}") - - # List all keys - keys = await storage.list_keys() - print(f"All keys: {keys}") - - # Delete data - await storage.delete(key) - print("Data deleted") - -# Run the async function -asyncio.run(main()) -``` - -### Using Sync Adapter - -If you need to use the async storage backend in synchronous code: - -```python -from ml4t.data.storage.async_filesystem import AsyncFileSystemBackend -from ml4t.data.storage.async_migration import AsyncStorageAdapter - -# Create async backend -async_backend = AsyncFileSystemBackend(data_root=Path("./data")) - -# Wrap with sync adapter -sync_storage = AsyncStorageAdapter(async_backend) - -# Use like a regular sync storage backend -key = sync_storage.write(data) -loaded_data = sync_storage.read(key) -``` - -## Migration Between Sync and Async - -### Sync to Async Migration - -```python -import asyncio -from ml4t.data.storage.filesystem import FileSystemBackend -from ml4t.data.storage.async_filesystem import AsyncFileSystemBackend -from ml4t.data.storage.async_migration import StorageMigrator - -async def migrate_to_async(): - # Create backends - sync_backend = FileSystemBackend(data_root=Path("./sync_data")) - async_backend = AsyncFileSystemBackend(data_root=Path("./async_data")) - - # Migrate all data - successful, failed = await StorageMigrator.sync_to_async( - sync_backend=sync_backend, - async_backend=async_backend, - batch_size=10 # Process 10 items at once - ) - - print(f"Migration complete: {successful} successful, {failed} failed") - -asyncio.run(migrate_to_async()) -``` - -### Async to Sync Migration - -```python -async def migrate_to_sync(): - # Create backends - async_backend = AsyncFileSystemBackend(data_root=Path("./async_data")) - sync_backend = FileSystemBackend(data_root=Path("./sync_data")) - - # Migrate all data - successful, failed = await StorageMigrator.async_to_sync( - async_backend=async_backend, - sync_backend=sync_backend, - prefix="equities/daily/", # Only migrate specific prefix - batch_size=5 - ) - - print(f"Migration complete: {successful} successful, {failed} failed") - -asyncio.run(migrate_to_sync()) -``` - -## Advanced Configuration - -### Lock Configuration - -```python -storage = AsyncFileSystemBackend( - data_root=Path("./data"), - lock_timeout=60.0, # Wait up to 60 seconds for lock - max_retries=5, # Retry up to 5 times - retry_base_delay=0.1, # Start with 100ms delay -) -``` - -### Concurrent Operations - -The async backend is designed for high concurrency: - -```python -async def concurrent_writes(): - storage = AsyncFileSystemBackend(data_root=Path("./data")) - - # Create multiple data objects - tasks = [] - for i in range(10): - data = create_sample_data(f"STOCK{i}") - task = storage.write(data) - tasks.append(task) - - # Execute all writes concurrently - keys = await asyncio.gather(*tasks) - print(f"Wrote {len(keys)} files concurrently") - -asyncio.run(concurrent_writes()) -``` - -### Concurrent Reads - -```python -async def concurrent_reads(): - storage = AsyncFileSystemBackend(data_root=Path("./data")) - - # List all available keys - keys = await storage.list_keys() - - # Read all data concurrently - tasks = [storage.read(key) for key in keys] - data_objects = await asyncio.gather(*tasks) - - print(f"Read {len(data_objects)} files concurrently") - -asyncio.run(concurrent_reads()) -``` - -## Performance Considerations - -### When to Use Async Storage - -- **High Concurrency**: When you need to handle many simultaneous requests -- **I/O Bound Operations**: When storage operations are the bottleneck -- **API Servers**: When building async web applications with FastAPI/aiohttp -- **Batch Processing**: When processing large datasets with many files - -### When to Use Sync Storage - -- **Simple Scripts**: For simple data loading scripts -- **Interactive Use**: For interactive analysis in Jupyter notebooks -- **Legacy Code**: When integrating with existing synchronous codebases - -## Error Handling - -```python -from ml4t.data.storage.async_filesystem import LockAcquisitionError, StorageError - -async def robust_operation(): - storage = AsyncFileSystemBackend(data_root=Path("./data")) - - try: - data = await storage.read("equities/daily/AAPL") - except StorageError as e: - print(f"Storage error: {e}") - except LockAcquisitionError as e: - print(f"Could not acquire lock: {e}") - except Exception as e: - print(f"Unexpected error: {e}") -``` - -## Testing - -The async storage backend includes comprehensive tests: - -```bash -# Run async storage tests -pytest tests/test_async_storage.py -v - -# Run with coverage -pytest tests/test_async_storage.py --cov=src/ml4t-data/storage/async_filesystem -``` - -## Implementation Details - -### Lock Management - -- Each storage key gets its own `asyncio.Lock` for fine-grained locking -- Locks are shared across `AsyncFileLock` instances via a backend-level registry -- File locks serve as indicators while actual coordination happens via asyncio -- Automatic cleanup prevents lock leakage - -### File Format Compatibility - -The async storage backend uses the same file formats as the sync backend: - -- **Data files**: `.parquet` format using Polars -- **Metadata files**: `.json` format with proper datetime serialization -- **Directory structure**: Same hierarchical organization - -### Migration Safety - -- Migration utilities preserve data integrity -- Batch processing prevents memory issues with large datasets -- Progress callbacks for monitoring long-running migrations -- Error recovery and reporting for failed items - -## Future Enhancements - -- Database backend support (PostgreSQL, MongoDB) -- S3/cloud storage integration -- Distributed locking for multi-node deployments -- Advanced caching with Redis -- Compression during async I/O operations diff --git a/docs/FEATURES.md b/docs/FEATURES.md deleted file mode 100644 index 24fcb031..00000000 --- a/docs/FEATURES.md +++ /dev/null @@ -1,363 +0,0 @@ -# ML4T-Data: Quick Reference Guide - -## At a Glance - -| Metric | Value | -|--------|-------| -| **Status** | Pre-release (0.1.0) - breaking changes acceptable | -| **Data Providers** | 20 source/provider adapters, plus synthetic and mock providers for testing | -| **Test Files** | 73 (50+ unit, 10 integration, 5+ acceptance) | -| **Lines of Code** | ~18,900 in src/ | -| **Python Version** | 3.9+ | -| **Storage** | Hive-partitioned and flat Parquet layouts | -| **Maturity** | 70-75% complete | - ---- - -## Core Features Checklist - -### Providers (20+ adapters) -- [x] Yahoo Finance (unlimited, free) -- [x] EODHD (500 calls/day) -- [x] CoinGecko (free, no key) -- [x] Binance (crypto spot/futures) -- [x] DataBento (institutional, trial) -- [x] Wiki Prices (historical 1962-2018) -- [x] Tiingo, Finnhub, Polygon, Twelve Data, CryptoCompare, Oanda -- [x] FRED, AQR, Fama-French, Kalshi, Polymarket, Polymarket US, ForecastEx, OKX, Binance public bulk, ITCH sample -- [x] Synthetic and mock providers for testing - -### Storage (Complete) -- [x] Hive-partitioned Parquet with date partition pruning -- [x] Metadata tracking (last updates, row counts) -- [x] Atomic writes with file locking -- [x] Transaction support with rollback -- [x] Migration system for schema evolution - -### Data Quality (Complete) -- [x] OHLCV validator (8 rules) -- [x] Anomaly detection (3 detectors) -- [x] Cross-provider validation -- [x] Deduplication & gap detection -- [x] Validation reports with JSON export - -### Updates (Complete) -- [x] Incremental updates (only fetch new) -- [x] Gap detection & filling -- [x] Resume capability on failure -- [x] Multiple strategies (INCREMENTAL, APPEND_ONLY, FULL_REFRESH, BACKFILL) - -### CLI (Complete) -- [x] `fetch` - Get data from provider -- [x] `update-all` - Automated incremental updates -- [x] `list` - Show stored data -- [x] `validate` - Data quality checks -- [x] `export` - CSV, JSON, Excel, Parquet - -### Utilities (Complete) -- [x] Rate limiting (global + per-provider) -- [x] Circuit breaker (5 failures → 5min timeout) -- [x] Retry with exponential backoff -- [x] File locking (concurrent access) -- [x] Gap optimization - -### Configuration (Complete) -- [x] YAML/JSON support -- [x] Environment variable override -- [x] Type validation (Pydantic) -- [x] Provider availability checks - -### Asset Management (Complete) -- [x] Asset class registry (8 classes) -- [x] Contract specifications -- [x] Symbol validation -- [x] Schema per asset type - -### Futures (Partial ⚠️) -- [x] Parser (Quandl CHRIS format) -- [x] Roll strategies (3 types) -- [x] Adjustment methods (3 types) -- [x] Continuous contract builder -- [ ] Time-based rolling with expiration calendar (TODO) - ---- - -## File Organization - -``` -src/ml4t/data/ -├── providers/ # 13 provider implementations -│ ├── base.py # Template Method pattern -│ ├── yahoo.py, coingecko.py, binance.py, ... (12 live) -│ └── wiki_prices.py # Historical fallback -├── storage/ # Data persistence -│ ├── hive.py # Hive-partitioned (primary) -│ ├── metadata_tracker.py # Update tracking -│ ├── migration.py # Schema evolution -│ └── backend.py # Atomic generation publication -├── validation/ # Data quality -│ ├── ohlcv.py # OHLC invariant checks -│ ├── anomaly.py # 3 anomaly detectors -│ └── report.py # Quality reports -├── update_manager.py # Incremental updates -├── data_manager.py # Unified interface -├── cli_interface.py # CLI commands -├── config/ # Configuration system -├── utils/ # Utilities (rate limit, gaps, locking, etc.) -├── futures/ # Futures handling (partial) -├── assets/ # Asset management -├── core/ # Models, schemas, exceptions -├── export/ # Export formats (CSV, JSON, Excel) -├── sessions/ # Session management -├── calendar/ # Trading calendars -└── security/ # Path validation -``` - ---- - -## Quick Start Examples - -### Fetch Data -```python -from ml4t.data.providers import YahooFinanceProvider - -provider = YahooFinanceProvider() -data = provider.fetch_ohlcv("AAPL", "2024-01-01", "2024-12-31") -print(data.head()) -``` - -### Store & Retrieve -```python -from ml4t.data.storage.hive import HiveStorage, StorageConfig -from ml4t.data.data_manager import DataManager - -storage = HiveStorage(config=StorageConfig(base_path="~/ml4t-data")) -manager = DataManager(storage=storage) - -# Store data and retain the canonical key -key = manager.import_data(data, "AAPL", provider="yahoo") - -# Retrieve later -retrieved = storage.read(key).collect() -``` - -### Validate Data -```python -from ml4t.data.validation import OHLCVValidator - -validator = OHLCVValidator() -report = validator.validate(data) - -if report.has_errors: - print(f"Found {len(report.violations)} issues") - fixed = validator.fix_auto_fixable(data) -``` - -### Detect Anomalies -```python -from ml4t.data.anomaly import AnomalyManager - -manager = AnomalyManager() -report = manager.analyze(data, symbol="AAPL") - -critical = report.get_critical_anomalies() -if critical: - print(f"Alert: {len(critical)} critical anomalies found") -``` - -### CLI Usage -```bash -# Fetch single symbol -ml4t-data fetch -s AAPL --start 2024-01-01 --end 2024-12-31 - -# Fetch from file -ml4t-data fetch -f symbols.txt --start 2024-01-01 --end 2024-12-31 - -# Automated incremental updates -ml4t-data update-all -c ml4t-data.yaml - -# Validate data -ml4t-data validate --symbol AAPL - -# Export to CSV -ml4t-data export --symbol AAPL --output data.csv -``` - ---- - -## Known Gaps - -### Critical Issues -None identified - -### Important Issues -1. **Binance Integration Tests** - Missing (requires VPN in US) -2. **Async Storage** - Implemented but under-tested -3. **Type Checking** - Reduce the remaining suppressed ty rules - -### Nice-to-Have -1. WebSocket streaming (Phase 2) -2. Parallel symbol updates -3. Time-based futures rolling with expiration calendar - -### Won't Fix -1. Alpha Vantage - Free tier too limited (25 calls/day) -2. Python <3.9 - Using modern syntax - ---- - -## Performance Notes - -### Best Practices -```python -# Good ✅ - Reuse provider (5ms overhead per call) -provider = YahooFinanceProvider() -for symbol in symbols: - data = provider.fetch_ohlcv(symbol, start, end) - -# Bad ❌ - New instance per call (35ms overhead per call) -for symbol in symbols: - provider = YahooFinanceProvider() # Don't do this! - data = provider.fetch_ohlcv(symbol, start, end) -``` - -### Benchmark Results -- **Provider Init**: 30-35ms (httpx.Client setup) -- **Per-call Overhead**: ~5ms (negligible) -- **Storage Query**: Hive storage prunes partitions for date-filtered reads -- **Metadata Lookup**: O(1) efficiency - ---- - -## Integration Points - -### Uses ml4t-data -- `ml4t-features` - Loads OHLCV for feature engineering -- `ml4t-eval` - Validates backtest data -- `ml4t-backtest` - Gets historical prices -- `ml4t-book` (third edition) - 6 example notebooks - -### Coordinate Breaking Changes With -- Parent-level context in the `ml4t/libraries/` workspace -- Multi-library workflows coordinated at the workspace level - ---- - -## Testing - -**Run Tests**: -```bash -# Unit tests only (fast, ~1 min) -pytest - -# With coverage -pytest --cov=src/ml4t/data --cov-report=html - -# Integration tests (requires API keys, slow, ~15 min) -pytest tests/integration/ -v -s - -# Specific provider -pytest tests/integration/test_yahoo.py -v - -# Specific test -pytest tests/test_config.py::test_storage_config -v -``` - -**Test Markers**: -- `@pytest.mark.slow` - Slow tests (excluded by default) -- `@pytest.mark.paid_tier` - Requires paid API tier -- `@pytest.mark.integration` - External API calls -- `@pytest.mark.requires_api_key` - API key needed - ---- - -## Configuration Example - -```yaml -# ml4t-data.yaml -storage: - path: ~/ml4t-data - backend: hive - -datasets: - # Free tier (unlimited) - sp500_daily: - provider: yahoo - symbols_file: config/symbols/sp500.txt - frequency: daily - update_strategy: incremental - - # Free tier (20 calls/day) - nasdaq100_daily: - provider: eodhd - symbols_file: config/symbols/nasdaq100.txt - frequency: daily - - # Crypto - major_crypto: - provider: binance - symbols: - - BTC - - ETH - - SOL - frequency: daily -``` - ---- - -## Common Commands - -### Development -```bash -# Format code -ruff format . - -# Lint -ruff check . - -# Type check -uv run ty check - -# Pre-commit hooks -pre-commit run --all-files - -# Build -uv build - -# Install locally -pip install -e . -``` - -### Data Operations -```bash -# Fetch and save -ml4t-data fetch -s BTC -s ETH --provider cryptocompare \ - --start 2024-01-01 --end 2024-12-31 -o crypto.parquet - -# Update daily from config -ml4t-data update-all -c ml4t-data.yaml --dry-run -ml4t-data update-all -c ml4t-data.yaml - -# List what's stored -ml4t-data list -c ml4t-data.yaml - -# Export to different format -ml4t-data export -s AAPL --output data.csv -ml4t-data export -s AAPL --output data.xlsx -``` - ---- - -## Documentation Links - -- **Full Inventory**: See comprehensive feature list above -- **README**: `../README.md` -- **Project Map**: See AGENTS.md files at each package level -- **Book Integration**: 6 example notebooks (ML4T third edition) -- **Performance Analysis**: `PERFORMANCE_BENCHMARKS.md` and `PERFORMANCE_ANALYSIS.md` - ---- - -**Generated**: 2025-11-27 -**Library Status**: Pre-release (0.1.0) -**Confidence Level**: HIGH (comprehensive codebase analysis) diff --git a/docs/INSTALLATION.md b/docs/INSTALLATION.md deleted file mode 100644 index fd51914c..00000000 --- a/docs/INSTALLATION.md +++ /dev/null @@ -1,304 +0,0 @@ -# ML4T Data Installation Guide - -## Quick Start - -### Standard Installation - -```bash -# Clone or navigate to ml4t-data -cd /path/to/ml4t-data - -# Create virtual environment -python3 -m venv .venv - -# Activate virtual environment -source .venv/bin/activate # Linux/Mac -# or -.venv\Scripts\activate # Windows - -# Install ml4t-data in editable mode with all dependencies -pip install -e . -``` - -### Verify Installation - -```bash -# Check ml4t-data CLI is available -ml4t-data --version - -# Verify session management dependencies -python -c "import pandas_market_calendars; print(f'✓ pandas-market-calendars {pandas_market_calendars.__version__}')" - -# Check providers -ml4t-data providers -``` - -## Dependencies - -ML4T Data has several categories of dependencies: - -### Core Dependencies (Always Installed) - -These are installed automatically with `pip install -e .`: - -- **Data Processing**: polars, pandas, numpy -- **HTTP/Networking**: httpx, tenacity, pybreaker -- **Configuration**: pyyaml, click, python-dotenv, pydantic-settings -- **Utilities**: structlog, platformdirs, filelock, rich -- **Session Management**: pandas-market-calendars (≥4.3.0) - -### Provider-Specific Dependencies (Optional) - -Install only the providers you need: - -```bash -# Yahoo Finance -pip install -e ".[yahoo]" - -# DataBento -pip install -e ".[databento]" - -# OANDA -pip install -e ".[oanda]" - -# CryptoCompare -pip install -e ".[cryptocompare]" - -# EODHD -pip install -e ".[eodhd]" - -# Binance -pip install -e ".[binance]" - -# Install multiple providers -pip install -e ".[yahoo,databento,cryptocompare]" - -# Install ALL providers -pip install -e ".[all]" -``` - -### Development Dependencies (Optional) - -For contributing to ml4t-data: - -```bash -pip install -e ".[dev]" -``` - -Includes the pytest test stack, ruff, ty, pre-commit, package verification tools, and the -provider SDKs used by the test suite. - -## Updating Installation - -If you've already installed ml4t-data but the venv is missing new dependencies (like `pandas-market-calendars`), you need to reinstall: - -```bash -# Option 1: Reinstall in editable mode (preserves existing packages) -pip install -e . --force-reinstall --no-deps -pip install -e . - -# Option 2: Recreate venv from scratch (clean slate) -rm -rf .venv -python3 -m venv .venv -source .venv/bin/activate -pip install -e . -``` - -## Session Management Requirement - -**Important**: Session management features (session date assignment, session completion) require `pandas-market-calendars`. - -This is a **core dependency** and should be installed automatically. If you see: - -``` -ModuleNotFoundError: No module named 'pandas_market_calendars' -``` - -Then your installation is incomplete. Reinstall ml4t-data: - -```bash -pip install -e . -``` - -## Platform-Specific Notes - -### Linux/macOS - -Standard installation should work out of the box: - -```bash -python3 -m venv .venv -source .venv/bin/activate -pip install -e . -``` - -### Windows - -Use PowerShell or Command Prompt: - -```powershell -python -m venv .venv -.venv\Scripts\activate -pip install -e . -``` - -### Docker - -If running in Docker container: - -```dockerfile -FROM python:3.12-slim - -WORKDIR /app -COPY . /app - -RUN pip install -e . - -CMD ["ml4t-data", "--help"] -``` - -## Troubleshooting - -### "No module named 'pip'" - -Your venv is broken. Recreate it: - -```bash -rm -rf .venv -python3 -m venv .venv -source .venv/bin/activate -pip install -e . -``` - -### "ml4t-data: command not found" - -The ml4t-data CLI is not in PATH. Either: - -1. Activate the virtual environment: `source .venv/bin/activate` -2. Use the full path: `.venv/bin/ml4t-data` -3. Reinstall: `pip install -e .` - -### "pandas-market-calendars not found" - -This is a core dependency. Reinstall: - -```bash -pip install -e . --force-reinstall -``` - -### Import Errors for Optional Dependencies - -If you see errors like: - -``` -ImportError: yfinance is required for YahooProvider -``` - -Install the provider-specific dependencies: - -```bash -pip install -e ".[yahoo]" -``` - -### Slow Installation - -If `pip install` is slow, try using a faster resolver: - -```bash -pip install -e . --use-feature=fast-deps -``` - -Or use a local mirror/cache: - -```bash -pip install -e . --no-index --find-links=/path/to/packages -``` - -## Verifying Your Installation - -Run this verification script: - -```bash -python -c " -import sys -print(f'Python: {sys.version}') - -try: - import polars - print(f'✓ polars {polars.__version__}') -except ImportError as e: - print(f'✗ polars: {e}') - -try: - import pandas_market_calendars as mcal - print(f'✓ pandas-market-calendars {mcal.__version__}') -except ImportError as e: - print(f'✗ pandas-market-calendars: {e}') - -try: - from ml4t-data import DataManager - print('✓ ml4t-data.DataManager') -except ImportError as e: - print(f'✗ ml4t-data.DataManager: {e}') - -try: - from ml4t.data.sessions import SessionAssigner - print('✓ ml4t-data.sessions.SessionAssigner') -except ImportError as e: - print(f'✗ ml4t-data.sessions.SessionAssigner: {e}') - -print('\n✓ All core dependencies installed correctly') -" -``` - -Expected output: - -``` -Python: 3.12.x -✓ polars 0.20.x -✓ pandas-market-calendars 4.3.x -✓ ml4t-data.DataManager -✓ ml4t-data.sessions.SessionAssigner - -✓ All core dependencies installed correctly -``` - -## Next Steps - -After installation: - -1. **Configure API keys** (for providers that need them): - ```bash - export DATABENTO_API_KEY="your_key" - export CRYPTOCOMPARE_API_KEY="your_key" - ``` - -2. **Test the CLI**: - ```bash - ml4t-data --help - ml4t-data providers - ml4t-data fetch --symbol BTC --start 2024-01-01 --end 2024-12-31 - ``` - -3. **Run examples**: - ```bash - python examples/wyden_cme_sessions_complete_workflow.py - python examples/nasdaq_bars_sessions.py - ``` - -4. **Read documentation**: - - `README.md` - Overview and quick start - - `docs/SESSION_MANAGEMENT.md` - Session date assignment - - `docs/PROVIDERS.md` - Provider-specific guides - -## Support - -If installation issues persist: - -1. Check Python version: `python --version` (requires ≥3.9) -2. Check pip version: `pip --version` (update with `pip install --upgrade pip`) -3. Try creating a fresh venv in a different location -4. Check for conflicting system packages -5. Review error messages for specific missing dependencies - -For provider-specific issues, see `docs/PROVIDERS.md`. diff --git a/docs/INTEGRATION_TESTING.md b/docs/INTEGRATION_TESTING.md deleted file mode 100644 index e67e0944..00000000 --- a/docs/INTEGRATION_TESTING.md +++ /dev/null @@ -1,241 +0,0 @@ -# ML4T Data Integration Testing Guide - -## Overview - -The ML4T Data integration testing suite provides comprehensive validation of all data providers against real APIs while optimizing for minimal costs and maximum coverage. - -## Test Architecture - -### Test Categories - -1. **Unit Tests** - Mock-based tests for individual components -2. **Integration Tests** - Real API validation with minimal data -3. **Performance Tests** - Baseline measurements and regression detection -4. **Expensive Tests** - Full dataset validation (nightly/manual only) - -### Provider Coverage - -| Provider | API Type | Cost Model | Test Strategy | -|----------|----------|------------|---------------| -| CryptoCompare | REST | Free tier (10 req/sec) | Single day, 2 symbols | -| Databento | REST | Pay per request | Daily bars, specific contracts | -| OANDA | REST | Free practice account | Hourly bars, major pairs | - -## Setup - -### Local Development - -1. **Install dependencies**: -```bash -uv pip install -e . -uv pip install pytest pytest-cov pytest-asyncio -``` - -2. **Configure API keys**: -```bash -cp .env.example .env -# Edit .env and add your API keys -source .env -``` - -3. **Run setup script**: -```bash -./scripts/setup_test_env.sh -``` - -### CI/CD Configuration - -GitHub Actions workflow automatically: -- Runs minimal tests on PRs -- Runs full suite on main branch -- Runs expensive tests nightly -- Tracks performance baselines - -## Running Tests - -### Quick Test Commands - -```bash -# Run all integration tests -pytest tests/integration/ -v - -# Run minimal tests (CI mode) -CI=true pytest tests/integration/ -v -m "integration and not expensive" - -# Run specific provider -pytest tests/integration/ -v -k cryptocompare - -# Run with coverage -pytest tests/integration/ -v --cov=ml4t-data.providers --cov-report=html - -# Run performance benchmarks -pytest tests/integration/test_real_api_integration.py::TestRealAPIIntegration::test_performance_baselines -v -``` - -### Test Markers - -- `@pytest.mark.integration` - Requires real API access -- `@pytest.mark.expensive` - High API costs (skip in CI) -- `@pytest.mark.skipif` - Conditional execution based on API keys - -## Cost Optimization - -### Strategies - -1. **Data Minimization** - - Single day requests (1 bar) for basic tests - - Maximum 3-day ranges for multi-day tests - - Limited symbol sets (2-3 per provider) - -2. **Smart Test Selection** - - Skip expensive tests in CI - - Run full suite only on main branch - - Nightly runs for comprehensive validation - -3. **Caching** - - Response caching for repeated requests - - Fixture reuse across test sessions - - Performance baseline storage - -### Cost Estimates - -| Provider | Test Type | Requests | Estimated Cost | -|----------|-----------|----------|----------------| -| CryptoCompare | Minimal | 5 | $0 (free tier) | -| CryptoCompare | Full | 20 | $0 (free tier) | -| Databento | Minimal | 3 | ~$0.01 | -| Databento | Full | 10 | ~$0.05 | -| OANDA | All | Unlimited | $0 (practice) | - -**Monthly CI Estimate**: < $5 with nightly runs - -## Performance Baselines - -### Target Metrics - -| Provider | Avg Response | Max Response | Throughput | -|----------|--------------|--------------|------------| -| CryptoCompare | < 2s | < 10s | 10 req/s | -| Databento | < 3s | < 15s | 5 req/s | -| OANDA | < 1s | < 5s | 120 req/s | - -### Monitoring - -Performance is tracked via: -- Test execution times -- API response latencies -- Memory usage patterns -- Error rates - -## Troubleshooting - -### Common Issues - -1. **Missing API Keys** - - Check `.env` file exists - - Verify key format (Databento starts with "db-") - - Source environment: `source .env` - -2. **Rate Limiting** - - Built-in rate limiters should prevent this - - If occurs, check provider limits - - Reduce parallel test execution - -3. **Empty Responses** - - Weekend/holiday dates return empty - - Invalid symbols return empty - - Check provider documentation - -4. **Test Failures** - - Verify API keys are valid - - Check network connectivity - - Review provider status pages - -### Debug Mode - -```bash -# Verbose output with logging -pytest tests/integration/ -vv -s --log-cli-level=DEBUG - -# Run single test -pytest tests/integration/test_real_api_integration.py::TestRealAPIIntegration::test_cryptocompare_integration -v - -# Capture warnings -pytest tests/integration/ -v -W default -``` - -## CI/CD Integration - -### GitHub Actions Secrets - -Required secrets: -``` -ALPACA_API_KEY -ALPACA_API_SECRET -CRYPTOCOMPARE_API_KEY -DATABENTO_API_KEY -OANDA_API_KEY -``` - -### Workflow Triggers - -- **Push to main**: Full integration tests -- **Pull Request**: Minimal tests only -- **Schedule**: Nightly expensive tests -- **Manual**: On-demand full suite - -### Test Reports - -- Coverage reports uploaded as artifacts -- Performance benchmarks stored -- API usage tracked in logs - -## Best Practices - -1. **Always use minimal data** for basic validation -2. **Mark expensive tests** with appropriate decorator -3. **Handle empty responses** gracefully -4. **Log API usage** for cost tracking -5. **Cache responses** where appropriate -6. **Document known limitations** -7. **Update baselines** regularly - -## Contributing - -When adding new providers: - -1. Implement provider class inheriting from `BaseProvider` -2. Add integration tests following existing patterns -3. Document API costs and limits -4. Update CI configuration if needed -5. Add to cost optimization strategy - -## Monitoring & Alerts - -### Metrics to Track - -- Test execution time trends -- API error rates -- Coverage percentages -- Cost per test run - -### Alert Thresholds - -- Coverage drops below 60% -- Performance degrades >50% -- API errors exceed 5% -- Monthly costs exceed $10 - -## Future Enhancements - -- [ ] Response caching layer -- [ ] Parallel test execution optimization -- [ ] Historical performance tracking -- [ ] Automated cost reporting -- [ ] Provider health monitoring -- [ ] Test data generation tools - ---- - -*Last updated: 2025-08-28* -*Maintained by: Stefan Jansen * diff --git a/docs/SESSION_MANAGEMENT.md b/docs/SESSION_MANAGEMENT.md deleted file mode 100644 index 7685ef1a..00000000 --- a/docs/SESSION_MANAGEMENT.md +++ /dev/null @@ -1,528 +0,0 @@ -# Session Management in ML4T Data - -Complete guide to assigning and using session dates for futures and intraday data. - -## Table of Contents - -1. [Overview](#overview) -2. [Why Session Dates Matter](#why-session-dates-matter) -3. [Use Cases](#use-cases) -4. [Supported Exchanges](#supported-exchanges) -5. [Quick Start](#quick-start) -6. [Working with ML4T Data Storage](#working-with-ml4t-data-storage) -7. [Standalone Session Assignment](#standalone-session-assignment) -8. [Cross-Validation with Sessions](#cross-validation-with-sessions) -9. [Complete Examples](#complete-examples) - ---- - -## Overview - -Session management in ml4t-data provides tools for: -- **Assigning session dates** to intraday data based on exchange calendars -- **Completing sessions** by filling gaps in minute-level data -- **Cross-validation** that respects session boundaries (prevents data leakage) - -### What is a Session Date? - -A **session date** is the calendar date assigned to each trading bar based on the exchange's trading hours. - -**Key characteristics**: -- For CME futures: Session starts Sunday 5pm CT, ends Friday 4pm CT (23 hours/day) -- Session date = **date when the session ENDS** (e.g., Monday 4pm CT → session_date = Monday) -- Overnight bars (e.g., Monday 11pm CT) still belong to Monday's session -- Critical for time-series cross-validation to avoid data leakage - ---- - -## Why Session Dates Matter - -### Problem: Naive Date-Based Splits Cause Data Leakage - -```python -# ❌ WRONG: Using calendar date creates leakage -df_train = df.filter(pl.col("timestamp").dt.date() < "2024-06-01") -df_test = df.filter(pl.col("timestamp").dt.date() >= "2024-06-01") - -# Problem: Monday's session starts Sunday 5pm -# → Sunday evening bars (5pm-11:59pm) are in df_test -# → But Monday morning bars (12am-4pm) are in df_train -# → Same trading session is split across train/test! -``` - -### Solution: Session-Based Splits Prevent Leakage - -```python -# ✅ CORRECT: Using session_date keeps sessions intact -df_train = df.filter(pl.col("session_date") < "2024-06-01") -df_test = df.filter(pl.col("session_date") >= "2024-06-01") - -# Result: All bars from same trading session stay together -# → Sunday 5pm → Monday 4pm (all marked as session_date=Monday) -# → No overlap between train and test sessions -``` - ---- - -## Use Cases - -### 1. Time-Series Cross-Validation - -Use session dates with `GroupKFold` to prevent data leakage: - -```python -from sklearn.model_selection import GroupKFold - -# Each session stays in one fold -gkf = GroupKFold(n_splits=5) -for train_idx, test_idx in gkf.split(X, y, groups=df["session_date"]): - # Train and test contain different sessions (no overlap) - pass -``` - -### 2. Session-Level Analysis - -Aggregate metrics by trading session: - -```python -session_stats = df.group_by("session_date").agg([ - pl.col("volume").sum().alias("daily_volume"), - pl.col("close").last().alias("session_close"), - (pl.col("close").last() - pl.col("close").first()).alias("session_return") -]) -``` - -### 3. Gap Filling for Complete Sessions - -CME futures have 1-hour daily maintenance breaks (4pm-5pm CT). Fill gaps: - -```python -df_complete = manager.complete_sessions( - df, - exchange="CME", - fill_gaps=True, # Fill missing minutes - zero_volume=True # Mark filled bars with volume=0 -) -``` - -### 4. Feature Engineering - -Build features that respect session boundaries: - -```python -# Rolling features within sessions (don't cross session boundaries) -df = df.with_columns([ - pl.col("close") - .rolling_mean(window_size=60) - .over("session_date") - .alias("close_ma60_session") -]) -``` - ---- - -## Supported Exchanges - -ML4T Data uses [pandas_market_calendars](https://github.com/rsheftel/pandas_market_calendars) for exchange calendars: - -| Exchange | Code | Calendar Name | Trading Hours (Local Time) | -|----------|------|---------------|----------------------------| -| CME (Globex Crypto) | CME | CME_Globex_Crypto | Sun 5pm - Fri 4pm CT (23h/day) | -| CME (Equity) | CME | CME_Equity | Sun 5pm - Fri 4pm CT (23h/day) | -| NASDAQ | NASDAQ | NASDAQ | 9:30am - 4:00pm ET | -| NYSE | NYSE | NYSE | 9:30am - 4:00pm ET | -| LSE | LSE | LSE | 8:00am - 4:30pm GMT | -| Tokyo | TSE | TSE | 9:00am - 3:00pm JST | -| Hong Kong | HKEX | HKEX | 9:30am - 4:00pm HKT | - -**Note**: CME futures have two calendar types: -- `CME_Globex_Crypto`: For BTC/ETH futures -- `CME_Equity`: For equity index futures (ES, NQ, etc.) - ---- - -## Quick Start - -### Method 1: Using DataManager (Recommended for ml4t-data storage) - -```python -from ml4t-data import DataManager -from ml4t.data.storage.backend import StorageConfig -from ml4t.data.storage.hive import HiveStorage - -# Initialize with your storage -storage = HiveStorage(config=StorageConfig(base_path="./data")) -manager = DataManager(storage=storage) - -# Load data -df = storage.read("crypto_futures_1min_BTC").collect() - -# Assign sessions -df_with_sessions = manager.assign_sessions(df, exchange="CME") - -# Complete sessions (fill gaps) -df_complete = manager.complete_sessions( - df_with_sessions, - exchange="CME", - fill_gaps=True, - zero_volume=True -) -``` - -### Method 2: Using SessionAssigner Directly - -```python -from ml4t.data.sessions import SessionAssigner -import polars as pl - -# Load your data -df = pl.read_parquet("my_data.parquet") - -# Assign sessions -assigner = SessionAssigner.from_exchange("CME") -df_with_sessions = assigner.assign_sessions(df) -``` - ---- - -## Working with ML4T Data Storage - -### Example: Wyden CME Crypto Futures - -Complete workflow for production data: - -```python -from pathlib import Path -import polars as pl -from ml4t-data import DataManager -from ml4t.data.storage.backend import StorageConfig -from ml4t.data.storage.hive import HiveStorage - -# Wyden data location -DATA_PATH = Path.home() / "clients/wyden/long-short/data" - -# Initialize -storage = HiveStorage(config=StorageConfig(base_path=str(DATA_PATH))) -manager = DataManager(storage=storage) - -# Read BTC futures -df = storage.read("crypto_futures_1min_BTC").collect() -print(f"Loaded {len(df):,} rows") # 2.79M rows - -# Assign CME sessions -df_with_sessions = manager.assign_sessions(df, exchange="CME") -print(f"Assigned {df_with_sessions['session_date'].n_unique():,} sessions") - -# Complete sessions -df_complete = manager.complete_sessions( - df_with_sessions, - exchange="CME", - fill_gaps=True, - zero_volume=True -) -print(f"Filled {len(df_complete) - len(df):,} gaps") - -# Save -df_complete.write_parquet("btc_futures_with_sessions.parquet") -``` - -**Full script**: `examples/wyden_cme_sessions_complete_workflow.py` - ---- - -## Standalone Session Assignment - -For data NOT stored in ml4t-data (external parquet/CSV files): - -### Command-Line Tool - -```bash -# Assign sessions to any parquet file -python examples/assign_sessions_standalone.py \ - --input nq_bars.parquet \ - --output nq_bars_with_sessions.parquet \ - --exchange CME \ - --timestamp-column datetime - -# Overwrite input file -python examples/assign_sessions_standalone.py \ - --input data.parquet \ - --exchange NASDAQ \ - --inplace -``` - -### Python API - -```python -from examples.assign_sessions_standalone import assign_sessions_to_file - -df = assign_sessions_to_file( - input_path="nq_bars.parquet", - output_path="nq_bars_with_sessions.parquet", - exchange="CME", - timestamp_column="datetime" -) - -print(df.select(["datetime", "session_date"]).head()) -``` - ---- - -## Cross-Validation with Sessions - -### GroupKFold (Recommended) - -**Why**: Ensures each session appears in exactly one fold. - -```python -from sklearn.model_selection import GroupKFold -import numpy as np - -# Load data with sessions -df = pl.read_parquet("data_with_sessions.parquet") - -# Prepare data -X = df.select([...feature_columns...]).to_numpy() -y = df["target"].to_numpy() -groups = df["session_date"].to_numpy() - -# Create folds -gkf = GroupKFold(n_splits=5) - -for fold, (train_idx, test_idx) in enumerate(gkf.split(X, y, groups)): - # Get session info - train_sessions = np.unique(groups[train_idx]) - test_sessions = np.unique(groups[test_idx]) - - # Verify no overlap - assert len(set(train_sessions) & set(test_sessions)) == 0 - - print(f"Fold {fold + 1}:") - print(f" Train: {len(train_sessions)} sessions, {len(train_idx):,} bars") - print(f" Test: {len(test_sessions)} sessions, {len(test_idx):,} bars") -``` - -### Time-Series Split with Sessions - -```python -from sklearn.model_selection import TimeSeriesSplit - -# Sort by session date -df = df.sort("session_date") - -# Get unique sessions -unique_sessions = df["session_date"].unique().sort() - -# Split sessions (not individual bars) -tscv = TimeSeriesSplit(n_splits=5) - -for fold, (train_sessions_idx, test_sessions_idx) in enumerate(tscv.split(unique_sessions)): - train_sessions = unique_sessions[train_sessions_idx] - test_sessions = unique_sessions[test_sessions_idx] - - # Filter data by sessions - train_data = df.filter(pl.col("session_date").is_in(train_sessions)) - test_data = df.filter(pl.col("session_date").is_in(test_sessions)) - - print(f"Fold {fold + 1}:") - print(f" Train: {len(train_sessions)} sessions, {len(train_data):,} bars") - print(f" Test: {len(test_sessions)} sessions, {len(test_data):,} bars") -``` - ---- - -## Complete Examples - -### 1. CME Crypto Futures (Time Bars) - -**File**: `examples/wyden_cme_sessions_complete_workflow.py` - -```bash -python examples/wyden_cme_sessions_complete_workflow.py -``` - -Demonstrates: -- Loading data from ml4t-data storage -- Assigning CME sessions -- Completing sessions (filling gaps) -- Session statistics -- Cross-validation setup - -### 2. NASDAQ 100 Futures (Trade/Volume Bars) - -**File**: `examples/nasdaq_bars_sessions.py` - -```bash -python examples/nasdaq_bars_sessions.py -``` - -Demonstrates: -- Working with bar data (not time-based) -- Irregular timestamps -- Session assignment for external data -- Cross-validation with bar data - -### 3. Standalone CLI Tool - -**File**: `examples/assign_sessions_standalone.py` - -```bash -# Help -python examples/assign_sessions_standalone.py --help - -# Assign sessions -python examples/assign_sessions_standalone.py \ - --input ~/clients/chimera/bias_strategy/data/processed/nqu25_processed.parquet \ - --output ~/clients/chimera/bias_strategy/data/processed/nqu25_with_sessions.parquet \ - --exchange CME \ - --timestamp-column datetime -``` - ---- - -## API Reference - -### DataManager Methods - -#### `assign_sessions()` - -```python -def assign_sessions( - self, - df: pl.DataFrame, - exchange: str | None = None, - calendar: str | None = None, -) -> pl.DataFrame: - """Assign session_date column to DataFrame. - - Args: - df: DataFrame with timestamp column - exchange: Exchange code (CME, NYSE, NASDAQ, etc.) - calendar: Calendar name override (e.g., "CME_Globex_Crypto") - - Returns: - DataFrame with session_date column added - """ -``` - -#### `complete_sessions()` - -```python -def complete_sessions( - self, - df: pl.DataFrame, - exchange: str | None = None, - fill_gaps: bool = True, - zero_volume: bool = True, -) -> pl.DataFrame: - """Complete trading sessions by filling gaps. - - Args: - df: DataFrame with timestamp and session_date columns - exchange: Exchange code (CME, NYSE, etc.) - fill_gaps: Fill missing minutes within sessions - zero_volume: Set volume=0 for filled bars - - Returns: - DataFrame with complete sessions - """ -``` - -### SessionAssigner Class - -```python -from ml4t.data.sessions import SessionAssigner - -# From exchange code -assigner = SessionAssigner.from_exchange("CME") - -# From calendar name -assigner = SessionAssigner("CME_Globex_Crypto") - -# Assign sessions -df_with_sessions = assigner.assign_sessions( - df, - start_date="2024-01-01", # Optional - end_date="2024-12-31" # Optional -) -``` - ---- - -## Troubleshooting - -### Missing pandas_market_calendars - -**Error**: `ModuleNotFoundError: No module named 'pandas_market_calendars'` - -**Solution**: -```bash -pip install pandas-market-calendars -``` - -### Wrong Calendar for CME Futures - -**Problem**: Using wrong calendar results in incorrect sessions. - -**Solution**: Use the right calendar: -- Crypto futures (BTC, ETH): `CME_Globex_Crypto` -- Equity futures (ES, NQ): `CME_Equity` - -```python -# Correct for BTC/ETH -assigner = SessionAssigner("CME_Globex_Crypto") - -# Correct for NQ -assigner = SessionAssigner("CME_Equity") -``` - -### Timestamp Column Not Found - -**Error**: `ValueError: DataFrame must have 'timestamp' column` - -**Solution**: Rename your timestamp column: -```python -df = df.rename({"datetime": "timestamp"}) -df_with_sessions = assigner.assign_sessions(df) -df_with_sessions = df_with_sessions.rename({"timestamp": "datetime"}) -``` - -### Session Dates are NULL - -**Problem**: All session_date values are NULL. - -**Possible causes**: -1. Data outside calendar date range -2. Timestamps are not timezone-aware -3. Wrong calendar for exchange - -**Solution**: Check your data: -```python -print(df["timestamp"].min(), df["timestamp"].max()) -print(df["timestamp"].dtype) -``` - ---- - -## Additional Resources - -- **pandas_market_calendars**: https://github.com/rsheftel/pandas_market_calendars -- **CME Trading Hours**: https://www.cmegroup.com/markets/equities/sp/e-mini-sandp500.html -- **Cross-Validation Best Practices**: See ml4t-data docs on time-series CV - ---- - -## Summary - -**Key Takeaways**: -1. ✅ Use session dates for cross-validation to prevent data leakage -2. ✅ Session date = date when session ENDS (e.g., Monday 4pm CT → Monday) -3. ✅ Complete sessions by filling gaps (CME has 1-hour maintenance breaks) -4. ✅ Use GroupKFold with session_date as groups -5. ✅ Works with any data: time bars, volume bars, trade bars - -**Next Steps**: -1. Run example scripts to understand workflows -2. Assign sessions to your data -3. Implement proper cross-validation with session groups -4. Build session-aware features for better models diff --git a/docs/crypto_providers.md b/docs/crypto_providers.md deleted file mode 100644 index 08cfb53c..00000000 --- a/docs/crypto_providers.md +++ /dev/null @@ -1,161 +0,0 @@ -# Cryptocurrency Data Providers - -ML4T Data supports multiple cryptocurrency data providers for fetching historical and real-time market data. - -## Available Providers - -### CryptoCompare -- **Type**: Free tier available (100,000 calls/month) -- **Markets**: Spot prices from multiple exchanges -- **Data**: OHLCV data with aggregated pricing -- **Frequencies**: minute, hourly, daily -- **API Key**: Required (free account available) - -### Binance -- **Type**: Free (no API key required for public data) -- **Markets**: Spot and Futures -- **Data**: OHLCV data directly from Binance -- **Frequencies**: 1m, 3m, 5m, 15m, 30m, 1h, 2h, 4h, 6h, 8h, 12h, 1d, 3d, 1w, 1M -- **Rate Limits**: 1200 weight per minute - -## Usage Examples - -### Loading Bitcoin Daily Data - -```bash -# Using CryptoCompare -ml4t-data load --provider cryptocompare --symbol BTC --start 2024-01-01 --end 2024-01-31 --frequency daily --asset-class crypto - -# Using Binance Spot -ml4t-data load --provider binance --symbol BTC --start 2024-01-01 --end 2024-01-31 --frequency daily --asset-class crypto - -# Using Binance Futures -ml4t-data load --provider binance_futures --symbol BTC --start 2024-01-01 --end 2024-01-31 --frequency daily --asset-class crypto -``` - -### Loading Minute-Level Data - -```bash -# Fetch minute data for Ethereum -ml4t-data load --provider binance --symbol ETH --start 2024-01-01 --end 2024-01-01 --frequency minute --asset-class crypto -``` - -### Incremental Updates - -```bash -# Update existing crypto data -ml4t-data update --symbol BTC --frequency daily --asset-class crypto --lookback-days 7 -``` - -## Symbol Formats - -The providers automatically normalize various symbol formats: - -- `BTC` → `BTCUSDT` (Binance) or `BTC/USD` (CryptoCompare) -- `BTC-USD` → normalized appropriately -- `BTC/USD` → normalized appropriately -- `BTCUSDT` → used directly for Binance - -## Python API - -```python -from ml4t.data.providers.cryptocompare import CryptoCompareProvider -from ml4t.data.providers.binance import BinanceProvider -from ml4t.data.pipeline import Pipeline -from ml4t.data.storage.filesystem import FileSystemBackend -from ml4t.data.core.config import Config - -# Initialize provider -provider = BinanceProvider(market="spot") - -# Initialize storage and config -storage = FileSystemBackend(data_root="./data") -config = Config() - -# Create pipeline -pipeline = Pipeline(provider=provider, storage=storage, config=config) - -# Load data -key = pipeline.run_load( - symbol="BTC", - start="2024-01-01", - end="2024-01-31", - frequency="daily", - asset_class="crypto" -) - -# Read the data -data_obj = storage.read(key) -df = data_obj.data # Polars DataFrame -``` - -## 24/7 Market Considerations - -Cryptocurrency markets operate 24/7, unlike traditional equity markets. The ML4T Data system handles this by: - -1. **No Weekend Gaps**: Gap detection doesn't flag weekends as missing data for crypto -2. **UTC Timezone**: All crypto data uses UTC timestamps for consistency -3. **Extended Hours**: For intraday data, the end date is extended to 23:59:59 to capture the full day - -## Rate Limiting - -### CryptoCompare -- Free tier: 100,000 calls/month -- Automatic rate limiting with delays between requests -- Reads the required key from `CRYPTOCOMPARE_API_KEY` - -### Binance -- Public endpoints: 1200 weight per minute -- Automatic retry on rate limit (429) errors -- Built-in delays between requests (0.05s) - -## Data Storage - -Crypto data is stored in the same format as other asset classes: - -``` -data/ -├── crypto/ -│ ├── daily/ -│ │ ├── BTC.parquet -│ │ ├── ETH.parquet -│ │ └── ... -│ ├── minute/ -│ │ ├── BTC.parquet -│ │ └── ... -│ └── hourly/ -│ └── ... -``` - -## Metadata - -Each dataset includes metadata about: -- Provider used (e.g., "binance_spot", "cryptocompare") -- Symbol normalization -- Date range -- Update history -- Data quality metrics - -## Error Handling - -The providers handle common errors: -- Invalid symbols return empty DataFrames -- Rate limit errors trigger automatic retries with backoff -- Network errors are retried with exponential backoff -- API errors are logged with clear messages - -## Best Practices - -1. **Use appropriate frequencies**: Minute data generates large files; use daily/hourly when possible -2. **Respect rate limits**: Don't make parallel requests to the same provider -3. **Store API keys securely**: Use environment variables or secure configuration -4. **Monitor data quality**: Check for gaps and stale data regularly -5. **Use incremental updates**: After initial load, use `ml4t-data update` for efficiency - -## Future Enhancements - -- [ ] WebSocket support for real-time data -- [ ] Additional exchanges (Coinbase, Kraken, etc.) -- [ ] Order book data -- [ ] Tick-level data support -- [ ] Cross-exchange arbitrage metrics diff --git a/docs/export_guide.md b/docs/export_guide.md deleted file mode 100644 index 87cc43bd..00000000 --- a/docs/export_guide.md +++ /dev/null @@ -1,282 +0,0 @@ -# Data Export Guide - -ML4T Data provides flexible data export functionality to convert stored market data into various formats for analysis and integration with other tools. - -## Supported Formats - -- **CSV** - Comma-separated values for spreadsheet applications -- **JSON** - JavaScript Object Notation for web applications and APIs -- **Excel** - Microsoft Excel format with multiple sheet support - -## CLI Usage - -### Basic Export - -Export a single dataset to CSV: - -```bash -ml4t-data export \ - --symbol equities/daily/AAPL \ - --output ./exports/aapl.csv \ - --format csv \ - --storage-path ./data -``` - -The CLI supports CSV, JSON, and Parquet for one storage key. Use the Python API for -compression, transformations, batch export, patterns, and Excel workbooks. - -## Python API - -### Basic Export - -```python -from pathlib import Path - -from ml4t.data.export.manager import ExportManager -from ml4t.data.storage import create_storage - -# Initialize -storage = create_storage("./data", strategy="hive") -manager = ExportManager(storage=storage) -Path("./exports").mkdir(exist_ok=True) - -# Export single dataset -result = manager.export( - key="equities/daily/AAPL", - output_path="./exports/AAPL.csv", - format_type="csv" -) - -if result.success: - print(f"Exported to: {result.output_path}") - print(f"Rows: {result.rows_exported}") -``` - -### Batch Export - -```python -# Export multiple datasets to Excel -results = manager.export_batch( - keys=["equities/daily/AAPL", "equities/daily/GOOGL"], - output_path="./report.xlsx", - format_type="excel" -) -``` - -### Export with Transformations - -```python -# Export with filters and calculations -result = manager.export( - key="crypto/daily/BTC", - output_path="./btc_analysis.csv", - format_type="csv", - date_filter=("2024-01-01", "2024-03-31"), - columns=["timestamp", "close", "volume"], - add_returns=True, - add_volatility=True -) -``` - -### Pattern-Based Export - -```python -# Export all matching datasets -results = manager.export_pattern( - pattern="equities/daily/*", - output_path="./exports/", - format_type="csv" -) - -print(f"Exported {len(results)} datasets") -``` - -Requested columns are written in the supplied order. An export fails explicitly if any -requested column is absent. Pattern matching uses shell-style `*`, `?`, and bracket -expressions against complete storage keys. - -## Export Formats Details - -### CSV Format - -- Human-readable text format -- Compatible with Excel, Google Sheets, pandas -- Supports compression (gzip) -- One file per dataset - -Example output: -```csv -timestamp,open,high,low,close,volume -2024-01-01T00:00:00,100.0,105.0,99.0,104.0,1000000 -2024-01-02T00:00:00,104.0,106.0,103.0,105.0,1100000 -``` - -### JSON Format - -- Structured data format -- Ideal for web applications -- Supports metadata inclusion -- Can combine multiple datasets - -Example output: -```json -{ - "symbol": "AAPL", - "metadata": { - "exported_at": "2024-03-15T10:30:00", - "rows": 252, - "date_range": { - "start": "2024-01-01T00:00:00", - "end": "2024-12-31T00:00:00" - } - }, - "data": [ - { - "timestamp": "2024-01-01T00:00:00", - "open": 100.0, - "high": 105.0, - "low": 99.0, - "close": 104.0, - "volume": 1000000 - } - ] -} -``` - -### Excel Format - -- Native Excel format (.xlsx) -- Multiple sheets support -- Automatic formatting -- Metadata sheet included - -Features: -- Each dataset in separate sheet -- Sheet names from symbol names -- Auto-fit columns -- Number formatting - -## Performance Considerations - -### Large Datasets - -For datasets with millions of rows: - -1. Filter by date so storage reads only matching partitions. -2. Select columns so storage projects only required fields. -3. Use gzip compression when the smaller output justifies its additional memory use. - -### Memory Usage - -Date and column filters are applied by production storage before collection. The filtered -result is then held in memory while it is transformed and written. Gzip CSV export also -creates the CSV payload in memory, so narrow the date range and columns before exporting a -large dataset. - -```python -result = manager.export( - key="equities/minute/AAPL", - output_path="./large_export.csv", - format_type="csv", - date_filter=("2024-01-01", "2024-01-31"), - columns=["timestamp", "close", "volume"] -) -``` - -## Error Handling - -Export operations return detailed results: - -```python -result = manager.export(key="data/key", output_path="./out", format_type="csv") - -if result.success: - print(f"Success! Exported {result.rows_exported} rows") - print(f"File size: {result.file_size / 1024 / 1024:.2f} MB") - print(f"Duration: {result.duration_seconds:.2f} seconds") -else: - print(f"Export failed: {result.error}") -``` - -## Best Practices - -1. **Choose appropriate format**: - - CSV for data analysis in Python/R - - Excel for business reports - - JSON for web applications - -2. **Use compression when required**: - ```python - manager.export( - key="equities/minute/AAPL", - output_path="./exports/AAPL.csv.gz", - format_type="csv", - compression="gzip" - ) - ``` - -3. **Filter unnecessary data**: - ```python - manager.export( - key="crypto/minute/BTC", - output_path="./exports/BTC.csv", - format_type="csv", - date_filter=("2024-01-01", "2024-01-31") - ) - ``` - -4. **Batch similar exports**: - ```python - manager.export_pattern( - pattern="equities/daily/*", - output_path="./daily_report.xlsx", - format_type="excel" - ) - ``` - -5. **Validate exports**: - - Check row counts match expectations - - Verify date ranges are correct - - Test with small dataset first - -## Troubleshooting - -### Excel Export Issues - -Excel export uses the included openpyxl dependency. XlsxWriter is optional: - -```bash -uv add xlsxwriter -``` - -### Memory Errors - -For large datasets, filter at storage read time before exporting: - -```python -manager.export( - key="large/dataset", - output_path="./exports/data.csv", - format_type="csv", - date_filter=("2024-01-01", "2024-01-31"), - columns=["timestamp", "close"] -) -``` - -### Permission Errors - -Ensure output directory exists and is writable: - -```bash -mkdir -p ./exports -chmod 755 ./exports -``` - -## Future Enhancements - -- [ ] HDF5 format support -- [ ] Parquet pass-through export -- [ ] Custom date/time formatting -- [ ] Export scheduling -- [ ] Export to cloud storage (S3, GCS) -- [ ] Streaming exports for real-time data diff --git a/docs/providers/crypto.md b/docs/providers/crypto.md index 8419ac2b..83edff78 100644 --- a/docs/providers/crypto.md +++ b/docs/providers/crypto.md @@ -17,7 +17,7 @@ closures, wash trading, and differences between spot, futures, perpetuals, and o | [CryptoCompare API](https://developers.cryptocompare.com/documentation) | Aggregated and exchange-specific crypto market data | REST and streaming APIs | API key for supported usage | Release qualification depends on live credential validation; aggregation methodology matters | `CryptoCompareProvider` | | [Kaiko Market Data](https://www.kaiko.com/products/market-data) | Centralized and decentralized venues, spot and derivatives, trades and order books | API, streaming, and cloud delivery | Institutional license | Venue and instrument history depend on the contracted product | No | | [Tardis.dev historical data](https://docs.tardis.dev/historical-data-details/overview) | Raw and normalized messages for centralized crypto exchanges, including closed venues | API and downloadable files | Commercial plan; limited samples | Reconstruction requires exchange-specific message semantics and snapshot handling | No | -| [CoinAPI Market Data](https://www.coinapi.io/products/market-data-api) | Multi-exchange spot and derivatives market data | REST, WebSocket, FIX, and files | API key; plan-dependent | Normalized symbols and aggregate feeds can hide exchange-specific contract details | No | +| [CoinAPI Market Data](https://www.coinapi.io/products/market-data-api/docs) | Multi-exchange spot and derivatives market data | REST, WebSocket, FIX, and files | API key; plan-dependent | Normalized symbols and aggregate feeds can hide exchange-specific contract details | No | On-chain metrics, developer activity, and social signals belong in the [alternative-data reference](alternative_data.md). This page covers tradable market data. diff --git a/docs/providers/cryptocompare.md b/docs/providers/cryptocompare.md index 11a1bbb2..1b71c3ae 100644 --- a/docs/providers/cryptocompare.md +++ b/docs/providers/cryptocompare.md @@ -3,15 +3,15 @@ **Provider**: `CryptoCompareProvider` **Website**: [cryptocompare.com](https://www.cryptocompare.com) **API Key**: Required -**0.1.0 status**: Included for evaluation; not release-qualified +**0.2.0 status**: Adapter included; live contract unverified --- ## Overview -CryptoCompare account registration was unavailable during the 0.1.0 release review. The adapter is -included for evaluation, but it has no successful live contract evidence for this release. Do not -treat it as a release-qualified provider until a later release records a successful contract run. +The 0.2.0 package includes the adapter and its offline contract tests. Release validation did not +have a configured `CRYPTOCOMPARE_API_KEY`, so no successful live provider contract was recorded. +Verify access with your own account before relying on the adapter for a production dataset. **Best For**: Crypto historical data, alternative to Binance @@ -62,7 +62,8 @@ Get your API key at [cryptocompare.com/cryptopian/api-keys](https://www.cryptoco ## Rate Limits -Consult CryptoCompare's current terms before use. Access and limits were not verified for 0.1.0. +Consult CryptoCompare's current terms before use. Account access and limits were not verified for +the 0.2.0 release. --- diff --git a/docs/providers/databento.md b/docs/providers/databento.md index be1e0867..564d7b95 100644 --- a/docs/providers/databento.md +++ b/docs/providers/databento.md @@ -183,5 +183,5 @@ streaming, use `provider.client` directly. - [Equity](equities.md), [futures](futures.md), and [options](options.md) source references - [Databento Pricing](https://databento.com/pricing) -- [Databento Reference](databento_reference.md) - Detailed schema guide +- [Databento Documentation](https://databento.com/docs) - [Provider reference](index.md) diff --git a/docs/providers/databento_reference.md b/docs/providers/databento_reference.md deleted file mode 100644 index 37846b7c..00000000 --- a/docs/providers/databento_reference.md +++ /dev/null @@ -1,161 +0,0 @@ -# Databento Provider Reference - -## Official Documentation -**Main Documentation**: https://databento.com/docs -**Historical API Reference**: https://databento.com/docs/api-reference-historical/client?historical=python&live=python&reference=python -**Datasets & Venues**: https://databento.com/docs/venues-and-datasets?historical=python&live=python&reference=python -**Schemas & Data Formats**: https://databento.com/docs/schemas-and-data-formats?historical=python&live=python&reference=python - -## Key Concepts - -### Datasets -Databento organizes data by dataset, which typically corresponds to an exchange or data source: -- **XNAS.ITCH**: NASDAQ ITCH feed (equities) -- **GLBX.MDP3**: CME Globex MDP 3.0 (futures) -- **OPRA.PILLAR**: Options Price Reporting Authority -- **DBEQ.BASIC**: Databento Equities Basic - -### Schemas -Data schemas define the structure and granularity of data: -- **ohlcv-1m/1h/1d**: OHLC bars with volume at different frequencies -- **trades**: Individual trade records -- **tbbo**: Top of book bid/offer (quotes) -- **mbo**: Market by order (full order book) -- **mbp-1/10**: Market by price (aggregated order book levels) - -### Symbol Types (stype_in) -- **raw_symbol**: Exchange-native symbols (e.g., "ESH4" for March 2024 E-mini) -- **parent**: Root symbol (e.g., "ES" for all E-mini S&P contracts) -- **continuous**: Continuous contracts (e.g., "ES.v.0" for front month) - -### Continuous Futures Notation -Databento uses `.v.N` notation for continuous contracts: -- `ES.v.0`: Front month E-mini S&P 500 -- `CL.v.1`: Second month Crude Oil -- `NQ.v.0`: Front month E-mini NASDAQ - -### Session Dates for Non-Calendar Trading -Some exchanges have trading sessions that don't align with calendar dates: -- **CME Futures**: Trading day starts at 5:00 PM CT (previous calendar day) -- **Other Futures Markets**: May have different session start times -- **24/7 Markets (Crypto)**: Typically use calendar dates - -The provider supports configurable session date adjustment: -```python -# For CME futures with session starting at 5pm CT (10pm UTC summer time) -provider = DataBentoProvider( - dataset="GLBX.MDP3", - adjust_session_dates=True, - session_start_hour_utc=22 -) - -# For equities or crypto (calendar dates) -provider = DataBentoProvider( - dataset="XNAS.ITCH", - adjust_session_dates=False # Default -) -``` - -## Implementation Notes - -### Rate Limits -- Historical API: 100 requests/second (very generous) -- Real-time API: Varies by subscription -- Metadata API: 10 requests/second - -### Data Formats -- **DBN**: Databento's native binary format (most efficient) -- **CSV**: Text format (larger, slower) -- **JSON**: For small datasets only -- **Parquet**: Available through client conversion - -### Cost Considerations -- Billed per symbol-day of data -- Different schemas have different costs -- MBO/full book data is more expensive than OHLCV -- Check pricing at https://databento.com/pricing - -## Usage Examples - -### Basic OHLCV Fetch -```python -# For futures with session date adjustment -provider = DataBentoProvider( - api_key="YOUR_KEY", - dataset="GLBX.MDP3", - adjust_session_dates=True, # Enable for CME futures - session_start_hour_utc=22 # 5pm CT = 10pm UTC (summer) -) - -# For equities (no session adjustment needed) -provider = DataBentoProvider( - api_key="YOUR_KEY", - dataset="XNAS.ITCH", -) - -# Fetch data -df = provider.fetch_ohlcv("AAPL", "2024-01-01", "2024-01-31", "minute") -``` - -### Multiple Schemas -```python -# Fetch both trades and quotes -schemas = ["trades", "tbbo"] -data = provider.fetch_multiple_schemas( - symbol="AAPL", - start="2024-01-01", - end="2024-01-01", - schemas=schemas, -) -trades_df = data["trades"] -quotes_df = data["tbbo"] -``` - -### Continuous Futures -```python -# Fetch front month crude oil continuous contract -df = provider.fetch_continuous_futures( - root_symbol="CL", - start="2024-01-01", - end="2024-01-31", - frequency="daily", - version=0 # Front month -) -``` - -## Error Handling - -Common errors and solutions: - -1. **Authentication Error**: Check API key is valid and has permissions -2. **Dataset Not Found**: Verify dataset name and subscription -3. **Symbol Not Found**: Check symbol format and stype_in parameter -4. **Rate Limit**: Implement exponential backoff (handled by base provider) -5. **Invalid Date Range**: Ensure data exists for requested period - -## Testing - -When testing the Databento provider: -1. Use mock responses for unit tests (avoid API costs) -2. Create integration tests with `DATABENTO_API_KEY` env var -3. Test continuous contract handling -4. Test session date adjustment configuration (when enabled) -5. Test multiple schema fetching -6. Verify wrapper behavior with configured datasets; use the native client for advanced schemas - -## Migration from ML3T Pattern - -The ML4T Data Databento provider improves on the ML3T pattern by: -1. Using Polars throughout (no pandas conversion) -2. Supporting selected schemas in one call -3. Configurable session date handling (not hardcoded for CME) -4. Built-in continuous contract support -5. Comprehensive error handling with circuit breaker -6. Access to all Databento datasets through the exposed native client - -## References - -- [API Client Documentation](https://databento.com/docs/api-reference-historical/client) -- [Symbology Guide](https://databento.com/docs/symbology) -- [Data Quality Information](https://databento.com/docs/data-quality) -- [Changelog & Updates](https://databento.com/docs/changelog) diff --git a/docs/providers/eodhd.md b/docs/providers/eodhd.md index c69a24f6..0d54e0fb 100644 --- a/docs/providers/eodhd.md +++ b/docs/providers/eodhd.md @@ -54,7 +54,8 @@ provider.close() | Hong Kong | .HK | 0700.HK, 9988.HK | | Australia | .AU | BHP.AU, CBA.AU | -See [EODHD Exchange List](https://eodhd.com/financial-apis/exchanges-api-list-of-tickers-and-டexchange-codes) for all 60+ exchanges. +See the [EODHD exchange list](https://eodhd.com/financial-apis/exchanges-api-list-of-tickers-and-trading-hours) +for current coverage. --- diff --git a/docs/providers/index.md b/docs/providers/index.md index 3148daa0..f8211cea 100644 --- a/docs/providers/index.md +++ b/docs/providers/index.md @@ -35,7 +35,7 @@ For the wider vendor landscape, including sources the library does not wrap, sta | [Finnhub](finnhub.md) | US quotes; premium OHLCV | 60 requests/minute | Thread | Yes | | [Binance](binance.md) | Crypto | Unlimited | Native | No | | [OKX](okx.md) | Crypto Perpetuals | No geo-limits | Native | No | -| [CryptoCompare](cryptocompare.md) | Crypto | Unverified for 0.1.0 | Native | Required | +| [CryptoCompare](cryptocompare.md) | Crypto | Live contract unverified for 0.2.0 | Native | Required | | [Oanda](oanda.md) | Forex | Demo only | Thread | Yes | ## Async Support @@ -95,8 +95,8 @@ provider's capabilities before passing it to `DataManager` or `async_batch_load( Provider access, coverage, retention, and redistribution terms can differ by account tier. Confirm the provider page and the provider's current terms before selecting it for a production dataset. -CryptoCompare is included for evaluation but is not release-qualified for 0.1.0 because account -registration was unavailable during release validation and no live contract evidence was obtained. +CryptoCompare is included in 0.2.0, but release validation had no configured API key and therefore +recorded no successful live contract. Verify access with your account before production use. ## Authentication diff --git a/docs/providers/mock.md b/docs/providers/mock.md deleted file mode 100644 index 26b25386..00000000 --- a/docs/providers/mock.md +++ /dev/null @@ -1,68 +0,0 @@ -# Mock Provider - -**Provider**: `MockProvider` -**API Key**: Not required -**Free Tier**: N/A (for testing) - ---- - -## Overview - -Mock provider for unit testing. Returns predefined data or raises configured errors. - -**Best For**: Unit tests, integration tests - ---- - -## Quick Start - -```python -from ml4t.data.providers import MockProvider -import polars as pl - -# Create with predefined data -test_data = pl.DataFrame({ - "timestamp": ["2024-01-01", "2024-01-02"], - "symbol": ["TEST", "TEST"], - "open": [100.0, 101.0], - "high": [102.0, 103.0], - "low": [99.0, 100.0], - "close": [101.0, 102.0], - "volume": [1000000.0, 1100000.0], -}) - -provider = MockProvider(data=test_data) -df = provider.fetch_ohlcv("TEST", "2024-01-01", "2024-01-02") -``` - ---- - -## Testing Error Handling - -```python -from ml4t.data.core.exceptions import SymbolNotFoundError - -# Configure to raise errors -provider = MockProvider(error=SymbolNotFoundError("TEST")) - -try: - df = provider.fetch_ohlcv("TEST", "2024-01-01", "2024-01-02") -except SymbolNotFoundError: - print("Correctly caught error") -``` - ---- - -## Use Cases - -1. **Unit Tests**: Test without network -2. **Error Handling Tests**: Verify exception handling -3. **Performance Tests**: Known data size -4. **CI/CD Pipelines**: No external dependencies - ---- - -## See Also - -- [Synthetic Provider](synthetic.md) -- [Provider reference](index.md) diff --git a/docs/providers/polymarket.md b/docs/providers/polymarket.md index e6ad974c..ecfbe04d 100644 --- a/docs/providers/polymarket.md +++ b/docs/providers/polymarket.md @@ -303,7 +303,7 @@ print(combined.group_by("symbol").agg(pl.col("close").last())) ## API Documentation Links -- **CLOB Timeseries**: https://docs.polymarket.com/developers/clob-api/price-history +- **CLOB Timeseries**: https://docs.polymarket.com/api-reference/markets/get-a-tokens-price-history - **Gamma Markets API**: https://docs.polymarket.com/developers/gamma-markets-api/get-markets - **Data API Trades**: https://docs.polymarket.com/api-reference/core/get-trades-for-a-user-or-markets - **py-clob-client**: https://github.com/Polymarket/py-clob-client diff --git a/docs/providers/synthetic.md b/docs/providers/synthetic.md index 59221c65..bc460d32 100644 --- a/docs/providers/synthetic.md +++ b/docs/providers/synthetic.md @@ -94,5 +94,4 @@ constructing the provider. ## See Also -- [Mock Provider](mock.md) - [Provider reference](index.md) diff --git a/docs/user-guide/exporting.md b/docs/user-guide/exporting.md new file mode 100644 index 00000000..6ae99f51 --- /dev/null +++ b/docs/user-guide/exporting.md @@ -0,0 +1,100 @@ +# Exporting Stored Data + +The command-line interface exports one stored dataset to CSV, JSON, or Parquet. The Python export +manager adds Excel output, batch export, pattern selection, and transformations. + +## Command-Line Export + +The CLI derives the storage key from the asset class, frequency, and symbol: + +```bash +ml4t-data export \ + --symbol AAPL \ + --asset-class equities \ + --frequency daily \ + --storage-path ./data \ + --output ./exports/aapl.csv \ + --format csv +``` + +Run `ml4t-data export --help` for the accepted values. Create the output directory before running +the command. + +## Python Export Manager + +`ExportManager` accepts any storage backend that implements `exists()`, `read()`, and `list_keys()`. + +```python +from pathlib import Path + +from ml4t.data.export.manager import ExportManager +from ml4t.data.storage import create_storage + +storage = create_storage("./data", strategy="hive") +manager = ExportManager(storage) +Path("./exports").mkdir(exist_ok=True) + +result = manager.export( + key="equities/daily/AAPL", + output_path="./exports/aapl.csv", + format_type="csv", +) +if not result.success: + raise RuntimeError(result.error) +``` + +`format_type` accepts `csv`, `json`, `excel`, or `xlsx`. The returned `ExportResult` records the +output path, row count, file size, duration, and any error. + +## Filter and Transform + +Export options are validated by `ExportConfig`: + +```python +result = manager.export( + key="equities/daily/AAPL", + output_path="./exports/aapl.csv.gz", + format_type="csv", + date_filter=("2024-01-01", "2024-03-31"), + columns=["timestamp", "close", "volume"], + add_returns=True, + add_volatility=True, + compression="gzip", +) +``` + +Date and column filters are passed to production storage before collection when the backend accepts +them. Requested columns retain their supplied order, and a missing column fails the export. + +CSV supports optional gzip compression. JSON can include export metadata. Excel supports one sheet +per dataset and requires the dependencies installed with ML4T Data. + +## Batch and Pattern Export + +Export explicit keys to one Excel workbook: + +```python +results = manager.export_batch( + keys=["equities/daily/AAPL", "equities/daily/MSFT"], + output_path="./exports/equities.xlsx", + format_type="excel", +) +``` + +Or select complete storage keys with shell-style patterns: + +```python +results = manager.export_pattern( + pattern="equities/daily/*", + output_path="./exports/", + format_type="csv", +) +``` + +Patterns support `*`, `?`, and bracket expressions. Two keys that resolve to the same export symbol +produce an explicit failure rather than overwriting one another. + +## Large Datasets + +Use `date_filter` and `columns` to reduce the data before it is collected. Export transformations +and file serialization operate in memory, including gzip CSV output. diff --git a/docs/user-guide/incremental-updates.md b/docs/user-guide/incremental-updates.md index da5c2bce..54489cd0 100644 --- a/docs/user-guide/incremental-updates.md +++ b/docs/user-guide/incremental-updates.md @@ -1,336 +1,127 @@ -# Incremental Updates & Data Management +# Incremental Updates -This document describes the incremental update system implemented in Sprint 004, including gap detection, file locking, chunked storage, and metadata tracking. +`ml4t-data update` creates a dataset when it is absent and refreshes it when it already exists. The +stored key is `{asset_class}/{frequency}/{symbol}`. -## Overview +## Create or Update One Dataset -The incremental update system allows efficient data updates without re-downloading entire datasets. Key features include: - -- **Incremental Updates**: Only fetch new data since last update -- **Gap Detection**: Identify and fill missing data points -- **File Locking**: Safe concurrent access to data files -- **Chunked Storage**: Split large datasets into manageable time-based chunks -- **Metadata Tracking**: Monitor dataset health and update history - -## CLI Commands - -### Initial Data Load - -First time loading data for a symbol: +The first call can define an explicit initial range: ```bash -# Load historical data -ml4t-data load --provider yahoo --symbol AAPL --start 2023-01-01 --end 2024-01-01 - -# Load with specific frequency -ml4t-data load -p yahoo -s AAPL --start 2023-01-01 --end 2024-01-01 -f daily +ml4t-data update \ + --symbol AAPL \ + --provider yahoo \ + --asset-class equities \ + --frequency daily \ + --initial-start 2023-01-01 \ + --initial-end 2024-01-01 \ + --storage-path ./data ``` -### Incremental Updates +When the key does not exist, `update` performs an initial load. Without `--initial-start`, it asks +for `--initial-load-days` of history, subject to the provider's declared history limit. -Update existing data with new data: +Subsequent calls overlap the stored data by seven days by default: ```bash -# Basic update (uses existing provider) -ml4t-data update --symbol AAPL - -# Update with options -ml4t-data update -s AAPL --lookback-days 10 --fill-gaps --show-status - -# Update without gap filling -ml4t-data update -s AAPL --no-fill-gaps - -# Update with different provider -ml4t-data update -s AAPL --provider yahoo +ml4t-data update \ + --symbol AAPL \ + --asset-class equities \ + --frequency daily \ + --lookback-days 7 \ + --storage-path ./data ``` -Options: -- `--lookback-days/-l`: Days to look back for validation (default: 7) -- `--fill-gaps/--no-fill-gaps`: Enable/disable gap filling (default: enabled) -- `--show-status`: Display detailed update status and history -- `--provider/-p`: Override provider (uses existing if not specified) +The overlap lets revised observations replace their stored values. The update merges on timestamp, +keeps the newest row for a duplicate timestamp, sorts the result, and publishes the replacement +through the storage backend's atomic generation mechanism. -### Health Monitoring +If `--provider` is omitted for an existing dataset, the update reuses the provider recorded in its +metadata. Supplying the option overrides that value. -Check health status of all datasets: - -```bash -# Basic health check -ml4t-data status +## Gap Handling -# Detailed health check -ml4t-data --verbose status - -# Custom staleness threshold -ml4t-data --verbose status --stale-days 3 -``` - -Output shows: -- Total datasets and their health status (✅ healthy, ⚠️ stale, ❌ error) -- Total rows across all datasets -- Breakdown by asset class -- Individual dataset details (with --verbose) - -## Workflow Examples - -### Example 1: Daily Stock Data Updates +Gap detection and forward filling are enabled by default. Disable them when missing periods must +remain explicit: ```bash -# Initial load of AAPL data -ml4t-data load -p yahoo -s AAPL --start 2023-01-01 --end 2024-01-01 - -# Daily update (run via cron) -ml4t-data update -s AAPL --show-status - -# Check health weekly -ml4t-data --verbose status -``` - -### Example 2: Multiple Symbol Management - -```bash -# Load multiple symbols -for symbol in AAPL GOOGL MSFT NVDA; do - ml4t-data load -p yahoo -s $symbol --start 2023-01-01 --end 2024-01-01 -done - -# Update all symbols -for symbol in AAPL GOOGL MSFT NVDA; do - ml4t-data update -s $symbol -done - -# Check overall health -ml4t-data status -``` - -### Example 3: Crypto Data (24/7 Trading) - -```bash -# Load crypto data -ml4t-data load -p yahoo -s BTC-USD --start 2023-01-01 --end 2024-01-01 -a crypto - -# Update with gap detection (important for 24/7 markets) -ml4t-data update -s BTC-USD -a crypto --fill-gaps --show-status -``` - -## Technical Details - -### Gap Detection - -The system distinguishes between expected and unexpected gaps: - -- **Stock Markets**: Weekends and after-hours are expected gaps -- **Crypto Markets**: 24/7 trading, all gaps are unexpected -- **Configurable Tolerance**: 10% default tolerance for gap detection - -Gap filling methods: -- `forward`: Use last known value (default) -- `backward`: Use next known value -- `interpolate`: Linear interpolation -- `zero`: Fill with zeros - -### File Locking - -Thread-safe and process-safe file locking ensures data integrity: - -- Uses `filelock` library for cross-platform compatibility -- Automatic lock acquisition and release -- Configurable timeout (default: 30 seconds) -- Prevents corruption during concurrent reads/writes - -### Chunked Storage - -Large datasets are split into time-based chunks: - -- **Monthly chunks** (default): 30-day periods -- **Weekly chunks**: 7-day periods -- **Quarterly chunks**: 90-day periods -- **Yearly chunks**: 365-day periods - -Benefits: -- Efficient incremental updates (only update relevant chunks) -- Parallel processing capability -- Reduced memory usage for large datasets -- Fast time-range queries - -### Metadata Tracking - -Each dataset maintains metadata including: - -- Update history (last 100 updates) -- Health status (healthy/stale/error) -- Data range and row count -- Provider information -- Error tracking - -Health checks consider: -- Days since last update -- Data currency (how far behind current date) -- Recent error frequency - -## Architecture - +ml4t-data update --symbol AAPL --no-fill-gaps --storage-path ./data ``` -┌─────────────┐ ┌──────────────┐ ┌─────────────┐ -│ Provider │────▶│ Pipeline │────▶│ Storage │ -└─────────────┘ └──────────────┘ └─────────────┘ - │ │ - ▼ ▼ - ┌──────────────┐ ┌─────────────┐ - │ Gap Detector │ │ File Lock │ - └──────────────┘ └─────────────┘ - │ │ - ▼ ▼ - ┌──────────────┐ ┌─────────────┐ - │ Metadata │ │ Chunks │ - │ Tracker │ │ Storage │ - └──────────────┘ └─────────────┘ -``` - -### Components -1. **Pipeline** (`src/ml4t-data/pipeline.py`) - - Orchestrates data flow - - `run_load()`: Full data load - - `run_update()`: Incremental update +The detector uses the requested frequency and distinguishes crypto, which trades continuously, from +other asset classes. It treats common overnight and weekend intervals as expected for non-crypto +intraday data. This is a general interval check, not an exchange-calendar reconstruction. Use +[trading-session completion](sessions.md) when exact exchange schedules are required. -2. **Gap Detector** (`src/ml4t-data/utils/gaps.py`) - - Identifies missing data points - - Market hours awareness - - Multiple fill strategies +If a provider's history limit cannot reach the stored dataset, the updater records +`provider_history_limited` and skips gap filling across the unavailable interval. -3. **File Locking** (`src/ml4t-data/utils/locking.py`) - - Thread/process-safe access - - Automatic cleanup - - Timeout configuration +## Python API -4. **Chunked Storage** (`src/ml4t-data/storage/chunked.py`) - - Time-based data splitting - - Efficient updates - - Metadata indexing - -5. **Metadata Tracker** (`src/ml4t-data/storage/metadata_tracker.py`) - - Update history - - Health monitoring - - Summary statistics - -## Best Practices - -### Update Frequency - -- **Daily data**: Update once per day after market close -- **Minute data**: Update every few hours during market hours -- **Crypto**: Update more frequently (hourly or more) - -### Error Handling - -The system includes robust error handling: +Configure a storage backend and pass it to `DataManager`: ```python -# Automatic retries with exponential backoff -@with_retry(max_attempts=3, min_wait=1.0, max_wait=30.0) -def _fetch_data_with_retry(...) +from ml4t.data import DataManager +from ml4t.data.storage import create_storage + +storage = create_storage("./data", strategy="hive") +manager = DataManager(storage=storage) + +key = manager.update( + symbol="AAPL", + provider="yahoo", + frequency="daily", + asset_class="equities", + lookback_days=7, + fill_gaps=True, + initial_start="2023-01-01", + initial_end="2024-01-01", +) ``` -### Monitoring +`DataManager.update()` returns the storage key. It raises when storage is not configured, stored data +is empty, metadata is invalid, or the provider or storage operation fails. + +## Inspect Stored State -Set up monitoring using the status command: +List keys and inspect one dataset: ```bash -# Cron job for daily health check -0 9 * * * ml4t-data status --stale-days 2 >> /var/log/ml4t-data-status.log +ml4t-data list --storage-path ./data +ml4t-data info \ + --symbol AAPL \ + --asset-class equities \ + --frequency daily \ + --storage-path ./data ``` -### Storage Management - -Monitor disk usage as datasets grow: +Show the overall metadata health classification: ```bash -# Check data directory size -du -sh ~/.ml4t-data/data/ - -# List all datasets -ml4t-data list - -# Remove old data if needed (manual process) -rm -rf ~/.ml4t-data/data/equities/daily/OLD_SYMBOL +ml4t-data status --detailed --stale-days 3 --storage-path ./data ``` -## Troubleshooting - -### Common Issues - -1. **"No existing data found"** - - Run `ml4t-data load` first before using `update` - - Check the correct asset class and frequency - -2. **"Lock timeout"** - - Another process is accessing the file - - Check for stuck processes - - Increase timeout if needed - -3. **"Gaps detected"** - - Normal for some data sources - - Use `--fill-gaps` to automatically fill - - Check provider data quality +`status` classifies a dataset from its recorded end date, or its last update time when no end date +is present. It reports an error when the timestamp or row count metadata is missing or invalid. -4. **"Data is stale"** - - Run update more frequently - - Check provider connectivity - - Verify market hours settings +## Update Configured Datasets -### Debug Mode - -Enable debug logging for troubleshooting: +Use a YAML configuration for repeatable multi-dataset updates: ```bash -# Set log level in .env -echo "ML4T Data_LOG_LEVEL=DEBUG" >> .env - -# Run with verbose output -ml4t-data update -s AAPL --show-status +ml4t-data update-all --config ml4t-data.yaml --dry-run +ml4t-data update-all --config ml4t-data.yaml ``` -## Performance Considerations - -### Memory Usage - -- Chunked storage keeps memory usage low -- Each chunk is processed independently -- Typical chunk size: 10-50 MB - -### Disk Usage - -- Parquet compression reduces storage by 50-80% -- Monthly chunks balance size and performance -- Metadata overhead: ~1KB per dataset - -### Update Speed - -- Incremental updates request only missing date ranges -- Gap detection runs before provider requests -- File locking prevents concurrent writers from replacing each other's data - -## Future Enhancements - -Potential improvements for future sprints: - -1. **Parallel Updates**: Update multiple symbols concurrently -2. **Smart Scheduling**: Automatic update scheduling based on asset class -3. **Data Validation**: Detect and flag suspicious data points -4. **Compression Options**: Support for different compression algorithms -5. **Archive Storage**: Move old data to compressed archives -6. **Update Notifications**: Email/webhook alerts for failures -7. **Data Reconciliation**: Compare and sync with multiple providers - -## Summary - -The incremental update system provides efficient, reliable data management with: +Pass `--dataset NAME` to select one configured dataset. The dry run reports the planned work without +fetching or writing data. -- ✅ Minimal bandwidth usage (only fetch new data) -- ✅ Data integrity (file locking, gap detection) -- ✅ Scalability (chunked storage) -- ✅ Observability (health monitoring, update history) -- ✅ Flexibility (configurable options, multiple providers) +## Operational Notes -This foundation enables building robust quantitative trading systems with reliable, up-to-date market data. +- Run an update only after the provider is expected to have finalized the latest requested period. +- Reuse one storage directory and the same asset-class and frequency values so updates address the + intended key. +- Treat forward-filled rows as synthetic observations. For minute-session completion, the + `is_imputed` column identifies inserted rows; the general updater does not add that marker. +- Use `ml4t-data validate` after an update when schema, OHLC relationships, duplicates, or anomaly + checks are part of the ingestion contract. diff --git a/docs/user-guide/sessions.md b/docs/user-guide/sessions.md new file mode 100644 index 00000000..403ff923 --- /dev/null +++ b/docs/user-guide/sessions.md @@ -0,0 +1,122 @@ +# Trading Sessions + +Use session dates when an exchange trading day does not match a calendar day. A session date keeps +all observations from one exchange session together for aggregation, validation, and train/test +splits. + +ML4T Data uses +[`pandas_market_calendars`](https://github.com/rsheftel/pandas_market_calendars) for schedules. The +dependency is included with the package. + +## Assign Session Dates + +`SessionAssigner` requires a Polars `Datetime` column named `timestamp`. Naive timestamps are +interpreted as UTC; timezone-aware timestamps are converted to UTC before they are matched to a +schedule. + +```python +from datetime import datetime + +import polars as pl + +from ml4t.data.sessions import SessionAssigner + +bars = pl.DataFrame( + { + "timestamp": [ + datetime(2024, 1, 2, 15, 0), + datetime(2024, 1, 2, 16, 0), + ], + "close": [100.0, 101.0], + } +) + +assigner = SessionAssigner.from_exchange("NYSE") +assigned = assigner.assign_sessions( + bars, + bar_frequency="intraday", + outside_session="raise", +) +``` + +The result contains a `session_date` column. `outside_session` controls observations that do not +fall within an open interval: + +- `"null"` retains the row with a null session date. +- `"raise"` rejects the input and reports sample timestamps. +- `"drop"` removes the row. + +Set `bar_frequency="daily"` when timestamps are daily labels. The default, `"auto"`, treats a +series as daily when every non-null timestamp is midnight. + +## Select an Exchange Calendar + +`SessionAssigner.from_exchange()` recognizes these exchange codes: + +| Code | Calendar | +| --- | --- | +| `CME` | CME Globex Crypto | +| `NYSE` | NYSE | +| `NASDAQ` | NASDAQ | +| `LSE` | LSE | +| `TSE` | TSE | +| `HKEX` | HKEX | +| `ASX` | ASX | +| `SSE` | SSE | +| `TSX` | TSX | + +For another calendar exposed by `pandas_market_calendars`, pass its name directly: + +```python +assigner = SessionAssigner("CME Globex Crypto") +``` + +Calendar names and schedules can change upstream. Check the installed calendar before assigning a +large dataset. + +## Complete Minute Sessions + +`SessionCompleter` expands minute bars to the exchange schedule. It requires `timestamp`, `open`, +`high`, `low`, `close`, and `volume`. The result adds `session_date` and `is_imputed`. + +```python +from ml4t.data.sessions import SessionCompleter + +completer = SessionCompleter("NYSE") +complete = completer.complete_sessions( + bars, + fill_method="forward", + zero_volume=True, +) +``` + +The supported fill methods are `"forward"`, `"backward"`, and `"none"`. Forward and backward +filling apply only within a session. `zero_volume=True` sets volume to zero for inserted rows. + +Session completion rejects duplicate or non-minute-aligned timestamps and observations outside the +selected schedule. Use session assignment first when you need to inspect how individual timestamps +map to the exchange calendar. + +## Use Session Dates in Validation + +Group observations by `session_date` when a split must not divide one trading session: + +```python +from sklearn.model_selection import GroupKFold + +groups = assigned["session_date"].to_numpy() +features = assigned.select("close").to_numpy() +target = assigned["close"].to_numpy() + +for train_index, test_index in GroupKFold(n_splits=2).split(features, target, groups): + train = assigned[train_index] + test = assigned[test_index] +``` + +This groups rows by session. It does not make `GroupKFold` chronological. Use a time-ordered split +over unique session dates when the evaluation must also preserve time order. + +## Related APIs + +`DataManager.assign_sessions()` and `DataManager.complete_sessions()` expose the same operations +through the main facade. See the [API reference](../api/index.md) for their signatures. diff --git a/docs/yahoo_provider.md b/docs/yahoo_provider.md deleted file mode 100644 index 3b54a77b..00000000 --- a/docs/yahoo_provider.md +++ /dev/null @@ -1,144 +0,0 @@ -# Yahoo Finance Provider - -## Overview - -The Yahoo Finance provider fetches real-time and historical market data using the `yfinance` library. It provides free access to stock prices, volumes, and basic market data. - -## Features - -- **Free Access**: No API key required -- **Rate Limiting**: Built-in rate limiting to avoid overwhelming the API (default: 0.5 requests/second) -- **Multiple Frequencies**: Supports minute, hourly, daily, weekly, and monthly data -- **Auto-Adjustment**: Automatically adjusts for stock splits and dividends -- **Robust Error Handling**: Retry logic for transient failures - -## Usage - -### CLI Usage - -```bash -# Fetch daily data for Apple stock -ml4t-data load --provider yahoo --symbol AAPL --start 2024-01-01 --end 2024-01-31 - -# Fetch minute-level data -ml4t-data load --provider yahoo --symbol MSFT --start 2024-01-01 --end 2024-01-01 --frequency minute - -# Fetch weekly data -ml4t-data load --provider yahoo --symbol GOOGL --start 2023-01-01 --end 2024-01-01 --frequency weekly -``` - -### Python API Usage - -```python -from ml4t.data.providers.yahoo import YahooFinanceProvider -from ml4t.data.storage.filesystem import FileSystemBackend -from ml4t.data.pipeline import Pipeline -from ml4t.data.core.config import Config - -# Initialize provider -provider = YahooFinanceProvider( - max_requests_per_second=0.5, # Rate limit - enable_progress=False # Progress bars -) - -# Fetch data directly -df = provider.fetch_ohlcv( - symbol="AAPL", - start="2024-01-01", - end="2024-01-31", - frequency="daily" -) - -# Or use with pipeline -config = Config() -storage = FileSystemBackend(data_root=config.data_root) -pipeline = Pipeline(provider, storage, config) - -# Load data through pipeline -key = pipeline.run_load( - symbol="AAPL", - start="2024-01-01", - end="2024-01-31", - frequency="daily", - asset_class="equities" -) -``` - -## Supported Frequencies - -The provider maps standard frequency names to Yahoo Finance intervals: - -| Frequency | Yahoo Interval | Description | -|-----------|---------------|-------------| -| `minute` | `1m` | 1-minute bars | -| `5minute` | `5m` | 5-minute bars | -| `15minute` | `15m` | 15-minute bars | -| `30minute` | `30m` | 30-minute bars | -| `hourly` | `1h` | Hourly bars | -| `daily` | `1d` | Daily bars (default) | -| `weekly` | `1wk` | Weekly bars | -| `monthly` | `1mo` | Monthly bars | - -## Rate Limiting - -The provider includes automatic rate limiting to prevent API throttling: - -- Default rate: 0.5 requests per second (1 request every 2 seconds) -- Configurable via `max_requests_per_second` parameter -- Thread-safe implementation -- Automatic waiting when rate limit is reached - -## Limitations - -1. **Historical Data**: - - Minute data: Available for last 30 days - - Hourly data: Available for last 730 days (2 years) - - Daily/Weekly/Monthly: Full historical data available - -2. **API Limitations**: - - No official API documentation - - Subject to Yahoo Finance terms of service - - May experience occasional downtime - - Data quality varies for less liquid securities - -3. **Data Coverage**: - - Best coverage for US equities - - Limited international market support - - No options, futures, or forex data through this provider - -## Error Handling - -The provider includes robust error handling: - -- Automatic retry with exponential backoff for network errors -- Graceful handling of empty responses (e.g., invalid symbols) -- Detailed logging for debugging -- Returns empty DataFrame with correct schema on errors - -## Best Practices - -1. **Rate Limiting**: Always use rate limiting to avoid being blocked -2. **Batch Requests**: Process multiple symbols sequentially, not in parallel -3. **Cache Data**: Store fetched data locally to minimize API calls -4. **Error Handling**: Always check for empty DataFrames in your code -5. **Respect ToS**: Use data in compliance with Yahoo Finance terms of service - -## Troubleshooting - -### Common Issues - -1. **Empty Data**: Symbol may be invalid or delisted -2. **Rate Limit Errors**: Reduce `max_requests_per_second` -3. **Connection Errors**: Check internet connection; retry logic will handle transient issues -4. **Data Quality**: Some securities may have missing or incorrect data - -### Debug Logging - -Enable debug logging to troubleshoot issues: - -```python -import structlog -structlog.configure( - wrapper_class=structlog.make_filtering_bound_logger(logging.DEBUG) -) -``` diff --git a/mkdocs.yml b/mkdocs.yml index 19821e0a..e8f7f081 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -120,6 +120,8 @@ nav: - Incremental Updates: user-guide/incremental-updates.md - Data Quality: user-guide/data-quality.md - Storage: user-guide/storage.md + - Trading Sessions: user-guide/sessions.md + - Exporting Data: user-guide/exporting.md - Providers: - providers/index.md @@ -203,18 +205,5 @@ extra: # Copyright copyright: Copyright © 2024-2026 Stefan Jansen -# These pages remain searchable but are intentionally outside the primary navigation. -not_in_nav: | - /ASYNC_STORAGE.md - /FEATURES.md - /INSTALLATION.md - /INTEGRATION_TESTING.md - /SESSION_MANAGEMENT.md - /crypto_providers.md - /export_guide.md - /yahoo_provider.md - /providers/databento_reference.md - /providers/mock.md - validation: omitted_files: warn diff --git a/scripts/verify_documentation_identity.py b/scripts/verify_documentation_identity.py index 3d78e7cc..8fcac4d8 100644 --- a/scripts/verify_documentation_identity.py +++ b/scripts/verify_documentation_identity.py @@ -7,6 +7,8 @@ import urllib.error import urllib.parse import urllib.request +from collections.abc import Callable +from concurrent.futures import ThreadPoolExecutor from html.parser import HTMLParser from pathlib import Path @@ -32,10 +34,14 @@ class _ReferenceParser(HTMLParser): def __init__(self) -> None: super().__init__() self.references: list[str] = [] + self.content_links: list[str] = [] self.anchors: set[str] = set() + self._article_depth = 0 def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None: values = dict(attrs) + if tag == "article": + self._article_depth += 1 anchor = values.get("id") or (values.get("name") if tag == "a" else None) if anchor: self.anchors.add(anchor) @@ -48,6 +54,12 @@ def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None reference = values.get(attribute) if attribute else None if reference: self.references.append(reference) + if tag == "a" and self._article_depth: + self.content_links.append(reference) + + def handle_endtag(self, tag: str) -> None: + if tag == "article" and self._article_depth: + self._article_depth -= 1 def identity_failures( @@ -142,6 +154,56 @@ def site_link_failures(site: Path, site_path: str = "/docs/data/") -> list[str]: return failures +def external_content_urls( + site: Path, + site_root: str = "https://www.ml4trading.io/docs/data/", +) -> list[str]: + """Return distinct external destinations linked from rendered page content.""" + root = urllib.parse.urlsplit(site_root) + references: set[str] = set() + for page in _site_pages(site): + parser = _ReferenceParser() + parser.feed(page.read_text(encoding="utf-8")) + relative = page.relative_to(site).as_posix() + page_url = urllib.parse.urljoin( + site_root, + "" if relative == "index.html" else relative.removesuffix("index.html"), + ) + for reference in parser.content_links: + absolute = urllib.parse.urljoin(page_url, reference) + parsed = urllib.parse.urlsplit(absolute) + if parsed.scheme not in {"http", "https"}: + continue + if parsed.netloc == root.netloc and parsed.path.startswith(root.path): + continue + references.add( + urllib.parse.urlunsplit( + (parsed.scheme, parsed.netloc, parsed.path, parsed.query, "") + ) + ) + return sorted(references) + + +def external_link_failures( + site: Path, + *, + probe_url: Callable[[str], None] | None = None, +) -> list[str]: + """Return unavailable external destinations linked from rendered content.""" + urls = external_content_urls(site) + probe = probe_url or _probe_external_url + + def failure(url: str) -> str | None: + try: + probe(url) + except (OSError, urllib.error.URLError, ValueError) as error: + return f"{url}: {error}" + return None + + with ThreadPoolExecutor(max_workers=8) as executor: + return [result for result in executor.map(failure, urls) if result is not None] + + def page_reference_urls(html: str, *, page_url: str, site_root: str) -> list[str]: """Return distinct same-site documentation references from one deployed page.""" parser = _ReferenceParser() @@ -180,6 +242,34 @@ def _probe_url(url: str, attempts: int = 3, delay: float = 1.0) -> None: raise ValueError(str(error)) +def _probe_external_url(url: str, attempts: int = 3, delay: float = 1.0) -> None: + """Check a link without failing a release on access controls or upstream outages.""" + encoded_url = urllib.parse.quote(url, safe=":/?&=%#@+;,") + request = urllib.request.Request(encoded_url, headers={"User-Agent": "ml4t-link-verifier"}) + error: Exception | None = None + for attempt in range(attempts): + try: + with urllib.request.urlopen(request, timeout=30) as response: # noqa: S310 + if response.status >= 400: + raise ValueError(f"HTTP {response.status}") + response.read(1) + return + except urllib.error.HTTPError as caught: + caught.close() + if caught.code in {401, 403, 429}: + return + error = caught + if caught.code < 500: + break + if attempt + 1 == attempts: + return + except (OSError, urllib.error.URLError, ValueError) as caught: + error = caught + if attempt + 1 < attempts: + time.sleep(delay) + raise ValueError(str(error)) + + def deployed_link_failures(urls: list[str]) -> list[str]: """Return unavailable navigation, asset, and internal-link targets.""" site_root = urls[0] @@ -271,6 +361,7 @@ def main() -> int: ) ) failures.extend(site_link_failures(args.site)) + failures.extend(external_link_failures(args.site)) except (OSError, UnicodeDecodeError, urllib.error.URLError, ValueError) as error: failures.append(str(error)) if args.url is not None: diff --git a/tests/test_release_pipeline.py b/tests/test_release_pipeline.py index 053f133a..267a3bf0 100644 --- a/tests/test_release_pipeline.py +++ b/tests/test_release_pipeline.py @@ -25,6 +25,8 @@ ) from scripts.run_readme_quickstart import extract_quick_start from scripts.verify_documentation_identity import ( + external_content_urls, + external_link_failures, identity_failures, page_reference_urls, site_link_failures, @@ -218,6 +220,29 @@ def test_deployed_documentation_discovers_only_same_route_references() -> None: ] +def test_rendered_documentation_checks_external_content_links(tmp_path: Path) -> None: + site = tmp_path / "site" + site.mkdir() + (site / "index.html").write_text( + '' + '', + encoding="utf-8", + ) + + urls = external_content_urls(site) + + def probe(url: str) -> None: + if "broken.example" in url: + raise ValueError("unavailable") + + failures = external_link_failures(site, probe_url=probe) + + assert urls == ["https://broken.example/", "https://working.example/path"] + assert failures == ["https://broken.example/: unavailable"] + + def test_readme_link_check_rejects_missing_local_and_unavailable_remote_targets( tmp_path: Path, ) -> None: From af368c023ca24114d2c98cf1d6add0f8131db5de Mon Sep 17 00:00:00 2001 From: Stefan Jansen Date: Sat, 10 Oct 2026 05:18:37 -0400 Subject: [PATCH 5/5] fix: tolerate transient external link failures --- scripts/verify_documentation_identity.py | 3 +++ tests/test_release_pipeline.py | 18 ++++++++++++++++++ 2 files changed, 21 insertions(+) diff --git a/scripts/verify_documentation_identity.py b/scripts/verify_documentation_identity.py index 8fcac4d8..0703aa16 100644 --- a/scripts/verify_documentation_identity.py +++ b/scripts/verify_documentation_identity.py @@ -265,6 +265,9 @@ def _probe_external_url(url: str, attempts: int = 3, delay: float = 1.0) -> None return except (OSError, urllib.error.URLError, ValueError) as caught: error = caught + reason = caught.reason if isinstance(caught, urllib.error.URLError) else caught + if attempt + 1 == attempts and isinstance(reason, TimeoutError): + return if attempt + 1 < attempts: time.sleep(delay) raise ValueError(str(error)) diff --git a/tests/test_release_pipeline.py b/tests/test_release_pipeline.py index 267a3bf0..38e3c567 100644 --- a/tests/test_release_pipeline.py +++ b/tests/test_release_pipeline.py @@ -5,6 +5,8 @@ import io import re import tarfile +import urllib.error +import urllib.request import zipfile from copy import deepcopy from email.message import Message @@ -25,6 +27,7 @@ ) from scripts.run_readme_quickstart import extract_quick_start from scripts.verify_documentation_identity import ( + _probe_external_url, external_content_urls, external_link_failures, identity_failures, @@ -243,6 +246,21 @@ def probe(url: str) -> None: assert failures == ["https://broken.example/: unavailable"] +def test_external_link_probe_distinguishes_timeout_from_missing_route(monkeypatch) -> None: + def timeout(*args, **kwargs): + raise TimeoutError("timed out") + + monkeypatch.setattr(urllib.request, "urlopen", timeout) + _probe_external_url("https://slow.example/", attempts=1, delay=0) + + def missing(request, timeout): + raise urllib.error.HTTPError(request.full_url, 404, "Not Found", {}, None) + + monkeypatch.setattr(urllib.request, "urlopen", missing) + with pytest.raises(ValueError, match="404"): + _probe_external_url("https://missing.example/", attempts=1, delay=0) + + def test_readme_link_check_rejects_missing_local_and_unavailable_remote_targets( tmp_path: Path, ) -> None: