From c521bdc324a8affb3a38016d17f6a807bd942abd Mon Sep 17 00:00:00 2001 From: Thor Whalen <1906276+thorwhalen@users.noreply.github.com> Date: Sat, 25 Jul 2026 00:40:06 +0200 Subject: [PATCH 1/2] revert: remove umpyre package accidentally merged into odbcdol (aa5f68e) Merge aa5f68e ("Merge copilot/repair-ci-tests from umpyre") pulled an entire unrelated umpyre project into odbcdol's master: the umpyre/ package, its tests under a root tests/ dir, planning docs (PHASE_*/IMPLEMENTATION_*/etc.), a track-metrics action, an umpyre-config.yml, a stray metrics json, and umpyre's CHANGELOG. This removes all of it. odbcdol's own package (odbcdol/) is untouched; the umpyre work is preserved in git history and in the separate i2mint/umpyre repo. Also drops the now-obsolete legacy setup.cfg and the pyproject.toml.migrated artifact (pyproject.toml is the single source of truth). --- .github/umpyre-config.yml | 36 -- FAILURE_PROTECTION.md | 155 ------ IMPLEMENTATION_STATUS.md | 188 -------- IMPLEMENTATION_SUMMARY.md | 353 -------------- PHASE_2_PLAN.md | 432 ----------------- PHASE_3_PLAN.md | 602 ------------------------ QUICK_START.md | 212 --------- STORAGE_STRUCTURE.md | 225 --------- TESTING_GUIDE.md | 377 --------------- TEST_RESULTS.md | 129 ----- actions/track-metrics/action.yml | 74 --- json | 32 -- misc/CHANGELOG.md | 235 --------- pyproject.toml.migrated | 252 ---------- setup.cfg | 23 - test_on_astate.py | 432 ----------------- tests/test_base_collector.py | 128 ----- tests/test_config.py | 131 ------ tests/test_coverage_collector.py | 125 ----- tests/test_schema.py | 71 --- tests/test_umpyre_collector.py | 80 ---- umpyre/__init__.py | 50 -- umpyre/cli.py | 241 ---------- umpyre/collectors/__init__.py | 17 - umpyre/collectors/base.py | 101 ---- umpyre/collectors/coverage_collector.py | 170 ------- umpyre/collectors/umpyre_collector.py | 309 ------------ umpyre/collectors/wily_collector.py | 221 --------- umpyre/collectors/workflow_status.py | 165 ------- umpyre/config.py | 169 ------- umpyre/python_code_stats.py | 365 -------------- umpyre/schema.py | 97 ---- umpyre/storage/__init__.py | 25 - umpyre/storage/formats.py | 108 ----- umpyre/storage/git_branch.py | 226 --------- umpyre/storage/query_utils.py | 195 -------- 36 files changed, 6751 deletions(-) delete mode 100644 .github/umpyre-config.yml delete mode 100644 FAILURE_PROTECTION.md delete mode 100644 IMPLEMENTATION_STATUS.md delete mode 100644 IMPLEMENTATION_SUMMARY.md delete mode 100644 PHASE_2_PLAN.md delete mode 100644 PHASE_3_PLAN.md delete mode 100644 QUICK_START.md delete mode 100644 STORAGE_STRUCTURE.md delete mode 100644 TESTING_GUIDE.md delete mode 100644 TEST_RESULTS.md delete mode 100644 actions/track-metrics/action.yml delete mode 100644 json delete mode 100644 misc/CHANGELOG.md delete mode 100644 pyproject.toml.migrated delete mode 100644 setup.cfg delete mode 100644 test_on_astate.py delete mode 100644 tests/test_base_collector.py delete mode 100644 tests/test_config.py delete mode 100644 tests/test_coverage_collector.py delete mode 100644 tests/test_schema.py delete mode 100644 tests/test_umpyre_collector.py delete mode 100644 umpyre/__init__.py delete mode 100644 umpyre/cli.py delete mode 100644 umpyre/collectors/__init__.py delete mode 100644 umpyre/collectors/base.py delete mode 100644 umpyre/collectors/coverage_collector.py delete mode 100644 umpyre/collectors/umpyre_collector.py delete mode 100644 umpyre/collectors/wily_collector.py delete mode 100644 umpyre/collectors/workflow_status.py delete mode 100644 umpyre/config.py delete mode 100644 umpyre/python_code_stats.py delete mode 100644 umpyre/schema.py delete mode 100644 umpyre/storage/__init__.py delete mode 100644 umpyre/storage/formats.py delete mode 100644 umpyre/storage/git_branch.py delete mode 100644 umpyre/storage/query_utils.py diff --git a/.github/umpyre-config.yml b/.github/umpyre-config.yml deleted file mode 100644 index 77d4d48..0000000 --- a/.github/umpyre-config.yml +++ /dev/null @@ -1,36 +0,0 @@ -schema_version: "1.0" - -collectors: - workflow_status: - enabled: true - lookback_runs: 10 - - wily: - enabled: true - max_revisions: 5 - operators: [cyclomatic, maintainability] - - coverage: - enabled: true - source: pytest-cov - - umpyre_stats: - enabled: true - exclude_dirs: [tests, examples, scrap] - -storage: - branch: code-metrics - formats: [json, csv] - retention: - strategy: all - -visualization: - generate_plots: true - generate_readme: true - plot_metrics: [maintainability, coverage, loc] - -thresholds: - enabled: false - -aggregation: - enabled: false diff --git a/FAILURE_PROTECTION.md b/FAILURE_PROTECTION.md deleted file mode 100644 index 20e88ad..0000000 --- a/FAILURE_PROTECTION.md +++ /dev/null @@ -1,155 +0,0 @@ -# Failure Protection Strategy - -Umpyre implements **three layers of failure protection** to ensure metrics collection never breaks your CI/CD pipeline. - -## Why This Matters - -Metrics collection is **informational**, not critical. Your CI should: -- ✅ Always publish to PyPI successfully -- ✅ Always commit version bumps -- ✅ Always tag releases -- ⚠️ Try to collect metrics, but don't fail if it doesn't work - -## Three Layers of Protection - -### Layer 1: GitHub Actions `continue-on-error` - -In your CI workflow file, the metrics step has `continue-on-error: true`: - -```yaml -- name: Track Code Metrics - uses: i2mint/umpyre/actions/track-metrics@master - continue-on-error: true # Don't fail CI if metrics collection fails - with: - github-token: ${{ secrets.GITHUB_TOKEN }} -``` - -**What it does**: If the entire action fails (Python crashes, dependency issues, etc.), GitHub Actions will log the error but continue with the next step. - -### Layer 2: Action Error Handling - -The GitHub Action itself catches errors and always exits with code 0: - -```bash -set +e # Don't exit on error - -python -m umpyre.cli collect --config "${{ inputs.config-path }}" -EXIT_CODE=$? - -if [ $EXIT_CODE -eq 0 ]; then - echo "✅ Metrics collection completed successfully" -else - echo "⚠️ Metrics collection failed with exit code $EXIT_CODE" - echo "This is informational only - CI pipeline continues" -fi - -exit 0 # Always succeed -``` - -**What it does**: Even if umpyre CLI fails, the action reports it as a warning and exits successfully. - -### Layer 3: CLI Error Handling - -The umpyre CLI detects CI environment and returns exit code 0 on errors: - -```python -def cmd_collect(args): - try: - # ... collect metrics ... - return 0 - except Exception as e: - print(f"❌ Error during metrics collection: {e}") - - # In CI mode, log but don't fail - if os.getenv('CI') or os.getenv('GITHUB_ACTIONS'): - print("⚠️ Running in CI - treating as non-fatal") - return 0 - - return 1 # Fail in local development -``` - -**What it does**: -- **In CI**: Logs error, returns 0 (success) -- **Locally**: Returns 1 (failure) so developers know something's wrong - -## Behavior Summary - -| Scenario | Layer | CI Exit Code | Result | -|----------|-------|--------------|--------| -| Collection succeeds | - | 0 | ✅ Metrics stored | -| Collector error (missing coverage) | L3 | 0 | ⚠️ Partial metrics stored | -| CLI error (no git repo) | L3 | 0 | ⚠️ Error logged, CI continues | -| Action failure (Python crash) | L2 | 0 | ⚠️ Error logged, CI continues | -| Timeout or system error | L1 | 0 | ⚠️ Step marked as warning | - -## Local Development - -When running locally (not in CI), errors **will fail** with exit code 1: - -```bash -$ python -m umpyre.cli collect -❌ Error during metrics collection: ... -Collection failed after 0.02s -$ echo $? -1 -``` - -This helps you debug issues during development. - -## Testing Failure Modes - -### Test Layer 3 (CLI error handling): - -```bash -# Normal mode - should fail -cd /tmp/not-a-repo -python -m umpyre.cli collect -echo $? # Returns 1 - -# CI mode - should not fail -CI=true python -m umpyre.cli collect -echo $? # Returns 0 -``` - -### Test Layer 2 (Action error handling): - -Create a workflow with invalid config and check GitHub Actions logs - step will show warning but not fail. - -### Test Layer 1 (continue-on-error): - -Trigger any catastrophic failure (kill Python process, etc.) - CI will continue. - -## Best Practices - -1. **Always use `continue-on-error: true`** in CI workflows -2. **Place metrics collection AFTER critical steps** (publish, tag, commit) -3. **Monitor but don't alert on metrics failures** - treat as informational -4. **Test locally first** - fix issues before they hit CI - -## Collector-Level Error Handling - -Individual collectors also handle errors gracefully: - -```python -def collect(self) -> dict: - try: - # ... collect metrics ... - return metrics - except Exception as e: - return { - "num_functions": 0, - "error": str(e) - } -``` - -This ensures one broken collector doesn't fail the entire collection. - -## Gradual Degradation - -Umpyre fails gracefully at every level: -- Missing collector dependency? Skip that collector -- Coverage file not found? Return zero coverage -- Git command fails? Use environment variables -- Network error? Store partial metrics - -The goal: **Always provide some value, never break CI.** diff --git a/IMPLEMENTATION_STATUS.md b/IMPLEMENTATION_STATUS.md deleted file mode 100644 index 42de7ce..0000000 --- a/IMPLEMENTATION_STATUS.md +++ /dev/null @@ -1,188 +0,0 @@ -# Summary: Storage Structure Improvements & Phase Plans - -## What Changed - -### Storage Structure (Addressing Your Feedback) ✅ - -**Your concerns**: -1. ❌ `by_version/` folder = data duplication (not SSOT) -2. ❌ Monthly folders = unnecessary complexity - -**Solution Implemented**: -- **Flat structure**: Just `history/` (no nested `history/2025-11/` folders) -- **No duplication**: Removed `by_version/` folder entirely -- **Smart filename**: `YYYY_MM_DD_HH_MM_SS__shahash__version.json` - -### New Structure - -``` -code-metrics branch: -├── metrics.json # Latest snapshot -├── metrics.csv # Latest CSV -└── history/ # Flat! (SSOT) - ├── 2025_11_14_22_45_00__700e012__0.1.0.json - ├── 2025_11_14_22_50_15__abc1234__0.1.1.json - └── 2025_11_15_10_30_22__def5678__none.json # No version -``` - -### Filename Format - -`{YYYY_MM_DD_HH_MM_SS}__{commit_sha}__{version}.json` - -**Benefits**: -1. ✅ **Chronological**: Natural time-based sorting -2. ✅ **Unique**: Commit SHA prevents duplicates -3. ✅ **Parseable**: All metadata in filename -4. ✅ **SSOT**: No duplication anywhere -5. ✅ **Simple**: Flat structure, easy queries -6. ✅ **Shell-friendly**: Works with standard tools - -### Query Examples - -```bash -# Get all metrics from November 14 -ls history/2025_11_14_* - -# Find metrics for commit 700e012 -ls history/*__700e012__* - -# Find all v0.1.0 metrics -ls history/*__0.1.0.json - -# Get latest 10 metrics -ls -t history/ | head -10 - -# Exclude metrics without version -ls history/ | grep -v "__none.json" -``` - ---- - -## Phase Plans Created - -### PHASE_2_PLAN.md ✅ - -**Duration**: 10-13 hours - -**Components**: -1. **Plot Generation** (3h): Time-series charts (coverage, maintainability, LOC, complexity) -2. **README Auto-Generation** (2h): `METRICS.md` with tables, charts, badges -3. **Cross-Repo Aggregation** (4h): Organization-wide metrics aggregation -4. **Dashboard Generation** (4h): Interactive HTML dashboard with Plotly.js - -**Key Features**: -- Matplotlib/Plotly charts saved to `plots/` in code-metrics branch -- Shields.io badges for GitHub README -- Aggregate metrics across all i2mint repos -- Interactive dashboard hosted on GitHub Pages -- CLI commands: `visualize`, `aggregate`, `dashboard` - -### PHASE_3_PLAN.md ✅ - -**Duration**: 15-18 hours - -**Components**: -1. **Threshold Validation** (3h): Enforce quality standards (min/max/delta thresholds) -2. **Additional Collectors** (8h): - - BanditCollector (security scanning) - - InterrogateCollector (docstring coverage) - - MyPyCollector (type hint coverage) - - PylintCollector (code quality) -3. **Data Management** (3h): Pruning, compression, rotation -4. **Schema Migrations** (4h): Version migration system (1.0 → 1.1 → 1.2) - -**Key Features**: -- Config-driven thresholds with custom validators -- 4 new collectors for comprehensive code analysis -- Intelligent data pruning (keep last 30 days, then weekly, then monthly) -- gzip compression for old metrics -- Auto-migration on schema version changes - ---- - -## Current Status - -✅ **Phase 1**: COMPLETE (100%) -- All collectors working -- Storage system (flat, SSOT-compliant) -- PyPI version tracking -- CLI interface -- GitHub Action -- CI failure protection -- 34/34 tests passing -- **Production-ready!** - -📋 **Phase 2**: PLANNED (detailed plan ready) -- Visualization system -- Cross-repo aggregation -- Dashboard generation - -📋 **Phase 3**: PLANNED (detailed plan ready) -- Threshold validation -- Additional collectors (security, quality, types) -- Data management -- Schema migrations - ---- - -## Files Created/Updated - -### Created: -- `PHASE_2_PLAN.md` - Detailed Phase 2 implementation plan -- `PHASE_3_PLAN.md` - Detailed Phase 3 implementation plan -- `STORAGE_STRUCTURE.md` - Complete storage architecture docs - -### Updated: -- `umpyre/storage/git_branch.py` - Flat structure, no duplication -- `umpyre/collectors/umpyre_collector.py` - PyPI version extraction -- `misc/CHANGELOG.md` - Storage structure improvements - ---- - -## Next Steps (Your Call) - -**Option 1: Deploy Phase 1 Now** 🚀 -- Add CI snippet to your templates (you mentioned doing this) -- Start collecting metrics on production repos -- Gather feedback before Phase 2 - -**Option 2: Start Phase 2** 📊 -- Begin with plot generation (3 hours) -- Add README auto-generation (2 hours) -- Build incrementally - -**Option 3: Test Storage** 🧪 -- Run `collect` without `--no-store` on umpyre -- Verify new flat structure works -- Inspect generated filenames - ---- - -## Testing the New Structure - -```bash -# Test on umpyre itself (will create code-metrics branch) -cd /Users/thorwhalen/Dropbox/py/proj/i/umpyre -python -m umpyre.cli collect - -# Then check the structure -git checkout code-metrics -ls -lh history/ -# Should see: 2025_11_14_*__*__0.1.0.json - -# Parse filename -ls history/ | head -1 -# Example: 2025_11_14_15_03_23__700e012__0.1.0.json -# ^^^^^^^^^^^^^^^^^^ ^^^^^^^ ^^^^^ -# timestamp commit version -``` - ---- - -## Questions? - -- Want me to test the storage by running actual collection? -- Ready to start Phase 2 (which part first)? -- Need any clarification on the plans? - -**Meanwhile**, you're adding the CI snippet to your templates - perfect timing! The storage structure is now battle-ready and SSOT-compliant. 🎉 diff --git a/IMPLEMENTATION_SUMMARY.md b/IMPLEMENTATION_SUMMARY.md deleted file mode 100644 index ebf197d..0000000 --- a/IMPLEMENTATION_SUMMARY.md +++ /dev/null @@ -1,353 +0,0 @@ -# Umpyre Metrics Tracking System - Implementation Summary - -## Project Status: Phase 1 Complete ✅ - -**Implemented**: Core metrics tracking system with collectors, storage, CLI, and GitHub Action -**Remaining**: Phase 2 (Visualization/Aggregation) and Phase 3 (Advanced Features) - ---- - -## What Has Been Implemented - -### ✅ Phase 1: Complete MVP - -#### 1. Architecture & Configuration (Complete) -- **Config System** (`config.py`): YAML-based configuration with deep merge -- **Schema System** (`schema.py`): Versioned metric schema (v1.0) with migration support -- **Collector Registry** (`collectors/base.py`): Pluggable collector system with Mapping interface -- **Test Coverage**: 32 passing tests, 2 skipped - -#### 2. Core Collectors (Complete) -All collectors implement the `MetricCollector` base class with Mapping interface: - -- **WorkflowStatusCollector** ✅ - - Tracks GitHub CI/CD status via GitHub API - - Recent failure counts, last success timestamp - - Configurable lookback window (default: 10 runs) - -- **CoverageCollector** ✅ - - Parses pytest-cov and coverage.py reports - - Supports JSON and XML (Cobertura) formats - - Auto-detects coverage files in standard locations - -- **WilyCollector** ✅ - - Complexity metrics using wily - - Cyclomatic complexity and maintainability index - - Limited to 5 recent commits for performance - -- **UmpyreCollector** ✅ - - Uses existing `python_code_stats.py` module - - Function/class counts, line metrics, code ratios - - Note: Has some compatibility issues inherited from original code - -#### 3. Storage System (Complete) -- **Git Branch Storage** (`storage/git_branch.py`): - - Stores metrics in separate branch (default: `code-metrics`) - - Monthly history organization (`history/YYYY-MM/`) - - Concurrent commit handling with retry logic - - Shallow clones for performance - -- **Serialization** (`storage/formats.py`): - - JSON format (structured, human-readable) - - CSV format (flat, pandas-friendly) - - Automatic flattening of nested metrics - -#### 4. CLI Interface (Complete) -- **`umpyre collect`**: Collect and store metrics - - Auto-detects git commit info - - Supports custom config files - - Dry-run mode (`--no-store`) - - Environment variable integration (GITHUB_SHA, GITHUB_REPOSITORY) - -- **`umpyre validate`**: Placeholder for Phase 3 threshold validation - -#### 5. GitHub Action (Complete) -- Reusable composite action: `actions/track-metrics/action.yml` -- Auto-installs dependencies -- Integrates with GitHub Actions workflows -- Configurable via inputs (config path, storage branch, Python version) - -#### 6. Documentation (Complete) -- **README.md**: Comprehensive usage guide with examples -- **CHANGELOG.md**: Detailed record of changes -- **Example config**: `.github/umpyre-config.yml` - ---- - -## File Structure - -``` -umpyre/ -├── umpyre/ -│ ├── __init__.py # Main exports -│ ├── python_code_stats.py # Original (preserved) -│ ├── config.py # ✅ Config loading/validation -│ ├── schema.py # ✅ Versioned metric schema -│ ├── cli.py # ✅ Command-line interface -│ ├── collectors/ -│ │ ├── __init__.py -│ │ ├── base.py # ✅ Abstract Collector -│ │ ├── workflow_status.py # ✅ GitHub workflow tracker -│ │ ├── wily_collector.py # ✅ Complexity metrics -│ │ ├── coverage_collector.py # ✅ Test coverage -│ │ └── umpyre_collector.py # ✅ Code statistics -│ └── storage/ -│ ├── __init__.py -│ ├── git_branch.py # ✅ Git branch storage -│ └── formats.py # ✅ JSON/CSV serialization -├── actions/ -│ └── track-metrics/ -│ └── action.yml # ✅ GitHub Action -├── tests/ -│ ├── test_schema.py # ✅ 6 tests -│ ├── test_config.py # ✅ 9 tests -│ ├── test_base_collector.py # ✅ 8 tests -│ ├── test_umpyre_collector.py # ✅ 5 tests (2 skipped) -│ └── test_coverage_collector.py # ✅ 6 tests -├── misc/ -│ └── CHANGELOG.md # ✅ Detailed changes -├── .github/ -│ └── umpyre-config.yml # ✅ Example config -├── README.md # ✅ Complete documentation -└── pyproject.toml # ✅ Updated with CLI entry point -``` - ---- - -## How to Use - -### 1. Installation - -```bash -pip install umpyre -``` - -### 2. Local Usage - -```bash -# Collect metrics (dry run) -umpyre collect --no-store - -# Collect and store to code-metrics branch -umpyre collect - -# Custom config -umpyre collect --config my-config.yml -``` - -### 3. GitHub Actions Integration - -Add to your workflow after successful PyPI publish: - -```yaml -- name: Track Code Metrics - if: success() - uses: i2mint/umpyre/actions/track-metrics@master - with: - github-token: ${{ secrets.GITHUB_TOKEN }} -``` - -### 4. Configuration - -Create `.github/umpyre-config.yml`: - -```yaml -schema_version: "1.0" - -collectors: - workflow_status: - enabled: true - coverage: - enabled: true - umpyre_stats: - enabled: true - exclude_dirs: [tests, examples] - -storage: - branch: code-metrics - formats: [json, csv] -``` - ---- - -## Design Patterns Used - -- **Mapping Interface**: Collectors provide dict-like access -- **Registry Pattern**: Dynamic collector registration -- **Open-Closed Principle**: Config-driven extensibility -- **Lazy Evaluation**: Metrics collected on first access -- **Dependency Injection**: Collectors configured via constructor -- **Facade Pattern**: Clean abstractions over complex tools - ---- - -## Testing Strategy - -All core components have comprehensive tests: - -```bash -pytest tests/ -v -# 32 passed, 2 skipped -``` - -**Test Coverage:** -- Schema: Creation, validation, migration -- Config: Loading, merging, validation -- Collectors: Mapping interface, registration, error handling -- Storage: Serialization (JSON, CSV) - ---- - -## Known Limitations - -1. **UmpyreCollector**: Inherited compatibility issues from `python_code_stats.py` - - May fail on some directory structures - - Tries to execute `setup.py` during analysis - - 2 tests skipped due to these issues - -2. **WilyCollector**: Requires wily installation and git history - -3. **WorkflowStatusCollector**: Subject to GitHub API rate limits (5000 req/hour with auth) - ---- - -## What's NOT Implemented (Future Phases) - -### Phase 2: Visualization & Aggregation -- Plot generation (matplotlib/plotly) -- README auto-generation with embedded charts -- Cross-repository aggregation -- Organization-wide dashboard -- GitHub Pages deployment - -### Phase 3: Advanced Features -- Additional collectors (bandit, interrogate) -- Threshold validation system with custom validators -- Data pruning and compression utilities -- Schema migration tools -- Advanced retention policies - ---- - -## Testing Recommendations - -Before deploying to production repos, test on these repositories as specified: - -1. **https://github.com/thorwhalen/astate** - Small, stable repo -2. **https://github.com/thorwhalen/ps** - Larger test case - -### Test Checklist: -```bash -# 1. Clone test repo -git clone https://github.com/thorwhalen/astate -cd astate - -# 2. Install umpyre -pip install umpyre - -# 3. Create config -cat > .github/umpyre-config.yml << EOF -schema_version: "1.0" -collectors: - coverage: - enabled: true - umpyre_stats: - enabled: true -storage: - branch: code-metrics - formats: [json] -EOF - -# 4. Test dry run -umpyre collect --no-store - -# 5. Test actual storage -umpyre collect - -# 6. Verify metrics branch -git fetch origin code-metrics -git checkout code-metrics -ls -la # Should see metrics.json, history/ -``` - ---- - -## Next Steps - -### Immediate (Optional Enhancements): -1. Add bandit and interrogate collectors -2. Implement threshold validation -3. Add pruning/compression utilities - -### Phase 2 (Visualization): -1. Create plot generation module -2. Build README generator with charts -3. Implement cross-repo aggregation -4. Create dashboard template - -### Phase 3 (Production Hardening): -1. Add schema migration utilities -2. Implement data retention policies -3. Add error recovery mechanisms -4. Create migration guide for schema updates - ---- - -## Success Criteria Met ✅ - -- ✅ Metrics collection completes in < 30 seconds per repo -- ✅ Handles 200+ repositories without rate limiting (via GitHub API) -- ✅ Stores data reliably in git branches -- ✅ Schema is versioned and migrations prepared -- ✅ Easy to add new metric collectors (registry pattern) -- ✅ Works with existing CI without breaking changes -- ✅ Comprehensive documentation and examples - ---- - -## Example Output - -After running `umpyre collect`, the `code-metrics` branch contains: - -``` -code-metrics branch/ -├── metrics.json # Latest snapshot -├── metrics.csv # Flat format -└── history/ - └── 2025-11/ - └── 2025-11-14_120530_abc1234.json -``` - -**metrics.json** structure: -```json -{ - "schema_version": "1.0", - "timestamp": "2025-11-14T12:05:30Z", - "commit_sha": "abc1234...", - "metrics": { - "coverage": { - "line_coverage": 87.5, - "branch_coverage": 82.1 - }, - "umpyre_stats": { - "num_functions": 342, - "num_classes": 28, - "total_lines": 5420 - } - }, - "collection_duration_seconds": 8.3 -} -``` - ---- - -## Conclusion - -**Phase 1 is production-ready** for basic metrics tracking. The system is: -- Config-driven and extensible -- Well-tested (32 passing tests) -- Documented with examples -- Integrated with GitHub Actions -- Designed for 200+ repo scale - -**Ready for pilot deployment** on test repositories. Phases 2 and 3 can be added incrementally based on user feedback. diff --git a/PHASE_2_PLAN.md b/PHASE_2_PLAN.md deleted file mode 100644 index ba72fc0..0000000 --- a/PHASE_2_PLAN.md +++ /dev/null @@ -1,432 +0,0 @@ -# Phase 2: Visualization & Aggregation - -## Overview - -Build visualization and cross-repository aggregation capabilities to make metrics actionable and provide organization-wide insights. - -**Estimated Duration**: 10-13 hours -**Dependencies**: Phase 1 complete ✅ - ---- - -## 2.1 Time-Series Plot Generation (3 hours) - -### Goal -Generate visual charts showing metric trends over time. - -### Features - -**Core Plots**: -- Maintainability index trend -- Test coverage trend (line & branch) -- Lines of code growth -- Cyclomatic complexity over time -- Function/class count evolution - -**Implementation**: -```python -# umpyre/visualization/plots.py - -class PlotGenerator: - """Generate time-series plots from metrics history.""" - - def __init__(self, metrics_branch: str = "code-metrics"): - self.branch = metrics_branch - - def generate_coverage_plot(self) -> Path: - """Generate coverage trend plot.""" - # Parse history/*.json files - # Extract coverage metrics over time - # Plot with matplotlib/plotly - # Save as PNG/SVG - pass - - def generate_maintainability_plot(self) -> Path: - """Generate maintainability trend plot.""" - pass - - def generate_all_plots(self) -> dict[str, Path]: - """Generate all standard plots.""" - pass -``` - -**Technical Approach**: -1. Read all `history/*.json` files -2. Parse timestamps from filenames (`YYYY_MM_DD_HH_MM_SS__sha__version.json`) -3. Extract target metrics -4. Sort by timestamp -5. Generate matplotlib/plotly charts -6. Save to `plots/` directory in code-metrics branch - -**Output Location**: -``` -code-metrics branch: -├── plots/ -│ ├── coverage_trend.png -│ ├── maintainability_trend.png -│ ├── loc_growth.png -│ └── complexity_trend.png -``` - -**Config Options**: -```yaml -visualization: - enabled: true - generate_plots: true - plot_metrics: - - coverage - - maintainability - - loc - - complexity - plot_format: png # or svg - lookback_days: 90 # Only plot last 90 days -``` - ---- - -## 2.2 README Auto-Generation (2 hours) - -### Goal -Automatically generate `METRICS.md` in code-metrics branch with summary tables and embedded plots. - -### Features - -**Content Sections**: -1. **Latest Metrics Summary** (table) -2. **Trend Charts** (embedded images) -3. **Historical Comparison** (table: current vs 1 week ago vs 1 month ago) -4. **Badge Generation** (shields.io format for GitHub README) - -**Example Output**: -```markdown -# Code Metrics Report - -Generated: 2025-11-14 22:45:00 UTC -Commit: 700e012 -Version: 0.1.0 - -## Latest Metrics - -| Metric | Value | Change (7d) | -|--------|-------|-------------| -| Test Coverage | 85.2% | +2.1% ↑ | -| Maintainability | 72.5 | -1.2 ↓ | -| Lines of Code | 2,750 | +45 ↑ | -| Cyclomatic Complexity | 3.2 | 0.0 → | - -## Trends - -### Coverage Over Time -![Coverage Trend](plots/coverage_trend.png) - -### Maintainability Index -![Maintainability](plots/maintainability_trend.png) - -## Badges - -![Coverage](https://img.shields.io/badge/coverage-85.2%25-brightgreen) -![Maintainability](https://img.shields.io/badge/maintainability-72.5-yellow) -``` - -**Implementation**: -```python -# umpyre/visualization/readme_generator.py - -class ReadmeGenerator: - """Generate METRICS.md from metrics history.""" - - def generate_summary_table(self) -> str: - """Create latest metrics table.""" - pass - - def generate_trend_section(self) -> str: - """Embed plot images.""" - pass - - def generate_badges(self) -> str: - """Create shields.io badges.""" - pass - - def generate_full_readme(self) -> str: - """Generate complete METRICS.md.""" - pass -``` - -**Badge Format**: -``` -https://img.shields.io/badge/coverage-85.2%25-brightgreen -https://img.shields.io/badge/maintainability-72.5-yellow -https://img.shields.io/badge/complexity-3.2-green -``` - ---- - -## 2.3 Cross-Repository Aggregation (4 hours) - -### Goal -Aggregate metrics across multiple repositories to provide organization-wide insights. - -### Features - -**Aggregation Metrics**: -- Average coverage across all repos -- Total lines of code (organization-wide) -- Repos below threshold counts -- Trend analysis (improving vs declining repos) -- Top/bottom performers - -**Implementation**: -```python -# umpyre/collectors/aggregation_collector.py - -class AggregationCollector(MetricCollector): - """Aggregate metrics across multiple repositories.""" - - def __init__( - self, - org: str, - repos: list[str], - github_token: Optional[str] = None - ): - """ - Initialize aggregator. - - Args: - org: GitHub organization name - repos: List of repository names - github_token: GitHub API token - """ - self.org = org - self.repos = repos - self.token = github_token - - def collect(self) -> dict: - """ - Aggregate metrics from multiple repos. - - Returns: - Aggregated metrics dictionary - """ - # For each repo: - # 1. Clone code-metrics branch - # 2. Read latest metrics.json - # 3. Aggregate - - return { - "total_repos": len(self.repos), - "avg_coverage": 82.5, - "total_loc": 125000, - "repos_below_threshold": 3, - "trending_up": 8, - "trending_down": 2, - "per_repo_summary": [...], - } -``` - -**Storage**: -- Store aggregated metrics in special "metrics-dashboard" repository -- Or in `.github` repository with organization-wide metrics - -**Config**: -```yaml -aggregation: - enabled: true - org: i2mint - repos: - - umpyre - - py2store - - creek - - dol - # ... or use "all" to auto-discover - schedule: daily # Run aggregation daily -``` - ---- - -## 2.4 Dashboard Generation (4 hours) - -### Goal -Create interactive HTML dashboard with organization-wide metrics. - -### Features - -**Dashboard Sections**: -1. **Overview Cards**: Total repos, avg coverage, total LOC -2. **Interactive Charts**: Plotly.js for zoom/pan -3. **Repository Table**: Sortable, filterable list of all repos -4. **Trend Indicators**: Up/down arrows, color coding -5. **Drill-Down**: Click repo → see detailed metrics - -**Tech Stack**: -- Static HTML/CSS/JS (no server needed) -- Plotly.js for interactive charts -- GitHub Pages hosting -- Data embedded as JSON - -**Implementation**: -```python -# umpyre/visualization/dashboard.py - -class DashboardGenerator: - """Generate interactive HTML dashboard.""" - - def generate_html(self, aggregated_metrics: dict) -> str: - """ - Generate dashboard HTML. - - Args: - aggregated_metrics: Output from AggregationCollector - - Returns: - HTML string - """ - # Use Jinja2 template - # Embed metrics as JSON - # Include Plotly.js CDN - # Generate interactive charts - pass -``` - -**Example Dashboard**: -``` -https://i2mint.github.io/metrics-dashboard/ - -┌─────────────────────────────────────────┐ -│ i2mint Metrics Dashboard │ -├─────────────────────────────────────────┤ -│ 📊 Total Repos: 25 │ -│ ✅ Avg Coverage: 82.5% │ -│ 📝 Total LOC: 125,000 │ -└─────────────────────────────────────────┘ - -[Interactive Chart: Coverage by Repo] -[Interactive Chart: LOC Distribution] - -┌───────────┬──────────┬────────┬──────────┐ -│ Repo │ Coverage │ LOC │ Status │ -├───────────┼──────────┼────────┼──────────┤ -│ umpyre │ 85% │ 2,750 │ ✓ Good │ -│ py2store │ 78% │ 8,200 │ ⚠ Fair │ -│ creek │ 92% │ 1,500 │ ✓ Great │ -└───────────┴──────────┴────────┴──────────┘ -``` - ---- - -## CLI Integration - -Add new commands to `umpyre.cli`: - -```bash -# Generate plots for current repo -python -m umpyre.cli visualize - -# Generate plots + README -python -m umpyre.cli visualize --with-readme - -# Aggregate metrics across org -python -m umpyre.cli aggregate --org i2mint - -# Generate dashboard -python -m umpyre.cli dashboard --org i2mint --output-dir ./dashboard -``` - ---- - -## GitHub Action Integration - -```yaml -# .github/workflows/metrics-visualization.yml - -name: Generate Metrics Visualizations - -on: - schedule: - - cron: '0 0 * * 0' # Weekly - workflow_dispatch: - -jobs: - visualize: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - with: - ref: code-metrics - - - name: Generate Plots - uses: i2mint/umpyre/actions/visualize@master - with: - generate-plots: true - generate-readme: true - - - name: Commit Visualizations - run: | - git config user.name "umpyre-bot" - git config user.email "umpyre@automated" - git add plots/ METRICS.md - git commit -m "Update visualizations [skip ci]" - git push -``` - ---- - -## Testing Strategy - -1. **Unit Tests**: - - Test plot generation with mock data - - Test README generation - - Test aggregation logic - -2. **Integration Tests**: - - Test on real umpyre repository - - Verify plots are created - - Verify README is valid markdown - -3. **Visual Tests**: - - Manual inspection of generated plots - - Check dashboard rendering - ---- - -## Dependencies - -**New Python Packages**: -- `matplotlib` or `plotly` (for plotting) -- `jinja2` (for templating) -- `pygithub` (for GitHub API in aggregation) - -**Update pyproject.toml**: -```toml -[project.optional-dependencies] -visualization = [ - "matplotlib>=3.5.0", - "plotly>=5.0.0", - "jinja2>=3.0.0", -] -aggregation = [ - "pygithub>=2.0.0", -] -``` - ---- - -## Success Criteria - -- [ ] Time-series plots generated for all key metrics -- [ ] METRICS.md auto-generated with tables and charts -- [ ] Cross-repo aggregation working for 5+ repos -- [ ] Interactive dashboard deployed to GitHub Pages -- [ ] CLI commands functional -- [ ] GitHub Action workflow working -- [ ] Documentation updated -- [ ] Tests passing - ---- - -## Future Enhancements (Beyond Phase 2) - -- Real-time dashboard updates (webhook-triggered) -- Slack/Discord notifications for threshold violations -- Comparison views (repo A vs repo B) -- Custom metric definitions -- Export to CSV/Excel for analysis diff --git a/PHASE_3_PLAN.md b/PHASE_3_PLAN.md deleted file mode 100644 index 4fed43e..0000000 --- a/PHASE_3_PLAN.md +++ /dev/null @@ -1,602 +0,0 @@ -# Phase 3: Advanced Features & Production Hardening - -## Overview - -Add production-ready features including threshold validation, additional collectors, data management, and schema migrations. - -**Estimated Duration**: 15-18 hours -**Dependencies**: Phase 2 complete - ---- - -## 3.1 Threshold Validation System (3 hours) - -### Goal -Enforce quality standards by validating metrics against configurable thresholds. - -### Features - -**Threshold Types**: -- Minimum thresholds (e.g., coverage >= 80%) -- Maximum thresholds (e.g., complexity <= 10) -- Delta thresholds (e.g., coverage drop <= 5%) -- Custom validators (Python functions) - -**Actions**: -- `warn`: Log warning, continue (exit code 0) -- `fail`: Exit with code 1, fail CI -- `notify`: Send notification (Slack/email) -- `comment`: Comment on PR - -**Implementation**: -```python -# umpyre/validation/thresholds.py - -@dataclass -class Threshold: - """Threshold definition.""" - metric_path: str # e.g., "metrics.coverage.line_coverage" - operator: str # ">=", "<=", "==", "!=", ">", "<" - value: float - action: str # "warn", "fail", "notify", "comment" - message: Optional[str] = None - -class ThresholdValidator: - """Validate metrics against thresholds.""" - - def __init__(self, thresholds: list[Threshold]): - self.thresholds = thresholds - - def validate(self, metrics: dict) -> list[ThresholdViolation]: - """ - Validate metrics against all thresholds. - - Returns: - List of violations - """ - violations = [] - - for threshold in self.thresholds: - value = self._extract_metric(metrics, threshold.metric_path) - if not self._check_threshold(value, threshold): - violations.append( - ThresholdViolation( - threshold=threshold, - actual_value=value, - expected_value=threshold.value, - ) - ) - - return violations - - def _check_threshold(self, value: float, threshold: Threshold) -> bool: - """Check if value passes threshold.""" - ops = { - ">=": lambda a, b: a >= b, - "<=": lambda a, b: a <= b, - "==": lambda a, b: a == b, - "!=": lambda a, b: a != b, - ">": lambda a, b: a > b, - "<": lambda a, b: a < b, - } - return ops[threshold.operator](value, threshold.value) -``` - -**Config**: -```yaml -thresholds: - enabled: true - rules: - # Minimum coverage - - metric: metrics.coverage.line_coverage - operator: ">=" - value: 80.0 - action: fail - message: "Test coverage must be at least 80%" - - # Maximum complexity - - metric: metrics.wily.cyclomatic_avg - operator: "<=" - value: 10.0 - action: warn - message: "Average complexity should be below 10" - - # Delta threshold (vs previous commit) - - metric: metrics.coverage.line_coverage - operator: ">=" - value: -5.0 # Max 5% drop - delta: true - action: fail - message: "Coverage dropped by more than 5%" - - # Custom validator - - metric: custom - validator: "umpyre.validators.check_docstring_ratio" - action: warn -``` - -**Custom Validators**: -```python -# umpyre/validators.py - -def check_docstring_ratio(metrics: dict) -> tuple[bool, str]: - """ - Custom validator: check if docstring ratio is acceptable. - - Returns: - (passed, message) - """ - docs = metrics["metrics"]["umpyre_stats"]["docs_lines"] - total = metrics["metrics"]["umpyre_stats"]["total_lines"] - ratio = docs / total if total > 0 else 0 - - if ratio < 0.1: - return False, f"Docstring ratio {ratio:.1%} is below 10%" - return True, "Docstring ratio acceptable" -``` - -**CLI Integration**: -```bash -# Validate metrics against thresholds -python -m umpyre.cli validate - -# Validate and collect -python -m umpyre.cli collect --validate - -# Show violations only -python -m umpyre.cli validate --violations-only -``` - ---- - -## 3.2 Additional Collectors (8 hours) - -### 3.2.1 BanditCollector (2 hours) - -**Purpose**: Security vulnerability scanning - -```python -# umpyre/collectors/bandit_collector.py - -class BanditCollector(MetricCollector): - """Collect security metrics using Bandit.""" - - def collect(self) -> dict: - """ - Run bandit security scan. - - Returns: - { - "total_issues": 12, - "high_severity": 2, - "medium_severity": 5, - "low_severity": 5, - "confidence_high": 3, - "confidence_medium": 6, - "confidence_low": 3, - "files_scanned": 25, - "top_issues": [ - {"test_id": "B101", "count": 4, "severity": "medium"}, - ... - ] - } - """ - # Run: bandit -r . -f json - # Parse JSON output - # Aggregate by severity/confidence - pass -``` - -**Config**: -```yaml -collectors: - bandit: - enabled: true - config_file: .bandit # Optional - exclude_dirs: [tests, examples] - severity_threshold: medium # low, medium, high -``` - ---- - -### 3.2.2 InterrogateCollector (2 hours) - -**Purpose**: Docstring coverage analysis - -```python -# umpyre/collectors/interrogate_collector.py - -class InterrogateCollector(MetricCollector): - """Collect docstring coverage using interrogate.""" - - def collect(self) -> dict: - """ - Run interrogate to check docstring coverage. - - Returns: - { - "coverage": 78.5, - "functions_with_docs": 45, - "functions_without_docs": 12, - "classes_with_docs": 8, - "classes_without_docs": 2, - "files_analyzed": 15, - } - """ - # Run: interrogate . -vv --generate-badge . --quiet - # Parse output - pass -``` - ---- - -### 3.2.3 MyPyCollector (2 hours) - -**Purpose**: Type hint coverage and type checking - -```python -# umpyre/collectors/mypy_collector.py - -class MyPyCollector(MetricCollector): - """Collect type checking results using mypy.""" - - def collect(self) -> dict: - """ - Run mypy type checker. - - Returns: - { - "total_errors": 15, - "error_types": { - "type-error": 8, - "attr-defined": 4, - "arg-type": 3, - }, - "files_checked": 25, - "functions_typed": 45, - "functions_untyped": 12, - "type_coverage": 78.9, - } - """ - # Run: mypy . --show-error-codes --no-error-summary - # Parse output - # Calculate type coverage - pass -``` - ---- - -### 3.2.4 PylintCollector (2 hours) - -**Purpose**: Code quality scoring - -```python -# umpyre/collectors/pylint_collector.py - -class PylintCollector(MetricCollector): - """Collect code quality score using Pylint.""" - - def collect(self) -> dict: - """ - Run pylint code analysis. - - Returns: - { - "score": 8.5, # Out of 10 - "total_statements": 1250, - "convention": 12, - "refactor": 5, - "warning": 8, - "error": 2, - "files_analyzed": 15, - } - """ - # Run: pylint --output-format=json . - # Parse JSON - # Extract score and violations - pass -``` - ---- - -## 3.3 Data Management (3 hours) - -### Goal -Manage metrics history efficiently (pruning, compression, rotation). - -### 3.3.1 Pruning Strategy - -**Keep**: -- All metrics from last 30 days -- One metric per week for 30-90 days ago -- One metric per month for 90+ days ago - -```python -# umpyre/storage/pruning.py - -class MetricsPruner: - """Prune old metrics to save space.""" - - def prune(self, branch: str = "code-metrics"): - """ - Prune metrics history based on retention policy. - - Strategy: - - Keep all metrics < 30 days old - - Keep weekly snapshots for 30-90 days - - Keep monthly snapshots for 90+ days - """ - # Parse all history/*.json filenames - # Extract timestamps - # Apply retention policy - # Delete old files - pass -``` - -**Config**: -```yaml -storage: - retention: - strategy: tiered # all, tiered, custom - keep_days: 30 - weekly_after_days: 30 - monthly_after_days: 90 - prune_on_collect: false # Prune automatically after each collection -``` - ---- - -### 3.3.2 Compression - -**Goal**: gzip old metrics to save space - -```python -# umpyre/storage/compression.py - -class MetricsCompressor: - """Compress old metrics files.""" - - def compress_old_files( - self, - branch: str = "code-metrics", - older_than_days: int = 90 - ): - """ - Compress metrics older than N days. - - Changes: - - 2024_01_15_10_30_00__abc1234__0.1.0.json - → 2024_01_15_10_30_00__abc1234__0.1.0.json.gz - """ - # Find files older than N days - # gzip compress - # Delete original - pass -``` - ---- - -### 3.3.3 Rotation - -**Goal**: Auto-delete very old metrics - -```python -# umpyre/storage/rotation.py - -class MetricsRotator: - """Rotate out very old metrics.""" - - def rotate( - self, - branch: str = "code-metrics", - max_age_days: int = 365 - ): - """Delete metrics older than max_age_days.""" - # Parse timestamps - # Delete files older than threshold - pass -``` - -**Config**: -```yaml -storage: - rotation: - enabled: true - max_age_days: 365 - warn_before_delete: true -``` - ---- - -## 3.4 Schema Migrations (4 hours) - -### Goal -Handle schema version changes gracefully with automatic migrations. - -### Implementation - -```python -# umpyre/schema.py (enhanced) - -class MetricSchema: - """Versioned schema with migration support.""" - - version: str = "1.1" # Bump version - - @classmethod - def migrate(cls, data: dict, from_version: str) -> dict: - """ - Migrate data from old schema to current. - - Migration chain: - 1.0 → 1.1 → 1.2 → ... - """ - if from_version == cls.current_version(): - return data - - # Migration registry - migrations = { - "1.0": cls._migrate_1_0_to_1_1, - "1.1": cls._migrate_1_1_to_1_2, - } - - # Apply migrations in sequence - current_version = from_version - current_data = data - - while current_version != cls.current_version(): - if current_version not in migrations: - raise ValueError(f"No migration path from {current_version}") - - migrator = migrations[current_version] - current_data = migrator(current_data) - current_version = cls._next_version(current_version) - - return current_data - - @classmethod - def _migrate_1_0_to_1_1(cls, data: dict) -> dict: - """ - Migrate from schema 1.0 to 1.1. - - Changes in 1.1: - - Added pypi_version field - - Added collection_duration_seconds - """ - # Add new fields with defaults - if "metrics" in data: - for collector_name, metrics in data["metrics"].items(): - if "pypi_version" not in metrics: - metrics["pypi_version"] = None - - data["schema_version"] = "1.1" - return data - - @classmethod - def _migrate_1_1_to_1_2(cls, data: dict) -> dict: - """Future migration example.""" - # Apply changes for 1.2 - data["schema_version"] = "1.2" - return data -``` - -**Auto-Migration on Load**: -```python -# umpyre/storage/formats.py (enhanced) - -def load_metrics(filepath: Path) -> dict: - """ - Load metrics with automatic schema migration. - - If old schema detected, automatically migrates to current. - """ - with open(filepath) as f: - data = json.load(f) - - schema_version = data.get("schema_version", "1.0") - current_version = MetricSchema.current_version() - - if schema_version != current_version: - print(f"Migrating from schema {schema_version} to {current_version}") - data = MetricSchema.migrate(data, from_version=schema_version) - - return data -``` - -**Migration Tool**: -```bash -# Migrate all metrics in branch -python -m umpyre.cli migrate --branch code-metrics - -# Dry run (show what would change) -python -m umpyre.cli migrate --dry-run -``` - ---- - -## CLI Integration - -New commands for Phase 3: - -```bash -# Validate against thresholds -python -m umpyre.cli validate - -# Prune old metrics -python -m umpyre.cli prune --older-than 90 - -# Compress old metrics -python -m umpyre.cli compress - -# Migrate schema -python -m umpyre.cli migrate - -# Run all collectors (including new ones) -python -m umpyre.cli collect --all -``` - ---- - -## Testing Strategy - -1. **Threshold Tests**: - - Test all operators (>=, <=, ==, !=, >, <) - - Test delta thresholds - - Test custom validators - - Test actions (warn, fail, notify) - -2. **Collector Tests**: - - Test each new collector with sample repos - - Test error handling - - Test config options - -3. **Data Management Tests**: - - Test pruning logic - - Test compression/decompression - - Test rotation - -4. **Migration Tests**: - - Create sample data in old schema - - Verify migration to new schema - - Test migration chains (1.0 → 1.1 → 1.2) - ---- - -## Dependencies - -**New Python Packages**: -```toml -[project.optional-dependencies] -security = [ - "bandit>=1.7.0", -] -quality = [ - "interrogate>=1.5.0", - "mypy>=1.0.0", - "pylint>=2.15.0", -] -``` - ---- - -## Success Criteria - -- [ ] Threshold validation working with all operators -- [ ] 4 new collectors implemented (bandit, interrogate, mypy, pylint) -- [ ] Pruning/compression/rotation working -- [ ] Schema migration system working -- [ ] CLI commands functional -- [ ] All tests passing -- [ ] Documentation updated -- [ ] Phase 3 complete and production-ready - ---- - -## Future Enhancements (Beyond Phase 3) - -- Machine learning predictions (predict when coverage will drop) -- Anomaly detection (unusual metric changes) -- Automated PR comments with metrics -- Slack/Discord bot integration -- Web API for querying metrics -- Export to data warehouses (BigQuery, Snowflake) diff --git a/QUICK_START.md b/QUICK_START.md deleted file mode 100644 index ab10407..0000000 --- a/QUICK_START.md +++ /dev/null @@ -1,212 +0,0 @@ -# Umpyre Quick Start Guide - -## Installation - -```bash -pip install umpyre -``` - -## Basic Usage (5 minutes) - -### 1. Test Locally (Dry Run) - -```bash -# In any Python repo with tests -cd /path/to/your/repo - -# Collect metrics without storing -umpyre collect --no-store -``` - -You'll see output like: -``` -Collecting metrics for abc1234... -Metrics collected (not stored): -{ - "schema_version": "1.0", - "timestamp": "2025-11-14T12:00:00Z", - ... -} -Collection completed in 3.45s -``` - -### 2. Store Metrics Locally - -```bash -# This creates a 'code-metrics' branch -umpyre collect -``` - -Check the branch: -```bash -git fetch origin code-metrics # If you pushed -git checkout code-metrics -ls -la # See metrics.json, metrics.csv, history/ -``` - -### 3. Customize Configuration - -Create `.github/umpyre-config.yml`: - -```yaml -schema_version: "1.0" - -collectors: - coverage: - enabled: true - umpyre_stats: - enabled: true - exclude_dirs: [tests, examples, scrap] - -storage: - branch: code-metrics - formats: [json, csv] -``` - -Run again: -```bash -umpyre collect --config .github/umpyre-config.yml -``` - -### 4. Add to GitHub Actions - -In `.github/workflows/ci.yml`, add after your tests: - -```yaml -jobs: - test-and-publish: - runs-on: ubuntu-latest - steps: - # ... your existing steps ... - - - name: Track Code Metrics - if: success() # Only after successful tests - uses: i2mint/umpyre/actions/track-metrics@master - with: - github-token: ${{ secrets.GITHUB_TOKEN }} -``` - -Commit and push - metrics will be tracked automatically! - -## What Gets Tracked? - -By default (if tools are available): - -- ✅ **Test Coverage** (from pytest-cov or coverage.py) - - Line coverage % - - Branch coverage % - -- ✅ **Code Statistics** (built-in analyzer) - - Number of functions and classes - - Lines of code (total, empty, comments, docs) - - Code ratios - -- ✅ **CI Health** (from GitHub API, in Actions only) - - Last run status - - Recent failure count - -- ⚠️ **Complexity** (requires `pip install wily`) - - Cyclomatic complexity - - Maintainability index - -## Viewing Metrics - -### In Git Branch - -```bash -git checkout code-metrics -cat metrics.json # Latest snapshot -cat metrics.csv # Flat format for analysis -ls history/ # Historical records by month -``` - -### In Python - -```python -from umpyre.storage import deserialize_metrics - -metrics = deserialize_metrics("metrics.json", format="json") -print(metrics["metrics"]["coverage"]["line_coverage"]) -# 87.5 -``` - -### With Pandas - -```python -import pandas as pd - -# Load historical data -df = pd.read_csv("metrics.csv") -print(df.head()) - -# Or load multiple history files -import glob -import json - -history = [] -for file in glob.glob("history/*/*.json"): - with open(file) as f: - history.append(json.load(f)) - -df = pd.DataFrame(history) -``` - -## Common Use Cases - -### Track Coverage Over Time - -```bash -# Collect after each test run -pytest --cov=mypackage --cov-report=json -umpyre collect -``` - -### Only Track Specific Metrics - -`.github/umpyre-config.yml`: -```yaml -collectors: - coverage: - enabled: true - umpyre_stats: - enabled: false # Disable - wily: - enabled: false # Disable -``` - -### Custom Branch Name - -```yaml -storage: - branch: my-metrics # Instead of code-metrics -``` - -## Troubleshooting - -### "No coverage file found" -- Run tests with `--cov-report=json` or `--cov-report=xml` -- Coverage file should be in repo root - -### "wily not installed" -- Install: `pip install wily` -- Or disable in config: `wily: { enabled: false }` - -### "Not a git repository" -- Umpyre requires git for storage -- Initialize: `git init` - -### UmpyreCollector returns 0 -- Known issue with some directory structures -- Disable in config if problematic: `umpyre_stats: { enabled: false }` - -## Next Steps - -- 📖 Read full docs: `README.md` -- 🔧 See all config options: `.github/umpyre-config.yml` -- 🧪 Test on repos: `astate`, `ps` -- 📊 Coming soon: Visualization and dashboards! - -## Get Help - -- Issues: https://github.com/i2mint/umpyre/issues -- Docs: See `README.md` and `IMPLEMENTATION_SUMMARY.md` diff --git a/STORAGE_STRUCTURE.md b/STORAGE_STRUCTURE.md deleted file mode 100644 index 2d1c95a..0000000 --- a/STORAGE_STRUCTURE.md +++ /dev/null @@ -1,225 +0,0 @@ -# Metrics Storage Structure - -## Overview - -Metrics are stored in a separate `code-metrics` git branch with a **flat, chronologically-ordered structure** using parseable filenames: `YYYY_MM_DD_HH_MM_SS__shahash__version.json` - -This design follows **SSOT principles** (no data duplication) while enabling efficient querying by time, commit, or version. - -## Storage Location - -**Without** `--no-store` flag: Metrics are pushed to the `code-metrics` branch in the same repository. - -**With** `--no-store` flag: Metrics are collected and displayed but not stored. - -## Branch Structure - -``` -code-metrics branch: -├── metrics.json # Latest snapshot -├── metrics.csv # Latest CSV -└── history/ # Flat history (SSOT!) - ├── 2025_11_14_22_45_00__700e012__0.1.0.json - ├── 2025_11_14_22_50_15__abc1234__0.1.1.json - ├── 2025_11_15_10_30_22__def5678__0.1.1.json - ├── 2025_11_15_14_20_00__9a8b7c6__none.json # No version detected - └── 2025_11_16_09_00_00__1234567__0.2.0.json -``` - -## Filename Format - -**Pattern**: `{YYYY_MM_DD_HH_MM_SS}__{commit_sha[:7]}__{pypi_version}.json` - -**Components**: -1. **Timestamp**: `YYYY_MM_DD_HH_MM_SS` - Chronological ordering -2. **Commit SHA**: 7-character short hash - Uniqueness guarantee -3. **PyPI Version**: Semantic version or `none` - Easy filtering - -**Examples**: -- `2025_11_14_22_45_00__700e012__0.1.0.json` -- `2025_11_15_10_30_22__def5678__none.json` (no version found) - -## Key Design Decisions - -### 1. Flat Structure (No Nested Folders) ✅ - -**Why**: Simplicity and easy querying -- No need to know which month to look in -- Simple `ls` or glob patterns -- Easy to parse all metrics -- No directory traversal overhead - -### 2. Chronological Ordering ✅ - -**Why**: Natural time-based queries -- Files are already sorted by time (filename sorting) -- Easy to find "latest N metrics" -- Easy to filter by date range -- Works with standard shell tools - -### 3. Commit SHA for Uniqueness ✅ - -**Why**: One entry per commit -- Guarantees no duplicates (same commit = overwrite) -- Git-traceable (link to exact code state) -- Works even without version tags -- 7 chars enough for uniqueness in practice - -### 4. PyPI Version for Filtering ✅ - -**Why**: Easy version-based queries without duplication -- No separate `by_version/` folder (SSOT!) -- Simple grep/filter to find version -- Supports repos without versions (`none`) -- Parseable from filename - -### 5. No Data Duplication (SSOT) ✅ - -**Why**: Single source of truth -- One file = one commit's metrics -- No redundant storage in `by_version/` -- Easier to maintain consistency -- Smaller repository size - -## Querying Patterns - -### Query by Time Range - -```bash -# Get all metrics from November 2025 -git checkout code-metrics -ls history/2025_11_* - -# Get metrics from specific date -ls history/2025_11_14_* - -# Get latest 10 metrics -ls -t history/ | head -10 -``` - -### Query by Commit SHA - -```bash -# Find metrics for commit 700e012 -git checkout code-metrics -ls history/*__700e012__* - -# Or with grep -ls history/ | grep "700e012" -``` - -### Query by Version - -```bash -# Find all metrics for version 0.1.0 -git checkout code-metrics -ls history/*__0.1.0.json - -# Find all metrics with a version (exclude 'none') -ls history/ | grep -v "__none.json" - -# Get latest metric for version 0.1.0 -ls -t history/*__0.1.0.json | head -1 -``` - -### Parse Filename Components - -```python -import re -from pathlib import Path - -def parse_metric_filename(filename: str) -> dict: - """ - Parse metric filename into components. - - Example: "2025_11_14_22_45_00__700e012__0.1.0.json" - Returns: { - "timestamp": "2025-11-14T22:45:00", - "commit_sha": "700e012", - "pypi_version": "0.1.0", - } - """ - pattern = r'(\d{4}_\d{2}_\d{2}_\d{2}_\d{2}_\d{2})__(\w{7})__(.+)\.json' - match = re.match(pattern, filename) - - if not match: - raise ValueError(f"Invalid filename format: {filename}") - - timestamp_str, sha, version = match.groups() - - # Convert timestamp to ISO format - timestamp = timestamp_str.replace('_', '-', 2).replace('_', ':', 2).replace('_', 'T', 1) - - return { - "timestamp": timestamp, - "commit_sha": sha, - "pypi_version": None if version == "none" else version, - } - -# Usage -for file in Path("history").glob("*.json"): - info = parse_metric_filename(file.name) - print(f"Commit {info['commit_sha']} at {info['timestamp']} (v{info['pypi_version']})") -``` - -## Metrics Schema - -```json -{ - "schema_version": "1.0", - "timestamp": "2025-11-14T22:45:00Z", - "commit_sha": "700e012d85d1393de95d0634eec8efa224ff0bc9", - "commit_message": "Refactor collector...", - "python_version": "3.12", - "metrics": { - "umpyre_stats": { - "num_functions": 134, - "num_classes": 13, - "total_lines": 2750, - "pypi_version": "0.1.0", ← Used in filename! - "files_analyzed": 16, - ... - }, - "coverage": { ... }, - "wily": { ... } - }, - "collection_duration_seconds": 3.42 -} -``` - -## Benefits of This Structure - -1. **SSOT**: No data duplication (no `by_version/` folder) -2. **Chronological**: Files naturally sorted by time -3. **Parseable**: All info in filename (timestamp, commit, version) -4. **Unique**: One entry per commit (commit SHA) -5. **Queryable**: Easy to filter by time, commit, or version -6. **Simple**: Flat structure, no directory traversal -7. **Scalable**: Works with thousands of metrics files -8. **Shell-Friendly**: Standard tools (ls, grep, sort) work perfectly - -## CI Workflow Integration - -When running in CI (e.g., GitHub Actions): - -```yaml -- name: Track Code Metrics - uses: ./actions/track-metrics - with: - branch: code-metrics - continue-on-error: true # Never fails CI -``` - -On each CI run after PyPI publish: -1. Collects metrics (including pypi_version from pyproject.toml) -2. Stores as `history/{timestamp}__{commit_sha}__{version}.json` -3. Updates `metrics.json` (latest snapshot) -4. No duplication, no nested folders - -## Migration from Old Structure - -If you have old structure with monthly folders or `by_version/`: -1. Old metrics remain accessible -2. New metrics use flat structure -3. No breaking changes -4. Can run migration script to flatten old structure (optional) diff --git a/TESTING_GUIDE.md b/TESTING_GUIDE.md deleted file mode 100644 index 3b04e47..0000000 --- a/TESTING_GUIDE.md +++ /dev/null @@ -1,377 +0,0 @@ -# Testing Guide for Umpyre on astate and ps Repositories - -## Test Repositories - -1. **https://github.com/thorwhalen/astate** - Small, stable repo (recommended first test) -2. **https://github.com/thorwhalen/ps** - Larger repo for scale testing - ---- - -## Pre-Test Setup - -```bash -# Install umpyre in development mode -cd /path/to/umpyre -pip install -e . - -# Verify installation -umpyre --help -python -m pytest tests/ -v # Should pass 32 tests -``` - ---- - -## Test 1: astate Repository - -### Step 1: Clone and Setup - -```bash -cd ~/test_metrics # Or wherever you want to test -git clone https://github.com/thorwhalen/astate -cd astate - -# Check if it has tests and coverage -pytest --cov=astate --cov-report=json # Generate coverage data -ls -la coverage.json # Should exist -``` - -### Step 2: Create Config - -```bash -mkdir -p .github -cat > .github/umpyre-config.yml << 'EOF' -schema_version: "1.0" - -collectors: - coverage: - enabled: true - source: pytest-cov - - umpyre_stats: - enabled: true - exclude_dirs: [tests, examples, scrap] - - workflow_status: - enabled: false # Requires GitHub token - -storage: - branch: code-metrics-test - formats: [json, csv] -EOF -``` - -### Step 3: Dry Run - -```bash -# Test without storing -umpyre collect --no-store - -# Expected output: -# - Collecting metrics for ... -# - Metrics collected (not stored): -# - {JSON output} -# - Collection completed in X.XXs - -# Check for errors -echo $? # Should be 0 -``` - -### Step 4: Store Metrics - -```bash -# Create metrics branch -umpyre collect --config .github/umpyre-config.yml - -# Verify branch was created -git branch -a | grep code-metrics-test -# Should see: code-metrics-test - -# Check branch contents -git checkout code-metrics-test -ls -la -# Expected files: -# - metrics.json -# - metrics.csv -# - history/YYYY-MM/*.json -``` - -### Step 5: Verify Metrics - -```bash -# Check JSON structure -cat metrics.json | python -m json.tool | head -20 - -# Check CSV -cat metrics.csv | head -10 - -# Verify historical record -ls history/$(date +%Y-%m)/ -``` - -### Step 6: Test Multiple Collections - -```bash -# Switch back to main branch -git checkout main - -# Make a small change -echo "# Test comment" >> README.md -git add README.md -git commit -m "Test: trigger new metrics collection" - -# Collect again -umpyre collect - -# Verify new entry in history -git checkout code-metrics-test -ls history/$(date +%Y-%m)/ -# Should have 2 files now -``` - ---- - -## Test 2: ps Repository - -### Step 1: Clone and Setup - -```bash -cd ~/test_metrics -git clone https://github.com/thorwhalen/ps -cd ps - -# Run tests if available -pytest --cov=ps --cov-report=json -``` - -### Step 2: Create Config (Same as astate) - -```bash -mkdir -p .github -# Copy same config from astate test -``` - -### Step 3: Performance Test - -```bash -# Time the collection -time umpyre collect --no-store - -# Expected: < 30 seconds -# If slower, check which collector is taking time -``` - -### Step 4: Test with Wily (Optional) - -```bash -# Install wily -pip install wily - -# Update config to enable wily -cat >> .github/umpyre-config.yml << 'EOF' -collectors: - wily: - enabled: true - max_revisions: 3 # Keep it small for testing -EOF - -# Build wily cache (first time is slow) -wily build . --max-revisions 3 - -# Collect with wily -umpyre collect -``` - ---- - -## Test 3: GitHub Actions Integration (Optional) - -If you have write access to a fork: - -### Create Test Workflow - -```yaml -# .github/workflows/test-metrics.yml -name: Test Metrics Collection - -on: - push: - branches: [test-metrics] - -jobs: - test: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@v4 - - - name: Set up Python - uses: actions/setup-python@v5 - with: - python-version: '3.10' - - - name: Install dependencies - run: | - pip install -e . - pip install pytest pytest-cov - - - name: Run tests - run: | - pytest --cov=astate --cov-report=json - - - name: Track Metrics - if: success() - uses: i2mint/umpyre/actions/track-metrics@master - with: - github-token: ${{ secrets.GITHUB_TOKEN }} -``` - -### Push and Test - -```bash -git checkout -b test-metrics -git add .github/workflows/test-metrics.yml -git commit -m "Add metrics tracking workflow" -git push origin test-metrics - -# Check Actions tab in GitHub UI -# Verify code-metrics-test branch is created/updated -``` - ---- - -## Verification Checklist - -### For Each Repository: - -- [ ] `umpyre collect --no-store` runs without errors -- [ ] Collection completes in < 30 seconds -- [ ] `code-metrics-test` branch is created -- [ ] `metrics.json` exists and is valid JSON -- [ ] `metrics.csv` exists and is valid CSV -- [ ] `history/` directory has monthly subdirectories -- [ ] Multiple collections create separate history files -- [ ] Coverage metrics are present (if tests exist) -- [ ] Code stats metrics are present -- [ ] No sensitive data in metrics files -- [ ] Branch can be pushed to remote - -### Expected Metrics Structure: - -```json -{ - "schema_version": "1.0", - "timestamp": "2025-11-14T...", - "commit_sha": "...", - "metrics": { - "coverage": { - "line_coverage": , - "branch_coverage": - }, - "umpyre_stats": { - "num_functions": , - "num_classes": , - "total_lines": - } - } -} -``` - ---- - -## Common Issues and Solutions - -### Issue: "No coverage file found" - -**Solution:** -```bash -# Generate coverage data first -pytest --cov= --cov-report=json -# Then collect -umpyre collect -``` - -### Issue: "Not a git repository" - -**Solution:** -```bash -git init -git add . -git commit -m "Initial commit" -``` - -### Issue: UmpyreCollector returns 0 functions - -**Solution:** This is a known issue. Disable in config: -```yaml -collectors: - umpyre_stats: - enabled: false -``` - -### Issue: Collection is slow (>30s) - -**Check:** -- Disable wily if not needed -- Reduce `max_revisions` for wily -- Check if repository is very large - -### Issue: Branch push fails - -**Solution:** -```bash -# Pull first -git checkout code-metrics-test -git pull origin code-metrics-test -# Then try collect again -``` - ---- - -## Success Criteria - -✅ Both repositories successfully collect metrics -✅ Collection time < 30 seconds each -✅ Metrics stored correctly in git branches -✅ JSON and CSV formats both valid -✅ Historical tracking works across multiple runs -✅ No errors in dry-run mode -✅ No sensitive data exposed in metrics - ---- - -## Reporting Results - -After testing, document: - -1. **Performance:** - - Collection time for each repo - - Any timeouts or slow operations - -2. **Metrics Quality:** - - Which collectors worked - - Which had issues - - Accuracy of metrics - -3. **Usability:** - - Ease of setup - - Clarity of output - - Any confusing errors - -4. **Issues Found:** - - Error messages - - Unexpected behavior - - Suggestions for improvement - ---- - -## Next Steps After Testing - -If tests pass: -1. Document any issues found -2. Consider enabling on production repos -3. Plan Phase 2 (visualization) based on needs - -If tests fail: -1. Check error messages -2. Verify environment setup -3. Review logs in `IMPLEMENTATION_SUMMARY.md` -4. Open issues with details diff --git a/TEST_RESULTS.md b/TEST_RESULTS.md deleted file mode 100644 index f777fd4..0000000 --- a/TEST_RESULTS.md +++ /dev/null @@ -1,129 +0,0 @@ -# Umpyre Test Results - astate Repository - -**Test Date:** November 14, 2025 -**Repository:** `/Users/thorwhalen/Dropbox/py/proj/t/astate` -**Test Script:** `test_on_astate.py` - -## Executive Summary - -✅ **ALL TESTS PASSED** (10/10 passed, 1 warning) - -The umpyre metrics collection system successfully validated on the astate repository. Core functionality is working correctly: configuration loading, schema validation, metric collection, and serialization all work as expected. - -## Test Results - -### TEST 1: Registry ✅ -- **Status:** PASSED -- **Details:** Found 4 collectors registered: `umpyre_stats`, `workflow_status`, `wily`, `coverage` - -### TEST 2: Configuration ✅✅ -- **Status:** PASSED (2/2 checks) -- **Details:** - - Successfully loaded YAML configuration from `.umpyre.yml` - - Configuration correctly identified enabled collectors - -### TEST 3: Schema ✅✅ -- **Status:** PASSED (2/2 checks) -- **Details:** - - Sample metrics validated successfully - - All required schema fields present (`schema_version`, `timestamp`, `commit_sha`, `metrics`) - -### TEST 4: Individual Collectors ⚠️ -- **Status:** 1 warning -- **Details:** - - **Coverage Collector:** ⚠️ Warning - "No coverage file found" (expected - astate has no coverage reports) - - **Umpyre Stats Collector:** Skipped due to known issues with `setup.py` execution - -### TEST 5: Full Collection Pipeline ✅✅✅✅✅ -- **Status:** PASSED (5/5 checks) -- **Details:** - - Successfully collected metrics from 1 collector (coverage) - - Output schema validation passed - - JSON serialization successful (453 bytes) - - JSON deserialization round-trip successful - - **Performance:** 0.02s (well under 30s target) - -## Known Issues - -### 1. UmpyreCollector - FIXED ✅ -- **Severity:** ~~Medium~~ **RESOLVED** -- **Description:** ~~The `UmpyreCollector` attempted to execute `setup.py` during analysis, causing errors~~ **Fixed by switching to AST-based parsing** -- **Impact:** ~~Cannot collect code statistics~~ **Now works safely on all Python files** -- **Solution:** Reimplemented using Python's AST module instead of dynamic imports. No code execution occurs during analysis. -- **Status:** **FIXED** - All tests passing, works on astate and umpyre itself - -### 2. Coverage Collector - No Coverage Files -- **Severity:** Low (expected) -- **Description:** Coverage collector reports "No coverage file found" -- **Impact:** None - this is expected when a repository doesn't have coverage reports -- **Workaround:** Run pytest with coverage before collection, or skip this collector -- **Status:** Working as designed - -## What Works - -✅ **Configuration System** -- YAML config loading -- Config validation -- Collector enable/disable -- Config merging with defaults - -✅ **Schema System** -- Versioned schema (v1.0) -- Metric data creation -- Schema validation -- Timezone-aware timestamps - -✅ **Collector Registry** -- Dynamic collector registration -- Collector lookup by name -- List available collectors - -✅ **Serialization** -- JSON serialization to file -- JSON deserialization from file -- Round-trip data integrity - -✅ **Performance** -- Collection completed in 0.02s -- Well under the 30s target - -## Recommendations - -### For Immediate Use - -1. **All collectors now stable** - coverage, workflow_status, and umpyre_stats all work correctly -2. **umpyre_stats collector now safe** - Uses AST parsing, no code execution -3. **Run pytest with coverage first** - If you want coverage metrics: - ```bash - pytest --cov=. --cov-report=json --cov-report=xml - ``` - -### For astate Specifically - -The astate repository is a good test case because: -- It's a real Python project -- It has a standard structure -- Collection is very fast (0.02s) - -However, to get full metrics: -1. Generate coverage reports first -2. Ensure GitHub Actions are enabled (for workflow_status) -3. Install wily separately if you want complexity metrics - -### Next Steps - -1. ✅ **Validated on astate** - Core functionality confirmed -2. ✅ **Fixed UmpyreCollector** - Now uses safe AST-based parsing -3. 🔲 **Test on ps repository** - Test with larger codebase -4. 🔲 **Test with actual coverage** - Run tests with coverage enabled -5. 🔲 **Test storage system** - Validate git branch storage (needs git push permissions) - -## Conclusion - -The umpyre system is **production-ready for Phase 1 features**: -- ✅ All core collectors working (coverage, workflow_status, umpyre_stats) -- ✅ UmpyreCollector fixed with safe AST-based parsing -- ✅ Coverage collector requires pre-generated coverage reports (by design) -- ⚠️ Storage to git branches not yet tested (would need write permissions) - -All core architectural components (config, schema, collectors, serialization) are working correctly and meet performance requirements. **34/34 unit tests passing.** diff --git a/actions/track-metrics/action.yml b/actions/track-metrics/action.yml deleted file mode 100644 index dfec82c..0000000 --- a/actions/track-metrics/action.yml +++ /dev/null @@ -1,74 +0,0 @@ -name: Track Code Metrics -description: Collect and store code quality metrics for CI/CD pipelines - -inputs: - config-path: - description: 'Path to umpyre config file' - default: '.github/umpyre-config.yml' - required: false - - storage-branch: - description: 'Branch to store metrics' - default: 'code-metrics' - required: false - - github-token: - description: 'GitHub token for API access' - required: true - - force-run: - description: 'Run even if workflow failed' - default: 'false' - required: false - - python-version: - description: 'Python version to use' - default: '3.10' - required: false - -runs: - using: composite - steps: - - name: Setup Python - uses: actions/setup-python@v5 - with: - python-version: ${{ inputs.python-version }} - - - name: Install umpyre and dependencies - shell: bash - run: | - pip install umpyre coverage wily bandit interrogate - - - name: Collect and Store Metrics - shell: bash - run: | - echo "::group::Collecting Code Metrics" - - # Set up error handling - continue even if collection fails - set +e - - # Run umpyre collect - if [ -f "${{ inputs.config-path }}" ]; then - python -m umpyre.cli collect --config "${{ inputs.config-path }}" - EXIT_CODE=$? - else - python -m umpyre.cli collect - EXIT_CODE=$? - fi - - # Report status but don't fail - if [ $EXIT_CODE -eq 0 ]; then - echo "✅ Metrics collection completed successfully" - else - echo "⚠️ Metrics collection failed with exit code $EXIT_CODE" - echo "This is informational only - CI pipeline continues" - fi - - echo "::endgroup::" - - # Always exit successfully to not break CI - exit 0 - -branding: - icon: 'bar-chart-2' - color: 'blue' diff --git a/json b/json deleted file mode 100644 index c482900..0000000 --- a/json +++ /dev/null @@ -1,32 +0,0 @@ -{ - "collection_duration_seconds": null, - "commit_message": null, - "commit_sha": "26a381140d20ccf873fe9473637ae060727f9435", - "metrics": { - "coverage": { - "branch_coverage": 0.0, - "error": "argument should be a str or an os.PathLike object where __fspath__ returns a str, not 'Config'", - "line_coverage": 0.0, - "statements_covered": 0, - "statements_total": 0 - }, - "umpyre_stats": { - "comment_lines": 0, - "comment_lines_ratio": 0.0, - "docs_lines": 0, - "empty_lines": 0, - "empty_lines_ratio": 0.0, - "error": "'Config' object has no attribute 'endswith'", - "function_lines": 0, - "function_lines_ratio": 0.0, - "mean_lines_per_function": 0.0, - "num_classes": 0, - "num_functions": 0, - "total_lines": 0 - } - }, - "python_version": null, - "schema_version": "1.0", - "timestamp": "2025-11-14T20:37:33.502900Z", - "workflow_status": {} -} \ No newline at end of file diff --git a/misc/CHANGELOG.md b/misc/CHANGELOG.md deleted file mode 100644 index 947289d..0000000 --- a/misc/CHANGELOG.md +++ /dev/null @@ -1,235 +0,0 @@ -# Changelog - -All notable changes to the umpyre project are recorded here. - -## 2025-11-14 - Simplified Storage Structure (SSOT Compliance) - -### Changed -- **Storage structure**: Completely redesigned for SSOT principles and simplicity - - **Flat structure**: Removed monthly folders (`history/2025-11/`) - now just `history/` - - **Removed data duplication**: Eliminated `by_version/` folder (was redundant) - - **New filename format**: `YYYY_MM_DD_HH_MM_SS__shahash__version.json` - - Chronologically ordered (sorts naturally by time) - - All info parseable from filename (timestamp, commit, version) - - Example: `2025_11_14_22_45_00__700e012__0.1.0.json` - - **Benefits**: Single source of truth, easier querying, simpler maintenance - -### Rationale -- User feedback: "Is it a good idea to have a separate by_version folder? We'll be repeating data there no? That's not very SSOT." -- Flat structure is simpler than nested monthly directories -- Filename contains all indexing info (no need for duplicate folders) -- Shell-friendly: `ls`, `grep`, `sort` work perfectly - -## 2025-11-14 - Added PyPI Version Tracking and Commit-Based Indexing - -### Added -- **PyPI version extraction**: `UmpyreCollector` now automatically detects and includes `pypi_version` in metrics - - Checks `pyproject.toml` (preferred), `setup.py` (fallback), or `__init__.py` (last resort) - - Uses regex to extract version strings (e.g., "0.1.0", "1.2.3-beta") - - Returns `null` if no version found -- **Commit-SHA indexing**: Storage now uses commit hash as primary index for uniqueness - - Historical files named as `{commit_sha[:7]}.json` (e.g., `700e012.json`) - - Guarantees one entry per commit (prevents duplicates) - - Timestamp removed from filename for cleaner structure -- **By-version secondary index**: New `by_version/` directory for easy version-based queries - - Files named as `{pypi_version}.json` (e.g., `0.1.0.json`) - - Contains latest metrics for each published version - - Enables comparison across releases - -### Changed -- **Storage structure**: Now uses dual indexing system - ``` - code-metrics/ - ├── history/2025-11/{commit_sha}.json ← Primary index - └── by_version/{version}.json ← Secondary index - ``` -- **Root path handling**: Fixed `UmpyreCollector.__init__` to convert `root_path` to `Path` object (was causing type errors) - -### Documentation -- Added `STORAGE_STRUCTURE.md` with comprehensive storage design documentation -- Explains uniqueness guarantees, querying patterns, and CI integration - -# Changelog - -All notable changes to the umpyre project are documented here. - -## 2025-11-14 - Fixed UmpyreCollector with AST-based Parsing - -### Changed -- **UmpyreCollector**: Completely reimplemented using Python's AST (Abstract Syntax Tree) module instead of dynamic imports - - Previously used `py2store.sources.Attrs.module_from_path()` which dynamically imported files, causing setup.py and other files to execute - - Now uses safe AST parsing that analyzes code structure without executing it - - Works on all Python files including setup.py, conf.py, and scripts - - Maintains same metrics output format for backward compatibility - -### Added -- `_analyze_file()` method: AST-based analysis of individual Python files -- `_should_analyze()` method: Improved filtering logic for files to analyze -- `files_analyzed` metric: Now tracks how many files were successfully analyzed -- Error tracking: Records parsing errors per file without failing entire collection - -### Fixed -- **Critical**: UmpyreCollector no longer crashes when encountering setup.py or other executable scripts -- All 5 umpyre_collector tests now pass (previously 2 were skipped) -- Total test suite: 34/34 tests passing - -### Performance -- AST parsing is faster than dynamic imports -- No subprocess overhead -- No risk of code execution side effects - -## 2025-11-14 - Phase 1 Implementation Complete - -### Added - -**Architecture & Configuration** -- Implemented config-driven architecture with YAML configuration support -- Created versioned schema system (`schema.py`) for metrics with migration support -- Built pluggable collector system with Mapping interface pattern -- Added comprehensive configuration loading with deep merge and validation - -**Metric Collectors** -- `WorkflowStatusCollector`: Track GitHub CI/CD workflow status via API - - Last run status (success/failure/other) - - Recent failure counts with configurable lookback - - Last successful run timestamp -- `WilyCollector`: Complexity metrics using wily - - Cyclomatic complexity (configurable operators) - - Maintainability index - - Limited to recent commits for performance (default: 5 revisions) -- `CoverageCollector`: Test coverage from pytest-cov or coverage.py - - Line and branch coverage percentages - - Supports JSON and XML (Cobertura) report formats - - Auto-detection of coverage files -- `UmpyreCollector`: Code statistics using existing `python_code_stats` module - - Function/class counts - - Line metrics (total, empty, comments, docs) - - Code ratios and averages - -**Storage System** -- Git branch-based storage (`GitBranchStorage`) - - Stores metrics in separate branch (default: `code-metrics`) - - Supports JSON and CSV formats - - Monthly history organization - - Handles concurrent commits with retry logic -- Serialization formats module with JSON/CSV support -- Flat CSV format for easy pandas integration - -**CLI Interface** -- `umpyre collect`: Collect and store metrics - - Config-driven collection - - Auto-detects git commit info - - Dry-run mode (`--no-store`) - - Environment variable support (GITHUB_SHA, GITHUB_REPOSITORY) -- `umpyre validate`: Placeholder for threshold validation (Phase 3) - -**GitHub Action** -- Reusable composite action at `actions/track-metrics/action.yml` -- Integrates with GitHub CI/CD workflows -- Auto-installs dependencies (umpyre, coverage, wily, bandit, interrogate) -- Configurable Python version and config path - -**Testing** -- Comprehensive test suite for all core components -- Test coverage for config, schema, collectors, and storage -- 23+ passing tests with pytest - -### Design Patterns Used - -- **Facade Pattern**: Collectors provide clean Mapping interface -- **Registry Pattern**: Collector registry for extensibility -- **Open-Closed Principle**: Config-driven, easily extensible without code changes -- **Lazy Evaluation**: Collectors cache results on first access -- **Dependency Injection**: Collectors accept configuration as constructor args - -### Configuration Example - -```yaml -schema_version: "1.0" - -collectors: - workflow_status: - enabled: true - lookback_runs: 10 - wily: - enabled: true - max_revisions: 5 - operators: [cyclomatic, maintainability] - coverage: - enabled: true - source: pytest-cov - umpyre_stats: - enabled: true - exclude_dirs: [tests, examples, scrap] - -storage: - branch: code-metrics - formats: [json, csv] - retention: - strategy: all - -visualization: - generate_plots: true - generate_readme: true - plot_metrics: [maintainability, coverage, loc] - -thresholds: - enabled: false - -aggregation: - enabled: false -``` - -### Usage Example - -```bash -# Install umpyre -pip install umpyre - -# Collect metrics (uses .github/umpyre-config.yml if present) -python -m umpyre.cli collect - -# Collect with custom config -python -m umpyre.cli collect --config my-config.yml - -# Dry run (don't store) -python -m umpyre.cli collect --no-store -``` - -### In GitHub Actions - -```yaml -- name: Track Code Metrics - if: success() # Only after successful publish - uses: i2mint/umpyre/actions/track-metrics@master - with: - github-token: ${{ secrets.GITHUB_TOKEN }} - config-path: .github/umpyre-config.yml -``` - -### Known Limitations - -- `UmpyreCollector` may have compatibility issues with some directory structures (inherits from `python_code_stats.py`) -- `WilyCollector` requires wily installation and git history -- GitHub API rate limiting applies to `WorkflowStatusCollector` (5000 req/hour with auth) - -### Pending (Future Phases) - -**Phase 2**: Visualization & Aggregation -- Plot generation (matplotlib/plotly) -- README auto-generation with embedded charts -- Cross-repository aggregation -- Dashboard generation for organizations - -**Phase 3**: Advanced Features -- Additional collectors (bandit for security, interrogate for docstrings) -- Threshold validation system with custom validators -- Data pruning and compression -- Schema migration utilities - -### Technical Details - -- Python 3.10+ required -- Dependencies: py2store, pandas, pyyaml, requests -- Test framework: pytest -- Schema version: 1.0 diff --git a/pyproject.toml.migrated b/pyproject.toml.migrated deleted file mode 100644 index 6417c82..0000000 --- a/pyproject.toml.migrated +++ /dev/null @@ -1,252 +0,0 @@ -[build-system] -requires = [ - "hatchling", -] -build-backend = "hatchling.build" - -[project] -name = "odbcdol" -version = "0.0.3" -description = "odbc (through pyodbc) with a simple (dict-like or list-like) interface" -readme = "README.md" -requires-python = ">=3.10" -keywords = [] -authors = [] -dependencies = [ - "dol", - "pyodbc", -] - -[project.license] -text = "Apache-2.0" - -[project.urls] -Homepage = "https://github.com/i2mint/odbcdol" - -[project.optional-dependencies] -dev = [ - "pytest>=7.0", - "pytest-cov>=4.0", - "ruff>=0.1.0", -] -docs = [ - "sphinx>=6.0", - "sphinx-rtd-theme>=1.0", -] - -# External system dependencies (PEP 725/804 compliant) -[external] -# Unix ODBC library - provides core ODBC functionality -unixodbc = "dep:generic/unixodbc" -# Microsoft SQL Server ODBC driver - enables SQL Server connectivity -msodbcsql18 = "dep:vendor/microsoft/msodbcsql18" - -# Operational metadata for external dependencies -[tool.wads.external.ops.unixodbc] -rationale = "Required by pyodbc for ODBC database connectivity" -url = "https://www.unixodbc.org/" - -[tool.wads.external.ops.unixodbc.check] -linux = [ - "odbcinst --version", - "test -f /usr/lib/x86_64-linux-gnu/libodbc.so || test -f /usr/lib/libodbc.so", -] -macos = [ - "odbcinst --version", - "test -f /opt/homebrew/lib/libodbc.dylib || test -f /usr/local/lib/libodbc.dylib", -] - -[tool.wads.external.ops.unixodbc.install] -linux = [ - "sudo apt-get update", - "sudo apt-get install -y unixodbc unixodbc-dev", -] -macos = [ - "brew install unixodbc", -] - -[tool.wads.external.ops.msodbcsql18] -rationale = "Microsoft SQL Server ODBC driver for enterprise database connectivity" -url = "https://docs.microsoft.com/en-us/sql/connect/odbc/linux-mac/installing-the-microsoft-odbc-driver-for-sql-server" - -[tool.wads.external.ops.msodbcsql18.check] -linux = [ - "test -f '/opt/microsoft/msodbcsql18/lib64/libmsodbcsql-18.*.so.*'", - "odbcinst -d -q | grep -i 'ODBC Driver 18 for SQL Server'", -] -macos = [ - "test -f '/opt/microsoft/msodbcsql18/lib/libmsodbcsql.18.dylib'", - "odbcinst -d -q | grep -i 'ODBC Driver 18 for SQL Server'", -] -windows = [ - "Get-OdbcDriver | Where-Object {$_.Name -like '*ODBC Driver 18 for SQL Server*'}", -] - -[tool.wads.external.ops.msodbcsql18.install] -linux = [ - "curl https://packages.microsoft.com/keys/microsoft.asc | sudo apt-key add -", - "curl https://packages.microsoft.com/config/ubuntu/$(lsb_release -rs)/prod.list | sudo tee /etc/apt/sources.list.d/mssql-release.list", - "sudo apt-get update", - "sudo ACCEPT_EULA=Y apt-get install -y msodbcsql18", -] -macos = [ - "brew tap microsoft/mssql-release https://github.com/Microsoft/homebrew-mssql-release", - "brew update", - "HOMEBREW_NO_ENV_FILTERING=1 ACCEPT_EULA=Y brew install msodbcsql18 mssql-tools18", -] -windows = [ - "choco install sqlserver-odbcdriver --version=18.3.2.1 -y", -] - -[tool.ruff] -line-length = 88 -target-version = "py310" -exclude = [ - "**/*.ipynb", - ".git", - ".venv", - "build", - "dist", - "tests", - "examples", - "scrap", -] - -[tool.ruff.lint] -select = [ - "D100", -] -ignore = [ - "D203", - "E501", - "B905", -] - -[tool.ruff.lint.pydocstyle] -convention = "google" - -[tool.ruff.lint.per-file-ignores] -"**/tests/*" = [ - "D", -] -"**/examples/*" = [ - "D", -] -"**/scrap/*" = [ - "D", -] - -[tool.pytest.ini_options] -minversion = "6.0" -testpaths = [ - "tests", -] -doctest_optionflags = [ - "NORMALIZE_WHITESPACE", - "ELLIPSIS", -] - -# Comprehensive CI configuration -[tool.wads.ci] -project_name = "odbcdol" - -[tool.wads.ci.commands] -pre_test = [ - "odbcinst --version", # Verify ODBC installation -] -test = [ - "pytest -v --tb=short", -] -post_test = [] -lint = [ - "ruff check .", -] -format = [ - "ruff format .", -] - -[tool.wads.ci.env] -# Environment variables for testing -ACCEPT_EULA = "Y" # Required for Microsoft ODBC driver installation - -# System dependencies (legacy format - will be deprecated) -[tool.wads.ci.env.install] -linux = [ - "sudo apt-get update", - "sudo apt-get install -y unixodbc unixodbc-dev", - "curl https://packages.microsoft.com/keys/microsoft.asc | sudo apt-key add -", - "curl https://packages.microsoft.com/config/ubuntu/$(lsb_release -rs)/prod.list | sudo tee /etc/apt/sources.list.d/mssql-release.list", - "sudo apt-get update", - "sudo ACCEPT_EULA=Y apt-get install -y msodbcsql18", -] -macos = [ - "brew install unixodbc", - "brew tap microsoft/mssql-release https://github.com/Microsoft/homebrew-mssql-release", - "brew update", - "HOMEBREW_NO_ENV_FILTERING=1 ACCEPT_EULA=Y brew install msodbcsql18 mssql-tools18", -] - -[tool.wads.ci.quality.ruff] -enabled = true -line_length = 88 -target_version = "py310" - -[tool.wads.ci.quality.black] -enabled = false - -[tool.wads.ci.quality.mypy] -enabled = false - -[tool.wads.ci.testing] -python_versions = [ - "3.10", - "3.12", -] -pytest_args = [ - "-v", - "--tb=short", -] -coverage_enabled = true -coverage_threshold = 0 -coverage_report_format = [ - "term", - "xml", -] -exclude_paths = [ - "examples", - "scrap", -] -test_on_windows = true -# Allow Windows tests to fail gracefully due to ODBC driver complexity -allow_windows_failure = true - -[tool.wads.ci.build] -sdist = true -wheel = true - -[tool.wads.ci.publish] -enabled = true -# Publish on successful tests across all platforms -require_tests_pass = true - -[tool.wads.ci.docs] -enabled = true -builder = "epythet" -ignore_paths = [ - "tests/", - "scrap/", - "examples/", -] - -# Additional CI settings for ODBC-specific requirements -[tool.wads.ci.matrix] -# Test with different Python versions and ODBC configurations -include_combinations = [ - { python = "3.10", odbc = "unixodbc" }, - { python = "3.12", odbc = "unixodbc" }, -] - -[tool.wads.ci.artifacts] -# Store ODBC connection test results -store_test_results = true -store_coverage_reports = true diff --git a/setup.cfg b/setup.cfg deleted file mode 100644 index 4ae7e29..0000000 --- a/setup.cfg +++ /dev/null @@ -1,23 +0,0 @@ -[metadata] -name = odbcdol -version = 0.0.3 -url = https://github.com/i2mint/odbcdol -platforms = any -description_file = README.md -root_url = https://github.com/i2mint/ -license = apache-2.0 - -description = odbc (through pyodbc) with a simple (dict-like or list-like) interface -long_description = file:README.md -long_description_content_type = text/markdown -keywords = -display_name = odbcdol - -[options] -packages = find: -include_package_data = True -zip_safe = False -install_requires = - dol - pyodbc - diff --git a/test_on_astate.py b/test_on_astate.py deleted file mode 100644 index 1ad5a50..0000000 --- a/test_on_astate.py +++ /dev/null @@ -1,432 +0,0 @@ -#!/usr/bin/env python -""" -Automated testing script for umpyre on astate repository. - -This script validates the umpyre metrics collection system by: -1. Running metrics collection on astate -2. Verifying storage outputs -3. Checking metric data validity -4. Measuring performance -""" - -import sys -import os -import time -import json -import csv -import tempfile -import shutil -from pathlib import Path -from datetime import datetime - -# Add umpyre to path -umpyre_root = Path(__file__).parent -sys.path.insert(0, str(umpyre_root)) - -from umpyre import Config, MetricSchema, registry -from umpyre.collectors.base import MetricCollector -from umpyre.storage.formats import serialize_metrics, deserialize_metrics - - -class TestResult: - """Track test results.""" - - def __init__(self): - self.passed = [] - self.failed = [] - self.warnings = [] - self.start_time = None - self.end_time = None - - def pass_test(self, name: str, details: str = ""): - self.passed.append((name, details)) - print(f"✅ {name}") - if details: - print(f" {details}") - - def fail_test(self, name: str, error: str): - self.failed.append((name, error)) - print(f"❌ {name}") - print(f" Error: {error}") - - def warn(self, message: str): - self.warnings.append(message) - print(f"⚠️ {message}") - - def start(self): - self.start_time = time.time() - - def stop(self): - self.end_time = time.time() - - @property - def duration(self): - if self.start_time and self.end_time: - return self.end_time - self.start_time - return 0 - - def summary(self): - print("\n" + "=" * 70) - print("TEST SUMMARY") - print("=" * 70) - print(f"Passed: {len(self.passed)}") - print(f"Failed: {len(self.failed)}") - print(f"Warnings: {len(self.warnings)}") - print(f"Duration: {self.duration:.2f}s") - - if self.failed: - print("\n❌ FAILED TESTS:") - for name, error in self.failed: - print(f" - {name}: {error}") - - if self.warnings: - print("\n⚠️ WARNINGS:") - for warning in self.warnings: - print(f" - {warning}") - - print("\n" + "=" * 70) - return len(self.failed) == 0 - - -def test_registry(): - """Test that collectors are registered.""" - result = TestResult() - - collectors = registry.list_collectors() - if len(collectors) >= 3: # At least workflow_status, coverage, wily - result.pass_test( - "Collector registry", - f"Found {len(collectors)} collectors: {', '.join(collectors)}", - ) - else: - result.fail_test( - "Collector registry", - f"Expected at least 3 collectors, found {len(collectors)}", - ) - - return result - - -def test_config(repo_path: Path): - """Test configuration loading.""" - result = TestResult() - - try: - # Test with minimal config - config_data = { - 'repo_path': str(repo_path), - 'collectors': { - 'workflow_status': {'enabled': True}, - 'coverage': {'enabled': True}, - }, - } - - # Write temporary config - config_file = repo_path / '.umpyre.yml' - import yaml - - with open(config_file, 'w') as f: - yaml.dump(config_data, f) - - config = Config(config_path=str(config_file)) - - result.pass_test("Config loading", f"Loaded config from {config_file}") - - # Test config access - if config.is_collector_enabled('workflow_status'): - result.pass_test("Config collector check", "workflow_status is enabled") - else: - result.fail_test( - "Config collector check", "workflow_status should be enabled" - ) - - # Cleanup - config_file.unlink() - - except Exception as e: - result.fail_test("Config loading", str(e)) - - return result - - -def test_schema(): - """Test metric schema validation.""" - result = TestResult() - - try: - # Create sample metric data - sample_metrics = { - 'workflow_status': { - 'last_run_status': 'success', - 'last_run_conclusion': 'success', - } - } - - metric_data = MetricSchema.create_metric_data( - commit_sha='abc123', - metrics=sample_metrics, - commit_message='Test commit', - python_version='3.10', - ) - - # Validate - is_valid = MetricSchema.validate(metric_data) - - if is_valid: - result.pass_test( - "Schema validation", "Sample metrics validated successfully" - ) - else: - result.fail_test("Schema validation", "Validation failed") - - # Check required fields - required_fields = ['schema_version', 'timestamp', 'commit_sha', 'metrics'] - missing = [f for f in required_fields if f not in metric_data] - - if not missing: - result.pass_test("Schema structure", "All required fields present") - else: - result.fail_test("Schema structure", f"Missing fields: {missing}") - - except Exception as e: - result.fail_test("Schema validation", str(e)) - - return result - - -def test_collectors_on_astate(repo_path: Path): - """Test collectors on actual astate repository.""" - result = TestResult() - result.start() - - # Test each collector with proper initialization - test_specs = { - 'coverage': { - 'description': 'Test coverage', - 'init_kwargs': {'repo_path': str(repo_path), 'source': 'pytest-cov'}, - }, - 'umpyre_stats': { - 'description': 'Code statistics (AST-based)', - 'init_kwargs': { - 'root_path': str(repo_path), - 'exclude_dirs': ['tests', 'examples', 'docsrc'], - }, - }, - } - - for collector_name, spec in test_specs.items(): - description = spec['description'] - try: - if collector_name not in registry.list_collectors(): - result.warn(f"Collector '{collector_name}' not registered, skipping") - continue - - collector_class = registry.get(collector_name) - collector = collector_class(**spec['init_kwargs']) - - # Collect metrics - start = time.time() - metrics = dict(collector) - duration = time.time() - start - - if metrics: - # Check for errors - if 'error' in metrics: - result.warn( - f"{description}: Collector returned error: {metrics['error']}" - ) - else: - result.pass_test( - f"{description} collector", - f"Collected {len(metrics)} metrics in {duration:.2f}s", - ) - else: - result.warn( - f"{description}: No metrics collected (may be normal if data not available)" - ) - - except Exception as e: - result.fail_test(f"{description} collector", str(e)) - - result.stop() - return result - - -def test_full_collection(repo_path: Path): - """Test full metrics collection pipeline.""" - result = TestResult() - result.start() - - try: - # Manually collect metrics from collectors - all_metrics = {} - - # Coverage collector - try: - from umpyre.collectors.coverage_collector import CoverageCollector - - collector = CoverageCollector(repo_path=str(repo_path)) - all_metrics['coverage'] = collector.to_dict() - except Exception as e: - all_metrics['coverage'] = {'error': str(e)} - - # Umpyre stats collector - try: - from umpyre.collectors.umpyre_collector import UmpyreCollector - - collector = UmpyreCollector( - root_path=str(repo_path), exclude_dirs=['tests', 'docsrc'] - ) - all_metrics['umpyre_stats'] = collector.to_dict() - except Exception as e: - all_metrics['umpyre_stats'] = {'error': str(e)} - - if all_metrics: - result.pass_test( - "Full collection pipeline", - f"Collected metrics from {len(all_metrics)} collectors", - ) - - # Create proper schema structure - import subprocess - - try: - commit_sha = subprocess.check_output( - ['git', 'rev-parse', 'HEAD'], cwd=repo_path, text=True - ).strip() - except: - commit_sha = 'unknown' - - metrics_data = MetricSchema.create_metric_data( - commit_sha=commit_sha, metrics=all_metrics - ) - - # Validate schema - is_valid = MetricSchema.validate(metrics_data) - if is_valid: - result.pass_test("Output schema validation", "Metrics data is valid") - else: - result.fail_test("Output schema validation", "Validation failed") - - # Test serialization to temp file - import tempfile - - with tempfile.NamedTemporaryFile( - mode='w', suffix='.json', delete=False - ) as f: - temp_json = Path(f.name) - - try: - serialize_metrics(metrics_data, temp_json, 'json') - result.pass_test( - "JSON serialization", - f"Serialized to {temp_json.stat().st_size} bytes", - ) - - # Test deserialization - recovered = deserialize_metrics(temp_json, 'json') - if recovered == metrics_data: - result.pass_test("JSON deserialization", "Round-trip successful") - else: - result.warn("JSON deserialization: Data changed during round-trip") - finally: - temp_json.unlink() - - else: - result.fail_test("Full collection pipeline", "No metrics collected") - - except Exception as e: - result.fail_test("Full collection pipeline", str(e)) - import traceback - - traceback.print_exc() - - result.stop() - - # Check performance requirement (<30s) - if result.duration < 30: - result.pass_test( - "Performance requirement", - f"Completed in {result.duration:.2f}s (target: <30s)", - ) - else: - result.fail_test( - "Performance requirement", f"Took {result.duration:.2f}s (target: <30s)" - ) - - return result - - -def main(): - """Run all tests.""" - print("=" * 70) - print("UMPYRE AUTOMATED TEST SUITE") - print("=" * 70) - print() - - # Check astate path - astate_path = Path("/Users/thorwhalen/Dropbox/py/proj/t/astate") - - if not astate_path.exists(): - print(f"❌ astate repository not found at: {astate_path}") - print(" Please update the path in the script") - return 1 - - if not (astate_path / '.git').exists(): - print(f"❌ {astate_path} is not a git repository") - return 1 - - print(f"✅ Found astate repository at: {astate_path}") - print() - - all_results = [] - - # Run tests - print("TEST 1: Registry") - print("-" * 70) - all_results.append(test_registry()) - print() - - print("TEST 2: Configuration") - print("-" * 70) - all_results.append(test_config(astate_path)) - print() - - print("TEST 3: Schema") - print("-" * 70) - all_results.append(test_schema()) - print() - - print("TEST 4: Individual Collectors") - print("-" * 70) - all_results.append(test_collectors_on_astate(astate_path)) - print() - - print("TEST 5: Full Collection Pipeline") - print("-" * 70) - all_results.append(test_full_collection(astate_path)) - print() - - # Overall summary - print("\n" + "=" * 70) - print("OVERALL RESULTS") - print("=" * 70) - - total_passed = sum(len(r.passed) for r in all_results) - total_failed = sum(len(r.failed) for r in all_results) - total_warnings = sum(len(r.warnings) for r in all_results) - - print(f"Total Passed: {total_passed}") - print(f"Total Failed: {total_failed}") - print(f"Total Warnings: {total_warnings}") - - if total_failed == 0: - print("\n🎉 ALL TESTS PASSED!") - return 0 - else: - print("\n❌ SOME TESTS FAILED") - return 1 - - -if __name__ == '__main__': - sys.exit(main()) diff --git a/tests/test_base_collector.py b/tests/test_base_collector.py deleted file mode 100644 index 2a0ef44..0000000 --- a/tests/test_base_collector.py +++ /dev/null @@ -1,128 +0,0 @@ -"""Tests for base collector.""" - -import pytest -from umpyre.collectors.base import MetricCollector, CollectorRegistry - - -class SimpleCollector(MetricCollector): - """Test collector implementation.""" - - def collect(self) -> dict: - return { - "lines": 100, - "functions": 10, - "classes": 5, - } - - -class ErrorCollector(MetricCollector): - """Collector that raises an error.""" - - def collect(self) -> dict: - raise RuntimeError("Collection failed") - - -def test_collector_mapping_interface(): - """Should provide Mapping interface.""" - collector = SimpleCollector() - - # __getitem__ - assert collector["lines"] == 100 - assert collector["functions"] == 10 - - # __iter__ - keys = list(collector) - assert "lines" in keys - assert "functions" in keys - - # __len__ - assert len(collector) == 3 - - # dict methods - assert "lines" in collector - assert collector.get("lines") == 100 - assert collector.get("missing", "default") == "default" - - -def test_collector_lazy_collection(): - """Should lazily collect metrics on first access.""" - collector = SimpleCollector() - - # Not collected yet - assert collector._cached_metrics is None - - # Access triggers collection - _ = collector["lines"] - assert collector._cached_metrics is not None - - # Subsequent accesses use cache - assert collector["functions"] == 10 - - -def test_collector_to_dict(): - """Should export metrics as dictionary.""" - collector = SimpleCollector() - data = collector.to_dict() - - assert isinstance(data, dict) - assert data == {"lines": 100, "functions": 10, "classes": 5} - - -def test_collector_refresh(): - """Should force re-collection of metrics.""" - collector = SimpleCollector() - - # Collect once - _ = collector["lines"] - cached = collector._cached_metrics - - # Refresh - collector.refresh() - assert collector._cached_metrics is None - - # Access again - should re-collect - _ = collector["lines"] - assert collector._cached_metrics is not None - assert collector._cached_metrics is not cached # New object - - -def test_collector_error_handling(): - """Should propagate collection errors.""" - collector = ErrorCollector() - - with pytest.raises(RuntimeError, match="Collection failed"): - _ = collector["any_key"] - - -def test_collector_registry(): - """Should register and retrieve collectors.""" - registry = CollectorRegistry() - - # Register - registry.register("simple", SimpleCollector) - - # Retrieve - CollectorClass = registry.get("simple") - assert CollectorClass is SimpleCollector - - # List - assert "simple" in registry.list_collectors() - - -def test_registry_unknown_collector(): - """Should raise error for unknown collector.""" - registry = CollectorRegistry() - - with pytest.raises(KeyError, match="Unknown collector"): - registry.get("nonexistent") - - -def test_registry_invalid_type(): - """Should reject non-collector classes.""" - registry = CollectorRegistry() - - class NotACollector: - pass - - with pytest.raises(TypeError, match="must be a MetricCollector subclass"): - registry.register("invalid", NotACollector) diff --git a/tests/test_config.py b/tests/test_config.py deleted file mode 100644 index 47a4470..0000000 --- a/tests/test_config.py +++ /dev/null @@ -1,131 +0,0 @@ -"""Tests for config module.""" - -import pytest -import tempfile -from pathlib import Path -from umpyre.config import Config, ConfigError - - -def test_config_default_values(): - """Should provide default configuration.""" - config = Config() - - assert config.storage_branch == "code-metrics" - assert "json" in config.storage_formats - assert config.retention_strategy == "all" - - -def test_config_from_dict(): - """Should create config from dictionary.""" - custom = { - "storage": { - "branch": "custom-metrics", - "formats": ["csv"], - } - } - - config = Config.from_dict(custom) - assert config.storage_branch == "custom-metrics" - assert config.storage_formats == ["csv"] - - -def test_config_from_yaml_file(): - """Should load config from YAML file.""" - yaml_content = """ -collectors: - wily: - enabled: false - max_revisions: 10 - -storage: - branch: test-metrics -""" - - with tempfile.NamedTemporaryFile(mode='w', suffix='.yml', delete=False) as f: - f.write(yaml_content) - f.flush() - - config = Config.from_file(f.name) - - assert config.storage_branch == "test-metrics" - assert config.is_collector_enabled("wily") is False - assert config.collector_config("wily")["max_revisions"] == 10 - - Path(f.name).unlink() - - -def test_config_deep_merge(): - """Should deep merge configurations properly.""" - config = Config.from_dict( - { - "collectors": { - "wily": { - "max_revisions": 3, # Override default - # Keep other wily defaults - }, - "coverage": { - "enabled": False, # Override default - }, - } - } - ) - - # Should have overridden value - assert config.get("collectors", "wily", "max_revisions") == 3 - # Should keep default - assert "cyclomatic" in config.get("collectors", "wily", "operators") - # Should have override - assert config.is_collector_enabled("coverage") is False - - -def test_config_get_nested(): - """Should retrieve nested config values.""" - config = Config() - - # Existing path - assert config.get("collectors", "wily", "enabled") is True - - # Missing path with default - assert config.get("missing", "path", default="fallback") == "fallback" - - # Missing path without default - assert config.get("missing", "path") is None - - -def test_config_is_collector_enabled(): - """Should check if collector is enabled.""" - config = Config() - - assert config.is_collector_enabled("workflow_status") is True - assert config.is_collector_enabled("nonexistent") is False - - -def test_config_collector_config(): - """Should get collector-specific configuration.""" - config = Config() - - wily_config = config.collector_config("wily") - assert wily_config["enabled"] is True - assert wily_config["max_revisions"] == 5 - - missing_config = config.collector_config("nonexistent") - assert missing_config == {} - - -def test_config_to_dict(): - """Should export config as dictionary.""" - config = Config.from_dict({"storage": {"branch": "test"}}) - exported = config.to_dict() - - assert isinstance(exported, dict) - assert exported["storage"]["branch"] == "test" - - -def test_config_validation(): - """Should validate required config sections.""" - # This should work - has required sections via defaults - Config() - - # Missing required sections should raise error - # (though our implementation merges with defaults, so this is hard to trigger) - # Future: could add more strict validation diff --git a/tests/test_coverage_collector.py b/tests/test_coverage_collector.py deleted file mode 100644 index c9818d0..0000000 --- a/tests/test_coverage_collector.py +++ /dev/null @@ -1,125 +0,0 @@ -"""Tests for CoverageCollector.""" - -import tempfile -import json -from pathlib import Path -import pytest - -from umpyre.collectors.coverage_collector import CoverageCollector -from umpyre.collectors import registry - - -def test_coverage_collector_json_report(): - """Should parse coverage.py JSON report.""" - with tempfile.TemporaryDirectory() as tmpdir: - # Create a mock coverage.json file - coverage_file = Path(tmpdir) / "coverage.json" - coverage_data = { - "totals": { - "num_statements": 100, - "covered_lines": 85, - "num_branches": 20, - "covered_branches": 16, - } - } - coverage_file.write_text(json.dumps(coverage_data)) - - collector = CoverageCollector( - repo_path=tmpdir, coverage_file=str(coverage_file) - ) - metrics = collector.collect() - - assert metrics["line_coverage"] == 85.0 - assert metrics["branch_coverage"] == 80.0 - assert metrics["statements_covered"] == 85 - assert metrics["statements_total"] == 100 - assert "error" not in metrics - - -def test_coverage_collector_xml_report(): - """Should parse Cobertura XML report.""" - with tempfile.TemporaryDirectory() as tmpdir: - # Create a mock coverage.xml file - coverage_file = Path(tmpdir) / "coverage.xml" - xml_content = ''' - - -''' - coverage_file.write_text(xml_content) - - collector = CoverageCollector( - repo_path=tmpdir, coverage_file=str(coverage_file) - ) - metrics = collector.collect() - - assert metrics["line_coverage"] == 87.5 - assert metrics["branch_coverage"] == 80.0 - assert metrics["statements_covered"] == 87 - assert metrics["statements_total"] == 100 - - -def test_coverage_collector_no_file(): - """Should handle missing coverage file gracefully.""" - with tempfile.TemporaryDirectory() as tmpdir: - collector = CoverageCollector(repo_path=tmpdir) - metrics = collector.collect() - - # Should return empty metrics with error - assert metrics["line_coverage"] == 0.0 - assert "error" in metrics - assert "No coverage file found" in metrics["error"] - - -def test_coverage_collector_auto_detect(): - """Should auto-detect coverage file.""" - with tempfile.TemporaryDirectory() as tmpdir: - # Place coverage.json in standard location - coverage_file = Path(tmpdir) / "coverage.json" - coverage_data = { - "totals": { - "num_statements": 50, - "covered_lines": 40, - "num_branches": 0, - "covered_branches": 0, - } - } - coverage_file.write_text(json.dumps(coverage_data)) - - collector = CoverageCollector(repo_path=tmpdir) - metrics = collector.collect() - - assert metrics["line_coverage"] == 80.0 - assert "error" not in metrics - - -def test_coverage_collector_registered(): - """Should be registered in global registry.""" - from umpyre.collectors import registry as global_registry - - assert "coverage" in global_registry.list_collectors() - CollectorClass = global_registry.get("coverage") - assert CollectorClass.__name__ == "CoverageCollector" - - -def test_coverage_collector_zero_statements(): - """Should handle zero statements gracefully.""" - with tempfile.TemporaryDirectory() as tmpdir: - coverage_file = Path(tmpdir) / "coverage.json" - coverage_data = { - "totals": { - "num_statements": 0, - "covered_lines": 0, - "num_branches": 0, - "covered_branches": 0, - } - } - coverage_file.write_text(json.dumps(coverage_data)) - - collector = CoverageCollector( - repo_path=tmpdir, coverage_file=str(coverage_file) - ) - metrics = collector.collect() - - # Should not divide by zero - assert metrics["line_coverage"] == 0.0 - assert metrics["branch_coverage"] == 0.0 diff --git a/tests/test_schema.py b/tests/test_schema.py deleted file mode 100644 index fcfdd91..0000000 --- a/tests/test_schema.py +++ /dev/null @@ -1,71 +0,0 @@ -"""Tests for schema module.""" - -import pytest -from datetime import datetime, timezone -from umpyre.schema import MetricSchema - - -def test_schema_current_version(): - """Schema should have a current version.""" - assert MetricSchema.current_version() == "1.0" - - -def test_schema_create_metric_data(): - """Should create standardized metric data structure.""" - metrics = { - "complexity": {"cyclomatic_avg": 3.2}, - "coverage": {"line_coverage": 87.5}, - } - - data = MetricSchema.create_metric_data( - commit_sha="abc123", - metrics=metrics, - commit_message="Test commit", - python_version="3.10", - ) - - assert data["schema_version"] == "1.0" - assert data["commit_sha"] == "abc123" - assert data["commit_message"] == "Test commit" - assert data["python_version"] == "3.10" - assert data["metrics"] == metrics - assert "timestamp" in data - assert data["timestamp"].endswith("Z") - - -def test_schema_validate_valid_data(): - """Should validate correct metric data.""" - data = { - "schema_version": "1.0", - "timestamp": datetime.now(timezone.utc).isoformat().replace('+00:00', 'Z'), - "commit_sha": "abc123", - "metrics": {}, - } - - assert MetricSchema.validate(data) is True - - -def test_schema_validate_missing_fields(): - """Should raise error for missing required fields.""" - data = { - "schema_version": "1.0", - # Missing timestamp, commit_sha, metrics - } - - with pytest.raises(ValueError, match="Missing required fields"): - MetricSchema.validate(data) - - -def test_schema_migrate_same_version(): - """Should return data unchanged for same version.""" - data = {"schema_version": "1.0", "test": "value"} - result = MetricSchema.migrate(data, "1.0") - assert result == data - - -def test_schema_migrate_unknown_version(): - """Should raise error for unknown source version.""" - data = {"schema_version": "0.5"} - - with pytest.raises(ValueError, match="No migration path"): - MetricSchema.migrate(data, "0.5") diff --git a/tests/test_umpyre_collector.py b/tests/test_umpyre_collector.py deleted file mode 100644 index 9558cc5..0000000 --- a/tests/test_umpyre_collector.py +++ /dev/null @@ -1,80 +0,0 @@ -"""Tests for UmpyreCollector.""" - -import tempfile -from pathlib import Path -import pytest - -from umpyre.collectors.umpyre_collector import UmpyreCollector - - -def test_umpyre_collector_basic(): - """Should collect basic code statistics using AST parsing.""" - # Test with real umpyre package itself - import umpyre - import os - - umpyre_path = os.path.dirname(umpyre.__file__) - collector = UmpyreCollector(root_path=umpyre_path) - metrics = collector.collect() - - # Should have collected metrics from umpyre package itself - assert metrics["num_functions"] > 0 - assert metrics["total_lines"] > 0 - assert metrics["files_analyzed"] > 0 - # Should not have errors (AST parsing is safe) - assert "error" not in metrics - - -def test_umpyre_collector_exclude_dirs(): - """Should exclude specified directories using AST parsing.""" - import umpyre - import os - - # Test with real umpyre package, excluding tests directory - umpyre_root = os.path.dirname(os.path.dirname(umpyre.__file__)) - - # Collect without exclusions - collector1 = UmpyreCollector(root_path=umpyre_root, exclude_dirs=[]) - metrics1 = collector1.collect() - - # Collect with tests excluded - collector2 = UmpyreCollector(root_path=umpyre_root, exclude_dirs=["tests"]) - metrics2 = collector2.collect() - - # With tests excluded, should have fewer or equal functions and files - assert metrics2["num_functions"] <= metrics1["num_functions"] - assert metrics2["files_analyzed"] < metrics1["files_analyzed"] - - -def test_umpyre_collector_empty_directory(): - """Should handle empty directory gracefully.""" - with tempfile.TemporaryDirectory() as tmpdir: - collector = UmpyreCollector(root_path=tmpdir) - metrics = collector.collect() - - # Should return zeros - assert metrics["num_functions"] == 0 - assert metrics["total_lines"] == 0 - - -def test_umpyre_collector_mapping_interface(): - """Should work as Mapping.""" - import umpyre - import os - - umpyre_path = os.path.dirname(umpyre.__file__) - collector = UmpyreCollector(root_path=umpyre_path) - - # Access via mapping - assert collector["num_functions"] >= 0 - assert "total_lines" in collector - assert len(collector) > 0 - - -def test_umpyre_collector_registered(): - """Should be registered in global registry.""" - from umpyre.collectors import registry as global_registry - - assert "umpyre_stats" in global_registry.list_collectors() - CollectorClass = global_registry.get("umpyre_stats") - assert CollectorClass.__name__ == "UmpyreCollector" diff --git a/umpyre/__init__.py b/umpyre/__init__.py deleted file mode 100644 index 64dde53..0000000 --- a/umpyre/__init__.py +++ /dev/null @@ -1,50 +0,0 @@ -"""Umpyre - Code analysis and quality metrics tracking.""" - -# Original python_code_stats exports -from umpyre.python_code_stats import ( - modules_info_gen, - modules_info_df, - modules_info_df_stats, - stats_of, - get_objs, -) - -# New metrics tracking exports -from umpyre.config import Config -from umpyre.schema import MetricSchema -from umpyre.collectors import ( - MetricCollector, - CollectorRegistry, - registry, - UmpyreCollector, - WorkflowStatusCollector, - WilyCollector, - CoverageCollector, -) -from umpyre.storage import ( - GitBranchStorage, - serialize_metrics, - deserialize_metrics, -) - -__all__ = [ - # Original exports - "modules_info_gen", - "modules_info_df", - "modules_info_df_stats", - "stats_of", - "get_objs", - # New exports - "Config", - "MetricSchema", - "MetricCollector", - "CollectorRegistry", - "registry", - "UmpyreCollector", - "WorkflowStatusCollector", - "WilyCollector", - "CoverageCollector", - "GitBranchStorage", - "serialize_metrics", - "deserialize_metrics", -] diff --git a/umpyre/cli.py b/umpyre/cli.py deleted file mode 100644 index 3292701..0000000 --- a/umpyre/cli.py +++ /dev/null @@ -1,241 +0,0 @@ -"""Command-line interface for umpyre.""" - -import argparse -import sys -import os -import time -from pathlib import Path -from typing import Optional - -from umpyre.config import Config -from umpyre.schema import MetricSchema -from umpyre.collectors import registry -from umpyre.storage import GitBranchStorage - - -def _collect_metrics(config: Config, repo_path: str) -> dict: - """Collect metrics from enabled collectors.""" - all_metrics = {} - - for collector_name in registry.list_collectors(): - if not config.is_collector_enabled(collector_name): - continue - - try: - collector_config = config.collector_config(collector_name) - CollectorClass = registry.get(collector_name) - - # Initialize collector with config - if collector_name == "workflow_status": - # Requires repo and token - repo = os.getenv("GITHUB_REPOSITORY") or collector_config.get("repo") - if not repo: - print(f"Skipping {collector_name}: no repository specified") - continue - collector = CollectorClass( - repo=repo, lookback_runs=collector_config.get("lookback_runs", 10) - ) - elif collector_name == "wily": - collector = CollectorClass( - repo_path=repo_path, - max_revisions=collector_config.get("max_revisions", 5), - operators=collector_config.get( - "operators", ["cyclomatic", "maintainability"] - ), - ) - elif collector_name == "coverage": - collector = CollectorClass( - repo_path=repo_path, - source=collector_config.get("source", "pytest-cov"), - ) - elif collector_name == "umpyre_stats": - collector = CollectorClass( - root_path=repo_path, - exclude_dirs=collector_config.get( - "exclude_dirs", ["tests", "examples"] - ), - ) - else: - collector = CollectorClass() - - # Collect metrics - metrics = collector.to_dict() - all_metrics[collector_name] = metrics - - except Exception as e: - print(f"Error collecting {collector_name}: {e}") - all_metrics[collector_name] = {"error": str(e)} - - return all_metrics - - -def cmd_collect(args): - """Collect and store metrics.""" - start_time = time.time() - - try: - # Load config - config = Config(config_path=args.config if args.config else None) - - # Get repo info - repo_path = args.repo_path or os.getcwd() - commit_sha = ( - args.commit or os.getenv("GITHUB_SHA") or _get_current_commit(repo_path) - ) - commit_message = args.message or _get_commit_message(repo_path, commit_sha) - - print(f"Collecting metrics for {commit_sha[:7]}...") - - # Collect metrics - metrics = _collect_metrics(config, repo_path) - - # Create standardized metric data - duration = time.time() - start_time - metric_data = MetricSchema.create_metric_data( - commit_sha=commit_sha, - metrics=metrics, - commit_message=commit_message, - python_version=_get_python_version(), - collection_duration=duration, - ) - - # Store metrics - if args.no_store: - print("Metrics collected (not stored):") - import json - - print(json.dumps(metric_data, indent=2)) - else: - storage = GitBranchStorage(repo_path) - storage.store_metrics( - metric_data, - commit_sha, - branch=config.storage_branch, - formats=config.storage_formats, - ) - print(f"Metrics stored to branch '{config.storage_branch}'") - - print(f"Collection completed in {duration:.2f}s") - return 0 - - except Exception as e: - duration = time.time() - start_time - print(f"❌ Error during metrics collection: {e}", file=sys.stderr) - print(f"Collection failed after {duration:.2f}s", file=sys.stderr) - - # In CI mode, log but don't fail - if os.getenv('CI') or os.getenv('GITHUB_ACTIONS'): - print("⚠️ Running in CI - treating as non-fatal", file=sys.stderr) - return 0 - - return 1 - - -def cmd_validate(args): - """Validate metrics against thresholds.""" - # Load config - config = Config(config_path=args.config if args.config else None) - - # Collect metrics - repo_path = args.repo_path or os.getcwd() - metrics = _collect_metrics(config, repo_path) - - # TODO: Implement threshold validation - print("Threshold validation not yet implemented") - return 0 - - -def _get_current_commit(repo_path: str) -> str: - """Get current git commit SHA.""" - import subprocess - - result = subprocess.run( - ["git", "rev-parse", "HEAD"], - cwd=repo_path, - capture_output=True, - text=True, - check=True, - ) - return result.stdout.strip() - - -def _get_commit_message(repo_path: str, commit_sha: str) -> str: - """Get commit message.""" - import subprocess - - result = subprocess.run( - ["git", "log", "-1", "--pretty=%B", commit_sha], - cwd=repo_path, - capture_output=True, - text=True, - check=True, - ) - return result.stdout.strip() - - -def _get_python_version() -> str: - """Get Python version string.""" - import sys - - return f"{sys.version_info.major}.{sys.version_info.minor}" - - -def main(): - """Main CLI entry point.""" - parser = argparse.ArgumentParser( - description="Umpyre - Code metrics collection and tracking" - ) - - subparsers = parser.add_subparsers(dest="command", help="Command to run") - - # collect command - collect_parser = subparsers.add_parser("collect", help="Collect and store metrics") - collect_parser.add_argument( - "--config", help="Path to config file (.github/umpyre-config.yml)" - ) - collect_parser.add_argument( - "--repo-path", help="Path to repository (default: current directory)" - ) - collect_parser.add_argument( - "--commit", help="Commit SHA (default: current HEAD or GITHUB_SHA)" - ) - collect_parser.add_argument( - "--message", help="Commit message (default: auto-detect from git)" - ) - collect_parser.add_argument( - "--no-store", action="store_true", help="Collect but don't store (dry-run)" - ) - collect_parser.set_defaults(func=cmd_collect) - - # validate command - validate_parser = subparsers.add_parser( - "validate", help="Validate metrics against thresholds" - ) - validate_parser.add_argument("--config", help="Path to config file") - validate_parser.add_argument("--repo-path", help="Path to repository") - validate_parser.set_defaults(func=cmd_validate) - - args = parser.parse_args() - - if not args.command: - parser.print_help() - return 1 - - try: - return args.func(args) - except Exception as e: - print(f"Error: {e}", file=sys.stderr) - - # In CI mode, be more forgiving - if os.getenv('CI') or os.getenv('GITHUB_ACTIONS'): - print("⚠️ Running in CI - treating error as non-fatal", file=sys.stderr) - return 0 - - import traceback - - traceback.print_exc() - return 1 - - -if __name__ == "__main__": - sys.exit(main()) diff --git a/umpyre/collectors/__init__.py b/umpyre/collectors/__init__.py deleted file mode 100644 index e678214..0000000 --- a/umpyre/collectors/__init__.py +++ /dev/null @@ -1,17 +0,0 @@ -"""Collectors package - metric collection implementations.""" - -from umpyre.collectors.base import MetricCollector, CollectorRegistry, registry -from umpyre.collectors.umpyre_collector import UmpyreCollector -from umpyre.collectors.workflow_status import WorkflowStatusCollector -from umpyre.collectors.wily_collector import WilyCollector -from umpyre.collectors.coverage_collector import CoverageCollector - -__all__ = [ - "MetricCollector", - "CollectorRegistry", - "registry", - "UmpyreCollector", - "WorkflowStatusCollector", - "WilyCollector", - "CoverageCollector", -] diff --git a/umpyre/collectors/base.py b/umpyre/collectors/base.py deleted file mode 100644 index d7600b5..0000000 --- a/umpyre/collectors/base.py +++ /dev/null @@ -1,101 +0,0 @@ -"""Base collector class with Mapping interface.""" - -from abc import ABC, abstractmethod -from collections.abc import Mapping -from typing import Any - - -class MetricCollector(Mapping, ABC): - """ - Abstract base for metric collectors using Mapping interface. - - Collectors implement the Mapping protocol to provide dict-like access - to collected metrics, enabling flexible composition and iteration. - - Example: - >>> class SimpleCollector(MetricCollector): - ... def collect(self) -> dict: - ... return {'lines': 100, 'functions': 10} - >>> collector = SimpleCollector() - >>> collector['lines'] - 100 - >>> list(collector) - ['lines', 'functions'] - """ - - def __init__(self): - """Initialize collector.""" - self._cached_metrics = None - - @abstractmethod - def collect(self) -> dict: - """ - Collect metrics and return as dictionary. - - Returns: - Dictionary of metric names to values - """ - raise NotImplementedError("Subclasses must implement collect()") - - def _ensure_collected(self): - """Lazily collect metrics on first access.""" - if self._cached_metrics is None: - self._cached_metrics = self.collect() - - def __getitem__(self, key: str) -> Any: - """Get specific metric value.""" - self._ensure_collected() - return self._cached_metrics[key] - - def __iter__(self): - """Iterate over metric names.""" - self._ensure_collected() - return iter(self._cached_metrics) - - def __len__(self) -> int: - """Return number of metrics.""" - self._ensure_collected() - return len(self._cached_metrics) - - def to_dict(self) -> dict: - """Export all metrics as dictionary.""" - self._ensure_collected() - return self._cached_metrics.copy() - - def refresh(self): - """Force re-collection of metrics.""" - self._cached_metrics = None - - -class CollectorRegistry: - """Registry for managing available collectors.""" - - def __init__(self): - """Initialize empty registry.""" - self._collectors = {} - - def register(self, name: str, collector_class: type[MetricCollector]): - """ - Register a collector class. - - Args: - name: Collector identifier - collector_class: MetricCollector subclass - """ - if not issubclass(collector_class, MetricCollector): - raise TypeError(f"{collector_class} must be a MetricCollector subclass") - self._collectors[name] = collector_class - - def get(self, name: str) -> type[MetricCollector]: - """Get collector class by name.""" - if name not in self._collectors: - raise KeyError(f"Unknown collector: {name}") - return self._collectors[name] - - def list_collectors(self) -> list[str]: - """List all registered collector names.""" - return list(self._collectors.keys()) - - -# Global collector registry -registry = CollectorRegistry() diff --git a/umpyre/collectors/coverage_collector.py b/umpyre/collectors/coverage_collector.py deleted file mode 100644 index a623fdf..0000000 --- a/umpyre/collectors/coverage_collector.py +++ /dev/null @@ -1,170 +0,0 @@ -"""Collector for test coverage metrics.""" - -import os -import json -import xml.etree.ElementTree as ET -from pathlib import Path -from typing import Optional - -from umpyre.collectors.base import MetricCollector, registry - - -class CoverageCollector(MetricCollector): - """ - Collect test coverage metrics from coverage reports. - - Supports: - - coverage.py JSON reports - - coverage.py XML reports (Cobertura format) - - Tracks: - - Line coverage percentage - - Branch coverage percentage (if available) - - Statements covered/total - - Example: - >>> # This requires coverage report files - >>> # Skip in doctest as it requires actual test runs - >>> True # doctest: +SKIP - """ - - def __init__( - self, - repo_path: Optional[str] = None, - source: str = "pytest-cov", - coverage_file: Optional[str] = None, - ): - """ - Initialize collector. - - Args: - repo_path: Path to repository (defaults to current directory) - source: Coverage source ('pytest-cov' or 'coverage') - coverage_file: Explicit path to coverage report (auto-detect if None) - """ - super().__init__() - self.repo_path = repo_path or os.getcwd() - self.source = source - self.coverage_file = coverage_file - - def collect(self) -> dict: - """ - Collect coverage metrics from report files. - - Returns: - Dictionary with: - - line_coverage: Line coverage percentage - - branch_coverage: Branch coverage percentage (may be 0 if not tracked) - - statements_covered: Number of statements covered - - statements_total: Total number of statements - """ - try: - # Try to find coverage file - coverage_file = self._find_coverage_file() - - if not coverage_file: - return self._empty_metrics(error="No coverage file found") - - # Parse based on file type - if coverage_file.suffix == ".json": - return self._parse_json_report(coverage_file) - elif coverage_file.suffix == ".xml": - return self._parse_xml_report(coverage_file) - else: - return self._empty_metrics( - error=f"Unsupported coverage file format: {coverage_file.suffix}" - ) - - except Exception as e: - return self._empty_metrics(error=str(e)) - - def _find_coverage_file(self) -> Optional[Path]: - """Auto-detect coverage report file.""" - if self.coverage_file: - path = Path(self.coverage_file) - return path if path.exists() else None - - # Common coverage file locations - repo = Path(self.repo_path) - candidates = [ - repo / "coverage.json", - repo / ".coverage.json", - repo / "coverage.xml", - repo / "htmlcov" / "coverage.json", - repo / ".pytest_cache" / "coverage.json", - ] - - for candidate in candidates: - if candidate.exists(): - return candidate - - return None - - def _parse_json_report(self, file_path: Path) -> dict: - """Parse coverage.py JSON report.""" - with open(file_path) as f: - data = json.load(f) - - # Extract totals - totals = data.get("totals", {}) - - # Calculate coverage percentages - num_statements = totals.get("num_statements", 0) - covered_lines = totals.get("covered_lines", 0) - num_branches = totals.get("num_branches", 0) - covered_branches = totals.get("covered_branches", 0) - - line_coverage = ( - (covered_lines / num_statements * 100) if num_statements > 0 else 0 - ) - branch_coverage = ( - (covered_branches / num_branches * 100) if num_branches > 0 else 0 - ) - - return { - "line_coverage": round(line_coverage, 2), - "branch_coverage": round(branch_coverage, 2), - "statements_covered": covered_lines, - "statements_total": num_statements, - } - - def _parse_xml_report(self, file_path: Path) -> dict: - """Parse Cobertura XML report.""" - tree = ET.parse(file_path) - root = tree.getroot() - - # Cobertura format has coverage attributes at root level - line_rate = float(root.get("line-rate", 0)) - branch_rate = float(root.get("branch-rate", 0)) - - # Convert rates (0-1) to percentages - line_coverage = line_rate * 100 - branch_coverage = branch_rate * 100 - - # Try to get statement counts - lines_covered = int(root.get("lines-covered", 0)) - lines_valid = int(root.get("lines-valid", 0)) - - return { - "line_coverage": round(line_coverage, 2), - "branch_coverage": round(branch_coverage, 2), - "statements_covered": lines_covered, - "statements_total": lines_valid, - } - - @staticmethod - def _empty_metrics(error: Optional[str] = None) -> dict: - """Return empty metrics.""" - metrics = { - "line_coverage": 0.0, - "branch_coverage": 0.0, - "statements_covered": 0, - "statements_total": 0, - } - if error: - metrics["error"] = error - return metrics - - -# Register collector -registry.register("coverage", CoverageCollector) diff --git a/umpyre/collectors/umpyre_collector.py b/umpyre/collectors/umpyre_collector.py deleted file mode 100644 index 99ef835..0000000 --- a/umpyre/collectors/umpyre_collector.py +++ /dev/null @@ -1,309 +0,0 @@ -"""Collector using AST-based code analysis (safe, no code execution).""" - -import os -import ast -import re -from pathlib import Path -from typing import Optional - -from umpyre.collectors.base import MetricCollector, registry - - -class UmpyreCollector(MetricCollector): - """ - Collect code statistics using AST parsing (no code execution). - - Provides metrics like: - - Number of functions and classes - - Lines of code (total, empty, comments, docs) - - Code ratios (comment/empty/function lines) - - Mean lines per function - - Uses AST parsing instead of dynamic imports for safety. - - Example: - >>> import tempfile - >>> from pathlib import Path - >>> # Create a simple Python file for testing - >>> with tempfile.TemporaryDirectory() as tmpdir: - ... test_file = Path(tmpdir) / "test.py" - ... test_file.write_text('''def hello(): - ... \"\"\"Say hello.\"\"\" - ... return "hello" - ... - ... class Greeter: - ... \"\"\"A greeter.\"\"\" - ... pass - ... ''') # doctest: +SKIP - ... collector = UmpyreCollector(tmpdir) - ... metrics = collector.collect() - ... metrics['num_functions'] >= 1 and metrics['num_classes'] >= 1 - True - """ - - def __init__( - self, root_path: Optional[str] = None, exclude_dirs: Optional[list[str]] = None - ): - """ - Initialize collector. - - Args: - root_path: Root directory to analyze (defaults to current directory) - exclude_dirs: List of directory names to exclude (e.g., ['tests', 'examples']) - """ - super().__init__() - self.root_path = Path(root_path or os.getcwd()) - self.exclude_dirs = exclude_dirs or [] - - def _analyze_file(self, filepath: Path) -> dict: - """ - Analyze a single Python file using AST (no code execution). - - Args: - filepath: Path to Python file - - Returns: - Dictionary with file metrics - """ - try: - with open(filepath, 'r', encoding='utf-8') as f: - content = f.read() - - lines = content.split('\n') - tree = ast.parse(content, filename=str(filepath)) - - # Count functions and classes - num_functions = 0 - num_classes = 0 - function_lines = 0 - docs_lines = 0 - - for node in ast.walk(tree): - if isinstance(node, ast.FunctionDef): - num_functions += 1 - # Count lines in function (end_lineno - lineno) - if hasattr(node, 'end_lineno') and node.end_lineno: - function_lines += node.end_lineno - node.lineno + 1 - # Extract docstring lines - docstring = ast.get_docstring(node) - if docstring: - docs_lines += len(docstring.split('\n')) - - elif isinstance(node, ast.ClassDef): - num_classes += 1 - # Extract class docstring - docstring = ast.get_docstring(node) - if docstring: - docs_lines += len(docstring.split('\n')) - - # Count empty and comment lines - empty_lines = sum(1 for line in lines if not line.strip()) - comment_lines = sum(1 for line in lines if line.strip().startswith('#')) - - return { - 'num_functions': num_functions, - 'num_classes': num_classes, - 'total_lines': len(lines), - 'empty_lines': empty_lines, - 'comment_lines': comment_lines, - 'docs_lines': docs_lines, - 'function_lines': function_lines, - } - - except Exception as e: - # Return zeros on error but track the file - return { - 'num_functions': 0, - 'num_classes': 0, - 'total_lines': 0, - 'empty_lines': 0, - 'comment_lines': 0, - 'docs_lines': 0, - 'function_lines': 0, - 'error': str(e), - } - - def _should_analyze(self, file_path: Path) -> bool: - """ - Check if file should be analyzed. - - Args: - file_path: Path to check - - Returns: - True if file should be analyzed - """ - # Skip excluded directories - for exclude_dir in self.exclude_dirs: - if exclude_dir in file_path.parts: - return False - - return True - - def _extract_pypi_version(self) -> Optional[str]: - """ - Extract PyPI version from package files. - - Checks in order: - 1. pyproject.toml (version = "x.y.z") - 2. setup.py (__version__ or version=) - 3. __init__.py (__version__) - - Returns: - Version string or None if not found - """ - # Try pyproject.toml first - pyproject = self.root_path / "pyproject.toml" - if pyproject.exists(): - try: - content = pyproject.read_text() - # Look for version = "x.y.z" in [project] or [tool.poetry] section - match = re.search( - r'version\s*=\s*["\']([\d\.]+(?:[-\w\.]*)?)["\']', content - ) - if match: - return match.group(1) - except Exception: - pass - - # Try setup.py - setup_py = self.root_path / "setup.py" - if setup_py.exists(): - try: - content = setup_py.read_text() - # Look for __version__ = "x.y.z" or version="x.y.z" - match = re.search( - r'(?:__version__|version)\s*=\s*["\']([\d\.]+(?:[-\w\.]*)?)["\']', - content, - ) - if match: - return match.group(1) - except Exception: - pass - - # Try package __init__.py - # Look for {root_path}/{package_name}/__init__.py - for init_file in self.root_path.rglob("__init__.py"): - # Only check top-level package (max 1 level deep) - if len(init_file.relative_to(self.root_path).parts) <= 2: - try: - content = init_file.read_text() - match = re.search( - r'__version__\s*=\s*["\']([\d\.]+(?:[-\w\.]*)?)["\']', content - ) - if match: - return match.group(1) - except Exception: - continue - - return None - - def collect(self) -> dict: - """ - Collect code statistics from root directory using AST parsing. - - Returns: - Dictionary with aggregated metrics: - - num_functions: Total number of functions - - num_classes: Total number of classes - - total_lines: Total lines of code - - empty_lines: Number of empty lines - - comment_lines: Number of comment lines - - docs_lines: Number of documentation lines - - function_lines: Lines in functions - - empty_lines_ratio: Ratio of empty lines - - comment_lines_ratio: Ratio of comment lines - - function_lines_ratio: Ratio of function lines - - mean_lines_per_function: Average lines per function - - files_analyzed: Number of files successfully analyzed - """ - root = Path(self.root_path) - - # Aggregate metrics across all files - total_metrics = { - 'num_functions': 0, - 'num_classes': 0, - 'total_lines': 0, - 'empty_lines': 0, - 'comment_lines': 0, - 'docs_lines': 0, - 'function_lines': 0, - } - - files_analyzed = 0 - errors = [] - - try: - # Walk through all Python files - for filepath in root.rglob('*.py'): - if not self._should_analyze(filepath): - continue - - file_metrics = self._analyze_file(filepath) - - if 'error' in file_metrics: - errors.append(f"{filepath.name}: {file_metrics['error']}") - continue - - files_analyzed += 1 - - # Aggregate - for key in total_metrics: - total_metrics[key] += file_metrics.get(key, 0) - - # Calculate ratios - total_lines = total_metrics['total_lines'] - - result = { - **total_metrics, - 'empty_lines_ratio': ( - total_metrics['empty_lines'] / total_lines - if total_lines > 0 - else 0.0 - ), - 'comment_lines_ratio': ( - total_metrics['comment_lines'] / total_lines - if total_lines > 0 - else 0.0 - ), - 'function_lines_ratio': ( - total_metrics['function_lines'] / total_lines - if total_lines > 0 - else 0.0 - ), - 'mean_lines_per_function': ( - total_metrics['function_lines'] / total_metrics['num_functions'] - if total_metrics['num_functions'] > 0 - else 0.0 - ), - 'files_analyzed': files_analyzed, - 'pypi_version': self._extract_pypi_version(), - } - - if errors: - result['errors'] = errors[:10] # Keep first 10 errors - result['total_errors'] = len(errors) - - return result - - except Exception as e: - # Return empty metrics on catastrophic error - return { - "num_functions": 0, - "num_classes": 0, - "total_lines": 0, - "empty_lines": 0, - "comment_lines": 0, - "docs_lines": 0, - "function_lines": 0, - "empty_lines_ratio": 0.0, - "comment_lines_ratio": 0.0, - "function_lines_ratio": 0.0, - "mean_lines_per_function": 0.0, - "files_analyzed": 0, - "error": str(e), - } - - -# Register collector -registry.register("umpyre_stats", UmpyreCollector) diff --git a/umpyre/collectors/wily_collector.py b/umpyre/collectors/wily_collector.py deleted file mode 100644 index a0be7dd..0000000 --- a/umpyre/collectors/wily_collector.py +++ /dev/null @@ -1,221 +0,0 @@ -"""Collector for complexity metrics using wily.""" - -import os -import subprocess -import json -import tempfile -from pathlib import Path -from typing import Optional - -from umpyre.collectors.base import MetricCollector, registry - - -class WilyCollector(MetricCollector): - """ - Collect complexity metrics using wily. - - Tracks: - - Cyclomatic complexity (average) - - Maintainability index - - Total lines of code - - Files analyzed - - Note: Limited to recent commits (max_revisions) for performance. - - Example: - >>> # This requires wily to be installed and a git repo - >>> # Skip in doctest as it requires git setup - >>> True # doctest: +SKIP - """ - - def __init__( - self, - repo_path: Optional[str] = None, - max_revisions: int = 5, - operators: Optional[list[str]] = None, - ): - """ - Initialize collector. - - Args: - repo_path: Path to git repository (defaults to current directory) - max_revisions: Number of recent commits to analyze (for speed) - operators: Wily operators to use (defaults to ['cyclomatic', 'maintainability']) - """ - super().__init__() - self.repo_path = repo_path or os.getcwd() - self.max_revisions = max_revisions - self.operators = operators or ["cyclomatic", "maintainability"] - - def collect(self) -> dict: - """ - Collect complexity metrics using wily. - - Returns: - Dictionary with: - - cyclomatic_avg: Average cyclomatic complexity - - maintainability_index: Maintainability index score - - total_loc: Total lines of code - - files_analyzed: Number of files analyzed - """ - try: - # Check if wily is available - if not self._is_wily_available(): - return self._empty_metrics(error="wily not installed") - - # Check if repo is initialized - wily_cache = Path(self.repo_path) / ".wily" - if not wily_cache.exists(): - # Build wily cache (limited revisions for speed) - self._build_wily_cache() - - # Get latest metrics - metrics = self._get_latest_metrics() - - return metrics - - except Exception as e: - return self._empty_metrics(error=str(e)) - - def _is_wily_available(self) -> bool: - """Check if wily is installed.""" - try: - subprocess.run( - ["wily", "--version"], capture_output=True, check=True, timeout=5 - ) - return True - except (subprocess.CalledProcessError, FileNotFoundError): - return False - - def _build_wily_cache(self): - """Build wily cache with limited revisions.""" - cmd = [ - "wily", - "build", - self.repo_path, - "--max-revisions", - str(self.max_revisions), - ] - - # Add operators - for operator in self.operators: - cmd.extend(["--operators", operator]) - - result = subprocess.run( - cmd, - capture_output=True, - text=True, - cwd=self.repo_path, - timeout=60, - ) - - if result.returncode != 0: - raise RuntimeError(f"wily build failed: {result.stderr}") - - def _get_latest_metrics(self) -> dict: - """Extract metrics from wily report.""" - # Get report in JSON format - result = subprocess.run( - ["wily", "report", self.repo_path, "--format", "json"], - capture_output=True, - text=True, - cwd=self.repo_path, - timeout=30, - ) - - if result.returncode != 0: - raise RuntimeError(f"wily report failed: {result.stderr}") - - # Parse JSON output - try: - report = json.loads(result.stdout) - except json.JSONDecodeError: - # Fallback: parse text output - return self._parse_text_report() - - # Extract metrics from JSON - return self._extract_metrics_from_json(report) - - def _parse_text_report(self) -> dict: - """Parse text output as fallback (wily doesn't always output valid JSON).""" - result = subprocess.run( - ["wily", "report", self.repo_path], - capture_output=True, - text=True, - cwd=self.repo_path, - timeout=30, - ) - - # Simple parsing of key metrics from text - output = result.stdout - - # This is a simplified parser - actual implementation would be more robust - metrics = { - "cyclomatic_avg": 0.0, - "maintainability_index": 0.0, - "total_loc": 0, - "files_analyzed": 0, - } - - # Count files - metrics["files_analyzed"] = output.count(".py") - - return metrics - - def _extract_metrics_from_json(self, report: dict) -> dict: - """Extract metrics from JSON report.""" - # Wily JSON structure varies, this is a best-effort extraction - metrics = { - "cyclomatic_avg": 0.0, - "maintainability_index": 0.0, - "total_loc": 0, - "files_analyzed": 0, - } - - # Navigate report structure (this varies by wily version) - if isinstance(report, dict): - metrics["files_analyzed"] = len(report) - - # Calculate averages - cyclomatic_values = [] - maintainability_values = [] - loc_values = [] - - for file_data in report.values(): - if isinstance(file_data, dict): - if "cyclomatic" in file_data: - cyclomatic_values.append(file_data["cyclomatic"]) - if "maintainability" in file_data: - maintainability_values.append(file_data["maintainability"]) - if "loc" in file_data: - loc_values.append(file_data["loc"]) - - if cyclomatic_values: - metrics["cyclomatic_avg"] = sum(cyclomatic_values) / len( - cyclomatic_values - ) - if maintainability_values: - metrics["maintainability_index"] = sum(maintainability_values) / len( - maintainability_values - ) - if loc_values: - metrics["total_loc"] = sum(loc_values) - - return metrics - - @staticmethod - def _empty_metrics(error: Optional[str] = None) -> dict: - """Return empty metrics.""" - metrics = { - "cyclomatic_avg": 0.0, - "maintainability_index": 0.0, - "total_loc": 0, - "files_analyzed": 0, - } - if error: - metrics["error"] = error - return metrics - - -# Register collector -registry.register("wily", WilyCollector) diff --git a/umpyre/collectors/workflow_status.py b/umpyre/collectors/workflow_status.py deleted file mode 100644 index df84dac..0000000 --- a/umpyre/collectors/workflow_status.py +++ /dev/null @@ -1,165 +0,0 @@ -"""Collector for GitHub workflow status using GitHub API.""" - -import os -from datetime import datetime, timezone -from typing import Optional - -import requests - -from umpyre.collectors.base import MetricCollector, registry - - -class WorkflowStatusCollector(MetricCollector): - """ - Collect GitHub workflow status using GitHub API. - - Tracks: - - Last workflow run status (success/failure) - - Last successful run timestamp - - Recent failure count - - Workflow run URL - - Example: - >>> # Mock test (requires actual GitHub API) - >>> import os - >>> if os.getenv('GITHUB_TOKEN'): # doctest: +SKIP - ... collector = WorkflowStatusCollector( - ... repo="thorwhalen/astate", - ... token=os.getenv('GITHUB_TOKEN') - ... ) - ... metrics = collector.collect() - ... 'last_run_status' in metrics - ... else: - ... True # Skip if no token - True - """ - - def __init__( - self, - repo: str, - token: Optional[str] = None, - lookback_runs: int = 10, - workflow_name: Optional[str] = None, - ): - """ - Initialize collector. - - Args: - repo: Repository in format "owner/repo" - token: GitHub token (defaults to GITHUB_TOKEN env var) - lookback_runs: Number of recent runs to analyze - workflow_name: Specific workflow name to track (None = all workflows) - """ - super().__init__() - self.repo = repo - self.token = token or os.getenv("GITHUB_TOKEN") - self.lookback_runs = lookback_runs - self.workflow_name = workflow_name - - if not self.token: - raise ValueError("GitHub token required (set GITHUB_TOKEN env var)") - - def collect(self) -> dict: - """ - Collect workflow status from GitHub API. - - Returns: - Dictionary with: - - last_run_status: 'success', 'failure', or 'other' - - last_success_timestamp: ISO timestamp or None - - recent_failure_count: Number of failures in lookback window - - run_url: URL of last run - - total_runs_analyzed: Number of runs examined - """ - try: - runs = self._fetch_workflow_runs() - - if not runs: - return self._empty_metrics() - - # Analyze runs - last_run = runs[0] - last_run_status = self._map_conclusion(last_run.get("conclusion")) - - # Find last successful run - last_success = None - for run in runs: - if run.get("conclusion") == "success": - last_success = run.get("updated_at") - break - - # Count recent failures - failure_count = sum( - 1 - for run in runs - if run.get("conclusion") in ["failure", "cancelled", "timed_out"] - ) - - return { - "last_run_status": last_run_status, - "last_success_timestamp": last_success, - "recent_failure_count": failure_count, - "run_url": last_run.get("html_url"), - "total_runs_analyzed": len(runs), - } - - except Exception as e: - return { - "last_run_status": "error", - "last_success_timestamp": None, - "recent_failure_count": 0, - "run_url": None, - "total_runs_analyzed": 0, - "error": str(e), - } - - def _fetch_workflow_runs(self) -> list[dict]: - """Fetch recent workflow runs from GitHub API.""" - headers = { - "Authorization": f"Bearer {self.token}", - "Accept": "application/vnd.github.v3+json", - } - - # Fetch workflow runs - url = f"https://api.github.com/repos/{self.repo}/actions/runs" - params = { - "per_page": self.lookback_runs, - "status": "completed", # Only completed runs - } - - response = requests.get(url, headers=headers, params=params, timeout=10) - response.raise_for_status() - - data = response.json() - runs = data.get("workflow_runs", []) - - # Filter by workflow name if specified - if self.workflow_name: - runs = [run for run in runs if run.get("name") == self.workflow_name] - - return runs[: self.lookback_runs] - - @staticmethod - def _map_conclusion(conclusion: Optional[str]) -> str: - """Map GitHub conclusion to simplified status.""" - if conclusion == "success": - return "success" - elif conclusion in ["failure", "cancelled", "timed_out"]: - return "failure" - else: - return "other" - - @staticmethod - def _empty_metrics() -> dict: - """Return empty metrics when no data available.""" - return { - "last_run_status": "unknown", - "last_success_timestamp": None, - "recent_failure_count": 0, - "run_url": None, - "total_runs_analyzed": 0, - } - - -# Register collector -registry.register("workflow_status", WorkflowStatusCollector) diff --git a/umpyre/config.py b/umpyre/config.py deleted file mode 100644 index f1b3836..0000000 --- a/umpyre/config.py +++ /dev/null @@ -1,169 +0,0 @@ -"""Configuration loading and validation for umpyre.""" - -import os -from pathlib import Path -from typing import Any, Optional - -import yaml - - -class ConfigError(Exception): - """Raised when configuration is invalid.""" - - -class Config: - """Configuration manager for umpyre metrics collection.""" - - DEFAULT_CONFIG = { - "schema_version": "1.0", - "collectors": { - "workflow_status": { - "enabled": True, - "lookback_runs": 10, - }, - "wily": { - "enabled": True, - "max_revisions": 5, - "operators": ["cyclomatic", "maintainability"], - }, - "coverage": { - "enabled": True, - "source": "pytest-cov", - }, - "umpyre_stats": { - "enabled": True, - "exclude_dirs": ["tests", "examples", "scrap"], - }, - }, - "storage": { - "branch": "code-metrics", - "formats": ["json", "csv"], - "retention": { - "strategy": "all", # or: last_n_days, last_n_commits - }, - }, - "visualization": { - "generate_plots": True, - "generate_readme": True, - "plot_metrics": ["maintainability", "coverage", "loc"], - }, - "thresholds": { - "enabled": False, - }, - "aggregation": { - "enabled": False, - }, - } - - def __init__( - self, config_path: Optional[str] = None, config_dict: Optional[dict] = None - ): - """ - Initialize configuration. - - Args: - config_path: Path to YAML config file - config_dict: Config as dictionary (overrides file) - """ - self._config = self._load_config(config_path, config_dict) - self._validate() - - def _load_config( - self, config_path: Optional[str], config_dict: Optional[dict] - ) -> dict: - """Load configuration from file or dict, merging with defaults.""" - config = self.DEFAULT_CONFIG.copy() - - if config_path and Path(config_path).exists(): - with open(config_path) as f: - file_config = yaml.safe_load(f) or {} - config = self._deep_merge(config, file_config) - - if config_dict: - config = self._deep_merge(config, config_dict) - - return config - - @staticmethod - def _deep_merge(base: dict, override: dict) -> dict: - """Deep merge two dictionaries.""" - result = base.copy() - for key, value in override.items(): - if ( - key in result - and isinstance(result[key], dict) - and isinstance(value, dict) - ): - result[key] = Config._deep_merge(result[key], value) - else: - result[key] = value - return result - - def _validate(self): - """Validate configuration structure.""" - required_sections = ["collectors", "storage"] - for section in required_sections: - if section not in self._config: - raise ConfigError(f"Missing required config section: {section}") - - def get(self, *keys: str, default: Any = None) -> Any: - """ - Get nested config value. - - Args: - *keys: Nested keys (e.g., 'collectors', 'wily', 'enabled') - default: Default value if key not found - - Returns: - Config value or default - - >>> config = Config(config_dict={'collectors': {'wily': {'enabled': True}}}) - >>> config.get('collectors', 'wily', 'enabled') - True - >>> config.get('collectors', 'missing', default='default_value') - 'default_value' - """ - value = self._config - for key in keys: - if isinstance(value, dict) and key in value: - value = value[key] - else: - return default - return value - - def is_collector_enabled(self, collector_name: str) -> bool: - """Check if a collector is enabled.""" - return self.get("collectors", collector_name, "enabled", default=False) - - def collector_config(self, collector_name: str) -> dict: - """Get configuration for a specific collector.""" - return self.get("collectors", collector_name, default={}) - - @property - def storage_branch(self) -> str: - """Get the storage branch name.""" - return self.get("storage", "branch", default="code-metrics") - - @property - def storage_formats(self) -> list[str]: - """Get enabled storage formats.""" - return self.get("storage", "formats", default=["json"]) - - @property - def retention_strategy(self) -> str: - """Get data retention strategy.""" - return self.get("storage", "retention", "strategy", default="all") - - def to_dict(self) -> dict: - """Export config as dictionary.""" - return self._config.copy() - - @classmethod - def from_file(cls, path: str) -> "Config": - """Load config from YAML file.""" - return cls(config_path=path) - - @classmethod - def from_dict(cls, config: dict) -> "Config": - """Create config from dictionary.""" - return cls(config_dict=config) diff --git a/umpyre/python_code_stats.py b/umpyre/python_code_stats.py deleted file mode 100644 index 32e6c09..0000000 --- a/umpyre/python_code_stats.py +++ /dev/null @@ -1,365 +0,0 @@ -""" -Get stats about packages. Your own, or other's. -Things like... -(Note that these will probably not work as doctests, since results are sensisitive to other slight system differences -(such as python version etc.)) - ->>> import collections ->>> modules_info_df(collections) #doctest: +SKIP - lines empty_lines ... num_of_functions num_of_classes -collections.__init__ 1280 189 ... 1 9 -collections.abc 3 1 ... 0 25 - -[2 rows x 7 columns] ->>> modules_info_df_stats(collections.abc) #doctest: +SKIP -lines 1283.000000 -empty_lines 190.000000 -comment_lines 79.000000 -docs_lines 133.000000 -function_lines 138.000000 -num_of_functions 1.000000 -num_of_classes 34.000000 -empty_lines_ratio 0.148090 -comment_lines_ratio 0.061574 -function_lines_ratio 0.107560 -mean_lines_per_function 138.000000 -dtype: float64 ->>> stats_of(['urllib', 'json', 'collections']) #doctest: +SKIP - urllib json collections -empty_lines_ratio 0.157293 0.136503 0.148090 -comment_lines_ratio 0.075217 0.038344 0.061574 -function_lines_ratio 0.212391 0.448620 0.107560 -mean_lines_per_function 13.463768 41.785714 138.000000 -lines 4374.000000 1304.000000 1283.000000 -empty_lines 688.000000 178.000000 190.000000 -comment_lines 329.000000 50.000000 79.000000 -docs_lines 425.000000 218.000000 133.000000 -function_lines 929.000000 585.000000 138.000000 -num_of_functions 69.000000 14.000000 1.000000 -num_of_classes 55.000000 3.000000 34.000000 -""" -import re -import os -from types import FunctionType, ModuleType -from inspect import getsource -from functools import partial -import collections - -import pandas as pd -from dol import filt_iter -from dol.sources import Attrs -from dol.filesys import FileStringReader - -psep = os.path.sep - -DFLT_ON_ERROR = "ignore" # could be 'print', 'ignore', or 'raise' - -empty_line = re.compile(r"^\s*$") -comment_line_p = re.compile(r"^\s+#.+$") -line_p = re.compile("\n|\r|\n\r|\r\n") -only_py_ext = lambda path: path.endswith(".py") -no_test_folder = lambda path: "test" not in path.split(psep) - - -def lines(string): - return line_p.split(string) - - -def _num_lines_of_function_code(obj: FunctionType): - assert isinstance(obj, FunctionType) - return len(getsource(obj).split("\n")) - len( - (obj.__doc__ or "").split("\n") - ) - - -def _root_dir_and_name(root): - """ - The parent path and leaf name for the given root - :param root: module instance, dot path, or directory path - :return: - """ - if isinstance(root, str) and not os.path.dirname(root): - root = __import__( - root - ) # assume it's a dot path string of a module and try to import it - if isinstance(root, ModuleType): - root = os.path.dirname(root.__file__) - if root.endswith(psep): - root = root[:-1] - return os.path.dirname(root), os.path.basename(root) - - -# TODO: Must be a standard lib for this! -def _path_to_module_str(path, root_path): - """ - The dot-path module string for a path (given the root_path, assumed to be on the python path) - :param path: The path to the module. - :param root_path: The path that's assumed to be on the python path - :return: - """ - assert path.endswith(".py") - path = path[:-3] - - if root_path.endswith(psep): - root_path = root_path[:-1] - root_path, root_package = _root_dir_and_name(root_path) - len_root = len(root_path) + 1 - path_parts = path[len_root:].split(psep) - if path_parts[-1] == "__init__.py": - path_parts = path_parts[:-1] - return ".".join(path_parts) - - -def get_objs(root, k): - name = _path_to_module_str(k, root) - module_store = Attrs.module_from_path( - k, key_filt=lambda x: not x.startswith("__"), name=name - ) - - def obj_filt( - obj, - ): # to make sure we only analyze objects defined in module itself, not imported - obj_module = getattr(obj, "__module__", None) - if obj_module: - return obj_module == name - - objs = list(filter(obj_filt, (vv._source for vv in module_store.values()))) - return objs - - -def modules_info_gen(root, filepath_filt=only_py_ext, on_error=DFLT_ON_ERROR): - """ - Yields statistics (as dicts) of modules under the root module or directory of a python package. - - :param root: module instance, dot path, or directory path - :param filepath_filt: filepath filter function or regular expression - :param on_error: What to do when an error occurs when extracting information from a module. - Values are 'ignore', 'print', 'raise', or 'yield' - :return: A generator of dicts - """ - """Gives us statistics given the root module or directory of a python package""" - if isinstance(filepath_filt, str): - filt_pattern = re.compile(filepath_filt) - filepath_filt = filt_pattern.match - - parent_dir, dirname = _root_dir_and_name(root) - root = os.path.join(parent_dir, dirname) - - @filt_iter(filt=filepath_filt) - class PyCodeReader(FileStringReader): - pass - - pycode = PyCodeReader(root) - - for filepath, code_str in pycode.items(): - try: - name = _path_to_module_str(filepath, root) - module_store = Attrs.module_from_path( - filepath, key_filt=lambda x: not x.startswith("__"), name=name - ) - - def obj_filt( - obj, - ): # to make sure we only analyze objects defined in module itself, not imported - obj_module = getattr(obj, "__module__", None) - if obj_module: - return obj_module == name - - objs = list( - filter(obj_filt, (vv._source for vv in module_store.values())) - ) - yield _code_stats_dict_for_file_and_objects( - code_str, filepath, objs - ) - except Exception as e: - if on_error == "print": - print(f"Problem with {filepath}: {str(e)[:50]}\n") - elif on_error == "ignore": - pass - elif on_error == "yield": - yield {"filepath": filepath, "error": e} - else: - raise - - -def _code_stats_dict_for_file_and_objects(code_str, filepath, objs): - return { - "filepath": filepath, - "lines": len(lines(code_str)), - "empty_lines": sum( - bool(empty_line.match(line)) for line in lines(code_str) - ), - "comment_lines": sum( - bool(comment_line_p.match(line)) for line in lines(code_str) - ), - "docs_lines": sum(len(lines(obj.__doc__ or "")) for obj in objs), - "function_lines": sum( - _num_lines_of_function_code(obj) - for obj in objs - if isinstance(obj, FunctionType) - ), - "num_of_functions": sum(isinstance(obj, FunctionType) for obj in objs), - "num_of_classes": sum(isinstance(obj, type) for obj in objs), - } - - -def modules_info_df( - root, filepath_filt=only_py_ext, index_field=None, on_error=DFLT_ON_ERROR -): - """ - A pandas DataFrame of stats of the root (package or directory thereof). - :param root: module or directory path - :param filepath_filt: filepath filter function or regular expression - :param index_field: function or field string that should be used for the indexing of modules - :param on_error: What to do when an error occurs when extracting information from a module. - Values are 'ignore', 'print', or 'raise' - :return: A DataFrame whose rows contain information for each module - - >>> import urllib - >>> df = modules_info_df(urllib) - >>> df #doctest: +SKIP - lines empty_lines ... num_of_functions num_of_classes - urllib.error 78 18 ... 0 3 - urllib.request 2774 410 ... 23 28 - urllib.__init__ 1 1 ... 0 0 - urllib.response 81 24 ... 0 4 - urllib.robotparser 274 36 ... 0 4 - urllib.parse 1166 199 ... 46 16 - - [6 rows x 7 columns] - """ - - import pandas as pd - - d = list(modules_info_gen(root, filepath_filt, on_error=on_error)) - - if index_field is None: - dirpath, dirname = _root_dir_and_name(root) - root = os.path.join(dirpath, dirname) - index_field = lambda x: _path_to_module_str(x.pop("filepath"), root) - - if isinstance(index_field, str): - return pd.DataFrame(d).set_index(index_field) - elif callable(index_field): - index = [index_field(x) for x in d] - return pd.DataFrame(index=index, data=d) - - -def modules_info_df_stats( - root, filepath_filt=only_py_ext, index_field=None, on_error=DFLT_ON_ERROR -): - """ - A pandas Series of statistics over all modules of some root (package or directory thereof). - :param root: module or directory path - :param filepath_filt: filepath filter function or regular expression - :param index_field: function or field string that should be used for the indexing of modules - :param on_error: What to do when an error occurs when extracting information from a module. - Values are 'ignore', 'print', or 'raise' - :return: A Series whose rows containing statistics - - >>> import json - >>> df = modules_info_df_stats(json) - >>> df #doctest: +SKIP - lines 1301.000000 - empty_lines 178.000000 - comment_lines 50.000000 - docs_lines 218.000000 - function_lines 585.000000 - num_of_functions 14.000000 - num_of_classes 3.000000 - empty_lines_ratio 0.136818 - comment_lines_ratio 0.038432 - function_lines_ratio 0.449654 - mean_lines_per_function 41.785714 - dtype: float64 - >>> modules_info_df_stats('collections.abc') #doctest: +SKIP - lines 1276.000000 - empty_lines 190.000000 - comment_lines 73.000000 - docs_lines 133.000000 - function_lines 138.000000 - num_of_functions 1.000000 - num_of_classes 34.000000 - empty_lines_ratio 0.148903 - comment_lines_ratio 0.057210 - function_lines_ratio 0.108150 - mean_lines_per_function 138.000000 - dtype: float64 - """ - df = modules_info_df(root, filepath_filt, index_field, on_error=on_error) - df = df.sum() - cols = set(df.index.values) - for col in ["empty_lines", "comment_lines", "doc_lines", "function_lines"]: - if {col, "lines"}.issubset(cols): - df[f"{col}_ratio"] = df[col] / df["lines"] - if {"num_of_functions", "function_lines"}.issubset(cols): - df["mean_lines_per_function"] = ( - df["function_lines"] / df["num_of_functions"] - ) - - return df - - -def stats_of( - modules, - filepath_filt=only_py_ext, - index_field=None, - on_error=DFLT_ON_ERROR, -): - """ - A dataframe of stats of the input modules. - - :param modules: list of importable names - :param root: module or directory path - :param filepath_filt: filepath filter function or regular expression - :param index_field: function or field string that should be used for the indexing of modules - :param on_error: What to do when an error occurs when extracting information from a module. - Values are 'ignore', 'print', or 'raise' - :return: - - >>> df = stats_of(['urllib', 'json', 'collections']) # doctest: +SKIP - >>> assert set(df.index.values).issuperset( # doctest: +SKIP - ... {'empty_lines_ratio', 'comment_lines_ratio', 'function_lines_ratio', 'mean_lines_per_function'}) - >>> stats_of(['urllib', 'json', 'collections']) #doctest: +SKIP - urllib json collections - empty_lines_ratio 0.157293 0.136503 0.148090 - comment_lines_ratio 0.075217 0.038344 0.061574 - function_lines_ratio 0.212391 0.448620 0.107560 - mean_lines_per_function 13.463768 41.785714 138.000000 - lines 4374.000000 1304.000000 1283.000000 - empty_lines 688.000000 178.000000 190.000000 - comment_lines 329.000000 50.000000 79.000000 - docs_lines 425.000000 218.000000 133.000000 - function_lines 929.000000 585.000000 138.000000 - num_of_functions 69.000000 14.000000 1.000000 - num_of_classes 55.000000 3.000000 34.000000 - """ - - stats_func = partial( - modules_info_df_stats, - filepath_filt=filepath_filt, - index_field=index_field, - on_error=on_error, - ) - - if isinstance(modules, str): - modules = [modules] - df = pd.concat([stats_func(x) for x in modules], axis=1) - df.columns = modules - put_at_the_end = [ - x for x in df.index if x.endswith("lines") or x.startswith("num") - ] - put_at_the_front = [x for x in df.index if x not in put_at_the_end] - df = df.loc[put_at_the_front + put_at_the_end] - return df - - -if __name__ == "__main__": - from dol.util import ModuleNotFoundErrorNiceMessage - - with ModuleNotFoundErrorNiceMessage(): - import argh - - argh.dispatch_commands( - [modules_info_df, modules_info_df_stats, stats_of] - ) diff --git a/umpyre/schema.py b/umpyre/schema.py deleted file mode 100644 index f040cdc..0000000 --- a/umpyre/schema.py +++ /dev/null @@ -1,97 +0,0 @@ -"""Versioned schema for metrics storage and migration support.""" - -from dataclasses import dataclass, field, asdict -from datetime import datetime, timezone -from typing import Any, Optional - - -@dataclass -class MetricSchema: - """Versioned schema for metrics storage with migration support.""" - - version: str = "1.0" - - @classmethod - def current_version(cls) -> str: - """Return the current schema version.""" - return "1.0" - - @classmethod - def migrate(cls, data: dict, from_version: str) -> dict: - """ - Migrate data from old schema version to current version. - - Args: - data: Metric data in old format - from_version: Version of the input data - - Returns: - Migrated data in current schema format - """ - if from_version == cls.current_version(): - return data - - # Future: implement migration chains - # e.g., 1.0 -> 1.1 -> 1.2 - migrations = { - # "1.0": cls._migrate_1_0_to_1_1, - } - - if from_version not in migrations: - raise ValueError(f"No migration path from version {from_version}") - - return migrations[from_version](data) - - @classmethod - def validate(cls, data: dict) -> bool: - """ - Validate that data conforms to schema. - - Args: - data: Metric data to validate - - Returns: - True if valid, raises ValueError otherwise - """ - required_fields = {"schema_version", "timestamp", "commit_sha", "metrics"} - - if not all(field in data for field in required_fields): - missing = required_fields - set(data.keys()) - raise ValueError(f"Missing required fields: {missing}") - - return True - - @classmethod - def create_metric_data( - cls, - commit_sha: str, - metrics: dict, - commit_message: Optional[str] = None, - python_version: Optional[str] = None, - workflow_status: Optional[dict] = None, - collection_duration: Optional[float] = None, - ) -> dict: - """ - Create a standardized metric data structure. - - Args: - commit_sha: Git commit SHA - metrics: Dictionary of collected metrics - commit_message: Optional commit message - python_version: Optional Python version string - workflow_status: Optional workflow status dict - collection_duration: Optional collection time in seconds - - Returns: - Standardized metric dictionary - """ - return { - "schema_version": cls.current_version(), - "timestamp": datetime.now(timezone.utc).isoformat().replace('+00:00', 'Z'), - "commit_sha": commit_sha, - "commit_message": commit_message, - "python_version": python_version, - "workflow_status": workflow_status or {}, - "metrics": metrics, - "collection_duration_seconds": collection_duration, - } diff --git a/umpyre/storage/__init__.py b/umpyre/storage/__init__.py deleted file mode 100644 index 87c0356..0000000 --- a/umpyre/storage/__init__.py +++ /dev/null @@ -1,25 +0,0 @@ -"""Storage operations for metrics.""" - -from umpyre.storage.git_branch import GitBranchStorage, GitBranchStorageError -from umpyre.storage.formats import serialize_metrics, deserialize_metrics -from umpyre.storage.query_utils import ( - parse_metric_filename, - find_metrics_by_commit, - find_metrics_by_version, - get_latest_metric_for_version, - get_all_versions, - filter_by_date_range, -) - -__all__ = [ - "GitBranchStorage", - "GitBranchStorageError", - "serialize_metrics", - "deserialize_metrics", - "parse_metric_filename", - "find_metrics_by_commit", - "find_metrics_by_version", - "get_latest_metric_for_version", - "get_all_versions", - "filter_by_date_range", -] diff --git a/umpyre/storage/formats.py b/umpyre/storage/formats.py deleted file mode 100644 index 488008e..0000000 --- a/umpyre/storage/formats.py +++ /dev/null @@ -1,108 +0,0 @@ -"""JSON and CSV serialization for metrics.""" - -import json -import csv -from pathlib import Path -from typing import Any -from io import StringIO - - -def _serialize_json(data: dict, file_path: Path, indent: int = 2): - """Serialize metrics to JSON file.""" - with open(file_path, 'w') as f: - json.dump(data, f, indent=indent, sort_keys=True) - - -def _deserialize_json(file_path: Path) -> dict: - """Deserialize metrics from JSON file.""" - with open(file_path) as f: - return json.load(f) - - -def _serialize_csv(data: dict, file_path: Path): - """ - Serialize metrics to CSV file (flat format). - - Args: - data: Nested metric dictionary - file_path: Path to CSV file - """ - # Flatten nested structure - flat_data = _flatten_dict(data) - - with open(file_path, 'w', newline='') as f: - writer = csv.writer(f) - # Write header - writer.writerow(['metric', 'value']) - # Write data - for key, value in flat_data.items(): - writer.writerow([key, value]) - - -def _deserialize_csv(file_path: Path) -> dict: - """ - Deserialize metrics from CSV file. - - Returns: - Flat dictionary of metrics - """ - data = {} - with open(file_path, newline='') as f: - reader = csv.DictReader(f) - for row in reader: - data[row['metric']] = row['value'] - return data - - -def _flatten_dict(d: dict, parent_key: str = '', sep: str = '.') -> dict: - """ - Flatten nested dictionary. - - Example: - >>> _flatten_dict({'a': {'b': 1, 'c': 2}, 'd': 3}) - {'a.b': 1, 'a.c': 2, 'd': 3} - """ - items = [] - for k, v in d.items(): - new_key = f"{parent_key}{sep}{k}" if parent_key else k - if isinstance(v, dict): - items.extend(_flatten_dict(v, new_key, sep=sep).items()) - else: - items.append((new_key, v)) - return dict(items) - - -def serialize_metrics(data: dict, output_path: Path, format: str = 'json'): - """ - Serialize metrics to file. - - Args: - data: Metric dictionary - output_path: Path to output file - format: Format ('json' or 'csv') - """ - if format == 'json': - _serialize_json(data, output_path) - elif format == 'csv': - _serialize_csv(data, output_path) - else: - raise ValueError(f"Unsupported format: {format}") - - -def deserialize_metrics(file_path: Path, format: str = 'json') -> dict: - """ - Deserialize metrics from file. - - Args: - file_path: Path to input file - format: Format ('json' or 'csv') - - Returns: - Metric dictionary - """ - if format == 'json': - return _deserialize_json(file_path) - elif format == 'csv': - return _deserialize_csv(file_path) - else: - raise ValueError(f"Unsupported format: {format}") diff --git a/umpyre/storage/git_branch.py b/umpyre/storage/git_branch.py deleted file mode 100644 index 8eac21c..0000000 --- a/umpyre/storage/git_branch.py +++ /dev/null @@ -1,226 +0,0 @@ -"""Git branch-based storage operations.""" - -import subprocess -import tempfile -from datetime import datetime -from pathlib import Path -from typing import Optional - -from umpyre.storage.formats import serialize_metrics - - -class GitBranchStorage: - """ - Store metrics in a separate git branch. - - This provides versioned, branch-based storage without polluting - the main branch or requiring external artifacts. - - Example: - >>> storage = GitBranchStorage(repo_path="/path/to/repo") # doctest: +SKIP - >>> metrics = {"lines": 100, "functions": 10} # doctest: +SKIP - >>> storage.store_metrics(metrics, "abc123", branch="code-metrics") # doctest: +SKIP - """ - - def __init__(self, repo_path: str, remote: str = "origin"): - """ - Initialize storage. - - Args: - repo_path: Path to git repository - remote: Git remote name - """ - self.repo_path = Path(repo_path) - self.remote = remote - - if not (self.repo_path / ".git").exists(): - raise ValueError(f"Not a git repository: {repo_path}") - - def store_metrics( - self, - metrics: dict, - commit_sha: str, - branch: str = "code-metrics", - formats: Optional[list[str]] = None, - ): - """ - Store metrics to git branch. - - Args: - metrics: Metric dictionary to store - commit_sha: Git commit SHA - branch: Target branch name - formats: List of formats ('json', 'csv') - """ - formats = formats or ["json", "csv"] - - # Create temporary directory for branch checkout - with tempfile.TemporaryDirectory() as tmpdir: - tmpdir = Path(tmpdir) - - # Fetch and checkout metrics branch - self._fetch_branch(branch) - self._checkout_branch(branch, tmpdir) - - # Create history directory (flat structure) - history_dir = tmpdir / "history" - history_dir.mkdir(exist_ok=True) - - # Build filename: YYYY_MM_DD_HH_MM_SS__shahash__version.json - timestamp = datetime.now().strftime("%Y_%m_%d_%H_%M_%S") - commit_short = commit_sha[:7] - - # Extract pypi_version if available (use 'none' if not found) - pypi_version = ( - metrics.get('metrics', {}).get('umpyre_stats', {}).get('pypi_version') - or 'none' - ) - - # Filename format: timestamp__sha__version (parseable, chronological) - filename_base = f"{timestamp}__{commit_short}__{pypi_version}" - - for fmt in formats: - if fmt == "json": - # Latest snapshot - serialize_metrics(metrics, tmpdir / "metrics.json", format="json") - - # Historical record (flat structure, parseable filename) - history_file = history_dir / f"{filename_base}.json" - serialize_metrics(metrics, history_file, format="json") - - elif fmt == "csv": - serialize_metrics(metrics, tmpdir / "metrics.csv", format="csv") - - # Commit and push - self._commit_and_push(tmpdir, branch, commit_sha) - - def _fetch_branch(self, branch: str): - """Fetch metrics branch from remote.""" - try: - subprocess.run( - ["git", "fetch", self.remote, f"{branch}:{branch}"], - cwd=self.repo_path, - capture_output=True, - check=False, # May not exist yet - timeout=30, - ) - except subprocess.TimeoutExpired: - pass - - def _checkout_branch(self, branch: str, target_dir: Path): - """Checkout metrics branch to temporary directory.""" - # Check if branch exists locally - result = subprocess.run( - ["git", "rev-parse", "--verify", branch], - cwd=self.repo_path, - capture_output=True, - check=False, - ) - - if result.returncode == 0: - # Branch exists - clone it - subprocess.run( - [ - "git", - "clone", - "--branch", - branch, - "--depth", - "1", - str(self.repo_path), - str(target_dir), - ], - capture_output=True, - check=True, - timeout=60, - ) - else: - # Branch doesn't exist - create new - subprocess.run( - ["git", "clone", "--depth", "1", str(self.repo_path), str(target_dir)], - capture_output=True, - check=True, - timeout=60, - ) - subprocess.run( - ["git", "checkout", "--orphan", branch], - cwd=target_dir, - capture_output=True, - check=True, - ) - # Remove all files from new orphan branch - subprocess.run( - ["git", "rm", "-rf", "."], - cwd=target_dir, - capture_output=True, - check=False, - ) - - def _commit_and_push(self, work_dir: Path, branch: str, commit_sha: str): - """Commit changes and push to remote.""" - # Configure git - subprocess.run( - ["git", "config", "user.name", "umpyre-bot"], - cwd=work_dir, - capture_output=True, - check=True, - ) - subprocess.run( - ["git", "config", "user.email", "umpyre@automated"], - cwd=work_dir, - capture_output=True, - check=True, - ) - - # Add all files - subprocess.run( - ["git", "add", "."], - cwd=work_dir, - capture_output=True, - check=True, - ) - - # Check if there are changes - result = subprocess.run( - ["git", "diff", "--cached", "--quiet"], - cwd=work_dir, - capture_output=True, - check=False, - ) - - if result.returncode != 0: - # There are changes - commit them - commit_message = f"Update metrics for {commit_sha[:7]}" - subprocess.run( - ["git", "commit", "-m", commit_message], - cwd=work_dir, - capture_output=True, - check=True, - ) - - # Push with retries - for attempt in range(3): - try: - subprocess.run( - ["git", "push", self.remote, branch], - cwd=work_dir, - capture_output=True, - check=True, - timeout=60, - ) - break - except subprocess.CalledProcessError: - if attempt < 2: - # Pull and try again - subprocess.run( - ["git", "pull", "--rebase", self.remote, branch], - cwd=work_dir, - capture_output=True, - check=False, - ) - else: - raise - - -class GitBranchStorageError(Exception): - """Raised when git branch storage operations fail.""" diff --git a/umpyre/storage/query_utils.py b/umpyre/storage/query_utils.py deleted file mode 100644 index a27a436..0000000 --- a/umpyre/storage/query_utils.py +++ /dev/null @@ -1,195 +0,0 @@ -"""Helper utilities for working with umpyre metrics storage.""" - -import re -from datetime import datetime -from pathlib import Path -from typing import Optional - - -def parse_metric_filename(filename: str) -> dict: - """ - Parse metric filename into components. - - Args: - filename: Filename like "2025_11_14_22_45_00__700e012__0.1.0.json" - - Returns: - Dictionary with parsed components: - { - "timestamp": datetime object, - "timestamp_str": "2025-11-14T22:45:00", - "commit_sha": "700e012", - "pypi_version": "0.1.0" or None, - "has_version": bool, - } - - Example: - >>> info = parse_metric_filename("2025_11_14_22_45_00__700e012__0.1.0.json") - >>> info['commit_sha'] - '700e012' - >>> info['pypi_version'] - '0.1.0' - """ - pattern = r'(\d{4}_\d{2}_\d{2}_\d{2}_\d{2}_\d{2})__(\w{7})__(.+)\.json' - match = re.match(pattern, filename) - - if not match: - raise ValueError(f"Invalid filename format: {filename}") - - timestamp_str, sha, version = match.groups() - - # Parse timestamp - dt = datetime.strptime(timestamp_str, "%Y_%m_%d_%H_%M_%S") - timestamp_iso = dt.isoformat() - - # Parse version (None if 'none') - pypi_version = None if version == "none" else version - - return { - "timestamp": dt, - "timestamp_str": timestamp_iso, - "commit_sha": sha, - "pypi_version": pypi_version, - "has_version": pypi_version is not None, - } - - -def find_metrics_by_commit(history_dir: Path, commit_sha: str) -> Optional[Path]: - """ - Find metrics file for a specific commit. - - Args: - history_dir: Path to history directory - commit_sha: Commit SHA (7 chars or full) - - Returns: - Path to metrics file or None - - Example: - >>> metrics = find_metrics_by_commit(Path("history"), "700e012") # doctest: +SKIP - >>> print(metrics.name) # doctest: +SKIP - 2025_11_14_22_45_00__700e012__0.1.0.json - """ - short_sha = commit_sha[:7] - matches = list(history_dir.glob(f"*__{short_sha}__*.json")) - return matches[0] if matches else None - - -def find_metrics_by_version(history_dir: Path, version: str) -> list[Path]: - """ - Find all metrics files for a specific PyPI version. - - Args: - history_dir: Path to history directory - version: PyPI version string (e.g., "0.1.0") - - Returns: - List of paths (sorted by timestamp, latest first) - - Example: - >>> metrics = find_metrics_by_version(Path("history"), "0.1.0") # doctest: +SKIP - >>> for m in metrics: # doctest: +SKIP - ... print(m.name) - 2025_11_14_22_50_00__abc1234__0.1.0.json - 2025_11_14_22_45_00__700e012__0.1.0.json - """ - matches = list(history_dir.glob(f"*__{version}.json")) - # Sort by filename (chronological order), reverse for latest first - return sorted(matches, reverse=True) - - -def get_latest_metric_for_version(history_dir: Path, version: str) -> Optional[Path]: - """ - Get the most recent metrics file for a version. - - Args: - history_dir: Path to history directory - version: PyPI version string - - Returns: - Path to latest metrics file or None - - Example: - >>> latest = get_latest_metric_for_version(Path("history"), "0.1.0") # doctest: +SKIP - >>> info = parse_metric_filename(latest.name) # doctest: +SKIP - >>> print(f"Latest 0.1.0 metrics from {info['timestamp_str']}") # doctest: +SKIP - """ - matches = find_metrics_by_version(history_dir, version) - return matches[0] if matches else None - - -def get_all_versions(history_dir: Path) -> list[str]: - """ - Get all unique PyPI versions found in metrics. - - Args: - history_dir: Path to history directory - - Returns: - Sorted list of versions (excluding 'none') - - Example: - >>> versions = get_all_versions(Path("history")) # doctest: +SKIP - >>> print(versions) # doctest: +SKIP - ['0.1.0', '0.1.1', '0.2.0'] - """ - versions = set() - - for filepath in history_dir.glob("*.json"): - try: - info = parse_metric_filename(filepath.name) - if info['has_version']: - versions.add(info['pypi_version']) - except ValueError: - continue - - # Sort semantically (simple lexicographic works for most cases) - return sorted(versions) - - -def filter_by_date_range( - history_dir: Path, - start_date: Optional[datetime] = None, - end_date: Optional[datetime] = None, -) -> list[Path]: - """ - Get metrics files within a date range. - - Args: - history_dir: Path to history directory - start_date: Start datetime (inclusive) - end_date: End datetime (inclusive) - - Returns: - List of paths within date range (sorted chronologically) - - Example: - >>> from datetime import datetime, timedelta # doctest: +SKIP - >>> end = datetime.now() # doctest: +SKIP - >>> start = end - timedelta(days=7) # doctest: +SKIP - >>> recent = filter_by_date_range(Path("history"), start, end) # doctest: +SKIP - >>> print(f"Found {len(recent)} metrics from last 7 days") # doctest: +SKIP - """ - matches = [] - - for filepath in history_dir.glob("*.json"): - try: - info = parse_metric_filename(filepath.name) - timestamp = info['timestamp'] - - if start_date and timestamp < start_date: - continue - if end_date and timestamp > end_date: - continue - - matches.append(filepath) - except ValueError: - continue - - return sorted(matches) - - -if __name__ == "__main__": - import doctest - - doctest.testmod() From eff3455ee21d4e2c78a353961163ed143f6859f7 Mon Sep 17 00:00:00 2001 From: Thor Whalen <1906276+thorwhalen@users.noreply.github.com> Date: Sat, 25 Jul 2026 00:40:18 +0200 Subject: [PATCH 2/2] =?UTF-8?q?chore:=20wads=20health=20pass=20=E2=80=94?= =?UTF-8?q?=20SPDX=20license,=20classifiers,=20editorconfig,=20testpaths?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit - license table [project.license]{text=Apache-2.0} -> SPDX string license = "Apache-2.0" - add trove classifiers + keywords (odbc/pyodbc/sql-server/dol/...) - drop stray pandas dependency (was added only for the removed umpyre tests; odbcdol itself does not use pandas) - testpaths ["tests"] (the removed umpyre root dir) -> ["odbcdol"] so pytest + --doctest-modules collect odbcdol's own tests and module doctests - skip-guard the SQLServerPersister smoke test so it skips (instead of erroring) when no live SQL Server is reachable (CI / dev without a database) - add .editorconfig (wads template) --- .editorconfig | 17 +++++++++++++++++ odbcdol/tests/test_simple.py | 18 +++++++++++++++++- pyproject.toml | 19 +++++++++++++------ 3 files changed, 47 insertions(+), 7 deletions(-) create mode 100644 .editorconfig diff --git a/.editorconfig b/.editorconfig new file mode 100644 index 0000000..88bf4d0 --- /dev/null +++ b/.editorconfig @@ -0,0 +1,17 @@ +root = true + +[*] +charset = utf-8 +end_of_line = lf +insert_final_newline = true +trim_trailing_whitespace = true + +[*.{py,toml,yml,yaml}] +indent_style = space +indent_size = 4 + +[*.md] +trim_trailing_whitespace = false + +[Makefile] +indent_style = tab diff --git a/odbcdol/tests/test_simple.py b/odbcdol/tests/test_simple.py index 2f02572..ba12074 100644 --- a/odbcdol/tests/test_simple.py +++ b/odbcdol/tests/test_simple.py @@ -1,8 +1,24 @@ +"""Smoke test for ``SQLServerPersister``. + +Requires a live SQL Server reachable with the default connection parameters; +when none is available (e.g. CI without a database service, or a dev machine), +the test skips instead of failing. +""" + +import pytest + from odbcdol import SQLServerPersister +def _persister_or_skip(): + try: + return SQLServerPersister() + except Exception as e: # no live SQL Server available + pytest.skip(f"live SQL Server not available: {e}") + + def test_sqlserver_persister(): - sql_server_persister = SQLServerPersister() + sql_server_persister = _persister_or_skip() print("Fetching a Record") print(sql_server_persister[1]) diff --git a/pyproject.toml b/pyproject.toml index 5380b40..f386c22 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -8,12 +8,19 @@ version = "0.0.3" description = "odbc (through pyodbc) with a simple (dict-like or list-like) interface" readme = "README.md" requires-python = ">=3.10" -keywords = [] +keywords = ["odbc", "pyodbc", "sql-server", "database", "dol", "storage", "key-value"] authors = [] -dependencies = ["dol", "pyodbc", "pandas"] - -[project.license] -text = "Apache-2.0" +license = "Apache-2.0" +dependencies = ["dol", "pyodbc"] +classifiers = [ + "Development Status :: 3 - Alpha", + "Intended Audience :: Developers", + "Programming Language :: Python :: 3", + "Programming Language :: Python :: 3.10", + "Programming Language :: Python :: 3.12", + "Topic :: Software Development :: Libraries :: Python Modules", + "Topic :: Database", +] [project.urls] Homepage = "https://github.com/i2mint/odbcdol" @@ -50,7 +57,7 @@ convention = "google" [tool.pytest.ini_options] minversion = "6.0" -testpaths = ["tests"] +testpaths = ["odbcdol"] doctest_optionflags = ["NORMALIZE_WHITESPACE", "ELLIPSIS"] # ============================================================================