From db0db678879abd5cb4a33a481a5c38799b3f8738 Mon Sep 17 00:00:00 2001 From: vahid-ahmadi Date: Wed, 16 Sep 2026 13:13:36 +0100 Subject: [PATCH 1/2] Fix the Lint job, which is failing on main The Lint job installs ruff>=0.9.0 with no upper bound. ruff 0.16.7 formats Python inside markdown code blocks, which earlier versions left alone, so five documentation files under docs/ became unformatted without anyone changing them. make check-format fails on an untouched checkout of main, and therefore on every open pull request. Reformats the five files and gives the constraint an upper bound, so a future ruff release changes the lint result only when someone chooses to move the pin. --- .github/workflows/pr_code_changes.yaml | 2 +- changelog.d/lint-on-main.fixed.md | 1 + docs/imputation-benchmarking/cross-validation.md | 12 ++++++------ docs/imputation-benchmarking/preprocessing.md | 8 ++++---- docs/imputation-benchmarking/visualizations.md | 8 ++------ docs/models/imputer/implement-new-model.md | 4 +--- docs/use_cases/index.md | 12 ++++++------ 7 files changed, 21 insertions(+), 26 deletions(-) create mode 100644 changelog.d/lint-on-main.fixed.md diff --git a/.github/workflows/pr_code_changes.yaml b/.github/workflows/pr_code_changes.yaml index 592dbe51..3afaf568 100644 --- a/.github/workflows/pr_code_changes.yaml +++ b/.github/workflows/pr_code_changes.yaml @@ -16,7 +16,7 @@ jobs: uses: astral-sh/setup-uv@v8.1.0 - name: Install relevant dependencies run: | - uv pip install "ruff>=0.9.0" --system + uv pip install "ruff>=0.9.0,<0.17.0" --system - name: Check code formatting run: make check-format diff --git a/changelog.d/lint-on-main.fixed.md b/changelog.d/lint-on-main.fixed.md new file mode 100644 index 00000000..592ead5c --- /dev/null +++ b/changelog.d/lint-on-main.fixed.md @@ -0,0 +1 @@ +Reformats five documentation files so the Lint job passes again, and bounds the ruff version the job installs. diff --git a/docs/imputation-benchmarking/cross-validation.md b/docs/imputation-benchmarking/cross-validation.md index 8da745ff..c1b6f93e 100644 --- a/docs/imputation-benchmarking/cross-validation.md +++ b/docs/imputation-benchmarking/cross-validation.md @@ -41,23 +41,23 @@ Returns a dictionary containing separate results for each metric type: ```python { "quantile_loss": { - "results": pd.DataFrame, # rows: ["train", "test"], cols: quantiles (mean across folds) + "results": pd.DataFrame, # rows: ["train", "test"], cols: quantiles (mean across folds) "results_std": pd.DataFrame, # rows: ["train", "test"], cols: quantiles (std across folds) "mean_train": float, "mean_test": float, "std_train": float, "std_test": float, - "variables": List[str] # numerical variables evaluated + "variables": List[str], # numerical variables evaluated }, "log_loss": { - "results": pd.DataFrame, # rows: ["train", "test"], cols: quantiles + "results": pd.DataFrame, # rows: ["train", "test"], cols: quantiles "results_std": pd.DataFrame, # rows: ["train", "test"], cols: quantiles (std across folds) "mean_train": float, "mean_test": float, "std_train": float, "std_test": float, - "variables": List[str] # categorical variables evaluated - } + "variables": List[str], # categorical variables evaluated + }, } ``` @@ -77,7 +77,7 @@ results = cross_validate_model( data=diabetes_df, predictors=["age", "sex", "bmi", "bp"], imputed_variables=["s1", "s4"], - n_splits=5 + n_splits=5, ) # Check performance for numerical variables diff --git a/docs/imputation-benchmarking/preprocessing.md b/docs/imputation-benchmarking/preprocessing.md index 736d73ed..05693943 100644 --- a/docs/imputation-benchmarking/preprocessing.md +++ b/docs/imputation-benchmarking/preprocessing.md @@ -117,10 +117,10 @@ result = autoimpute( predictors=["age", "education"], imputed_variables=["income", "wealth"], preprocessing={ - "income": "log", # Log transform (positive values only) - "wealth": "asinh", # Asinh transform (handles zeros/negatives) - "age": "normalize" # Z-score normalization - } + "income": "log", # Log transform (positive values only) + "wealth": "asinh", # Asinh transform (handles zeros/negatives) + "age": "normalize", # Z-score normalization + }, ) ``` diff --git a/docs/imputation-benchmarking/visualizations.md b/docs/imputation-benchmarking/visualizations.md index 21dde021..fe8f87b8 100644 --- a/docs/imputation-benchmarking/visualizations.md +++ b/docs/imputation-benchmarking/visualizations.md @@ -83,11 +83,7 @@ comparison_viz = method_comparison_results( ) # Generate plot -fig = comparison_viz.plot( - title="Method comparison", - show_mean=True, - plot_type="bar" -) +fig = comparison_viz.plot(title="Method comparison", show_mean=True, plot_type="bar") fig.show() # Get summary statistics @@ -165,7 +161,7 @@ perf_viz = model_performance_results( results=cv_results, model_name="QRF", method_name="Cross-validation", - metric="quantile_loss" + metric="quantile_loss", ) fig = perf_viz.plot(title="QRF performance") diff --git a/docs/models/imputer/implement-new-model.md b/docs/models/imputer/implement-new-model.md index edf21254..794d6193 100644 --- a/docs/models/imputer/implement-new-model.md +++ b/docs/models/imputer/implement-new-model.md @@ -73,9 +73,7 @@ class NewModelResults(ImputerResults): except Exception as e: self.logger.error(f"Error during Model prediction: {str(e)}") - raise RuntimeError( - f"Failed to predict with Model: {str(e)}" - ) from e + raise RuntimeError(f"Failed to predict with Model: {str(e)}") from e ``` ## Implementing the main model class diff --git a/docs/use_cases/index.md b/docs/use_cases/index.md index 32c7af96..2c9ff07c 100644 --- a/docs/use_cases/index.md +++ b/docs/use_cases/index.md @@ -23,7 +23,7 @@ Before imputation, make sure both datasets have compatible variables. Identify c ```python # Identify common variables -common_variables = ['age', 'income', 'education', 'marital_status', 'region'] +common_variables = ["age", "income", "education", "marital_status", "region"] # Ensure variable formats match (example: education coding) education_mapping = { @@ -31,19 +31,19 @@ education_mapping = { 2: "high_school", 3: "some_college", 4: "bachelor", - 5: "graduate" + 5: "graduate", } # Apply standardization to both datasets for dataset in [scf_data, cps_data]: - dataset['education'] = dataset['education'].map(education_mapping) + dataset["education"] = dataset["education"].map(education_mapping) # Convert income to same units (thousands) - if 'income' in dataset.columns: - dataset['income'] = dataset['income'] / 1000 + if "income" in dataset.columns: + dataset["income"] = dataset["income"] / 1000 # Identify target variable in donor dataset -target_variable = ['networth'] +target_variable = ["networth"] ``` ## Performing imputation From 0bed4c6a2159b14c6d266e213277562c882b269d Mon Sep 17 00:00:00 2001 From: vahid-ahmadi Date: Mon, 21 Sep 2026 10:32:27 +0100 Subject: [PATCH 2/2] Exclude prose from the formatter instead of reformatting it The previous approach reformatted five documentation files and bounded ruff to <0.17.0. Per review, the bound was a no-op against the failure it targeted: 0.16.0 through 0.16.8 all format markdown and all satisfy it, so only the reformatting was doing any work - and the next formatter change inside 0.16.x would reopen the failure. Excludes docs/**/*.md via [tool.ruff] instead, which fixes the cause rather than the symptom, and drops the five reformatted files. That also removes the byte-identical docs hunks that collided with #216, #217 and #219. The dev extra is bounded to match CI, so a contributor running make format no longer reformats files CI then rejects. The cause was ruff 0.16.0, not 0.16.7: 0.13, 0.14 and 0.15 report '65 files already formatted' and do not scan markdown at all. --- .github/workflows/pr_code_changes.yaml | 2 +- changelog.d/lint-on-main.fixed.md | 2 +- docs/imputation-benchmarking/cross-validation.md | 12 ++++++------ docs/imputation-benchmarking/preprocessing.md | 8 ++++---- docs/imputation-benchmarking/visualizations.md | 8 ++++++-- docs/models/imputer/implement-new-model.md | 4 +++- docs/use_cases/index.md | 12 ++++++------ pyproject.toml | 8 +++++++- 8 files changed, 34 insertions(+), 22 deletions(-) diff --git a/.github/workflows/pr_code_changes.yaml b/.github/workflows/pr_code_changes.yaml index 3afaf568..1db90fa2 100644 --- a/.github/workflows/pr_code_changes.yaml +++ b/.github/workflows/pr_code_changes.yaml @@ -16,7 +16,7 @@ jobs: uses: astral-sh/setup-uv@v8.1.0 - name: Install relevant dependencies run: | - uv pip install "ruff>=0.9.0,<0.17.0" --system + uv pip install "ruff>=0.16.0,<0.17.0" --system - name: Check code formatting run: make check-format diff --git a/changelog.d/lint-on-main.fixed.md b/changelog.d/lint-on-main.fixed.md index 592ead5c..19045498 100644 --- a/changelog.d/lint-on-main.fixed.md +++ b/changelog.d/lint-on-main.fixed.md @@ -1 +1 @@ -Reformats five documentation files so the Lint job passes again, and bounds the ruff version the job installs. +The Lint job no longer fails on an untouched checkout: documentation prose is excluded from the formatter, and the ruff version is bounded so a future release cannot silently change what the job checks. diff --git a/docs/imputation-benchmarking/cross-validation.md b/docs/imputation-benchmarking/cross-validation.md index c1b6f93e..8da745ff 100644 --- a/docs/imputation-benchmarking/cross-validation.md +++ b/docs/imputation-benchmarking/cross-validation.md @@ -41,23 +41,23 @@ Returns a dictionary containing separate results for each metric type: ```python { "quantile_loss": { - "results": pd.DataFrame, # rows: ["train", "test"], cols: quantiles (mean across folds) + "results": pd.DataFrame, # rows: ["train", "test"], cols: quantiles (mean across folds) "results_std": pd.DataFrame, # rows: ["train", "test"], cols: quantiles (std across folds) "mean_train": float, "mean_test": float, "std_train": float, "std_test": float, - "variables": List[str], # numerical variables evaluated + "variables": List[str] # numerical variables evaluated }, "log_loss": { - "results": pd.DataFrame, # rows: ["train", "test"], cols: quantiles + "results": pd.DataFrame, # rows: ["train", "test"], cols: quantiles "results_std": pd.DataFrame, # rows: ["train", "test"], cols: quantiles (std across folds) "mean_train": float, "mean_test": float, "std_train": float, "std_test": float, - "variables": List[str], # categorical variables evaluated - }, + "variables": List[str] # categorical variables evaluated + } } ``` @@ -77,7 +77,7 @@ results = cross_validate_model( data=diabetes_df, predictors=["age", "sex", "bmi", "bp"], imputed_variables=["s1", "s4"], - n_splits=5, + n_splits=5 ) # Check performance for numerical variables diff --git a/docs/imputation-benchmarking/preprocessing.md b/docs/imputation-benchmarking/preprocessing.md index 05693943..736d73ed 100644 --- a/docs/imputation-benchmarking/preprocessing.md +++ b/docs/imputation-benchmarking/preprocessing.md @@ -117,10 +117,10 @@ result = autoimpute( predictors=["age", "education"], imputed_variables=["income", "wealth"], preprocessing={ - "income": "log", # Log transform (positive values only) - "wealth": "asinh", # Asinh transform (handles zeros/negatives) - "age": "normalize", # Z-score normalization - }, + "income": "log", # Log transform (positive values only) + "wealth": "asinh", # Asinh transform (handles zeros/negatives) + "age": "normalize" # Z-score normalization + } ) ``` diff --git a/docs/imputation-benchmarking/visualizations.md b/docs/imputation-benchmarking/visualizations.md index fe8f87b8..21dde021 100644 --- a/docs/imputation-benchmarking/visualizations.md +++ b/docs/imputation-benchmarking/visualizations.md @@ -83,7 +83,11 @@ comparison_viz = method_comparison_results( ) # Generate plot -fig = comparison_viz.plot(title="Method comparison", show_mean=True, plot_type="bar") +fig = comparison_viz.plot( + title="Method comparison", + show_mean=True, + plot_type="bar" +) fig.show() # Get summary statistics @@ -161,7 +165,7 @@ perf_viz = model_performance_results( results=cv_results, model_name="QRF", method_name="Cross-validation", - metric="quantile_loss", + metric="quantile_loss" ) fig = perf_viz.plot(title="QRF performance") diff --git a/docs/models/imputer/implement-new-model.md b/docs/models/imputer/implement-new-model.md index 794d6193..edf21254 100644 --- a/docs/models/imputer/implement-new-model.md +++ b/docs/models/imputer/implement-new-model.md @@ -73,7 +73,9 @@ class NewModelResults(ImputerResults): except Exception as e: self.logger.error(f"Error during Model prediction: {str(e)}") - raise RuntimeError(f"Failed to predict with Model: {str(e)}") from e + raise RuntimeError( + f"Failed to predict with Model: {str(e)}" + ) from e ``` ## Implementing the main model class diff --git a/docs/use_cases/index.md b/docs/use_cases/index.md index 2c9ff07c..32c7af96 100644 --- a/docs/use_cases/index.md +++ b/docs/use_cases/index.md @@ -23,7 +23,7 @@ Before imputation, make sure both datasets have compatible variables. Identify c ```python # Identify common variables -common_variables = ["age", "income", "education", "marital_status", "region"] +common_variables = ['age', 'income', 'education', 'marital_status', 'region'] # Ensure variable formats match (example: education coding) education_mapping = { @@ -31,19 +31,19 @@ education_mapping = { 2: "high_school", 3: "some_college", 4: "bachelor", - 5: "graduate", + 5: "graduate" } # Apply standardization to both datasets for dataset in [scf_data, cps_data]: - dataset["education"] = dataset["education"].map(education_mapping) + dataset['education'] = dataset['education'].map(education_mapping) # Convert income to same units (thousands) - if "income" in dataset.columns: - dataset["income"] = dataset["income"] / 1000 + if 'income' in dataset.columns: + dataset['income'] = dataset['income'] / 1000 # Identify target variable in donor dataset -target_variable = ["networth"] +target_variable = ['networth'] ``` ## Performing imputation diff --git a/pyproject.toml b/pyproject.toml index f1134e01..3e220a1d 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -35,7 +35,7 @@ dependencies = [ dev = [ "pytest>=8.0.0,<10.0.0", "pytest-cov>=6.0.0,<8.0.0", - "ruff>=0.9.0", + "ruff>=0.16.0,<0.17.0", "mypy>=1.2.3,<2.0.0", "build>=1.2.0,<2.0.0", "towncrier>=24.8.0", @@ -101,3 +101,9 @@ showcontent = true directory = "removed" name = "Removed" showcontent = true + +[tool.ruff] +# Documentation is prose, not a lint surface. ruff 0.16.0 began formatting +# Python inside markdown code blocks, which reformatted five docs files that +# nobody had edited and turned the Lint job red on an untouched checkout. +extend-exclude = ["docs/**/*.md"]