From d7cc47e3b15a560409504b2df4eb8c1dea71fa11 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Tue, 15 Sep 2026 10:05:38 -0400 Subject: [PATCH 1/3] Move the API bundle to the canonical PolicyEngine 6.0.0 tuple The wrapper's 6.0.0 release binds policyengine-core 3.32.5, policyengine-us 2.2.1, policyengine-uk 2.90.2 and spm-calculator 1.0.0, and its manifest certifies the SPM measurement with scenario ce_trend, county geography, county vintage 2020 and forecast content sha256 3d86d5c4. Both pip build paths move together: Cloud Run and the published GHCR image install with pip and do not read uv.lock, so the separate spm-calculator pin has to move with the wrapper or the image resolves a calculator the country model cannot use. The lock was refreshed from PyPI only; the resolved wheels are ac51a637 (policyengine), 0993a6c7 (policyengine-us) and e354937a (spm-calculator, not the later 1.0.0.post1 that country 2.2.1's range would also admit), and the lock's own project version catches up to pyproject's 3.56.1. The automatic bundle updater cannot make this change. It rewrites only the policyengine[models] pin and then runs uv lock, and 6.0.0's models extra pins spm-calculator 1.0.0 against this project's standalone 0.3.1, which is unsatisfiable. Its dispatched run for 6.0.0 failed on exactly that resolution and opened no PR. The housing-cap tests exercised the SPM geography requirement with households that receive no housing assistance, where the cap is not the reason the measurement is needed. They now request a controlled positive award where the requirement is the point, and assert the unassisted case succeeds with a zero housing resource, no measurement-year receipts and no geography. Ordinary household_benefits and household_net_income count the actual award rather than the capped SPM resource, so they need no SPM geography at all and do not move with a valid selection; the contract document says so. These tests skip on a legacy install, so they were run against an installed canonical bundle rather than relied on in CI. Co-Authored-By: Claude Fable 5.1 --- changelog.d/canonical-bundle-6-0-0.changed.md | 1 + docker/Dockerfile | 2 +- docs/canonical-spm.md | 24 +++- pyproject.toml | 4 +- tests/unit/test_bundle_update_pins.py | 24 ++-- tests/unit/test_cloud_run_deploy_scripts.py | 6 +- tests/unit/test_country_spm.py | 125 ++++++++++++++++-- uv.lock | 31 ++--- 8 files changed, 170 insertions(+), 47 deletions(-) create mode 100644 changelog.d/canonical-bundle-6-0-0.changed.md diff --git a/changelog.d/canonical-bundle-6-0-0.changed.md b/changelog.d/canonical-bundle-6-0-0.changed.md new file mode 100644 index 000000000..a9f849690 --- /dev/null +++ b/changelog.d/canonical-bundle-6-0-0.changed.md @@ -0,0 +1 @@ +Move the API bundle to PolicyEngine 6.0.0, which binds policyengine-core 3.32.5, policyengine-us 2.2.1, policyengine-uk 2.90.2 and spm-calculator 1.0.0, and raise the separate spm-calculator pin that both pip build paths install to 1.0.0 so the calculator stays compatible with the country model. Qualify the SPM housing cap against households that actually receive housing assistance, and record that ordinary household income counts the real award rather than the capped SPM resource. diff --git a/docker/Dockerfile b/docker/Dockerfile index 7b08cd132..baec3db82 100644 --- a/docker/Dockerfile +++ b/docker/Dockerfile @@ -1,4 +1,4 @@ FROM python:3.12 # Match the API bundle in pyproject.toml; the bundle updater changes both pins. # Exact bundled model requirements prevent pip from silently backtracking. -RUN pip install "policyengine[models]==5.2.0" spm-calculator==0.3.1 ipython +RUN pip install "policyengine[models]==6.0.0" spm-calculator==1.0.0 ipython diff --git a/docs/canonical-spm.md b/docs/canonical-spm.md index 0e36bee8b..ac51f5329 100644 --- a/docs/canonical-spm.md +++ b/docs/canonical-spm.md @@ -63,6 +63,20 @@ or read its provenance. SPM-dependent requests validate the required primitives when the country calculates them. `/calculate-full` and stored household replay request the full output set, which includes SPM dependencies. +Ordinary `household_benefits` and `household_net_income` count actual +`housing_assistance`; they do not use the SPM housing cap. A request for only +these household outputs can include a positive housing award without supplying +SPM geography. Changing a valid SPM selection does not change those outputs or +create measurement-year receipts. Other programs still use their own required +geographic inputs. + +`spm_unit_net_income` is the SPM resource aggregate and retains +`spm_unit_capped_housing_subsidy`. The country allocates household housing awards +to SPM units before applying this cap. Units receiving a positive allocation +need the SPM geography and composition used by the cap. A zero-allocation unit +returns a zero housing resource without evaluating the SPM measurement; this +does not exempt an explicit threshold or poverty calculation from its inputs. + ### Choosing a measurement, or not A measurement is chosen by a request that sends `spm`, or by the household whose @@ -118,8 +132,11 @@ inherit the certified defaults. On any other country the same routes reject an `spm` key at all, null included, with `SPM_SETTINGS_UNSUPPORTED`, as `POST /{country}/simulation` already did: there is no shape of it to correct. -Tax-only calculations can use periods outside the artifact's measurement years, -including a valid metro selection, without generating SPM receipts. When an SPM +Tax-only and ordinary household-income calculations can use periods outside the +artifact's measurement years, when their own formulas support those periods, +including a valid metro selection, without evaluating SPM amounts or adding +measurement-year receipts. This also holds for ordinary income with a positive +housing award. When an SPM dependency actually executes for an unsupported year, a request that chose the measurement returns a structured failure with the calculator's typed `SPM_YEAR_UNAVAILABLE` code; a request that chose nothing leaves that year's @@ -163,7 +180,8 @@ receipt beside its null cells. Read the values to learn which of them a measurement produced. Provenance comes from the actual simulation and includes artifact, scenario, years, geography, composition/storage methods and runtime versions. It remains in JSON form through stored replay and cache hits. A tax-only receipt -may have empty `years` and `geographies` because no SPM measurement was requested. +may have empty `years` and `geographies` because no SPM measurement was requested; +the same is true for an ordinary household-income-only calculation. Clients saving simulation outputs must retain this full response envelope. `POST /us/simulation` takes `population_id`, `population_type` (`household` or diff --git a/pyproject.toml b/pyproject.toml index b1f527291..bed6230fe 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -42,10 +42,10 @@ dependencies = [ "policyengine_canada==0.96.3", "policyengine-ng==0.5.1", "policyengine-il==0.1.0", - "policyengine[models]==5.2.0", + "policyengine[models]==6.0.0", # Cloud Run installs with pip, which does not read uv.lock. Keep the # calculator compatible with the country model in this bundle. - "spm-calculator==0.3.1", + "spm-calculator==1.0.0", "pydantic", "pymysql", "python-dotenv", diff --git a/tests/unit/test_bundle_update_pins.py b/tests/unit/test_bundle_update_pins.py index c13abb99b..460617024 100644 --- a/tests/unit/test_bundle_update_pins.py +++ b/tests/unit/test_bundle_update_pins.py @@ -16,12 +16,12 @@ def update_checkout(tmp_path): (tmp_path / "docker").mkdir() (tmp_path / "changelog.d").mkdir() (tmp_path / "pyproject.toml").write_text( - '[project]\ndependencies = ["policyengine[models]==5.2.0", ' - '"spm-calculator==0.3.1"]\n' + '[project]\ndependencies = ["policyengine[models]==6.0.0", ' + '"spm-calculator==1.0.0"]\n' ) (tmp_path / "docker/Dockerfile").write_text( - 'FROM python:3.12\nRUN pip install "policyengine[models]==5.2.0" ' - "spm-calculator==0.3.1 ipython\n" + 'FROM python:3.12\nRUN pip install "policyengine[models]==6.0.0" ' + "spm-calculator==1.0.0 ipython\n" ) (tmp_path / "uv.lock").write_text("# registry resolution is stubbed\n") binaries = tmp_path / "bin" @@ -39,7 +39,7 @@ def update_checkout(tmp_path): if name == "git" and args[:1] == ["ls-remote"]: sys.exit(2) if name == "uv" and args[:1] == ["run"]: - print("POLICYENGINE_VERSION=5.3.0") + print("POLICYENGINE_VERSION=6.1.0") """ for name in ("git", "gh", "uv"): path = binaries / name @@ -56,7 +56,7 @@ def run_update(root): env={ **os.environ, "PATH": f"{root / 'bin'}{os.pathsep}{os.environ['PATH']}", - "LATEST_OVERRIDE": "5.3.0", + "LATEST_OVERRIDE": "6.1.0", "BUNDLE_TEST_CALLS": str(log), }, capture_output=True, @@ -72,24 +72,24 @@ def test_automatic_update_changes_and_stages_both_image_pins(update_checkout): assert result.returncode == 0, result.stderr for name in ("pyproject.toml", "docker/Dockerfile"): text = (update_checkout / name).read_text() - assert "policyengine[models]==5.3.0" in text - assert "policyengine[models]==5.2.0" not in text - assert "spm-calculator==0.3.1" in text + assert "policyengine[models]==6.1.0" in text + assert "policyengine[models]==6.0.0" not in text + assert "spm-calculator==1.0.0" in text assert [ "git", "add", "pyproject.toml", "docker/Dockerfile", "uv.lock", - "changelog.d/update-policyengine-bundle-5.3.0.changed.md", + "changelog.d/update-policyengine-bundle-6.1.0.changed.md", ] in calls assert ["uv", "lock", "--upgrade-package", "policyengine"] in calls -@pytest.mark.parametrize("pin", ["5.1.0", "5.2.01", "5.2.0rc1"]) +@pytest.mark.parametrize("pin", ["5.9.0", "6.0.01", "6.0.0rc1"]) def test_mismatched_image_pin_fails_before_file_changes(update_checkout, pin): image = update_checkout / "docker/Dockerfile" - image.write_text(image.read_text().replace("5.2.0", pin)) + image.write_text(image.read_text().replace("6.0.0", pin)) before = { name: (update_checkout / name).read_bytes() for name in ("pyproject.toml", "docker/Dockerfile") diff --git a/tests/unit/test_cloud_run_deploy_scripts.py b/tests/unit/test_cloud_run_deploy_scripts.py index 479567428..caf92c30c 100644 --- a/tests/unit/test_cloud_run_deploy_scripts.py +++ b/tests/unit/test_cloud_run_deploy_scripts.py @@ -518,15 +518,15 @@ def test_cloud_run_dockerfile_runs_startup_with_bash(): assert 'CMD ["/bin/sh", "/app/start.sh"]' not in dockerfile -def test_active_images_pin_spm_calculator_for_the_legacy_country_bundle(): +def test_active_images_pin_spm_calculator_for_the_canonical_country_bundle(): """Both pip build paths must protect the current bundle's SPM behavior.""" import tomllib project = tomllib.loads((REPO / "pyproject.toml").read_text()) - assert "spm-calculator==0.3.1" in project["project"]["dependencies"] + assert "spm-calculator==1.0.0" in project["project"]["dependencies"] # The independently published GHCR image does not install this project. generic_image = (REPO / "docker/Dockerfile").read_text() - assert "spm-calculator==0.3.1" in generic_image + assert "spm-calculator==1.0.0" in generic_image bundle_pin = next( requirement for requirement in project["project"]["dependencies"] diff --git a/tests/unit/test_country_spm.py b/tests/unit/test_country_spm.py index a36d9f8a0..9d9f96dcb 100644 --- a/tests/unit/test_country_spm.py +++ b/tests/unit/test_country_spm.py @@ -18,6 +18,11 @@ "age", "employment_income", "state_code", + "housing_assistance", + "hud_ttp", + "spm_unit_tenure_type", + "household_benefits", + "household_net_income", "spm_unit_federal_tax", "spm_unit_net_income", "spm_unit_spm_threshold", @@ -25,12 +30,27 @@ AXIS_POINTS = 2 -def requested_household(variable="spm_unit_spm_threshold", *, axes=False): +def requested_household( + variable="spm_unit_spm_threshold", *, axes=False, housing_assistance=None +): household = { "people": {"you": {"age": {"2024": 40}}}, "households": {"household": {"members": ["you"], "state_code": {"2024": "CA"}}}, "spm_units": {"spm_unit": {"members": ["you"], variable: {"2024": None}}}, } + if variable in {"household_benefits", "household_net_income"}: + requested = household["spm_units"]["spm_unit"].pop(variable) + household["households"]["household"][variable] = requested + if housing_assistance is not None: + # A controlled program award tests the resource boundary independently + # of HUD eligibility, take-up or an arbitrary household's zero award. + household["spm_units"]["spm_unit"].update( + { + "housing_assistance": {"2024": housing_assistance}, + "hud_ttp": {"2024": 0}, + "spm_unit_tenure_type": {"2024": "RENTER"}, + } + ) if axes: household["axes"] = [ [ @@ -282,7 +302,9 @@ def test_real_country_state_only_spm_dependency_requires_geography( """A household that chose this measurement is told what it is missing.""" with pytest.raises(ValueError) as caught: real_canonical_country.calculate( - requested_household(variable), None, spm_requested=True + requested_household(variable, housing_assistance=12_000), + None, + spm_requested=True, ) assert spm.spm_error_detail(caught.value)["code"] == "SPM_GEOGRAPHY_REQUIRED" @@ -305,18 +327,66 @@ def test_real_country_state_only_dependency_is_null_when_nothing_was_chosen( inherited default was not a choice, so the dependant is unavailable rather than the request rejected. """ - result = real_canonical_country.calculate(requested_household(variable), None) + result = real_canonical_country.calculate( + requested_household(variable, housing_assistance=12_000), None + ) assert result.household["spm_units"]["spm_unit"][variable]["2024"] is None assert result.spm_config["geography_kind"] == "county" -@pytest.mark.parametrize("county", ["99999", "malformed"]) -def test_real_country_unknown_county_is_structured(real_canonical_country, county): +@pytest.mark.parametrize("variable", ["household_benefits", "household_net_income"]) +@pytest.mark.parametrize("axes", [False, True]) +@pytest.mark.parametrize("chosen", [False, True]) +def test_real_country_assisted_ordinary_income_needs_no_spm_geography( + real_canonical_country, variable, axes, chosen +): + result = real_canonical_country.calculate( + requested_household(variable, axes=axes, housing_assistance=12_000), + None, + spm={"geography_kind": "county"} if chosen else None, + spm_requested=chosen, + ) + values = result.household["households"]["household"][variable]["2024"] + assert np.asarray(values).shape == ((AXIS_POINTS,) if axes else ()) + assert np.isfinite(values).all() + assert result.spm_provenance["years"] == {} + assert result.spm_provenance["geographies"] == [] + + +@pytest.mark.parametrize( + "variable", ["spm_unit_capped_housing_subsidy", "spm_unit_net_income"] +) +@pytest.mark.parametrize("axes", [False, True]) +def test_real_country_zero_award_resource_needs_no_spm_geography( + real_canonical_country, variable, axes +): + result = real_canonical_country.calculate( + requested_household(variable, axes=axes, housing_assistance=0), + None, + spm={"geography_kind": "county"}, + spm_requested=True, + ) + values = result.household["spm_units"]["spm_unit"][variable]["2024"] + assert np.asarray(values).shape == ((AXIS_POINTS,) if axes else ()) + assert np.isfinite(values).all() + if variable == "spm_unit_capped_housing_subsidy": + np.testing.assert_array_equal(values, [0] * AXIS_POINTS if axes else 0) + assert result.spm_provenance["years"] == {} + assert result.spm_provenance["geographies"] == [] + + +@pytest.mark.parametrize( + "county,code", + [("99999", "SPM_GEOGRAPHY_UNAVAILABLE"), ("malformed", "SPM_GEOGRAPHY_REQUIRED")], +) +def test_real_country_unknown_county_is_structured( + real_canonical_country, county, code +): household = requested_household() household["households"]["household"]["county_fips"] = {"2024": county} with pytest.raises(ValueError) as caught: real_canonical_country.calculate(household, None, spm_requested=True) - assert spm.spm_error_detail(caught.value)["code"] == "SPM_GEOGRAPHY_UNAVAILABLE" + assert spm.spm_error_detail(caught.value)["code"] == code def test_real_country_unknown_area_is_structured(real_canonical_country): @@ -554,12 +624,14 @@ def observed_calculate(*args, **kwargs): assert "2024" in response.json["spm_provenance"]["years"] -def household_in_year(variable, year, *, axes=False): +def household_in_year(variable, year, *, axes=False, housing_assistance=None): """Move every input and requested output to the regression's annual period.""" return json.loads( - json.dumps(requested_household(variable, axes=axes)).replace( - '"2024"', f'"{year}"' - ) + json.dumps( + requested_household( + variable, axes=axes, housing_assistance=housing_assistance + ) + ).replace('"2024"', f'"{year}"') ) @@ -604,7 +676,9 @@ def test_real_http_unsupported_spm_year_is_structured_only_when_calculated( response = real_http_client.post( "/us/calculate", json={ - "household": household_in_year(variable, 2036, axes=axes), + "household": household_in_year( + variable, 2036, axes=axes, housing_assistance=12_000 + ), "spm": selection_for_geography(geography), }, ) @@ -615,6 +689,35 @@ def test_real_http_unsupported_spm_year_is_structured_only_when_calculated( assert "2036" in response.json["errors"][0]["message"] +@pytest.mark.parametrize("variable", ["household_benefits", "household_net_income"]) +@pytest.mark.parametrize("year", [2024, 2036]) +@pytest.mark.parametrize("axes", [False, True]) +def test_real_http_assisted_ordinary_income_is_independent_of_spm_selection( + real_http_client, variable, year, axes +): + values = [] + for geography in ("county", "national", "metro"): + response = real_http_client.post( + "/us/calculate", + json={ + "household": household_in_year( + variable, year, axes=axes, housing_assistance=12_000 + ), + "spm": selection_for_geography(geography), + }, + ) + assert response.status_code == 200, response.json + assert response.json["status"] == "ok" + value = response.json["result"]["households"]["household"][variable][str(year)] + assert np.asarray(value).shape == ((AXIS_POINTS,) if axes else ()) + assert np.isfinite(value).all() + assert response.json["spm_provenance"]["years"] == {} + assert response.json["spm_provenance"]["geographies"] == [] + values.append(value) + for value in values[1:]: + np.testing.assert_array_equal(value, values[0]) + + def test_uncertifiable_country_receipt_does_not_become_an_internal_failure(monkeypatch): """A receipt shape this API cannot read is a typed 400, not a 500. diff --git a/uv.lock b/uv.lock index 3b8ba1f26..0a32e2153 100644 --- a/uv.lock +++ b/uv.lock @@ -2687,7 +2687,7 @@ wheels = [ [[package]] name = "policyengine" -version = "5.2.0" +version = "6.0.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "h5py" }, @@ -2699,9 +2699,9 @@ dependencies = [ { name = "pydantic" }, { name = "requests" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/45/2d/5b09f414e26573748f9ae2e448d6f9f94196b7733e8400d727db9ed17868/policyengine-5.2.0.tar.gz", hash = "sha256:c1e7a3a8d7fb23401aedbe8a9fa2d6ed69e24ae9983b6a4a8a0d5112394b0bc4", size = 732330, upload-time = "2026-08-29T17:24:52.814Z" } +sdist = { url = "https://files.pythonhosted.org/packages/60/5d/6de7fd3c795963f1f061d54d2f4222679a17432ce935728888997b4f4730/policyengine-6.0.0.tar.gz", hash = "sha256:3f3b83cf5fb184e9789e0d5469816f51ed7892d065c9afd768a8ed3731f8f596", size = 743038, upload-time = "2026-09-15T13:37:28.983Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/f3/c5/677144df3800ae41dd587b292bd633c09edd5c998695ca15987c05730bc6/policyengine-5.2.0-py3-none-any.whl", hash = "sha256:e73307f787cf366dc16c080b708c39c28a58b0adf34abe51de5ecd97ecfab154", size = 226102, upload-time = "2026-08-29T17:24:51.377Z" }, + { url = "https://files.pythonhosted.org/packages/d3/2f/b9d6b0058287817513ae19a7d0fca0db7f54cf6aaefeb080fc99a8890a9c/policyengine-6.0.0-py3-none-any.whl", hash = "sha256:ac51a637881744703939d661c94d843aa773e363bf2ba1e418b26592fa210aec", size = 244487, upload-time = "2026-09-15T13:37:27.631Z" }, ] [package.optional-dependencies] @@ -2709,11 +2709,12 @@ models = [ { name = "policyengine-core" }, { name = "policyengine-uk" }, { name = "policyengine-us" }, + { name = "spm-calculator" }, ] [[package]] name = "policyengine-api" -version = "3.54.2" +version = "3.56.1" source = { editable = "." } dependencies = [ { name = "a2wsgi" }, @@ -2787,7 +2788,7 @@ requires-dist = [ { name = "mypy", marker = "extra == 'dev'", specifier = ">=1.15,<2" }, { name = "openai" }, { name = "packaging", specifier = ">=24,<27" }, - { name = "policyengine", extras = ["models"], specifier = "==5.2.0" }, + { name = "policyengine", extras = ["models"], specifier = "==6.0.0" }, { name = "policyengine-canada", specifier = "==0.96.3" }, { name = "policyengine-il", specifier = "==0.1.0" }, { name = "policyengine-ng", specifier = "==0.5.1" }, @@ -2801,7 +2802,7 @@ requires-dist = [ { name = "redis" }, { name = "rq" }, { name = "ruff", marker = "extra == 'dev'", specifier = ">=0.9.0" }, - { name = "spm-calculator", specifier = "==0.3.1" }, + { name = "spm-calculator", specifier = "==1.0.0" }, { name = "sqlalchemy", specifier = ">=2,<3" }, { name = "sqlmodel", specifier = ">=0.0.39,<0.1" }, { name = "streamlit" }, @@ -2841,7 +2842,7 @@ wheels = [ [[package]] name = "policyengine-core" -version = "3.30.1" +version = "3.32.5" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "dpath" }, @@ -2861,9 +2862,9 @@ dependencies = [ { name = "standard-imghdr" }, { name = "wheel" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/a1/9d/f71da3e348ddc52e058c76b72bbc4b2f20abcf0e7389ecd5aca5d550fd14/policyengine_core-3.30.1.tar.gz", hash = "sha256:a16f29fe51ec01a7be2171386f8bd26b9ba55ffc338adb70adcadafe33d9c55f", size = 500419, upload-time = "2026-07-19T15:50:41.288Z" } +sdist = { url = "https://files.pythonhosted.org/packages/12/54/cfd2138584de9cddaa4459bba7fdc381101ca8b8ffd38356a1b256850d5f/policyengine_core-3.32.5.tar.gz", hash = "sha256:f051c269b538fe77ee7868f7cfbc0138551960b1b677a0ad6fdc1a93f1660460", size = 420112, upload-time = "2026-09-10T04:16:48.889Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/b9/7e/07624d8add8889af1b4af796d36c38b2e2af494b1d78a064fab29e9bea00/policyengine_core-3.30.1-py3-none-any.whl", hash = "sha256:2dbcf5f590a0199a7b7c77fcbbda2ff6bc289f6c6169ca0af3beace35d8b3e63", size = 244846, upload-time = "2026-07-19T15:50:39.651Z" }, + { url = "https://files.pythonhosted.org/packages/c6/d9/72c3707465e9fe77e6f7a55ea64b49a287cbf875b78f6e862d945df26975/policyengine_core-3.32.5-py3-none-any.whl", hash = "sha256:9d8c162d5fe5c784ae16c885da3ffbfc39a536add0e111133144d8edd555935f", size = 245925, upload-time = "2026-09-10T04:16:47.153Z" }, ] [[package]] @@ -2909,7 +2910,7 @@ wheels = [ [[package]] name = "policyengine-us" -version = "1.764.6" +version = "2.2.1" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "microdf-python" }, @@ -2919,9 +2920,9 @@ dependencies = [ { name = "tables" }, { name = "tqdm" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/6e/cc/9e65a9586069beab1f02102a23f7a0e6332f1a9c2d590cb9f0e68f9ac987/policyengine_us-1.764.6.tar.gz", hash = "sha256:9817d48abc5b7d690b8abe33047005e5cf40b71654a85a2534bb5e3bcf921f79", size = 11009555, upload-time = "2026-07-06T13:13:07.254Z" } +sdist = { url = "https://files.pythonhosted.org/packages/2c/09/763b21371a62635e15e62b53fb696e299f01460907c27313d40f63031eb2/policyengine_us-2.2.1.tar.gz", hash = "sha256:d5300c91f3eae2e47a61f485fdfde91a892b80e95db3e46ff2c4bbd417d76501", size = 11946510, upload-time = "2026-09-15T04:45:50.33Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/e2/5a/0566666f30ad06e416f0cf25b24bdee73acd71897fb2de3c0ac43b59e3e3/policyengine_us-1.764.6-py3-none-any.whl", hash = "sha256:2b2c5c02c26ab212270948db1f30129728b625fc6109fc9ca0c604b714546909", size = 13117921, upload-time = "2026-07-06T13:13:03.476Z" }, + { url = "https://files.pythonhosted.org/packages/5e/09/86ac3968d2a18e28fce3ba1186fc21728e9f299b96a806d6f204f1e0aaed/policyengine_us-2.2.1-py3-none-any.whl", hash = "sha256:0993a6c73fcdfbe171a796aca9d302741bd8e00ab83388ad81c15a08740318c6", size = 14885537, upload-time = "2026-09-15T04:45:46.649Z" }, ] [[package]] @@ -3840,7 +3841,7 @@ wheels = [ [[package]] name = "spm-calculator" -version = "0.3.1" +version = "1.0.0" source = { registry = "https://pypi.org/simple" } dependencies = [ { name = "census" }, @@ -3850,9 +3851,9 @@ dependencies = [ { name = "requests" }, { name = "us" }, ] -sdist = { url = "https://files.pythonhosted.org/packages/54/3b/b805c7e3e18c5b5c00f61b60112f9690d084c910e2481bc020f35390d8fd/spm_calculator-0.3.1.tar.gz", hash = "sha256:41f2f4d00d8c03422a7d57b800052e7760b88e463a5884802f83ed58d35c18c1", size = 75945, upload-time = "2026-04-17T19:52:39.707Z" } +sdist = { url = "https://files.pythonhosted.org/packages/c2/5e/3f28b212c401795990250bb3daec741f699a8c7dc2a23a3db7005df6feb6/spm_calculator-1.0.0.tar.gz", hash = "sha256:a99aac8c2c0bf81a9455105bbf872366f1ea9066a99cd793714881cd3db846dc", size = 7150825, upload-time = "2026-09-11T15:45:12.732Z" } wheels = [ - { url = "https://files.pythonhosted.org/packages/8e/1b/29f705f8a96fc7f55f2c07dfcddbbae78efdc6f174d25d4a0560fc3f5cf9/spm_calculator-0.3.1-py3-none-any.whl", hash = "sha256:52c57ecc5a240ec941b0f2b0d93bc4fa437ef6250e233baed8e11916fa9c1150", size = 57826, upload-time = "2026-04-17T19:52:38.444Z" }, + { url = "https://files.pythonhosted.org/packages/19/a0/c484f69a0ebf88a9b9fd0ac28176f8df66550f46714f1b0c5ec0b360824e/spm_calculator-1.0.0-py3-none-any.whl", hash = "sha256:e354937a5e1a4045d4966ed594a528d8b02866fabaac9bb5672017004b627305", size = 7395384, upload-time = "2026-09-11T15:45:10.492Z" }, ] [[package]] From eb4f46e3126198a46c9c61a87045c1ac4fda23f6 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Tue, 15 Sep 2026 12:03:42 -0400 Subject: [PATCH 2/3] Re-pin the installed catalog's entity counts to the 6.0.0 bundle tests/integration/test_v2_catalog_installed.py bounds the catalog the v2 seed job publishes by asserting exact entity counts, and those counts are a property of the installed country models. policyengine-us 1.764.6 to 2.2.1 moves four of them: variables 6,649 to 7,046, parameter nodes 27,813 to 29,118, parameters 99,006 to 103,705 and parameter values 1,172,130 to 1,192,826. Models, model versions, datasets and regions are unchanged, as are the dataset names and the US fallback summaries the same test asserts. The new numbers were produced twice from the canonical bundle and agree exactly: by this repository's own Cross-database integration job on the previous commit, and locally against an installed policyengine 6.0.0 environment. Co-Authored-By: Claude Fable 5.1 --- tests/integration/test_v2_catalog_installed.py | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/tests/integration/test_v2_catalog_installed.py b/tests/integration/test_v2_catalog_installed.py index b09ce746a..f897bce6b 100644 --- a/tests/integration/test_v2_catalog_installed.py +++ b/tests/integration/test_v2_catalog_installed.py @@ -39,10 +39,10 @@ def test_installed_policyengine_catalog_is_complete_and_bounded() -> None: assert catalog.entity_counts() == { "models": 2, "model_versions": 2, - "variables": 6_649, - "parameter_nodes": 27_813, - "parameters": 99_006, - "parameter_values": 1_172_130, + "variables": 7_046, + "parameter_nodes": 29_118, + "parameters": 103_705, + "parameter_values": 1_192_826, "datasets": 2, "regions": 826, } From 7d357a29b038831b899e9fdfc006c2a2f10034e7 Mon Sep 17 00:00:00 2001 From: Max Ghenis Date: Tue, 15 Sep 2026 13:04:09 -0400 Subject: [PATCH 3/3] Answer the canonical SPM contract in the unit suite's test doubles MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The canonical PolicyEngine 6.0.0 bundle certifies an SPM measurement, so `normalize_spm_selection` resolves a selection for every US request where a legacy bundle returned None. Three boundaries that were unreachable before then activate on every request: the API requires the selected worker to advertise a matching capability before it submits, it hands the measurement to the country package, and it refuses a worker result or a stored household calculation that carries no receipt for what was measured. Ninety-five unit tests were written while those boundaries were dormant, and their doubles answer none of them: plain MagicMock gateways return an attribute instead of a capability, mocked `/versions` surfaces serve no registry, country doubles accept only the two-argument call, and worker results carry no provenance. On the previous commit the CI-order suite was 95 failed, 2186 passed under the 6.0.0 pins; it is now 2286 passed, 0 failed. The doubles now answer the real contract, held in one place in `tests/fixtures/spm.py`: the worker capability, the gateway `/versions` registry document, the country provenance receipt, the measurement arguments a country package receives, and the request options the service resolves. Every value is read from the installed bundle manifest through `normalize_spm_selection` rather than transcribed, so a double follows the pinned bundle instead of becoming a second copy of it that the next bump would contradict, and none of these files now names an artifact hash. On a bundle whose model predates the contract every helper returns the legacy shape; the economy-service cache-identity constants are then byte-identical to the ones this commit replaces. Tests that assert legacy behaviour by name pin an uncertified bundle instead of inheriting one from whatever is installed. That is `test_a_model_this_build_cannot_load_reads_as_no_canonical_model`, whose docstring already says "uncertified bundle"; `test_actual_settings_validator_rejects_explicit_settings_on_legacy_worker`, where a legacy manifest describes a legacy install only together with the model it names; `test_legacy_country_requests_do_not_receive_spm`; and `SimulationAPIModal`'s `TestRun` and `TestRunBudgetWindowBatch`, which assert the legacy gateway body — an explicit data artifact revision dropped, no measurement attached. The canonical translation of the same payloads is already covered against real HTTP in `tests/unit/services/test_worker_spm.py`. `test_spm_settings.py` gains the certified counterpart of the first of those: whatever stops the country import on a certified bundle must reach the caller as a typed configuration failure rather than a 500. Three expectations changed rather than doubles, because the canonical contract makes the old ones wrong. `test__given_request_context__then_all_calls_forward_request_id` asserted the exact set of gateway paths a sequence of calls touches. A certified bundle reads the worker registry before each submission, so `/versions` joins that set; the test's subject is that every call carries the request id, and the capability probe is one of them. `test__given_no_previous_impact__creates_new_simulation` asserted that the options written with the start claim are the options the caller sent. The service resolves the measurement into the options before anything is hashed, claimed or submitted, so the stored row records what was submitted; it now asserts the resolved options. The same resolution is why the cache-identity constants gain an `spm` segment. `test_household_under_policy_calculates_and_caches_json_as_an_object` asserted that the country is called with a household and a policy alone. A certified bundle also passes the resolved measurement and whether the saved household requested one, so the expected call carries both. Two doubles were malformed rather than incomplete under the canonical contract and are now well-formed: budget-window results that declared a three-year window while carrying zero or one annual impacts, which the API reads as a window with incomplete receipts; and `test__given_bundle_default_dataset_name__canonicalizes_setup_identity`, which built setup options against the module-level live HTTP client and so attempted a real network call once the capability probe became reachable. Verified locally against an installed canonical environment (policyengine 6.0.0, core 3.32.5, us 2.2.1, uk 2.90.2, spm-calculator 1.0.0; CPython 3.14.4 arm64). No assertion was weakened or removed, and nothing is skipped or xfailed. Co-Authored-By: Claude Opus 5 (1M context) --- .../test_simulation_gateway_contract.py | 15 ++ tests/fixtures/services/economy_service.py | 65 +++++++- tests/fixtures/spm.py | 154 ++++++++++++++++++ .../test_budget_window_in_flight_dedupe.py | 6 + tests/unit/libs/test_simulation_entrypoint.py | 32 +++- .../test_calculate_deprecated_inputs.py | 9 +- .../routes/test_calculate_error_statuses.py | 4 +- tests/unit/routes/test_canonical_spm.py | 10 +- .../test_economy_submission_identity.py | 30 ++++ ...st_household_and_user_policy_orm_routes.py | 19 ++- tests/unit/services/test_economy_service.py | 80 ++++----- .../test_household_calculation_service.py | 18 +- tests/unit/services/test_worker_spm.py | 21 ++- tests/unit/test_spm_settings.py | 40 ++++- 14 files changed, 440 insertions(+), 63 deletions(-) create mode 100644 tests/fixtures/spm.py diff --git a/tests/contract/test_simulation_gateway_contract.py b/tests/contract/test_simulation_gateway_contract.py index edbdd11a3..1aca986b7 100644 --- a/tests/contract/test_simulation_gateway_contract.py +++ b/tests/contract/test_simulation_gateway_contract.py @@ -16,6 +16,7 @@ MOCK_SIMULATION_PAYLOAD_WITH_TELEMETRY, MOCK_SUBMIT_RESPONSE_SUCCESS, ) +from tests.fixtures.spm import worker_versions_document @pytest.fixture(autouse=True) @@ -56,6 +57,13 @@ def test_gateway_comparison_submit_and_poll_contract(monkeypatch): monkeypatch.setenv("OLD_SIMULATION_GATEWAY_URL", "https://simulation.test") client = _client_for( { + # A certified bundle reads the worker's advertised SPM + # capability before it submits anything, so the registry read is + # part of the submission contract rather than a separate one. + ("GET", "/versions"): _response( + status_code=200, + json_data=worker_versions_document(app_name=MOCK_RESOLVED_APP_NAME), + ), ("POST", "/simulate/economy/comparison"): _response( status_code=202, json_data=MOCK_SUBMIT_RESPONSE_SUCCESS, @@ -88,6 +96,13 @@ def test_gateway_budget_window_submit_and_poll_contract(monkeypatch): monkeypatch.setenv("OLD_SIMULATION_GATEWAY_URL", "https://simulation.test") client = _client_for( { + # A certified bundle reads the worker's advertised SPM + # capability before it submits anything, so the registry read is + # part of the submission contract rather than a separate one. + ("GET", "/versions"): _response( + status_code=200, + json_data=worker_versions_document(app_name=MOCK_RESOLVED_APP_NAME), + ), ( "POST", "/simulate/economy/budget-window", diff --git a/tests/fixtures/services/economy_service.py b/tests/fixtures/services/economy_service.py index 9c2bcbf22..d3e2c9727 100644 --- a/tests/fixtures/services/economy_service.py +++ b/tests/fixtures/services/economy_service.py @@ -9,6 +9,14 @@ ) from policyengine_api.data.v1_models import ReformImpact +from tests.fixtures.spm import ( + INSTALLED_SPM_SELECTION, + spm_options, + spm_options_hash_segment, + spm_result_fields, + worker_spm_capability, +) + # Mock data constants MOCK_COUNTRY_ID = "us" MOCK_POLICY_ID = 123 @@ -21,10 +29,20 @@ MOCK_TIME_PERIOD = "2025" MOCK_API_VERSION = "1.0" MOCK_OPTIONS = {"option1": "value1", "option2": "value2"} +# A certified bundle resolves a measurement into the request options before +# anything is hashed, cached or submitted, so the doubles below carry it +# wherever the service would have written it. The selection itself is read from +# the installed bundle manifest, not written out, because its artifact hash +# belongs to the pinned bundle rather than to these tests. +MOCK_SPM_SELECTION = INSTALLED_SPM_SELECTION +MOCK_RESOLVED_OPTIONS = spm_options(MOCK_OPTIONS) +MOCK_SPM_OPTIONS_HASH_SEGMENT = spm_options_hash_segment() +MOCK_SPM_RESULT_FIELDS = spm_result_fields(years=[MOCK_TIME_PERIOD]) MOCK_DATA_VERSION = "faux-populace-us-2099-test-release" MOCK_LOOKUP_OPTIONS_HASH = ( "[option1=value1&option2=value2" - "&dataset=hf://policyengine/faux-populace-us/faux_populace_us_2099.h5@" + + MOCK_SPM_OPTIONS_HASH_SEGMENT + + "&dataset=hf://policyengine/faux-populace-us/faux_populace_us_2099.h5@" "faux-populace-us-2099-test-release" "&model_version=1.2.3&target=general" "&data_version=faux-populace-us-2099-test-release" @@ -57,10 +75,15 @@ "poverty_impact": {"baseline": 0.12, "reform": 0.10}, "budget_impact": {"baseline": 1000, "reform": 1200}, "inequality_impact": {"baseline": 0.45, "reform": 0.42}, + # A canonical worker returns what it measured alongside what it computed, + # and the service refuses a result it cannot certify against the selection + # that was submitted. + **MOCK_SPM_RESULT_FIELDS, } MOCK_SIM_CONFIG = { "country": MOCK_COUNTRY_ID, + "spm": MOCK_SPM_SELECTION, "reform": json.loads(MOCK_REFORM_POLICY_JSON), "baseline": json.loads(MOCK_BASELINE_POLICY_JSON), "region": MOCK_REGION, @@ -151,6 +174,7 @@ def mock_simulation_entrypoint(): mock_api.get_execution_result.return_value = MOCK_REFORM_IMPACT_DATA mock_api.run_budget_window_batch.return_value = mock_batch_execution mock_api.get_budget_window_batch_by_id.return_value = mock_batch_execution + mock_api.get_spm_capability.return_value = worker_spm_capability() with patch( "policyengine_api.services.economy_service.simulation_entrypoint", mock_api @@ -285,6 +309,44 @@ def create_mock_modal_execution( return mock_execution +def create_mock_simulation_gateway(): + """A gateway double that certifies the bundle this API actually runs. + + An unconfigured `MagicMock` answers `get_spm_capability` with an attribute + rather than a capability, which a certified bundle reads as a worker that + cannot run the measurement it resolved. + """ + gateway = MagicMock() + gateway.get_spm_capability.return_value = worker_spm_capability() + return gateway + + +def create_mock_budget_window_annual_impact(year, **fields): + """One year of a budget-window worker result, with its own receipt.""" + return {"year": str(year), **fields, **spm_result_fields(years=[year])} + + +def create_mock_budget_window_result(years, totals=None, **row_fields): + """A complete budget-window worker result for exactly the given years. + + A canonical worker returns one certified receipt per year and the API + refuses a window whose receipts do not cover the years it submitted, so a + partial `annualImpacts` list is a broken result rather than a small one. + """ + years = [str(year) for year in years] + return { + "kind": "budgetWindow", + "startYear": years[0], + "endYear": years[-1], + "windowSize": len(years), + "annualImpacts": [ + create_mock_budget_window_annual_impact(year, **row_fields) + for year in years + ], + "totals": {} if totals is None else totals, + } + + def create_mock_budget_window_batch_execution( batch_job_id=MOCK_MODAL_JOB_ID, status=MODAL_EXECUTION_STATUS_SUBMITTED, @@ -329,6 +391,7 @@ def mock_simulation_entrypoint_legacy(): mock_api.get_execution_by_id.return_value = mock_execution mock_api.get_execution_status.return_value = MODAL_EXECUTION_STATUS_RUNNING mock_api.get_execution_result.return_value = MOCK_REFORM_IMPACT_DATA + mock_api.get_spm_capability.return_value = worker_spm_capability() with patch( "policyengine_api.services.economy_service.simulation_entrypoint", mock_api diff --git a/tests/fixtures/spm.py b/tests/fixtures/spm.py new file mode 100644 index 000000000..58f0ef745 --- /dev/null +++ b/tests/fixtures/spm.py @@ -0,0 +1,154 @@ +"""Doubles for the canonical SPM boundary a certified bundle activates. + +A certified bundle resolves an SPM selection for every US request. The API then +refuses to submit that request unless the selected worker advertises a matching +capability, and refuses to serve a result that carries no receipt for what was +measured. Doubles written while the pinned bundle was uncertified answer +neither, because the boundary was dormant and never asked them. + +Every value here is read from the installed bundle manifest rather than written +down, so a double follows the pinned bundle instead of becoming a second copy +of it that the next bump would silently contradict. On a bundle whose US model +predates the canonical constructor the selection is `None`, the boundary stays +dormant, and every helper degrades to the legacy shapes. + +Resolving the selection at import time also imports `policyengine_us` during +collection, before any test patches `datetime.datetime`, so the country import +inside the boundary is a `sys.modules` hit rather than a fresh import running +under a patched standard library. +""" + +from __future__ import annotations + +import pytest + +from policyengine_api import spm +from policyengine_api.constants import POLICYENGINE_VERSION + +# The contract identifier `worker_spm.validate_worker_spm` requires. +SPM_CONTRACT_VERSION = "canonical-spm-v1" + +# What the installed bundle certifies for an unqualified US request, or None on +# a bundle whose model predates the contract. +INSTALLED_SPM_SELECTION = spm.normalize_spm_selection("us", None) + + +def worker_spm_capability(selection: dict | None = None) -> dict | None: + """What a worker running this API's own bundle advertises to `/versions`.""" + resolved = INSTALLED_SPM_SELECTION if selection is None else selection + if resolved is None: + return None + return {"contract_version": SPM_CONTRACT_VERSION, "defaults": dict(resolved)} + + +def worker_versions_document( + *, + bundle_version: str = POLICYENGINE_VERSION, + app_name: str = "test-worker-app", + country: str = "us", + country_version: str | None = None, + selection: dict | None = None, +) -> dict: + """The gateway registry document `get_spm_capability` reads. + + Keyed by wrapper version, as the deployed registry is: the capability + belongs to the bundle the worker runs, not to the country route that + resolves to its application. + """ + document: dict = { + "policyengine": {bundle_version: app_name, "latest": bundle_version}, + } + if country_version is not None: + document[country] = {country_version: app_name, "latest": country_version} + capability = worker_spm_capability(selection) + if capability is not None: + document["spm_capabilities"] = {bundle_version: capability} + return document + + +def spm_receipt(*, years, selection: dict | None = None) -> dict: + """One country provenance receipt covering the given calculation years.""" + resolved = INSTALLED_SPM_SELECTION if selection is None else selection + return { + "forecast_id": "test-only", + "forecast_sha256": resolved["forecast_content_sha256"], + "scenario": resolved["scenario"], + "geography_kind": resolved["geography_kind"], + "runtime_versions": {"policyengine-us": "test-only"}, + "years": {str(year): {"status": "forecast"} for year in years}, + "geographies": [], + "composition_method": "classified-inputs", + "storage_method": "formula", + } + + +def spm_result_fields(*, years, selection: dict | None = None) -> dict: + """The SPM half of a worker result, or nothing on an uncertified bundle.""" + resolved = INSTALLED_SPM_SELECTION if selection is None else selection + if resolved is None: + return {} + receipt = spm_receipt(years=years, selection=resolved) + return { + "spm_config": dict(resolved), + "spm_provenance": { + "baseline": [dict(receipt)], + "reform": [dict(receipt)], + }, + } + + +def household_receipt_fields(*, years, selection: dict | None = None) -> dict: + """The SPM half of a cached household calculation. + + A household carries one receipt for the calculation it is, where a worker + result carries a baseline and reform list for the comparison it ran. + """ + resolved = INSTALLED_SPM_SELECTION if selection is None else selection + if resolved is None: + return {} + return { + "spm_config": dict(resolved), + "spm_provenance": spm_receipt(years=years, selection=resolved), + } + + +def country_calculate_kwargs( + *, requested: bool = False, selection: dict | None = None +) -> dict: + """The measurement arguments the service passes to a country package.""" + resolved = INSTALLED_SPM_SELECTION if selection is None else selection + if resolved is None: + return {} + return {"spm": dict(resolved), "spm_requested": requested} + + +def spm_options(options: dict, selection: dict | None = None) -> dict: + """Request options as the service resolves them against the bundle.""" + resolved = INSTALLED_SPM_SELECTION if selection is None else selection + if resolved is None: + return dict(options) + return {**options, "spm": dict(resolved)} + + +def spm_options_hash_segment(selection: dict | None = None) -> str: + """The segment a resolved selection contributes to a cache identity. + + Options are serialized in sorted-key order, so `spm` follows every option a + caller sent. Derived rather than written out because the artifact hash it + carries belongs to the pinned bundle. + """ + resolved = INSTALLED_SPM_SELECTION if selection is None else selection + return "" if resolved is None else f"&spm={resolved}" + + +@pytest.fixture +def legacy_bundle(monkeypatch): + """Pin a bundle that predates the canonical SPM contract. + + Use this in tests that assert legacy behaviour by name: the resolution and + the transport contract both differ on an uncertified bundle, so leaving the + regime to whatever happens to be installed makes the assertion mean + whichever thing the environment chose. + """ + monkeypatch.setattr(spm, "_current_bundle", dict) + monkeypatch.setattr(spm, "_installed_country_implements_spm", lambda _: False) diff --git a/tests/integration/test_budget_window_in_flight_dedupe.py b/tests/integration/test_budget_window_in_flight_dedupe.py index 0529473aa..510754fa7 100644 --- a/tests/integration/test_budget_window_in_flight_dedupe.py +++ b/tests/integration/test_budget_window_in_flight_dedupe.py @@ -3,6 +3,8 @@ from flask import Flask from policyengine_api.runtime_cache.fake import InMemoryCacheBackend +from tests.fixtures.spm import worker_spm_capability + class FakeRedis(InMemoryCacheBackend): pass @@ -31,6 +33,10 @@ def test_budget_window_in_flight_dedupe_uses_existing_batch_without_live_db( fake_cache = BudgetWindowCache(client=FakeRedis()) simulation_entrypoint = MagicMock() + # A certified bundle asks the selected worker to certify the measurement + # before the service reaches its cache, so the gateway double answers the + # capability its own bundle binds instead of an unread attribute. + simulation_entrypoint.get_spm_capability.return_value = worker_spm_capability() simulation_entrypoint.resolve_app_name.side_effect = ( lambda country, version, **kwargs: ("test-budget-window-worker", version) ) diff --git a/tests/unit/libs/test_simulation_entrypoint.py b/tests/unit/libs/test_simulation_entrypoint.py index 2f29ad609..cfafea630 100644 --- a/tests/unit/libs/test_simulation_entrypoint.py +++ b/tests/unit/libs/test_simulation_entrypoint.py @@ -59,8 +59,12 @@ MOCK_SUBMIT_RESPONSE_SUCCESS, create_mock_httpx_response, ) +from tests.fixtures.spm import worker_versions_document # noqa: E402 -pytest_plugins = ("tests.fixtures.libs.simulation_entrypoint",) +pytest_plugins = ( + "tests.fixtures.libs.simulation_entrypoint", + "tests.fixtures.spm", +) GATEWAY_AUTH_TEST_ENV_VARS = ( "GATEWAY_AUTH_ISSUER", @@ -107,6 +111,14 @@ def _response(self, method, url, json=None): elif "/jobs/" in path: payload = MOCK_POLL_RESPONSE_RUNNING status_code = 202 + elif path == "/versions": + # The whole registry, which is where a certified bundle reads the + # selected worker's advertised SPM capability. + payload = worker_versions_document( + app_name=MOCK_RESOLVED_APP_NAME, + country_version="1.459.0", + ) + status_code = 200 elif "/versions/" in path: payload = { "latest": "1.459.0", @@ -475,6 +487,10 @@ def test__given_request_context__then_all_calls_forward_request_id( "/simulate/economy/budget-window", f"/jobs/{MOCK_MODAL_JOB_ID}", f"/budget-window-jobs/{MOCK_BATCH_JOB_ID}", + # Each submission first reads the worker registry to certify + # the measurement this bundle resolved; that call carries the + # request id like every other. + "/versions", "/versions/us", "/health", } @@ -483,7 +499,18 @@ def test__given_request_context__then_all_calls_forward_request_id( for request in requests ) + @pytest.mark.usefixtures("legacy_bundle") class TestRun: + """Submission translation for a bundle that predates canonical SPM. + + These cases describe the legacy gateway body: an explicit data artifact + revision is dropped, and no measurement is attached. A certified bundle + translates the same payload differently and is covered against real + HTTP in `tests/unit/services/test_worker_spm.py`. Pinning the bundle + keeps each contract asserted where it is named instead of letting the + installed package choose which one these cases mean. + """ + def test__given_valid_payload__then_returns_execution_with_job_id( self, mock_httpx_client, @@ -729,7 +756,10 @@ def test__given_policyengine_version__then_returns_registered_bundle_app( f"{api.base_url}/versions/policyengine" ) + @pytest.mark.usefixtures("legacy_bundle") class TestRunBudgetWindowBatch: + """The same legacy submission contract for a budget-window batch.""" + def test__given_valid_payload__then_returns_batch_execution( self, mock_httpx_client, diff --git a/tests/unit/routes/test_calculate_deprecated_inputs.py b/tests/unit/routes/test_calculate_deprecated_inputs.py index aeb0af3e8..86724b692 100644 --- a/tests/unit/routes/test_calculate_deprecated_inputs.py +++ b/tests/unit/routes/test_calculate_deprecated_inputs.py @@ -12,6 +12,8 @@ class DummyCountry: def __init__(self): self.household = None self.policy = None + self.spm = None + self.spm_requested = False self.metadata = { "variables": { "age": {"entity": "person"}, @@ -37,9 +39,14 @@ def __init__(self): }, } - def calculate(self, household, policy): + def calculate(self, household, policy, *, spm=None, spm_requested=False): + # A certified bundle resolves a measurement for every US request and + # passes it to the country, so a double that accepts only the legacy + # two-argument call turns that request into a 500. self.household = household self.policy = policy + self.spm = spm + self.spm_requested = spm_requested return {"household": household, "policy": policy} diff --git a/tests/unit/routes/test_calculate_error_statuses.py b/tests/unit/routes/test_calculate_error_statuses.py index 6f8f6cd19..c41671f42 100644 --- a/tests/unit/routes/test_calculate_error_statuses.py +++ b/tests/unit/routes/test_calculate_error_statuses.py @@ -20,7 +20,7 @@ class ParsingErrorCountry(DummyCountry): - def calculate(self, household, policy): + def calculate(self, household, policy, *, spm=None, spm_requested=False): raise SituationParsingError( ["people", "you", "employment_income", "2026"], "Can't deal with value: expected type number, received '{}'.", @@ -28,7 +28,7 @@ def calculate(self, household, policy): class CrashingCountry(DummyCountry): - def calculate(self, household, policy): + def calculate(self, household, policy, *, spm=None, spm_requested=False): raise RuntimeError("engine exploded") diff --git a/tests/unit/routes/test_canonical_spm.py b/tests/unit/routes/test_canonical_spm.py index e3c55a015..538b68cbe 100644 --- a/tests/unit/routes/test_canonical_spm.py +++ b/tests/unit/routes/test_canonical_spm.py @@ -23,6 +23,8 @@ from policyengine_api.services.simulation_service import SimulationService from policyengine_api.utils import hash_object +pytest_plugins = ("tests.fixtures.spm",) + HOUSEHOLD = {"people": {"you": {"age": {"2026": 40}}}} FORECAST_HASH = "a" * 64 @@ -294,7 +296,13 @@ def test_certification_checked_before_cached_response(certified, harness): @pytest.mark.parametrize("country_id", ["us", "uk"]) -def test_legacy_country_requests_do_not_receive_spm(harness, country_id): +def test_legacy_country_requests_do_not_receive_spm(legacy_bundle, harness, country_id): + """An uncertified bundle is pinned: this is the case being described. + + Left to the installed bundle, the US half of this test asserts the legacy + contract on whichever bundle the environment happens to carry, and says + nothing at all once that bundle is certified. + """ client, country = harness response = client.post(f"/{country_id}/calculate", json={"household": HOUSEHOLD}) assert response.status_code == 200 diff --git a/tests/unit/routes/test_economy_submission_identity.py b/tests/unit/routes/test_economy_submission_identity.py index 9b0de13c7..86c669fd1 100644 --- a/tests/unit/routes/test_economy_submission_identity.py +++ b/tests/unit/routes/test_economy_submission_identity.py @@ -20,11 +20,29 @@ from policyengine_api.services.budget_window_cache import BudgetWindowCache from policyengine_api.services.economy_service import EconomyService from policyengine_api.services.reform_impacts_service import ReformImpactsService +from tests.fixtures.spm import worker_versions_document from tests.integration.test_cloud_run_candidate import ( test_cloud_run_candidate_current_law_economy as run_smoke, ) +BUNDLE_VERSION = "5.2.0" + + +def _registry_response(app_name): + """The full worker registry, read once per submission by a certified bundle. + + The service's own selection is neutralized in this fixture, but the HTTP + consumer under test still validates the worker's advertised capability + before it posts, and reads it from `/versions` rather than the per-kind + route these tests otherwise serve. + """ + return httpx.Response( + 200, + json=worker_versions_document(bundle_version=BUNDLE_VERSION, app_name=app_name), + ) + + @pytest.fixture def economy(monkeypatch): # The transport/cache boundary needs no real bundle or population compute. @@ -91,6 +109,8 @@ def test_budget_submission_identity_is_verified_before_caching(economy, submitte def transport(request): nonlocal active_app + if request.url.path == "/versions": + return _registry_response(active_app) if request.url.path == "/versions/policyengine": return httpx.Response(200, json={"5.2.0": active_app}) assert request.method == "POST" @@ -144,6 +164,8 @@ def test_budget_submission_identity_mismatch_retains_the_batch_for_the_next_poll def transport(request): nonlocal active_app + if request.url.path == "/versions": + return _registry_response(active_app) if request.url.path == "/versions/policyengine": return httpx.Response(200, json={"5.2.0": active_app}) if request.method == "POST": @@ -201,6 +223,8 @@ def test_budget_submission_identity_mismatch_never_overwrites_another_claim(econ active_app = "worker-A" def transport(request): + if request.url.path == "/versions": + return _registry_response(active_app) if request.url.path == "/versions/policyengine": return httpx.Response(200, json={"5.2.0": active_app}) return httpx.Response( @@ -295,6 +319,8 @@ def test_nonce_floods_cannot_evict_the_shared_scope_entry(economy): shared_result = {"budget": {"budgetary_impact": 0}, "resolved_app_name": "worker-A"} def transport(request): + if request.url.path == "/versions": + return _registry_response("worker-A") if request.url.path == "/versions/policyengine": return httpx.Response(200, json={"5.2.0": "worker-A"}) posts.append(request) @@ -389,6 +415,8 @@ def test_candidate_smoke_cannot_pass_from_prior_result_when_submission_is_broken } def transport(request): + if request.url.path == "/versions": + return _registry_response("worker-A") if request.url.path == "/versions/policyengine": return httpx.Response(200, json={"5.2.0": "worker-A"}) assert request.method == "POST" @@ -451,6 +479,8 @@ def test_nonce_polling_reuses_its_job_but_another_nonce_submits_again(economy): submissions = [] def transport(request): + if request.url.path == "/versions": + return _registry_response("worker-A") if request.url.path == "/versions/policyengine": return httpx.Response(200, json={"5.2.0": "worker-A"}) if request.method == "POST": diff --git a/tests/unit/routes/test_household_and_user_policy_orm_routes.py b/tests/unit/routes/test_household_and_user_policy_orm_routes.py index ee38bea40..9c7f8e32e 100644 --- a/tests/unit/routes/test_household_and_user_policy_orm_routes.py +++ b/tests/unit/routes/test_household_and_user_policy_orm_routes.py @@ -27,6 +27,11 @@ from policyengine_api.services.household_calculation_service import ( HouseholdCalculationService, ) +from tests.fixtures.spm import ( + INSTALLED_SPM_SELECTION, + country_calculate_kwargs, + household_receipt_fields, +) def test_household_under_policy_returns_cached_json_object(orm_session_factory): @@ -65,8 +70,17 @@ def test_household_under_policy_returns_cached_json_object(orm_session_factory): policy_hash="policy-hash", country_package_version=COUNTRY_PACKAGE_VERSIONS["us"], policyengine_version=POLICYENGINE_VERSION, + # A certified bundle resolves a measurement before the cache read, + # and it is part of the identity: a stored calculation measured + # against a different artifact is a different calculation. + spm=INSTALLED_SPM_SELECTION, + ), + CachedHouseholdCalculation( + household=stored_result, + # A stored calculation without a receipt cannot be certified as the + # one this identity names, so the cache declines it. + **household_receipt_fields(years=["2026"]), ), - CachedHouseholdCalculation(household=stored_result), ) service = HouseholdCalculationService( primary_session_factory=orm_session_factory, @@ -131,9 +145,12 @@ def test_household_under_policy_calculates_and_caches_json_as_an_object( response = get_household_under_policy("us", "1", "2") assert response["result"] == calculated + # A certified bundle resolves the stored household's measurement and hands + # it to the country; a saved household that chose nothing is not a request. country.calculate.assert_called_once_with( {"people": {"you": {}}}, {"gov.example.parameter": 1}, + **country_calculate_kwargs(requested=False), ) diff --git a/tests/unit/services/test_economy_service.py b/tests/unit/services/test_economy_service.py index 17a2b687b..8faf30997 100644 --- a/tests/unit/services/test_economy_service.py +++ b/tests/unit/services/test_economy_service.py @@ -37,10 +37,13 @@ MOCK_REGION, MOCK_RESOLVED_APP_NAME, MOCK_RESOLVED_DATASET, + MOCK_RESOLVED_OPTIONS, MOCK_RUN_ID, MOCK_TIME_PERIOD, create_mock_budget_window_batch_execution, + create_mock_budget_window_result, create_mock_reform_impact, + create_mock_simulation_gateway, ) pytest_plugins = ("tests.fixtures.services.economy_service",) @@ -372,7 +375,10 @@ def test__given_no_previous_impact__creates_new_simulation( write_values = ( mock_reform_impacts_service.set_reform_impact.call_args.kwargs ) - assert write_values["options"] == MOCK_OPTIONS + # A certified bundle resolves the measurement into the request + # options before the claim is written, so the stored row records + # what was submitted rather than what the caller sent. + assert write_values["options"] == MOCK_RESOLVED_OPTIONS assert write_values["reform_impact_json"] == {} def test__given_existing_start_claim__does_not_submit_duplicate_simulation( @@ -513,7 +519,7 @@ def test__given_policies_created_through_orm__submits_decoded_json( reform_impacts = MagicMock() reform_impacts.get_all_reform_impacts_by_options_hash_prefix.return_value = [] - simulation_gateway = MagicMock() + simulation_gateway = create_mock_simulation_gateway() simulation_gateway.resolve_app_name.return_value = ( "policyengine-simulation-test", MOCK_MODEL_VERSION, @@ -997,30 +1003,22 @@ def test__given_completed_cached_result__returns_completed_batch_result( mock_simulation_entrypoint, mock_budget_window_cache, ): - completed_result = { - "kind": "budgetWindow", - "startYear": "2026", - "endYear": "2028", - "windowSize": 3, - "annualImpacts": [ - { - "year": "2026", - "taxRevenueImpact": 100, - "federalTaxRevenueImpact": 80, - "stateTaxRevenueImpact": 20, - "benefitSpendingImpact": -10, - "budgetaryImpact": 90, - } - ], - "totals": { + completed_result = create_mock_budget_window_result( + ["2026", "2027", "2028"], + totals={ "year": "Total", - "taxRevenueImpact": 100, - "federalTaxRevenueImpact": 80, - "stateTaxRevenueImpact": 20, - "benefitSpendingImpact": -10, - "budgetaryImpact": 90, + "taxRevenueImpact": 300, + "federalTaxRevenueImpact": 240, + "stateTaxRevenueImpact": 60, + "benefitSpendingImpact": -30, + "budgetaryImpact": 270, }, - } + taxRevenueImpact=100, + federalTaxRevenueImpact=80, + stateTaxRevenueImpact=20, + benefitSpendingImpact=-10, + budgetaryImpact=90, + ) mock_budget_window_cache.get_completed_result.return_value = ( completed_result ) @@ -1073,14 +1071,9 @@ def test__given_completed_batch_poll__caches_result_and_returns_completed( mock_simulation_entrypoint, mock_budget_window_cache, ): - completed_result = { - "kind": "budgetWindow", - "startYear": "2026", - "endYear": "2028", - "windowSize": 3, - "annualImpacts": [], - "totals": {}, - } + completed_result = create_mock_budget_window_result( + ["2026", "2027", "2028"] + ) mock_budget_window_cache.get_batch_job_id.return_value = "fc-budget-123" mock_simulation_entrypoint.get_budget_window_batch_by_id.return_value = ( create_mock_budget_window_batch_execution( @@ -1146,14 +1139,9 @@ def test__given_completed_batch_cache_write_fails__does_not_clear_batch_id( mock_simulation_entrypoint, mock_budget_window_cache, ): - completed_result = { - "kind": "budgetWindow", - "startYear": "2026", - "endYear": "2028", - "windowSize": 3, - "annualImpacts": [], - "totals": {}, - } + completed_result = create_mock_budget_window_result( + ["2026", "2027", "2028"] + ) mock_budget_window_cache.get_batch_job_id.return_value = "fc-budget-123" mock_budget_window_cache.set_completed_result.return_value = False mock_simulation_entrypoint.get_budget_window_batch_by_id.return_value = ( @@ -1444,9 +1432,9 @@ def test__given_reordered_options__uses_same_budget_window_cache_identity( mock_simulation_entrypoint, mock_budget_window_cache, ): - mock_budget_window_cache.get_completed_result.return_value = { - "kind": "budgetWindow" - } + mock_budget_window_cache.get_completed_result.return_value = ( + create_mock_budget_window_result(["2026", "2027", "2028"]) + ) economy_service.get_budget_window_economic_impact( **{ @@ -2380,7 +2368,11 @@ def test__given_bundle_default_dataset_name__omits_data(self): assert result is None def test__given_bundle_default_dataset_name__canonicalizes_setup_identity(self): - service = EconomyService() + # Building setup options consults the selected worker, so this needs + # a gateway double rather than the module's live HTTP client. + service = EconomyService( + simulation_entrypoint_=create_mock_simulation_gateway() + ) common_args = { "country_id": "us", "policy_id": MOCK_POLICY_ID, diff --git a/tests/unit/services/test_household_calculation_service.py b/tests/unit/services/test_household_calculation_service.py index 0efed99bc..6b9fb22ce 100644 --- a/tests/unit/services/test_household_calculation_service.py +++ b/tests/unit/services/test_household_calculation_service.py @@ -19,6 +19,7 @@ CalculationResult, HouseholdCalculationService, ) +from tests.fixtures.spm import INSTALLED_SPM_SELECTION, household_receipt_fields PACKAGE_ROOT = Path(__file__).parents[3] / "policyengine_api" @@ -88,6 +89,10 @@ def _identity() -> HouseholdCalculationIdentity: policy_hash="policy-hash", country_package_version=COUNTRY_PACKAGE_VERSIONS["us"], policyengine_version=POLICYENGINE_VERSION, + # A certified bundle resolves a measurement before every read and + # write, and a calculation measured against a different artifact is a + # different calculation rather than a stale one. + spm=INSTALLED_SPM_SELECTION, ) @@ -112,10 +117,11 @@ class Country: "parameters": {}, } - def calculate(self, household, policy): + def calculate(self, household, policy, *, spm=None, spm_requested=False): return CalculationResult( household=household, warnings=("employment_income could not be calculated",), + **household_receipt_fields(years=["2026"]), ) service = HouseholdCalculationService( @@ -144,11 +150,14 @@ class Country: "entities": {"person": {"plural": "people", "roles": {}}}, } - def calculate(self, household, policy): + def calculate(self, household, policy, *, spm=None, spm_requested=False): assert primary.active_scopes == 0 return CalculationResult( household={"people": {"you": {"net_income": {"2026": 42}}}}, warnings=("net_income could not be calculated",), + # The cache stores the country's receipt alongside the result, + # and declines to serve a canonical identity without one. + **household_receipt_fields(years=["2026"]), ) service = HouseholdCalculationService( @@ -178,11 +187,12 @@ def test_calculation_uses_local_cache_without_recomputing(orm_session_factory): CachedHouseholdCalculation( household=calculated, warnings=("net_income could not be calculated",), + **household_receipt_fields(years=["2026"]), ), ) country = SimpleNamespace( metadata={"variables": {}, "entities": {}}, - calculate=lambda *_: (_ for _ in ()).throw( + calculate=lambda *_, **__: (_ for _ in ()).throw( AssertionError("cache hit should not calculate") ), ) @@ -208,7 +218,7 @@ def test_failed_cache_write_does_not_invalidate_successful_calculation( "variables": {}, "entities": {"person": {"plural": "people", "roles": {}}}, }, - calculate=lambda *_: SimpleNamespace( + calculate=lambda *_, **__: SimpleNamespace( household={"people": {"you": {}}}, ), ) diff --git a/tests/unit/services/test_worker_spm.py b/tests/unit/services/test_worker_spm.py index 156d91147..bb3f1a1e9 100644 --- a/tests/unit/services/test_worker_spm.py +++ b/tests/unit/services/test_worker_spm.py @@ -36,12 +36,21 @@ def test_legacy_worker_selection_remains_usable(country_id): def test_actual_settings_validator_rejects_explicit_settings_on_legacy_worker( country_id, ): - with patch( - "policyengine_api.spm._current_bundle", - return_value={ - "policyengine_version": "5.2.0", - "packages": {"policyengine-us": {"version": "1.764.6"}}, - }, + # A legacy manifest describes a legacy install only together with the model + # it names. Pinning the manifest alone leaves the constructor probe reading + # whichever country package this build happens to have installed, and a + # canonical one there is the uncertified-canonical case, not this one. + with ( + patch( + "policyengine_api.spm._current_bundle", + return_value={ + "policyengine_version": "5.2.0", + "packages": {"policyengine-us": {"version": "1.764.6"}}, + }, + ), + patch( + "policyengine_api.spm._installed_country_implements_spm", return_value=False + ), ): assert validate_worker_spm(country_id) is None with pytest.raises(SPMValidationError) as error: diff --git a/tests/unit/test_spm_settings.py b/tests/unit/test_spm_settings.py index 615fba9d2..0bfb63352 100644 --- a/tests/unit/test_spm_settings.py +++ b/tests/unit/test_spm_settings.py @@ -9,6 +9,7 @@ from policyengine_api import spm +pytest_plugins = ("tests.fixtures.spm",) ARTIFACT_HASH = "a" * 64 LEGACY_BUNDLE = { @@ -545,12 +546,14 @@ def test_a_country_package_without_a_simulation_is_typed_not_internal( ids=["import", "attribute", "os", "type", "value"], ) def test_a_model_this_build_cannot_load_reads_as_no_canonical_model( - monkeypatch, failure + monkeypatch, legacy_bundle, failure ): """Whatever stopped the import, the answer is "no canonical constructor". Letting one escape would turn every US request on an uncertified bundle into - a 500 rather than the legacy behaviour that bundle actually has. + a 500 rather than the legacy behaviour that bundle actually has. The bundle + is pinned uncertified here because that is the case being described; a + certified bundle reaches its own import below and fails closed instead. """ def refuse(name): @@ -562,6 +565,39 @@ def refuse(name): assert spm.normalize_spm_selection("us", None) is None +@pytest.mark.parametrize( + "failure", + [ + ImportError("no distribution"), + AttributeError("no Simulation"), + OSError("a data file will not open"), + TypeError("an extension will not initialize"), + ValueError("a module refused its own configuration"), + ], + ids=["import", "attribute", "os", "type", "value"], +) +def test_a_model_a_certified_bundle_cannot_load_is_typed_not_internal( + monkeypatch, certified_bundle, failure +): + """The same failures on a certified bundle are typed, never a 500. + + A certified bundle never reaches the capability probe's tolerant answer: it + imports the country itself, and whatever stopped that import must reach the + caller as the same configuration failure an uncertified bundle reports for + an explicit selection. + """ + + def refuse(name): + raise failure + + monkeypatch.setattr(spm.importlib, "import_module", refuse) + + with pytest.raises(spm.SPMValidationError) as caught: + spm.normalize_spm_selection("us", None) + assert caught.value.code == "SPM_CONFIGURATION_UNAVAILABLE" + assert spm.spm_metadata("us") == {"available": False} + + def test_capability_probe_resolves_the_installed_package_by_name(monkeypatch): """Country ids and package names are not the same string for every country.""" from policyengine_api.constants import COUNTRIES, COUNTRY_PACKAGE_NAMES