diff --git a/changelog.d/us-humanitarian-immigration-statuses-767.added.md b/changelog.d/us-humanitarian-immigration-statuses-767.added.md new file mode 100644 index 00000000..86176b42 --- /dev/null +++ b/changelog.d/us-humanitarian-immigration-statuses-767.added.md @@ -0,0 +1 @@ +Impute humanitarian and temporary-protection immigration statuses (REFUGEE, ASYLEE, PAROLED_ONE_YEAR per program origin, TPS per country) in the US immigration stage, drawn to cited DHS/OHSS/CBP/CRS stocks, so the enacted H.R.1 §71109/§71301/§71302 and SNAP §10108 eligibility channels stop computing $0; gate each category against its target and add four release-blocking reform-coverage probes. diff --git a/docs/evidence/spec-engine/us-f0-coverage.json b/docs/evidence/spec-engine/us-f0-coverage.json index b202fe63..61814717 100644 --- a/docs/evidence/spec-engine/us-f0-coverage.json +++ b/docs/evidence/spec-engine/us-f0-coverage.json @@ -4,9 +4,9 @@ "version": 1 }, "country": "us", - "documentation_sha256": "4b39450dbdb8dafb83c3b627123b8026c6f82c660b66fe76f341a67c4f37c77b", + "documentation_sha256": "02ac100a7aaa1770930dd6005b2d7f26550560e769c4ffb8d642b76f1774dd84", "field_usage": { - "authored_normative_field_count": 32384, + "authored_normative_field_count": 32521, "claim_count": 49, "claims": [ { @@ -220,8 +220,8 @@ ], "mode": "legacy_behavior", "pointer_class": "all", - "pointer_count": 97, - "pointer_sha256": "cabaaa3d96c534f5ec37f51811c3f549d0ed20791e2892c868888c0da6d7d80e", + "pointer_count": 115, + "pointer_sha256": "545f626e7b9c6aed25a5ae8dd3156c96db2cd35e5637ec1ebe443432d1684277", "rationale": null, "relative_sink_prefix": null, "source_prefix": "/authored/spec~1imputation.yaml/transfer_execution", @@ -274,8 +274,8 @@ ], "mode": "compiler_semantic", "pointer_class": "all", - "pointer_count": 24488, - "pointer_sha256": "6382a3ed07016414c31462b8bedb24590d92b792405a89c927d85fa1bd2a8347", + "pointer_count": 24554, + "pointer_sha256": "18140dbc554c2a17a673fec5ae5839f504d4b9980c5d8f1c1ccb545342977cb4", "rationale": null, "relative_sink_prefix": null, "source_prefix": "/authored/spec~1imputation.yaml/producer_graph", @@ -514,8 +514,8 @@ ], "mode": "legacy_behavior", "pointer_class": "all", - "pointer_count": 1702, - "pointer_sha256": "be3e24b53955b9dabccfc7e779212b1e97daf0c0992be3c183e391bf695ed4ee", + "pointer_count": 1715, + "pointer_sha256": "78227f34c3f4ef946c06693dec9177dd906069f98016854bfd4d39c4cd4c51fa", "rationale": null, "relative_sink_prefix": "/source_manifest/stages", "source_prefix": "/authored/spec~1sources.yaml/stages", @@ -530,8 +530,8 @@ ], "mode": "legacy_behavior", "pointer_class": "all", - "pointer_count": 88, - "pointer_sha256": "ae011455154bae0df3913ca9a056058a909d913439b292cb46c11f63c7d0d9a3", + "pointer_count": 89, + "pointer_sha256": "8d281f8f1e2f9914684150b0424ff816d0ec09ed0d9053ea5a64e877a86cc35b", "rationale": null, "relative_sink_prefix": null, "source_prefix": "/authored/spec~1spine.yaml/pipeline_contract", @@ -544,8 +544,8 @@ "legacy_sinks": [], "mode": "compiler_semantic", "pointer_class": "all", - "pointer_count": 277, - "pointer_sha256": "dfa7ae701f62f7b1f06954798d29cfba3fc53a0be21e1751646d1f5bafd6071f", + "pointer_count": 316, + "pointer_sha256": "bb848f38e068858ac23e964a4ffbbf5542253da50515c72fe7753d8c56912486", "rationale": null, "relative_sink_prefix": null, "source_prefix": "/authored/spec~1spine.yaml/seed_site_bindings", @@ -736,8 +736,8 @@ "legacy_sinks": [], "mode": "compiler_semantic", "pointer_class": "all", - "pointer_count": 824, - "pointer_sha256": "7537385c3fd399a2dbb7dcd8ed7cf1ff2481ed336db621eafcbfd741d5792f40", + "pointer_count": 1032, + "pointer_sha256": "6d4af2d988ee87ed38a044c1007e854176af17be4782e802d86d960f10376f7a", "rationale": null, "relative_sink_prefix": null, "source_prefix": "/resolved/seed_protocol", @@ -750,8 +750,8 @@ "legacy_sinks": [], "mode": "compiler_semantic", "pointer_class": "all", - "pointer_count": 277, - "pointer_sha256": "368fc19a8f07c4abf38ebcf2fbc4414d33c403c0f1f9881c2d2e3f5d6160feb6", + "pointer_count": 316, + "pointer_sha256": "199ec98045871070860164b522892327ab840b48c0c6a53972ebc0ff0d7d091f", "rationale": null, "relative_sink_prefix": null, "source_prefix": "/resolved/seed_site_bindings", @@ -772,21 +772,21 @@ "verifier": "vintages" } ], - "configuration_field_count": 42154, - "consumed_field_count": 42154, + "configuration_field_count": 42538, + "consumed_field_count": 42538, "generation0_effect_counts": { - "legacy_behavior": 38476, - "no_generation0_effect": 3678 + "legacy_behavior": 38574, + "no_generation0_effect": 3964 }, "mode_counts": { - "compiler_semantic": 27715, + "compiler_semantic": 28067, "front_end_validation": 348, "identity_only": 103, - "legacy_behavior": 13988 + "legacy_behavior": 14020 }, "multiple_primary_use_field_count": 0, - "pointer_inventory_sha256": "3fc6b9480ea81b9635bd0db56e180c2daf32a5cd2006a70d350586c570f96754", - "resolved_binding_field_count": 9770, + "pointer_inventory_sha256": "faef18fd543699ba65a91f3ffaf6fdf38a807ffdc88bc3debfeb970881050d62", + "resolved_binding_field_count": 10017, "unused_field_count": 0 }, "inventory_coverage": { @@ -810,14 +810,14 @@ "primary_targets": 65, "producer_authored_outputs": 92, "producer_compiled_outputs": 227, - "producer_inputs": 2744, + "producer_inputs": 2750, "producer_nodes": 38, "producer_virtual_resources": 75, "release_rungs": 5, "resolved_references": 334, - "seed_owner_bindings": 112, + "seed_owner_bindings": 125, "seed_owner_rows": 54, - "seed_sites": 53, + "seed_sites": 66, "seed_streams": 14, "source_operators": 16, "source_stages": 37, @@ -1074,11 +1074,11 @@ "legacy_adapter.stacked_checkpoint_static_components" ], "expected": { - "schedule_sha256": "e59c019d3d454eac99ac0ac209b6c5b6faaf9bdfcaeee18c36a25be19bf7da2f" + "schedule_sha256": "88bc9243a3518982ae951c3de21bd55877e296ce4fcb183b9bee420d3a684b10" }, "failures": [], "observed": { - "schedule_sha256": "e59c019d3d454eac99ac0ac209b6c5b6faaf9bdfcaeee18c36a25be19bf7da2f" + "schedule_sha256": "88bc9243a3518982ae951c3de21bd55877e296ce4fcb183b9bee420d3a684b10" }, "status": "covered" }, @@ -1250,14 +1250,14 @@ "expected": { "edges": 71, "nodes": 38, - "schedule_sha256": "e59c019d3d454eac99ac0ac209b6c5b6faaf9bdfcaeee18c36a25be19bf7da2f", + "schedule_sha256": "88bc9243a3518982ae951c3de21bd55877e296ce4fcb183b9bee420d3a684b10", "waves": 6 }, "failures": [], "observed": { "edges": 71, "nodes": 38, - "schedule_sha256": "e59c019d3d454eac99ac0ac209b6c5b6faaf9bdfcaeee18c36a25be19bf7da2f", + "schedule_sha256": "88bc9243a3518982ae951c3de21bd55877e296ce4fcb183b9bee420d3a684b10", "waves": 6 }, "status": "covered" @@ -1275,11 +1275,11 @@ ], "expected": { "relation": "source rows preserved exactly", - "rows": 2744 + "rows": 2750 }, "failures": [], "observed": { - "rows": 2744 + "rows": 2750 }, "status": "covered" }, @@ -1321,7 +1321,7 @@ "expected": "complete execution-row and transition-authority object", "failures": [], "observed": { - "sha256": "503428f6e9d98f19ed3a6ada5bc9883ae44c1b5dc27f09e60b7ac98895a99bc0" + "sha256": "b15bf2dcf709cf6c7922d4b76e1d64c14a6ec2f9aa1185e6770505d798333c76" }, "status": "covered" }, @@ -1338,12 +1338,12 @@ ], "expected": { "nodes": 38, - "sha256": "271a7bb8d0b3f97ff344e0b7e68184fa74738a6585c24fc8781793db669f388b" + "sha256": "014f90315324f72f45ece928bccc44f34b4d1e95c0582f1d7d73620f482da15f" }, "failures": [], "observed": { "nodes": 38, - "sha256": "271a7bb8d0b3f97ff344e0b7e68184fa74738a6585c24fc8781793db669f388b" + "sha256": "014f90315324f72f45ece928bccc44f34b4d1e95c0582f1d7d73620f482da15f" }, "status": "covered" }, @@ -1361,12 +1361,12 @@ ], "expected": { "producer_count": 38, - "sha256": "afebb6725373abf5b8dd4fdb77bf2814cb6fcc569cb606c0c30963a8f65c0bab" + "sha256": "db175a952340b3ff2774215f320c4d19592c06b9959442639a6c561055f2442d" }, "failures": [], "observed": { "producer_count": 38, - "sha256": "afebb6725373abf5b8dd4fdb77bf2814cb6fcc569cb606c0c30963a8f65c0bab" + "sha256": "db175a952340b3ff2774215f320c4d19592c06b9959442639a6c561055f2442d" }, "status": "covered" }, @@ -1533,7 +1533,7 @@ "expected": { "groups": 12, "partition": "disjoint and exhaustive", - "sites": 53 + "sites": 66 }, "failures": [], "observed": { @@ -1550,6 +1550,19 @@ "pregnancy_assignment", "wic_claim_assignment", "snap_discretionary_exemption_assignment", + "immigration_humanitarian_paroled_one_year_afghanistan_assignment", + "immigration_humanitarian_paroled_one_year_ukraine_assignment", + "immigration_humanitarian_paroled_one_year_nicaragua_assignment", + "immigration_humanitarian_paroled_one_year_venezuela_assignment", + "immigration_humanitarian_refugee_assignment", + "immigration_humanitarian_asylee_assignment", + "immigration_humanitarian_deportation_withheld_assignment", + "immigration_humanitarian_tps_venezuela_assignment", + "immigration_humanitarian_tps_el_salvador_assignment", + "immigration_humanitarian_tps_honduras_assignment", + "immigration_humanitarian_tps_nicaragua_assignment", + "immigration_humanitarian_tps_nepal_assignment", + "immigration_humanitarian_tps_other_designated_assignment", "immigration_ead_workers_assignment", "immigration_ead_students_assignment", "ssi_take_up_assignment", @@ -1616,7 +1629,7 @@ "sipp_tip_training_cap" ] }, - "sites": 53 + "sites": 66 }, "status": "covered" }, @@ -1656,13 +1669,13 @@ "compiler_ir.node_slices" ], "expected": { - "map_sha256": "271c23eca777b63bbfecc5b0bdbfffee634e13f79dd37273172f5fadda4afefb", - "protocol_sha256": "870f9b43766d4c11f6dc1538bf97e360c6a85dd0d4a4a68c149595dfc0902f0a" + "map_sha256": "f5301b13c672bd21cb33e357f0c25c762acab3ef44ca15ed88895ff2ffebc0af", + "protocol_sha256": "61db02328d1844c9cb9c490599e33a25aea86eafadc7b940147585340f948d92" }, "failures": [], "observed": { - "map_sha256": "271c23eca777b63bbfecc5b0bdbfffee634e13f79dd37273172f5fadda4afefb", - "protocol_sha256": "870f9b43766d4c11f6dc1538bf97e360c6a85dd0d4a4a68c149595dfc0902f0a" + "map_sha256": "f5301b13c672bd21cb33e357f0c25c762acab3ef44ca15ed88895ff2ffebc0af", + "protocol_sha256": "61db02328d1844c9cb9c490599e33a25aea86eafadc7b940147585340f948d92" }, "status": "covered" }, @@ -1677,7 +1690,7 @@ "compiler_ir.seed_stream_map" ], "expected": { - "implementation_sha256": "870f9b43766d4c11f6dc1538bf97e360c6a85dd0d4a4a68c149595dfc0902f0a", + "implementation_sha256": "61db02328d1844c9cb9c490599e33a25aea86eafadc7b940147585340f948d92", "protocol": "legacy-v1", "streams": [ "build_model", @@ -1698,7 +1711,7 @@ }, "failures": [], "observed": { - "implementation_sha256": "870f9b43766d4c11f6dc1538bf97e360c6a85dd0d4a4a68c149595dfc0902f0a", + "implementation_sha256": "61db02328d1844c9cb9c490599e33a25aea86eafadc7b940147585340f948d92", "protocol": "legacy-v1", "streams": [ "build_model", @@ -1732,12 +1745,12 @@ ], "expected": { "relation": "all site fields preserved exactly", - "sites": 53 + "sites": 66 }, "failures": [], "observed": { - "sha256": "a0366cd518f42a240b589af3ef6580aac44ad7ed70969bcc669b365d315146ab", - "sites": 53 + "sha256": "5c22dfb6f3cd7c23c527e054ba6f5040ca2eb4235df8659e3f7c61ebe2ae6c4f", + "sites": 66 }, "status": "covered" }, @@ -1752,12 +1765,12 @@ "compiler_ir.seed_stream_map.sites.owners" ], "expected": { - "bindings": 112, - "coverage": "all 53 sites" + "bindings": 125, + "coverage": "all 66 sites" }, "failures": [], "observed": { - "bindings": 112 + "bindings": 125 }, "status": "covered" }, @@ -1775,12 +1788,12 @@ "legacy_adapter.imputation.late_producer_resource_semantics" ], "expected": { - "sha256": "cd5ba8924d64da5425ee14cca82a774e3f4b2bb5aabe06df291cc3cc457287a9", + "sha256": "b82d911a0263ebc1b6d5d8f82bb1076fd8d779bea49072ec42f02a8fe027cbe3", "stages": 37 }, "failures": [], "observed": { - "sha256": "cd5ba8924d64da5425ee14cca82a774e3f4b2bb5aabe06df291cc3cc457287a9", + "sha256": "b82d911a0263ebc1b6d5d8f82bb1076fd8d779bea49072ec42f02a8fe027cbe3", "stages": 37 }, "status": "covered" @@ -1840,11 +1853,11 @@ "legacy_adapter.stacked_checkpoint_static_components" ], "expected": { - "sha256": "e660a8ce42b69a39d29c5f0ec37264bc69d61b03f27adc386336ec8889531bb2" + "sha256": "406b2cf93a7fb94dc63f1a24a52cfb20362ab83195eb8ebf61d4ef43072280ec" }, "failures": [], "observed": { - "sha256": "e660a8ce42b69a39d29c5f0ec37264bc69d61b03f27adc386336ec8889531bb2" + "sha256": "406b2cf93a7fb94dc63f1a24a52cfb20362ab83195eb8ebf61d4ef43072280ec" }, "status": "covered" }, @@ -1887,7 +1900,7 @@ "alpha", "zeta" ], - "sha256": "b88f2d9c0f6f92c6cd81eb14d6b126afe59577b8bb392b394b2c6fbbafd195c5" + "sha256": "a53da3c26dfaf8f400fb74ef92f79555dd0c152319aa50317b52c5603d0af7b4" }, "failures": [], "observed": { @@ -1911,7 +1924,7 @@ "alpha", "zeta" ], - "sha256": "b88f2d9c0f6f92c6cd81eb14d6b126afe59577b8bb392b394b2c6fbbafd195c5" + "sha256": "a53da3c26dfaf8f400fb74ef92f79555dd0c152319aa50317b52c5603d0af7b4" }, "status": "covered" }, @@ -1975,7 +1988,7 @@ "take_up_contract", "us_qbi_reconciliation_contract" ], - "sha256": "04899daa491e8f089899c9df64cdb2ed44d61d11da2b4c733db6f20f38a1668a" + "sha256": "fbc3631977c43ba558f89b3f36c1b8d0d3958a593f93198c0c5b5833cd18999b" }, "status": "covered" }, @@ -2599,7 +2612,7 @@ "country": "us", "schema_id": "country_spec", "schema_version": 1, - "spec_sha256": "835b4d61de3b153b13c536e371dca6a16fad6d41e003d070f9804c7c251f3590" + "spec_sha256": "ae83c7a32a4a2970070f71b95aa7de771a56ab435f71004c11ee2e6b181fe07b" } }, "report_schema_version": 3, @@ -2609,7 +2622,7 @@ "country": "us", "schema_id": "country_spec", "schema_version": 1, - "spec_sha256": "835b4d61de3b153b13c536e371dca6a16fad6d41e003d070f9804c7c251f3590" + "spec_sha256": "ae83c7a32a4a2970070f71b95aa7de771a56ab435f71004c11ee2e6b181fe07b" }, "status": "pass" } diff --git a/docs/spec-engine.md b/docs/spec-engine.md index 62451739..4a4de604 100644 --- a/docs/spec-engine.md +++ b/docs/spec-engine.md @@ -172,7 +172,7 @@ run_provenance_identity: spec_binding: # {country, schema_id, schema_version, # canonicalizer_version, spec_sha256} authority_versions: # semantic contract version per named - # authority (stacked authority v10 is + # authority (stacked authority v11 is # the live example) — field, bump # rules, and precedence below code_inventory_digest # builder_code_identity + kernel set diff --git a/docs/us-multispine-operator-ordering.md b/docs/us-multispine-operator-ordering.md index 9a0054b8..c8762390 100644 --- a/docs/us-multispine-operator-ordering.md +++ b/docs/us-multispine-operator-ordering.md @@ -246,13 +246,13 @@ by_origin_battery and target checkpoint schema remains version 6. The capital-gains tail manifest uses schema version 2 and binds its support contract and receipt. The canonical stacked authority is version 11, the outer stacked checkpoint - materializer uses version 12, and the stacked pool stage checkpoint - materializer uses version 7. + materializer uses version 13, and the stacked pool stage checkpoint + materializer uses version 8. The outer base identity binds primary-QRF version 6, the ACS universe and QBI reconciliation contracts, the tail schema and support contract, and - late-producer registry schema version 16, including the signed static and + late-producer registry schema version 17, including the signed static and derivation-mode semantics of every virtual DAG resource. The companion pool - manifest uses schema version 9. + manifest uses schema version 10. Older outer authority or materializer payloads are stale; primary-QRF version 6 remains current. @@ -309,7 +309,7 @@ by_origin_battery final publication reject a missing, stale, or reissued authority; NON-CANONICAL test receipts cannot ship. - Outer materializer v10 also embeds one signed resource-semantics row for + Outer materializer v13 also embeds one signed resource-semantics row for every DAG producer. Static configs are exact; donor tables are bound by the declared canonical scalar-content codec; primary and transfer banks name their outer/stage identity derivations; and source receipts name their @@ -879,13 +879,13 @@ The lexically canonical waves have sizes `(1, 1, 17, 14, 3, 2)`: 5. Education; adult-care transfer; WIC transfer. 6. `source_finalizer` and education transfer. -Registry schema version 13 and execution-receipt schema version 3 bind the +Registry schema version 17 and execution-receipt schema version 4 bind the canonical input declarations, outputs, edges, waves, exact kind-specific virtual-resource bindings, content-hashed execution-row schema, and immutable transition authority. The schedule SHA-256 is -`dbae9f945966a58592915780be78137e011d060271af6c933870a55db297baab`; +`4965c1485c283dec3685f4ca82fa469d8b88a85f82ccd6b39e2adc84bc0e94d6`; the full payload SHA-256 is -`95ee19cd1b4d1cf321a32910c234ebc460aa47f9cc30e03fa8560ea6ae5e2eb8`. +`9e72ceed9365ddf8993f68a25037a15cf53375093a60e59d5ce01627a7ceb210`. Reversing registry iteration produces those same bytes. The virtual-resource payload ledger is independently versioned: ACS-universe @@ -911,7 +911,7 @@ and valid. Neither receipt authorizes an upstream null. | PUF raw predictor sources | Every filing-status, count, and income component is observed in its declared source universe. Raw WAGP/SEMP authority is present and agrees with mapped leaves; a cross-grain source collision is rejected. A null on any eligible member fails before coercion. | Structure supplies status/count; ACS-native or ASEC-carried earnings supply earnings; early transfer supplies interest, dividends, and gains. | No. ACS under-15 WAGP/SEMP blanks are an exact source-universe state, not transfer starvation; all other source nulls fail. | | PUF tax-unit features | Every clone-1 recipient has a finite feature vector. Post-aggregation NaN, `+inf`, and `-inf` are counted by named predictor and rejected before fitting; none is coerced or snapped to zero. | Universe-aware person sums plus tax-unit structural inputs. | No. Eligible member values must be complete; the only special case is an all-child unit whose numeric-zero predictor is explicitly owned and counted by the named universe-zero rule. | | Primary QRF banks and chain | Donor/recipient banks are immutable; target order and RNG prefix are contiguous; all targets complete; live recipient identity, source-universe receipt, and feature digest match before finalization. | The processed full PUF donor and strict recipient checkpoint initialized above. | No. Mutation or missing receipt invalidates the bank; it cannot resume under legacy semantics. | -| Outer pool checkpoint identity and resume | Primary-QRF schema v6, tail-manifest schema v2, late-registry schema v14/receipt schema v3, outer stacked materializer v10/authority v9, stacked pool-stage materializer v5, pool manifest schema v7, and the ACS-universe, QBI-mutation, tail-support, late-DAG, and signed virtual-resource-semantics identities must match exactly before any cached stage is discovered. The retiring legacy envelope remains manifest schema v4/materializer v3. | Fresh input pins, live stack receipt, scale controls, code identity, and all semantic contract identities. | No. An older stacked materializer or authority payload is stale; a self-consistent old receipt cannot reopen a checkpoint. Primary-QRF v6 remains current. | +| Outer pool checkpoint identity and resume | Primary-QRF schema v6, tail-manifest schema v2, late-registry schema v17/receipt schema v4, outer stacked materializer v13/authority v11, stacked pool-stage materializer v8, pool manifest schema v10, and the ACS-universe, QBI-mutation, tail-support, late-DAG, and signed virtual-resource-semantics identities must match exactly before any cached stage is discovered. The retiring legacy envelope remains manifest schema v4/materializer v3. | Fresh input pins, live stack receipt, scale controls, code identity, and all semantic contract identities. | No. An older stacked materializer or authority payload is stale; a self-consistent old receipt cannot reopen a checkpoint. Primary-QRF v6 remains current. | | Clone-2 capital-gains tail | Each filing status requires as many eligible recipient households as selected q99.5 donors. Eligibility requires unique single-tax-unit PUF-detail lineage and half-weight capacity for the global maximum assigned donor weight. An adequate status assigns every selected donor once; a thin status skips as a whole with a named, counted `insufficient_support` receipt. | Completed clone-1 QRF output and full PUF tail donors. At 1%, `SINGLE` and `HEAD_OF_HOUSEHOLD` attach, `JOINT` and `SEPARATE` skip, and zero-requirement `SURVIVING_SPOUSE` is `not_applicable`. | No widening or partial attachment is permitted. All 22 AGI bands provide nearest-first fallback only inside a status. Universe-aware PUF recipients remain eligible, including explicitly receipted empty-universe tax units. | | Late producer DAG | Before any callback, all declared inputs are filled on their required scopes or carry an input-specific counted absence receipt; numeric inputs are finite. The exact derived order, readiness rows, once-only source finalizer, and bounded transfer receipts must validate. | ACS earnings-universe materialization, primary PUF/tail, 16 source producers, and 19 bounded transfer groups execute in six derived waves. | No. The refusing producer names the unfilled input and its declared producing stage. A cycle fails at import with its path. | | Late transfer completion | Every declared PUF-clone or ASEC source-producer cell is nonnull; all complementary recipients are filled; the allowed count for both unmodeled and residual rows is zero. | Forty-three PUF and 29 source targets, with two overlaps, supply the 70-target late surface. | No. A missing producer or recipient value is terminal at this boundary. | diff --git a/packages/microcosm-build/src/microcosm/build/spec_engine/field_usage.py b/packages/microcosm-build/src/microcosm/build/spec_engine/field_usage.py index aa6039e3..638a4dd5 100644 --- a/packages/microcosm-build/src/microcosm/build/spec_engine/field_usage.py +++ b/packages/microcosm-build/src/microcosm/build/spec_engine/field_usage.py @@ -26,9 +26,9 @@ ) from .schemas import load_schema_registry -EXPECTED_AUTHORED_FIELD_COUNT = 32_384 -EXPECTED_RESOLVED_BINDING_FIELD_COUNT = 9_770 -EXPECTED_CONFIGURATION_FIELD_COUNT = 42_154 +EXPECTED_AUTHORED_FIELD_COUNT = 32_521 +EXPECTED_RESOLVED_BINDING_FIELD_COUNT = 10_017 +EXPECTED_CONFIGURATION_FIELD_COUNT = 42_538 class FieldUsageError(AssertionError): @@ -249,13 +249,9 @@ def is_stacked_geography_source_identity(pointer: str) -> bool: ) if claim.pointer_class == "stacked_geography_source_identity": - return [ - row for row in rows if is_stacked_geography_source_identity(row[0]) - ] + return [row for row in rows if is_stacked_geography_source_identity(row[0])] if claim.pointer_class == "source_validation": - return [ - row for row in rows if not is_stacked_geography_source_identity(row[0]) - ] + return [row for row in rows if not is_stacked_geography_source_identity(row[0])] raise FieldUsageError(f"{claim.id}: unknown pointer class {claim.pointer_class!r}") @@ -425,12 +421,12 @@ def _path_inventory(rows: Sequence[tuple[str, object]]) -> tuple[int, str]: "e1dd7dc5123ab0f39d08ea4939d98dd09a6fdb8e7449a7ca3125fb1ddbd5b4e9", ), "imputation_producer_graph": ( - 24_488, - "6382a3ed07016414c31462b8bedb24590d92b792405a89c927d85fa1bd2a8347", + 24_554, + "18140dbc554c2a17a673fec5ae5839f504d4b9980c5d8f1c1ccb545342977cb4", ), "imputation_transfer_execution": ( - 97, - "cabaaa3d96c534f5ec37f51811c3f549d0ed20791e2892c868888c0da6d7d80e", + 115, + "545f626e7b9c6aed25a5ae8dd3156c96db2cd35e5637ec1ebe443432d1684277", ), "imputation_waiver_records": ( 70, @@ -457,12 +453,12 @@ def _path_inventory(rows: Sequence[tuple[str, object]]) -> tuple[int, str]: "6a781915fd491d2c4b16d2b7d482f69cf362c904130093c59f9629f7a319269b", ), "resolved_seed_protocol": ( - 824, - "7537385c3fd399a2dbb7dcd8ed7cf1ff2481ed336db621eafcbfd741d5792f40", + 1_032, + "6d4af2d988ee87ed38a044c1007e854176af17be4782e802d86d960f10376f7a", ), "resolved_seed_site_bindings": ( - 277, - "368fc19a8f07c4abf38ebcf2fbc4414d33c403c0f1f9881c2d2e3f5d6160feb6", + 316, + "199ec98045871070860164b522892327ab840b48c0c6a53972ebc0ff0d7d091f", ), "resolved_vintage_authorities": ( 63, @@ -489,8 +485,8 @@ def _path_inventory(rows: Sequence[tuple[str, object]]) -> tuple[int, str]: "d6782c5de5bbed1bdc6bf653c4a6d4aadcad4ccc72d35e1092e130fcb04680a3", ), "source_stages": ( - 1_702, - "be3e24b53955b9dabccfc7e779212b1e97daf0c0992be3c183e391bf695ed4ee", + 1_715, + "78227f34c3f4ef946c06693dec9177dd906069f98016854bfd4d39c4cd4c51fa", ), "spine_assembly_household_mass_shares": ( 2, @@ -509,16 +505,16 @@ def _path_inventory(rows: Sequence[tuple[str, object]]) -> tuple[int, str]: "cf0000464013118955571dac2691d4cc1b900c97c6570349b489678cdf937649", ), "spine_pipeline_contract": ( - 88, - "ae011455154bae0df3913ca9a056058a909d913439b292cb46c11f63c7d0d9a3", + 89, + "8d281f8f1e2f9914684150b0424ff816d0ec09ed0d9053ea5a64e877a86cc35b", ), "spine_sampling": ( 17, "29c6c1b3243e178783e7ab139993ba3a9b42d62edd4e9e4e1f3b28688daf2c6d", ), "spine_seed_site_bindings": ( - 277, - "dfa7ae701f62f7b1f06954798d29cfba3fc53a0be21e1751646d1f5bafd6071f", + 316, + "bb848f38e068858ac23e964a4ffbbf5542253da50515c72fe7753d8c56912486", ), "spine_support_roles": ( 29, @@ -1125,7 +1121,9 @@ def _verify_source_pins(context: _VerificationContext, claim: UsageClaim) -> Non if isinstance(row, Mapping) and isinstance(row.get("id"), str) ] if len(ids) != len(rows) or len(ids) != len(set(ids)): - raise FieldUsageError("source_pins: source ids are not an exact unique registry") + raise FieldUsageError( + "source_pins: source ids are not an exact unique registry" + ) expected_refs: set[tuple[str, str, str]] = set() for index, value in enumerate(rows): @@ -1468,9 +1466,7 @@ def _verify_claim(context: _VerificationContext, claim: UsageClaim) -> None: "legacy": lambda: _verify_legacy(context, claim), "source_pins": lambda: _verify_source_pins(context, claim), "spine_channels": lambda: _verify_spine_channels(context, claim), - "spine_assembly_legacy": lambda: _verify_spine_assembly_legacy( - context, claim - ), + "spine_assembly_legacy": lambda: _verify_spine_assembly_legacy(context, claim), "spine_assembly_validation": lambda: _verify_spine_assembly_validation( context, claim ), diff --git a/packages/microcosm-build/src/microcosm/build/spec_engine/inventory_coverage.py b/packages/microcosm-build/src/microcosm/build/spec_engine/inventory_coverage.py index 6e214e98..6ecd560f 100644 --- a/packages/microcosm-build/src/microcosm/build/spec_engine/inventory_coverage.py +++ b/packages/microcosm-build/src/microcosm/build/spec_engine/inventory_coverage.py @@ -296,6 +296,19 @@ "pregnancy_assignment", "wic_claim_assignment", "snap_discretionary_exemption_assignment", + "immigration_humanitarian_paroled_one_year_afghanistan_assignment", + "immigration_humanitarian_paroled_one_year_ukraine_assignment", + "immigration_humanitarian_paroled_one_year_nicaragua_assignment", + "immigration_humanitarian_paroled_one_year_venezuela_assignment", + "immigration_humanitarian_refugee_assignment", + "immigration_humanitarian_asylee_assignment", + "immigration_humanitarian_deportation_withheld_assignment", + "immigration_humanitarian_tps_venezuela_assignment", + "immigration_humanitarian_tps_el_salvador_assignment", + "immigration_humanitarian_tps_honduras_assignment", + "immigration_humanitarian_tps_nicaragua_assignment", + "immigration_humanitarian_tps_nepal_assignment", + "immigration_humanitarian_tps_other_designated_assignment", "immigration_ead_workers_assignment", "immigration_ead_students_assignment", "ssi_take_up_assignment", @@ -348,20 +361,20 @@ EXPECTED_HASHES = { "acs_group_predictors": "a927bb7ecf3e84f54c93583ab79318654514ac546aefafba67da5285615fbd60", "acs_person_predictors": "878c788a6f037d7aca12b3586ea034eff04f3034ffa11935a736493042551f25", - "authority": "e660a8ce42b69a39d29c5f0ec37264bc69d61b03f27adc386336ec8889531bb2", + "authority": "406b2cf93a7fb94dc63f1a24a52cfb20362ab83195eb8ebf61d4ef43072280ec", "early_families": "4aa9f736fd76e83955477ad1667e58f48f264783f05bdc7f0102cd32d61323bd", - "full_checkpoint": "b88f2d9c0f6f92c6cd81eb14d6b126afe59577b8bb392b394b2c6fbbafd195c5", + "full_checkpoint": "a53da3c26dfaf8f400fb74ef92f79555dd0c152319aa50317b52c5603d0af7b4", "gap_fill_schedule": "1c31f9868f7884347cc19cf1ff65da43f950b9114941a715bab168246db414a7", - "graph_nodes": "271a7bb8d0b3f97ff344e0b7e68184fa74738a6585c24fc8781793db669f388b", + "graph_nodes": "014f90315324f72f45ece928bccc44f34b4d1e95c0582f1d7d73620f482da15f", "geography_assignment": "f49425ca8734ac559c73cf44f6458d86d3162a48956b98a27e6e758959361585", "late_families": "d91f9ff0eb52f43e7b6eed3d5c58c37abe1620c3a11021da15dae9c10e16d382", - "late_resource_semantics": "afebb6725373abf5b8dd4fdb77bf2814cb6fcc569cb606c0c30963a8f65c0bab", - "late_schedule": "e59c019d3d454eac99ac0ac209b6c5b6faaf9bdfcaeee18c36a25be19bf7da2f", + "late_resource_semantics": "db175a952340b3ff2774215f320c4d19592c06b9959442639a6c561055f2442d", + "late_schedule": "88bc9243a3518982ae951c3de21bd55877e296ce4fcb183b9bee420d3a684b10", "ownership": "5f64f0aac49e2313177564f71876bffc8c81b3ded4df701e70930e60e9c98356", "primary_tuples": "987b501c695e31f45521c4a178528f75ab3df22c09bc407b182213b2de99ee57", - "seed_map": "271c23eca777b63bbfecc5b0bdbfffee634e13f79dd37273172f5fadda4afefb", - "seed_protocol": "870f9b43766d4c11f6dc1538bf97e360c6a85dd0d4a4a68c149595dfc0902f0a", - "source_manifest": "cd5ba8924d64da5425ee14cca82a774e3f4b2bb5aabe06df291cc3cc457287a9", + "seed_map": "f5301b13c672bd21cb33e357f0c25c762acab3ef44ca15ed88895ff2ffebc0af", + "seed_protocol": "61db02328d1844c9cb9c490599e33a25aea86eafadc7b940147585340f948d92", + "source_manifest": "b82d911a0263ebc1b6d5d8f82bb1076fd8d779bea49072ec42f02a8fe027cbe3", "take_up": "fa186daea0f8dd641cc470e41d1a2953f887d45282ec990201298f47bedf8d4d", "tail": "ac92829c88a1a4fb6460d61190918d5d99c6c377fc8dd8f62f02b332d09bf59c", } @@ -429,14 +442,14 @@ "primary_targets": 65, "producer_authored_outputs": 92, "producer_compiled_outputs": 227, - "producer_inputs": 2_744, + "producer_inputs": 2_750, "producer_nodes": 38, "producer_virtual_resources": 75, "release_rungs": 5, "resolved_references": 334, - "seed_owner_bindings": 112, + "seed_owner_bindings": 125, "seed_owner_rows": 54, - "seed_sites": 53, + "seed_sites": 66, "seed_streams": 14, "source_operators": 16, "source_stages": 37, @@ -1005,7 +1018,7 @@ def add( "producer_inputs_exact", clauses={ "producer input rows differ": inputs_exact, - "input row count differs": input_count == 2744, + "input row count differs": input_count == 2750, }, homes=("/imputation/producer_graph/nodes/*/inputs",), consumers=( @@ -1013,7 +1026,7 @@ def add( "compiler_ir.node_slices", ), observed={"rows": input_count}, - expected={"rows": 2744, "relation": "source rows preserved exactly"}, + expected={"rows": 2750, "relation": "source rows preserved exactly"}, ) outputs_exact = set(expected_outputs) == set(compiled_by_id) and all( _json_equal( @@ -1695,9 +1708,7 @@ def add( "sha256": _operational_free_sha256(geography_assignment), "authority_roles": sorted(geography_authorities), "target_vintage": _mapping( - geography_authorities.get( - "congressional_district_vintage_crosswalk" - ), + geography_authorities.get("congressional_district_vintage_crosswalk"), "checkpoint geography crosswalk authority", ).get("target_vintage"), }, @@ -1773,7 +1784,7 @@ def add( "seed_site_definitions_exact", clauses={ "seed site definitions differ": site_definitions_exact, - "seed site count differs": len(protocol_sites) == 53, + "seed site count differs": len(protocol_sites) == 66, }, homes=("/bundle/seed_protocol",), consumers=("compiler_ir.seed_stream_map.sites", "compiler_ir.node_slices"), @@ -1781,7 +1792,7 @@ def add( "sites": len(protocol_sites), "sha256": sha256_json([site.to_wire() for site in protocol.sites]), }, - expected={"sites": 53, "relation": "all site fields preserved exactly"}, + expected={"sites": 66, "relation": "all site fields preserved exactly"}, ) binding_by_site = {binding.site: binding for binding in spec.seed_site_bindings} binding_exact = set(binding_by_site) == set(compiled_sites) and all( @@ -1796,7 +1807,7 @@ def add( "owner binding count differs": sum( len(site.owners) for site in compiled_sites.values() ) - == 112, + == 125, "one or more sites have no owner": all( site.owners for site in compiled_sites.values() ), @@ -1806,7 +1817,7 @@ def add( observed={ "bindings": sum(len(site.owners) for site in compiled_sites.values()) }, - expected={"bindings": 112, "coverage": "all 53 sites"}, + expected={"bindings": 125, "coverage": "all 66 sites"}, ) expected_owner_sites: dict[tuple[str, str], list[str]] = {} for site in protocol.sites: @@ -1850,7 +1861,7 @@ def add( "seed groups overlap": groups_disjoint, "seed groups do not cover the protocol exactly": set(grouped_ids) == set(protocol_sites), - "seed group cardinality differs": len(grouped_ids) == 53, + "seed group cardinality differs": len(grouped_ids) == 66, }, homes=("/bundle/seed_protocol", "/spine/seed_site_bindings"), consumers=("compiler_ir.seed_stream_map",), @@ -1860,7 +1871,7 @@ def add( }, expected={ "groups": len(EXPECTED_SEED_GROUPS), - "sites": 53, + "sites": 66, "partition": "disjoint and exhaustive", }, ) diff --git a/packages/microcosm-build/src/microcosm/build/spec_engine/schema/imputation.schema.json b/packages/microcosm-build/src/microcosm/build/spec_engine/schema/imputation.schema.json index 4a4a09a5..dad932d2 100644 --- a/packages/microcosm-build/src/microcosm/build/spec_engine/schema/imputation.schema.json +++ b/packages/microcosm-build/src/microcosm/build/spec_engine/schema/imputation.schema.json @@ -339,6 +339,22 @@ "immigration_status_model_target": { "type": "string" }, + "immigration_required_predictors": { + "type": "array", + "items": { + "type": "string" + }, + "minItems": 1, + "uniqueItems": true + }, + "immigration_baseline_excluded_statuses": { + "type": "array", + "items": { + "type": "string" + }, + "minItems": 1, + "uniqueItems": true + }, "immigration_status_targets": { "type": "array", "items": { @@ -397,6 +413,62 @@ ] } }, + "required": [ + "activation", + "contract" + ] + }, + "humanitarian_immigration": { + "type": "object", + "additionalProperties": false, + "properties": { + "activation": { + "type": "object", + "additionalProperties": false, + "properties": { + "all_targets": { + "type": "array", + "items": { + "type": "string" + } + } + }, + "required": [ + "all_targets" + ] + }, + "contract": { + "type": "object", + "additionalProperties": false, + "properties": { + "candidate_shortfall": { + "type": "string" + }, + "mutable_rows": { + "type": "string" + }, + "selection": { + "type": "string" + }, + "target_basis": { + "type": "string" + }, + "targets": { + "type": "array", + "items": { + "type": "string" + } + } + }, + "required": [ + "candidate_shortfall", + "mutable_rows", + "selection", + "target_basis", + "targets" + ] + } + }, "required": [ "activation", "contract" @@ -526,6 +598,7 @@ }, "required": [ "adult_care", + "humanitarian_immigration", "schedule_d_capital_gain_distributions" ] }, @@ -776,6 +849,8 @@ "donor_combined_components", "group_optional_names", "housing", + "immigration_baseline_excluded_statuses", + "immigration_required_predictors", "immigration_status_model_target", "immigration_status_targets", "post_transfer_features", diff --git a/packages/microcosm-build/src/microcosm/build/spec_engine/schema/sources.schema.json b/packages/microcosm-build/src/microcosm/build/spec_engine/schema/sources.schema.json index c6f16826..ff05c8a2 100644 --- a/packages/microcosm-build/src/microcosm/build/spec_engine/schema/sources.schema.json +++ b/packages/microcosm-build/src/microcosm/build/spec_engine/schema/sources.schema.json @@ -1594,9 +1594,270 @@ "source", "target" ] + }, + "humanitarian_status_stocks": { + "type": "object", + "additionalProperties": false, + "properties": { + "asylee": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + }, + "deportation_withheld": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + }, + "paroled_one_year": { + "type": "object", + "additionalProperties": false, + "properties": { + "afghanistan": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + }, + "nicaragua": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + }, + "ukraine": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + }, + "venezuela": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + } + }, + "required": [ + "afghanistan", + "nicaragua", + "ukraine", + "venezuela" + ] + }, + "refugee": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + }, + "tps": { + "type": "object", + "additionalProperties": false, + "properties": { + "el_salvador": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + }, + "honduras": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + }, + "nepal": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + }, + "nicaragua": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + }, + "other_designated": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + }, + "venezuela": { + "type": "object", + "additionalProperties": false, + "properties": { + "source": { + "type": "string", + "x-spec-surface": "operational" + }, + "target": { + "type": "number" + } + }, + "required": [ + "source", + "target" + ] + } + }, + "required": [ + "el_salvador", + "honduras", + "nepal", + "nicaragua", + "other_designated", + "venezuela" + ] + } + }, + "required": [ + "asylee", + "deportation_withheld", + "paroled_one_year", + "refugee", + "tps" + ] } }, "required": [ + "humanitarian_status_stocks", "kind", "seed_from_build_config", "time_period_from_build_config", diff --git a/packages/microcosm-build/src/microcosm/build/spec_engine/schema/spine.schema.json b/packages/microcosm-build/src/microcosm/build/spec_engine/schema/spine.schema.json index dc4c7955..251360cf 100644 --- a/packages/microcosm-build/src/microcosm/build/spec_engine/schema/spine.schema.json +++ b/packages/microcosm-build/src/microcosm/build/spec_engine/schema/spine.schema.json @@ -288,7 +288,7 @@ "const": "populace_us_stacked_pool_checkpoint_identity" }, "schema_version": {"const": 1}, - "materializer_version": {"const": 12}, + "materializer_version": {"const": 13}, "pipeline": {"const": "us-stacked-pool"} } }, diff --git a/packages/microcosm-build/src/microcosm/build/spec_engine/seeds.py b/packages/microcosm-build/src/microcosm/build/spec_engine/seeds.py index 04fb87ad..981b2a63 100644 --- a/packages/microcosm-build/src/microcosm/build/spec_engine/seeds.py +++ b/packages/microcosm-build/src/microcosm/build/spec_engine/seeds.py @@ -742,6 +742,31 @@ def _stable_site( ), candidate_universe="all_person_rows_before_abawd_eligibility_mask", ), + *( + _stable_site( + f"immigration_humanitarian_{label.replace(':', '_')}_assignment", + salt=f"immigration:{label}", + key_grammar=("source_year:source_person_id_if_present", "else_person_id"), + candidate_universe=( + "all_person_rows_then_category_origin_window_pool_mask" + ), + ) + for label in ( + "paroled_one_year:afghanistan", + "paroled_one_year:ukraine", + "paroled_one_year:nicaragua", + "paroled_one_year:venezuela", + "refugee", + "asylee", + "deportation_withheld", + "tps:venezuela", + "tps:el_salvador", + "tps:honduras", + "tps:nicaragua", + "tps:nepal", + "tps:other_designated", + ) + ), _stable_site( "immigration_ead_workers_assignment", salt="immigration:ead_workers", diff --git a/packages/microcosm-build/src/microcosm/build/us/release_input_coverage_manifest.json b/packages/microcosm-build/src/microcosm/build/us/release_input_coverage_manifest.json index b352ee60..7d9ef8cd 100644 --- a/packages/microcosm-build/src/microcosm/build/us/release_input_coverage_manifest.json +++ b/packages/microcosm-build/src/microcosm/build/us/release_input_coverage_manifest.json @@ -1414,6 +1414,103 @@ "parameter_changes": {}, "period": 2024, "reason": "PolicyEngine-US multiplies HUD HAP by the restored SPM-unit take-up leaf after eligibility. Microcosm keeps that leaf exactly equal to source-backed housing-assistance receipt, so neutralizing it must remove the assistance paid to measured/imputed recipients. A 6,000-household production-ingredient smoke scored $202.795 million baseline-minus-neutralized; the $100 million floor is below that observed subset effect but far above numerical noise. A default-only or absent carry makes the source reconciliation or this uniquely isolating probe fail." + }, + { + "binding_inputs": [ + "immigration_status_str" + ], + "budget_measure": "medicaid", + "effect_direction": "reform_minus_baseline", + "expected_sign": "positive", + "id": "hr1_medicaid_humanitarian_eligibility_restoration", + "issue": "PolicyEngine/microcosm#767", + "min_abs_effect": 50000000.0, + "name": "H.R.1 SS71109 Medicaid humanitarian-status restoration", + "parameter_changes": { + "gov.hhs.medicaid.eligibility.eligible_immigration_statuses": { + "2026-10-01.2100-12-31": [ + "CITIZEN", + "LEGAL_PERMANENT_RESIDENT", + "REFUGEE", + "ASYLEE", + "DEPORTATION_WITHHELD", + "CUBAN_HAITIAN_ENTRANT", + "CONDITIONAL_ENTRANT", + "PAROLED_ONE_YEAR" + ] + } + }, + "period": 2027, + "reason": "H.R.1 SS71109 narrows the Medicaid qualified-immigrant list to citizens, LPRs, and Cuban/Haitian entrants from 2026-10-01; the probe restores the pre-narrowing list at 2027 law, which binds only through the REFUGEE/ASYLEE/DEPORTATION_WITHHELD/PAROLED_ONE_YEAR values of immigration_status_str. The source stage draws roughly 160k refugees (74% reporting Medicaid on the 2024 ASEC), 155k asylees, and 405k parolees to cited DHS/OHSS stocks, so restoring their eligibility must re-enroll anchored takers and move person-level medicaid dollars by far more than the floor. A ~$0 score means the humanitarian statuses regressed to zero records (microcosm #767's silent-zero failure) or the engine channel broke." + }, + { + "binding_inputs": [ + "immigration_status_str" + ], + "budget_measure": "aca_ptc", + "effect_direction": "reform_minus_baseline", + "expected_sign": "positive", + "id": "hr1_aca_below_fpl_exception_restoration", + "issue": "PolicyEngine/microcosm#767", + "min_abs_effect": 5000000.0, + "name": "H.R.1 SS71302 ACA below-FPL immigrant exception restoration", + "parameter_changes": { + "gov.aca.below_fpl_immigration_exception_in_effect": { + "2026-01-01.2026-12-31": true + } + }, + "period": 2026, + "reason": "H.R.1 SS71302 repeals the 26 USC 36B(c)(1)(B) below-FPL lawfully-present exception from 2026. Restoring it for 2026 binds, at 2026 law, only through tax units containing a non-citizen who is ACA-lawfully-present yet Medicaid-ineligible by status - on this file exactly the TPS population the source stage draws to the CRS RS20844 per-country stocks (865k weighted persons; the other humanitarian classes remain Medicaid-qualified at 2026 annual parameters). Below-poverty TPS units with marketplace take-up must regain premium tax credits above the floor; ~$0 means the TPS records vanished or the exception channel broke." + }, + { + "binding_inputs": [ + "immigration_status_str" + ], + "budget_measure": "aca_ptc", + "effect_direction": "reform_minus_baseline", + "expected_sign": "positive", + "id": "hr1_aca_lawful_presence_restoration", + "issue": "PolicyEngine/microcosm#767", + "min_abs_effect": 10000000.0, + "name": "H.R.1 SS71301 ACA lawful-presence list restoration", + "parameter_changes": { + "gov.aca.ineligible_immigration_statuses": { + "2027-01-01.2100-12-31": [ + "DACA", + "UNDOCUMENTED" + ] + } + }, + "period": 2027, + "reason": "H.R.1 SS71301 adds TPS/REFUGEE/ASYLEE/DEPORTATION_WITHHELD/PAROLED_ONE_YEAR to the ACA ineligible-status list from 2027-01-01. The probe restores the pre-H.R.1 list (DACA and UNDOCUMENTED only) at 2027, re-qualifying every humanitarian status the source stage draws (~1.6M weighted persons across parole, refugee, asylee, and TPS). Units among them with marketplace take-up must regain premium tax credits above the floor; ~$0 means the humanitarian statuses regressed or the lawful-presence channel broke." + }, + { + "binding_inputs": [ + "immigration_status_str" + ], + "budget_measure": "snap", + "effect_direction": "reform_minus_baseline", + "expected_sign": "positive", + "id": "hr1_snap_humanitarian_eligibility_restoration", + "issue": "PolicyEngine/microcosm#767", + "min_abs_effect": 25000000.0, + "name": "H.R.1 SS10108 SNAP humanitarian-status restoration", + "parameter_changes": { + "gov.usda.snap.eligibility.eligible_immigration_statuses": { + "2025-07-01.2100-12-31": [ + "CITIZEN", + "LEGAL_PERMANENT_RESIDENT", + "REFUGEE", + "ASYLEE", + "DEPORTATION_WITHHELD", + "CUBAN_HAITIAN_ENTRANT", + "CONDITIONAL_ENTRANT", + "PAROLED_ONE_YEAR" + ] + } + }, + "period": 2026, + "reason": "H.R.1 SS10108 narrows SNAP alien eligibility to citizens, LPRs, Cuban/Haitian entrants, and COFA citizens from 2025-07-01 (modeled 2025-07). Restoring the pre-H.R.1 list at 2026 re-includes the REFUGEE/ASYLEE/DEPORTATION_WITHHELD/PAROLED_ONE_YEAR members the source stage draws to cited DHS/OHSS stocks; is_snap_excluded_member stops excluding them, so SNAP allotments for their households must rise above the floor. ~$0 means the humanitarian statuses regressed to zero records or the SNAP immigration channel broke." } ], "schema_version": 1, diff --git a/packages/microcosm-build/src/microcosm/build/us/source_stages.json b/packages/microcosm-build/src/microcosm/build/us/source_stages.json index 5304c435..401468cb 100644 --- a/packages/microcosm-build/src/microcosm/build/us/source_stages.json +++ b/packages/microcosm-build/src/microcosm/build/us/source_stages.json @@ -1866,20 +1866,20 @@ { "stage": "immigration_status", "survey": "CPS ASEC + published unauthorized-population estimates", - "source": "https://www.pewresearch.org/short-reads/2024/07/22/what-we-know-about-unauthorized-immigrants-living-in-the-us/", + "source": "https://www.pewresearch.org/race-and-ethnicity/2025/08/21/u-s-unauthorized-immigrant-population-reached-a-record-14-million-in-2023/", "grain": "person", "artifacts": [ { "kind": "public_microdata", "format": "census_asec", "vintage": "build_year", - "locator": "Census CPS ASEC person file citizenship (PRCITSHP), entry year (PEINUSYR), nativity (PENATVTY), and program-participation fields" + "locator": "Census CPS ASEC person file citizenship (PRCITSHP), entry year (PEINUSYR), nativity (PENATVTY), labor-force status (A_LFSR), and program-participation fields" }, { "kind": "administrative_table", "format": "published_estimate", "vintage": "latest_available", - "locator": "Pew Research Center unauthorized-immigrant population and worker estimates; Higher Ed Immigration Portal undocumented-student estimates" + "locator": "Pew Research Center unauthorized-immigrant population and worker estimates; Higher Ed Immigration Portal undocumented-student estimates; OHSS refugee and asylee annual flow reports; DHS OAW/U4U parole reports and CBP CHNV releases; CRS RS20844 TPS designations table; EOIR adjudication statistics via CRS IN12501" } ], "operations": [ @@ -1893,16 +1893,74 @@ "seed_from_build_config": true, "time_period_from_build_config": true, "undocumented_workers": { - "target": 8300000, - "source": "https://www.pewresearch.org/short-reads/2024/07/22/what-we-know-about-unauthorized-immigrants-living-in-the-us/" + "target": 9700000, + "source": "https://www.pewresearch.org/race-and-ethnicity/2025/08/21/u-s-unauthorized-immigrant-population-reached-a-record-14-million-in-2023/" }, "undocumented_students": { "target": 408000, "source": "https://www.higheredimmigrationportal.org/research/undocumented-students-in-higher-education-updated-march-2021/" }, "undocumented_population_anchor": { - "value": 11000000, - "source": "https://www.pewresearch.org/short-reads/2024/07/22/what-we-know-about-unauthorized-immigrants-living-in-the-us/" + "value": 14000000, + "source": "https://www.pewresearch.org/race-and-ethnicity/2025/08/21/u-s-unauthorized-immigrant-population-reached-a-record-14-million-in-2023/" + }, + "humanitarian_status_stocks": { + "paroled_one_year": { + "afghanistan": { + "target": 73566, + "source": "https://www.dhs.gov/sites/default/files/2023-06/PLCY%20-%20AFG%202503%20Update%202_0.pdf" + }, + "ukraine": { + "target": 158000, + "source": "https://www.dhs.gov/sites/default/files/2024-12/2024_1104_dmo_plcy_uniting_for_ukraine_process_overview_and_assessment.pdf" + }, + "nicaragua": { + "target": 93070, + "source": "https://www.cbp.gov/newsroom/national-media-release/cbp-releases-december-2024-monthly-update" + }, + "venezuela": { + "target": 117330, + "source": "https://www.cbp.gov/newsroom/national-media-release/cbp-releases-december-2024-monthly-update" + } + }, + "refugee": { + "target": 160000, + "source": "https://ohss.dhs.gov/topics/immigration/refugees/annual-flow-report/fy-24-refugees-flow-report" + }, + "asylee": { + "target": 155000, + "source": "https://ohss.dhs.gov/topics/immigration/asylees/annual-flow-report/fy24-asylees-flow-report" + }, + "deportation_withheld": { + "target": 0, + "source": "https://crsreports.congress.gov/product/pdf/IN/IN12501" + }, + "tps": { + "venezuela": { + "target": 605015, + "source": "https://www.everycrsreport.com/reports/RS20844.html" + }, + "el_salvador": { + "target": 170125, + "source": "https://www.everycrsreport.com/reports/RS20844.html" + }, + "honduras": { + "target": 51225, + "source": "https://www.everycrsreport.com/reports/RS20844.html" + }, + "nicaragua": { + "target": 2910, + "source": "https://www.everycrsreport.com/reports/RS20844.html" + }, + "nepal": { + "target": 7160, + "source": "https://www.everycrsreport.com/reports/RS20844.html" + }, + "other_designated": { + "target": 21215, + "source": "https://www.everycrsreport.com/reports/RS20844.html" + } + } } } ], @@ -1910,7 +1968,7 @@ "ssn_card_type", "immigration_status_str" ], - "notes": "Citizenship is measured (PRCITSHP), not imputed. Non-citizens with any ASEC-UA legal-status indicator (Van Hook et al., SSRN 4662801) become OTHER_NON_CITIZEN. The CPS carries no work-authorization variable, so among residual non-citizens, workers and students spill to NON_CITIZEN_VALID_EAD in deterministic seeded order until the remaining undocumented worker count matches Pew's 8.3M and the undocumented student count matches the Higher Ed Immigration Portal's ~408k \u2014 the only forced margins. The total undocumented population is emergent (\u224813M on the 2024 ASEC, within the range of published 2023-24 estimates) and is gated against Pew's 11.0M 2022 anchor with a coarse plausibility band rather than forced; reconciling the level is the calibration lane's job. Immigration-status tags carry only statutory tests the data supports: DACA (arrival cohort among EAD holders) and CUBAN_HAITIAN_ENTRANT (nativity plus post-1980 arrival); other documented non-citizens stay LEGAL_PERMANENT_RESIDENT, the modal true status \u2014 blanket REFUGEE/TPS labels are deliberately not fabricated because they would mislabel millions and over-grant refugee-class benefit exemptions." + "notes": "Citizenship is measured (PRCITSHP), not imputed. Non-citizens with any ASEC-UA legal-status indicator (Van Hook et al., SSRN 4662801) become OTHER_NON_CITIZEN. Humanitarian/temporary-protection statuses (microcosm #767) are drawn between the indicators and the EAD spill, sequentially and mutually exclusively in decreasing order of target precision \u2014 PAROLED_ONE_YEAR per program origin (OAW Afghans 73,566 as of 2022-03-31; U4U Ukrainians over 158,000 as of 2023-09-30; CHNV Nicaraguans 93,070 and Venezuelans 117,330 paroled through 2024-12), REFUGEE (FY2023+FY2024 admissions, 60,050+100,060 OHSS \u2014 refugees must apply for LPR adjustment after one year, so the not-yet-adjusted stock is about the trailing two years), ASYLEE (FY2022-FY2024 grants 35,080+51,410+68,060=154,550 OHSS, rounded to 155,000; 88% of 2015-22 affirmative adult grantees had adjusted by end-2024, so about three trailing years), then TPS per country (CRS RS20844 Table 1 as of 2025-03-31, with required-arrival cutoffs binding at PEINUSYR granularity: El Salvador <=2001, Honduras/Nicaragua <=1999, Nepal <=2015). Candidates are restricted by program origin countries (Census Appendix I PENATVTY codes) and arrival windows; Cuba/Haiti-born are excluded everywhere (the CUBAN_HAITIAN_ENTRANT class is the better statutory label and keeps H.R.1 eligibility \u2014 Haiti TPS 330,735 rides that label), the DACA statutory cohort is excluded (dual TPS/DACA holders stay DACA), and Ukraine/Afghanistan TPS registrants (101,150/8,105) are carried by the parole draw since they are overwhelmingly the same U4U/OAW parolees. Parole targets are gross admissions: cohort members who later won asylum keep the parole label, which is channel-equivalent under H.R.1 (both classes lose Medicaid/SNAP/PTC on identical dates). DEPORTATION_WITHHELD carries an explicit zero target: EOIR grants about 2.2k withholdings a year (CRS IN12501: under 1% of 666,177 FY2024 decisions) and publishes no stock, so the accumulated population is below the CPS's resolution for a defensible draw. CONDITIONAL_ENTRANT is deliberately not emitted (the INA 203(a)(7) class closed in 1980). REFUGEE/ASYLEE draw from the indicator-documented pool only; parole/TPS may draw residual-pool people, who then receive NON_CITIZEN_VALID_EAD SSN cards (both programs confer employment authorization). Pew's 2025 report supersedes its earlier series and estimates a coherent 2023 pair of 14.0M unauthorized immigrants and 9.7M in the labor force. Its modeled universe is broader than the engine's UNDOCUMENTED enum: it retains DACA recipients, parolees, TPS holders, similar temporary protections, and the residual-EAD subset of the engine's broad CUBAN_HAITIAN_ENTRANT class, while excluding refugees, people already granted asylum, and assumed-documented Cuban/Haitian entrants. The worker EAD spill binds that modeled universe to 9.7M using actual ASEC labor-force status (A_LFSR codes 1-4 at ages 16+) restored from the same pinned official ASEC archives by exact PERIDNUM join, never prior-year WSAL_VAL/SEMP_VAL earnings. The student spill binds the corresponding broad residual universe to the Higher Ed Immigration Portal's ~408k students. The total modeled Pew-defined unauthorized population is emergent and gated against the 14.0M 2023 anchor with a coarse band; per-category humanitarian masses are gated against their cited targets with a coarse band (the ASEC undercovers 2022-24 arrivals \u2014 the Ukraine parole pool saturates below its admin count); reconciling levels is the calibration lane's job. DACA (arrival cohort among EAD holders) and CUBAN_HAITIAN_ENTRANT (nativity plus post-1980 arrival) keep their statutory tests; other documented non-citizens stay LEGAL_PERMANENT_RESIDENT, the modal true status." }, { "stage": "hours_worked", diff --git a/packages/microcosm-build/src/microcosm/build/us/spec/imputation.yaml b/packages/microcosm-build/src/microcosm/build/us/spec/imputation.yaml index 16ae791a..1548aca4 100644 --- a/packages/microcosm-build/src/microcosm/build/us/spec/imputation.yaml +++ b/packages/microcosm-build/src/microcosm/build/us/spec/imputation.yaml @@ -342,7 +342,11 @@ waiver_records: - dependent_child_status - pregnancy_status transfer_execution: - schema_version: 2 + schema_version: 3 + immigration_required_predictors: + - __acs_transfer_is_us_citizen + - __acs_transfer_birth_country_code + - __acs_transfer_arrival_year group_optional_names: __acs_transfer_employment_income: __acs_transfer_group_employment_income_sum __acs_transfer_interest_dividend_rental_income: __acs_transfer_group_investment_income_sum @@ -405,6 +409,14 @@ transfer_execution: - ssn_card_type - immigration_status_str immigration_status_model_target: __acs_transfer_immigration_status_pair + immigration_baseline_excluded_statuses: + - ASYLEE + - CUBAN_HAITIAN_ENTRANT + - DACA + - DEPORTATION_WITHHELD + - PAROLED_ONE_YEAR + - REFUGEE + - TPS discrete_numeric_targets: - first_home_mortgage_origination_year - second_home_mortgage_origination_year @@ -450,6 +462,19 @@ transfer_execution: tax_unit_role: tax_unit_role_input tax_unit_link: person_tax_unit_id mutable_rows: newly_imputed_expense_cells_only + humanitarian_immigration: + activation: + all_targets: + - ssn_card_type + - immigration_status_str + contract: + targets: + - ssn_card_type + - immigration_status_str + mutable_rows: newly_imputed_paired_cells_only + target_basis: manifest_residual_after_immutable_asec_mass + selection: stable_person_hash_in_manifest_draw_order + candidate_shortfall: error schedule_d_capital_gain_distributions: activation: derive_schedule_d: true @@ -2776,7 +2801,7 @@ families: output_coverage_scope: whole_pool producer_graph: graph_schema_version: 2 - schedule_payload_schema_version: 16 + schedule_payload_schema_version: 17 external_stages: - post_clone_input_surface scope_coverage: @@ -9516,6 +9541,15 @@ producer_graph: producing_stage: post_clone_input_surface required_scope: asec_source tolerated_absence_receipts: [] + - alternatives: + - - column: A_LFSR + entity: person + value_kind: finite_numeric + column: '@effective:raw_person:A_LFSR' + entity: person + producing_stage: post_clone_input_surface + required_scope: asec_source + tolerated_absence_receipts: [] - alternatives: - - column: A_MARITL entity: person @@ -9669,15 +9703,6 @@ producer_graph: producing_stage: post_clone_input_surface required_scope: asec_source tolerated_absence_receipts: [] - - alternatives: - - - column: SEMP_VAL - entity: person - value_kind: finite_numeric - column: '@effective:raw_person:SEMP_VAL' - entity: person - producing_stage: post_clone_input_surface - required_scope: asec_source - tolerated_absence_receipts: [] - alternatives: - - column: SPM_CAPHOUSESUB entity: person @@ -9705,15 +9730,6 @@ producer_graph: producing_stage: post_clone_input_surface required_scope: asec_source tolerated_absence_receipts: [] - - alternatives: - - - column: WSAL_VAL - entity: person - value_kind: finite_numeric - column: '@effective:raw_person:WSAL_VAL' - entity: person - producing_stage: post_clone_input_surface - required_scope: asec_source - tolerated_absence_receipts: [] - alternatives: - - column: '@resolved_weight' entity: person @@ -9808,6 +9824,15 @@ producer_graph: - - column: person_id entity: person value_kind: non_null + - - column: source_household_id + entity: person + value_kind: non_null + - column: source_person_id + entity: person + value_kind: non_null + - column: source_year + entity: person + value_kind: non_null - - column: source_person_id entity: person value_kind: non_null @@ -24952,6 +24977,84 @@ producer_graph: producing_stage: post_clone_input_surface required_scope: whole_pool tolerated_absence_receipts: [] + - alternatives: + - - column: YOEP + entity: person + value_kind: column_present + column: '@effective:immigration_acs_arrival' + entity: person + producing_stage: post_clone_input_surface + required_scope: acs_source + tolerated_absence_receipts: [] + - alternatives: + - - column: CIT + entity: person + value_kind: finite_numeric + column: '@effective:immigration_acs_citizenship' + entity: person + producing_stage: post_clone_input_surface + required_scope: acs_source + tolerated_absence_receipts: [] + - alternatives: + - - column: POBP + entity: person + value_kind: finite_numeric + column: '@effective:immigration_acs_origin' + entity: person + producing_stage: post_clone_input_surface + required_scope: acs_source + tolerated_absence_receipts: [] + - alternatives: + - - column: PEINUSYR + entity: person + value_kind: finite_numeric + column: '@effective:immigration_asec_arrival' + entity: person + producing_stage: post_clone_input_surface + required_scope: asec_source + tolerated_absence_receipts: [] + - alternatives: + - - column: PRCITSHP + entity: person + value_kind: finite_numeric + column: '@effective:immigration_asec_citizenship' + entity: person + producing_stage: post_clone_input_surface + required_scope: asec_source + tolerated_absence_receipts: [] + - alternatives: + - - column: PENATVTY + entity: person + value_kind: finite_numeric + column: '@effective:immigration_asec_origin' + entity: person + producing_stage: post_clone_input_surface + required_scope: asec_source + tolerated_absence_receipts: [] + - alternatives: + - - column: person_id + entity: person + value_kind: non_null + - - column: source_household_id + entity: person + value_kind: non_null + - column: source_person_id + entity: person + value_kind: non_null + - column: source_year + entity: person + value_kind: non_null + - - column: source_person_id + entity: person + value_kind: non_null + - column: source_year + entity: person + value_kind: non_null + column: '@effective:immigration_stable_person_lineage' + entity: person + producing_stage: post_clone_input_surface + required_scope: whole_pool + tolerated_absence_receipts: [] - alternatives: - - column: is_female entity: person @@ -25271,7 +25374,7 @@ producer_graph: - alternatives: - - column: person_id entity: person - value_kind: finite_numeric + value_kind: non_null column: person_id entity: person producing_stage: primary_puf_qrf @@ -33375,10 +33478,10 @@ producer_graph: - first_home_mortgage_origination_year - health_savings_account_ald execution_receipt_contract: - version: 3 + version: 4 row_binding: declared_globally_reconciled_input_and_scope_exact_output_source_and_primary_callback_resource_receipt_and_previous_execution_sha256 virtual_resource_binding: exact_kind_specific_semantic_payload_and_sha256 - top_binding: entry_and_output_frame_sha256_execution_chain_source_completion_and_nineteen_transfer_groups + top_binding: entry_and_output_frame_sha256_execution_chain_source_completion_nineteen_transfer_groups_and_constrained_immigration_reconciliation transition_authority: authority_id: us_stacked_late_producer_transition metadata_key: us_late_producer_transition_authority diff --git a/packages/microcosm-build/src/microcosm/build/us/spec/sources.yaml b/packages/microcosm-build/src/microcosm/build/us/spec/sources.yaml index a160533f..72347037 100644 --- a/packages/microcosm-build/src/microcosm/build/us/spec/sources.yaml +++ b/packages/microcosm-build/src/microcosm/build/us/spec/sources.yaml @@ -99,7 +99,7 @@ sources: stage_asset: id: source_stages path: microcosm.build.us/source_stages.json - sha256: dc58a0d700f0add7b658cec774df6e9587303beb58a1f432a35a18dcd1ac4097 + sha256: 788b815f748abb2061c41efb0cec4cc4952435dbcb4448ecf21ccac85a5ccec2 stage_manifest: version: 1 country: us @@ -1884,19 +1884,21 @@ stages: formula-owned employment_income_last_year; the signed self-employment leaf and availability flag persist. - stage: immigration_status survey: CPS ASEC + published unauthorized-population estimates - source: https://www.pewresearch.org/short-reads/2024/07/22/what-we-know-about-unauthorized-immigrants-living-in-the-us/ + source: https://www.pewresearch.org/race-and-ethnicity/2025/08/21/u-s-unauthorized-immigrant-population-reached-a-record-14-million-in-2023/ grain: person artifacts: - kind: public_microdata format: census_asec vintage: build_year locator: Census CPS ASEC person file citizenship (PRCITSHP), entry year (PEINUSYR), nativity (PENATVTY), - and program-participation fields + labor-force status (A_LFSR), and program-participation fields - kind: administrative_table format: published_estimate vintage: latest_available locator: Pew Research Center unauthorized-immigrant population and worker estimates; Higher Ed Immigration - Portal undocumented-student estimates + Portal undocumented-student estimates; OHSS refugee and asylee annual flow reports; DHS OAW/U4U + parole reports and CBP CHNV releases; CRS RS20844 TPS designations table; EOIR adjudication statistics + via CRS IN12501 operations: - kind: read_table table: person @@ -1905,29 +1907,96 @@ stages: seed_from_build_config: true time_period_from_build_config: true undocumented_workers: - target: 8300000 - source: https://www.pewresearch.org/short-reads/2024/07/22/what-we-know-about-unauthorized-immigrants-living-in-the-us/ + target: 9700000 + source: https://www.pewresearch.org/race-and-ethnicity/2025/08/21/u-s-unauthorized-immigrant-population-reached-a-record-14-million-in-2023/ undocumented_students: target: 408000 source: https://www.higheredimmigrationportal.org/research/undocumented-students-in-higher-education-updated-march-2021/ undocumented_population_anchor: - value: 11000000 - source: https://www.pewresearch.org/short-reads/2024/07/22/what-we-know-about-unauthorized-immigrants-living-in-the-us/ + value: 14000000 + source: https://www.pewresearch.org/race-and-ethnicity/2025/08/21/u-s-unauthorized-immigrant-population-reached-a-record-14-million-in-2023/ + humanitarian_status_stocks: + paroled_one_year: + afghanistan: + target: 73566 + source: https://www.dhs.gov/sites/default/files/2023-06/PLCY%20-%20AFG%202503%20Update%202_0.pdf + ukraine: + target: 158000 + source: https://www.dhs.gov/sites/default/files/2024-12/2024_1104_dmo_plcy_uniting_for_ukraine_process_overview_and_assessment.pdf + nicaragua: + target: 93070 + source: https://www.cbp.gov/newsroom/national-media-release/cbp-releases-december-2024-monthly-update + venezuela: + target: 117330 + source: https://www.cbp.gov/newsroom/national-media-release/cbp-releases-december-2024-monthly-update + refugee: + target: 160000 + source: https://ohss.dhs.gov/topics/immigration/refugees/annual-flow-report/fy-24-refugees-flow-report + asylee: + target: 155000 + source: https://ohss.dhs.gov/topics/immigration/asylees/annual-flow-report/fy24-asylees-flow-report + deportation_withheld: + target: 0 + source: https://crsreports.congress.gov/product/pdf/IN/IN12501 + tps: + venezuela: + target: 605015 + source: https://www.everycrsreport.com/reports/RS20844.html + el_salvador: + target: 170125 + source: https://www.everycrsreport.com/reports/RS20844.html + honduras: + target: 51225 + source: https://www.everycrsreport.com/reports/RS20844.html + nicaragua: + target: 2910 + source: https://www.everycrsreport.com/reports/RS20844.html + nepal: + target: 7160 + source: https://www.everycrsreport.com/reports/RS20844.html + other_designated: + target: 21215 + source: https://www.everycrsreport.com/reports/RS20844.html outputs: - ssn_card_type - immigration_status_str notes: 'Citizenship is measured (PRCITSHP), not imputed. Non-citizens with any ASEC-UA legal-status - indicator (Van Hook et al., SSRN 4662801) become OTHER_NON_CITIZEN. The CPS carries no work-authorization - variable, so among residual non-citizens, workers and students spill to NON_CITIZEN_VALID_EAD in deterministic - seeded order until the remaining undocumented worker count matches Pew''s 8.3M and the undocumented - student count matches the Higher Ed Immigration Portal''s ~408k — the only forced margins. The total - undocumented population is emergent (≈13M on the 2024 ASEC, within the range of published 2023-24 - estimates) and is gated against Pew''s 11.0M 2022 anchor with a coarse plausibility band rather than - forced; reconciling the level is the calibration lane''s job. Immigration-status tags carry only statutory - tests the data supports: DACA (arrival cohort among EAD holders) and CUBAN_HAITIAN_ENTRANT (nativity - plus post-1980 arrival); other documented non-citizens stay LEGAL_PERMANENT_RESIDENT, the modal true - status — blanket REFUGEE/TPS labels are deliberately not fabricated because they would mislabel millions - and over-grant refugee-class benefit exemptions.' + indicator (Van Hook et al., SSRN 4662801) become OTHER_NON_CITIZEN. Humanitarian/temporary-protection + statuses (microcosm #767) are drawn between the indicators and the EAD spill, sequentially and mutually + exclusively in decreasing order of target precision — PAROLED_ONE_YEAR per program origin (OAW Afghans + 73,566 as of 2022-03-31; U4U Ukrainians over 158,000 as of 2023-09-30; CHNV Nicaraguans 93,070 and + Venezuelans 117,330 paroled through 2024-12), REFUGEE (FY2023+FY2024 admissions, 60,050+100,060 OHSS + — refugees must apply for LPR adjustment after one year, so the not-yet-adjusted stock is about the + trailing two years), ASYLEE (FY2022-FY2024 grants 35,080+51,410+68,060=154,550 OHSS, rounded to 155,000; + 88% of 2015-22 affirmative adult grantees had adjusted by end-2024, so about three trailing years), + then TPS per country (CRS RS20844 Table 1 as of 2025-03-31, with required-arrival cutoffs binding + at PEINUSYR granularity: El Salvador <=2001, Honduras/Nicaragua <=1999, Nepal <=2015). Candidates + are restricted by program origin countries (Census Appendix I PENATVTY codes) and arrival windows; + Cuba/Haiti-born are excluded everywhere (the CUBAN_HAITIAN_ENTRANT class is the better statutory label + and keeps H.R.1 eligibility — Haiti TPS 330,735 rides that label), the DACA statutory cohort is excluded + (dual TPS/DACA holders stay DACA), and Ukraine/Afghanistan TPS registrants (101,150/8,105) are carried + by the parole draw since they are overwhelmingly the same U4U/OAW parolees. Parole targets are gross + admissions: cohort members who later won asylum keep the parole label, which is channel-equivalent + under H.R.1 (both classes lose Medicaid/SNAP/PTC on identical dates). DEPORTATION_WITHHELD carries + an explicit zero target: EOIR grants about 2.2k withholdings a year (CRS IN12501: under 1% of 666,177 + FY2024 decisions) and publishes no stock, so the accumulated population is below the CPS''s resolution + for a defensible draw. CONDITIONAL_ENTRANT is deliberately not emitted (the INA 203(a)(7) class closed + in 1980). REFUGEE/ASYLEE draw from the indicator-documented pool only; parole/TPS may draw residual-pool + people, who then receive NON_CITIZEN_VALID_EAD SSN cards (both programs confer employment authorization). + Pew''s 2025 report supersedes its earlier series and estimates a coherent 2023 pair of 14.0M unauthorized + immigrants and 9.7M in the labor force. Its modeled universe is broader than the engine''s UNDOCUMENTED + enum: it retains DACA recipients, parolees, TPS holders, similar temporary protections, and the residual-EAD + subset of the engine''s broad CUBAN_HAITIAN_ENTRANT class, while excluding refugees, people already + granted asylum, and assumed-documented Cuban/Haitian entrants. The worker EAD spill binds that modeled + universe to 9.7M using actual ASEC labor-force status (A_LFSR codes 1-4 at ages 16+) restored from + the same pinned official ASEC archives by exact PERIDNUM join, never prior-year WSAL_VAL/SEMP_VAL + earnings. The student spill binds the corresponding broad residual universe to the Higher Ed Immigration + Portal''s ~408k students. The total modeled Pew-defined unauthorized population is emergent and gated + against the 14.0M 2023 anchor with a coarse band; per-category humanitarian masses are gated against + their cited targets with a coarse band (the ASEC undercovers 2022-24 arrivals — the Ukraine parole + pool saturates below its admin count); reconciling levels is the calibration lane''s job. DACA (arrival + cohort among EAD holders) and CUBAN_HAITIAN_ENTRANT (nativity plus post-1980 arrival) keep their statutory + tests; other documented non-citizens stay LEGAL_PERMANENT_RESIDENT, the modal true status.' - stage: hours_worked survey: Census CPS ASEC source: https://www.census.gov/programs-surveys/cps.html diff --git a/packages/microcosm-build/src/microcosm/build/us/spec/spine.yaml b/packages/microcosm-build/src/microcosm/build/us/spec/spine.yaml index e932429a..0fc8b8d3 100644 --- a/packages/microcosm-build/src/microcosm/build/us/spec/spine.yaml +++ b/packages/microcosm-build/src/microcosm/build/us/spec/spine.yaml @@ -3,7 +3,7 @@ pipeline_contract: artifact_protocol: artifact_kind: populace_us_stacked_pool_checkpoint_identity schema_version: 1 - materializer_version: 12 + materializer_version: 13 pipeline: us-stacked-pool stacked_operator_order: - assemble_stacked_spine @@ -17,6 +17,7 @@ pipeline_contract: - materialize_multispine_agreement_outputs - stacked_completeness_gate - by_origin_battery + - us_immigration_composition_gate pre_clone_source_operator_order: - derive_us_cps_carried_inputs - with_us_hours_worked_inputs @@ -330,6 +331,58 @@ seed_site_bindings: owners: - kind: source_stage id: snap_abawd_discretionary_exemption +- site: immigration_humanitarian_paroled_one_year_afghanistan_assignment + owners: + - kind: source_stage + id: immigration_status +- site: immigration_humanitarian_paroled_one_year_ukraine_assignment + owners: + - kind: source_stage + id: immigration_status +- site: immigration_humanitarian_paroled_one_year_nicaragua_assignment + owners: + - kind: source_stage + id: immigration_status +- site: immigration_humanitarian_paroled_one_year_venezuela_assignment + owners: + - kind: source_stage + id: immigration_status +- site: immigration_humanitarian_refugee_assignment + owners: + - kind: source_stage + id: immigration_status +- site: immigration_humanitarian_asylee_assignment + owners: + - kind: source_stage + id: immigration_status +- site: immigration_humanitarian_deportation_withheld_assignment + owners: + - kind: source_stage + id: immigration_status +- site: immigration_humanitarian_tps_venezuela_assignment + owners: + - kind: source_stage + id: immigration_status +- site: immigration_humanitarian_tps_el_salvador_assignment + owners: + - kind: source_stage + id: immigration_status +- site: immigration_humanitarian_tps_honduras_assignment + owners: + - kind: source_stage + id: immigration_status +- site: immigration_humanitarian_tps_nicaragua_assignment + owners: + - kind: source_stage + id: immigration_status +- site: immigration_humanitarian_tps_nepal_assignment + owners: + - kind: source_stage + id: immigration_status +- site: immigration_humanitarian_tps_other_designated_assignment + owners: + - kind: source_stage + id: immigration_status - site: immigration_ead_workers_assignment owners: - kind: source_stage diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/__init__.py b/packages/microcosm-build/src/microcosm/build/us_runtime/__init__.py index 518baacb..df16b8c5 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/__init__.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/__init__.py @@ -210,8 +210,12 @@ from microcosm.build.us_runtime.education_assistance_source import ( ASEC_EDUCATION_ASSISTANCE_ARCHIVES, ASEC_EDUCATION_ASSISTANCE_INCOME_YEARS, + ASEC_EDUCATION_ASSISTANCE_SOURCE_COLUMNS, + ASEC_LABOR_FORCE_STATUS_COLUMN, + ASEC_LABOR_FORCE_STATUS_VALID_CODES, fetch_asec_education_assistance_source, fill_asec_education_assistance_source, + fill_asec_labor_force_status_source, load_asec_education_assistance_sources, ) from microcosm.build.us_runtime.education_inputs import ( @@ -371,16 +375,23 @@ with_us_housing_inputs, ) from microcosm.build.us_runtime.immigration import ( + HUMANITARIAN_STATUS_CATEGORIES, IMMIGRATION_STATUS_VALUES, SSN_CARD_TYPE_VALUES, US_IMMIGRATION_NONCONSTANT_PERSON_COLUMNS, US_IMMIGRATION_OUTPUT_COLUMNS, US_IMMIGRATION_REQUIRED_SOURCE_COLUMNS, US_IMMIGRATION_STAGE_NAME, + HumanitarianDraw, + ImmigrationControls, UndocumentedControls, derive_us_immigration_status_from_manifest, + reconcile_us_immigration_humanitarian_transfer, us_immigration_composition_gate, us_immigration_composition_summary, + us_immigration_controls, + us_immigration_evidence_features, + us_immigration_humanitarian_draw_mask, us_immigration_stage_spec, with_us_immigration_inputs, ) @@ -1150,12 +1161,15 @@ "US_JCT_TAX_EXPENDITURE_TARGET_SPECS", "US_JCT_TAX_EXPENDITURE_TARGET_REFERENCES", "SOI_VARIABLE_MAP", + "HUMANITARIAN_STATUS_CATEGORIES", "IMMIGRATION_STATUS_VALUES", "SSN_CARD_TYPE_VALUES", "US_IMMIGRATION_NONCONSTANT_PERSON_COLUMNS", "US_IMMIGRATION_OUTPUT_COLUMNS", "US_IMMIGRATION_REQUIRED_SOURCE_COLUMNS", "US_IMMIGRATION_STAGE_NAME", + "HumanitarianDraw", + "ImmigrationControls", "UndocumentedControls", "US_HOURS_WORKED_NONCONSTANT_PERSON_COLUMNS", "US_HOURS_WORKED_OUTPUT_COLUMNS", @@ -1464,8 +1478,12 @@ "US_EDUCATION_INPUTS_OWNED_OUTPUT_COLUMNS", "ASEC_EDUCATION_ASSISTANCE_ARCHIVES", "ASEC_EDUCATION_ASSISTANCE_INCOME_YEARS", + "ASEC_EDUCATION_ASSISTANCE_SOURCE_COLUMNS", + "ASEC_LABOR_FORCE_STATUS_COLUMN", + "ASEC_LABOR_FORCE_STATUS_VALID_CODES", "fetch_asec_education_assistance_source", "fill_asec_education_assistance_source", + "fill_asec_labor_force_status_source", "load_asec_education_assistance_sources", "ASEC_PUBLIC_ASSISTANCE_TYPE_AUDIT_PINS", "ASEC_PUBLIC_ASSISTANCE_TYPE_INCOME_YEARS", @@ -1760,8 +1778,12 @@ "us_prior_year_income_summary", "with_us_prior_year_income_inputs", "derive_us_immigration_status_from_manifest", + "reconcile_us_immigration_humanitarian_transfer", "us_immigration_composition_gate", "us_immigration_composition_summary", + "us_immigration_controls", + "us_immigration_evidence_features", + "us_immigration_humanitarian_draw_mask", "us_immigration_stage_spec", "with_us_immigration_inputs", "US_TAKE_UP_SHARE_BAND", @@ -2166,14 +2188,16 @@ def to_manifest(self) -> dict[str, object]: US_IMMIGRATION_STAGE_NAME: DonorSpec( survey="CPS ASEC + published unauthorized-population estimates", source=( - "https://www.pewresearch.org/short-reads/2024/07/22/" - "what-we-know-about-unauthorized-immigrants-living-in-the-us/" + "https://www.pewresearch.org/race-and-ethnicity/2025/08/21/" + "u-s-unauthorized-immigrant-population-reached-a-record-" + "14-million-in-2023/" ), notes=( "SSN card type and immigration status from ASEC citizenship, " - "entry-year, nativity, and program-participation fields via the " - "ASEC-UA residual method (SSRN 4662801), targeted to published " - "undocumented population/worker/student control totals." + "entry-year, nativity, labor-force, and program-participation " + "fields via the ASEC-UA residual method (SSRN 4662801), with " + "Pew's broad 2023 unauthorized population and labor-force " + "definitions applied consistently." ), ), US_HOURS_WORKED_STAGE_NAME: DonorSpec( diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/acs_pums.py b/packages/microcosm-build/src/microcosm/build/us_runtime/acs_pums.py index dd0456fc..423f6a04 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/acs_pums.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/acs_pums.py @@ -83,6 +83,9 @@ "SSIP", "RETP", "INTP", + "CIT", + "POBP", + "YOEP", "PWGTP", ) _PERSON_OPTIONAL: tuple[str, ...] = () diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/acs_transfer.py b/packages/microcosm-build/src/microcosm/build/us_runtime/acs_transfer.py index a3fb00a6..d6760c64 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/acs_transfer.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/acs_transfer.py @@ -31,9 +31,14 @@ from microcosm.build.gates import FitWeightRecord from microcosm.build.serialization_dtypes import canonicalize_frame_string_dtypes +from microcosm.build.us_runtime.acs_pums import ACS_2024_1YR_VINTAGE from microcosm.build.us_runtime.capital_gain_distributions import ( capital_gain_distribution_shares_asset_identity, ) +from microcosm.build.us_runtime.immigration import ( + reconcile_us_immigration_humanitarian_transfer, + us_immigration_evidence_features, +) from microcosm.build.us_runtime.puf_support import ( PUF_TAX_DETAIL_DEFAULT_PERSON_OUTPUTS, PUF_TAX_DETAIL_DEFAULT_TAX_UNIT_OUTPUTS, @@ -182,11 +187,38 @@ def _pregnancy_structural_policy_identity(*, enabled: bool) -> dict[str, object] ).hexdigest() return payload + _IMMIGRATION_STATUS_TARGETS = ( "ssn_card_type", "immigration_status_str", ) _IMMIGRATION_STATUS_MODEL_TARGET = "__acs_transfer_immigration_status_pair" +_IMMIGRATION_CITIZENSHIP_FEATURE = "__acs_transfer_is_us_citizen" +_IMMIGRATION_ORIGIN_FEATURE = "__acs_transfer_birth_country_code" +_IMMIGRATION_ARRIVAL_FEATURE = "__acs_transfer_arrival_year" +_IMMIGRATION_REQUIRED_FEATURES = ( + _IMMIGRATION_CITIZENSHIP_FEATURE, + _IMMIGRATION_ORIGIN_FEATURE, + _IMMIGRATION_ARRIVAL_FEATURE, +) +_HUMANITARIAN_IMMIGRATION_STATUSES = frozenset( + { + "PAROLED_ONE_YEAR", + "REFUGEE", + "ASYLEE", + "DEPORTATION_WITHHELD", + "TPS", + } +) +_EVIDENCE_DERIVED_IMMIGRATION_STATUSES = frozenset( + { + "CUBAN_HAITIAN_ENTRANT", + "DACA", + } +) +_CONSTRAINED_IMMIGRATION_STATUSES = ( + _HUMANITARIAN_IMMIGRATION_STATUSES | _EVIDENCE_DERIVED_IMMIGRATION_STATUSES +) # RNTP/GRNTP are not pre-subsidy rent, so this leaf remains donor-transferable. _ADDITIONAL_PERSON_TRANSFER_TARGETS = ("pre_subsidy_rent",) @@ -286,8 +318,9 @@ def acs_transfer_execution_contract_identity( ) adult_care_enabled = _ADULT_CARE_EXPENSE in requested_targets payload: dict[str, object] = { - "schema_version": 2, + "schema_version": 3, "person_required_predictors": list(ACS_PERSON_TRANSFER_PREDICTORS), + "immigration_required_predictors": list(_IMMIGRATION_REQUIRED_FEATURES), "person_optional_predictors": list(ACS_OPTIONAL_PERSON_TRANSFER_PREDICTORS), "group_required_predictors": list(ACS_GROUP_TRANSFER_PREDICTORS), "group_optional_names": dict(sorted(_GROUP_OPTIONAL_NAMES.items())), @@ -312,6 +345,9 @@ def acs_transfer_execution_contract_identity( "tenure_codes": dict(sorted(_TENURE_CODES.items())), "immigration_status_targets": list(_IMMIGRATION_STATUS_TARGETS), "immigration_status_model_target": _IMMIGRATION_STATUS_MODEL_TARGET, + "immigration_baseline_excluded_statuses": sorted( + _CONSTRAINED_IMMIGRATION_STATUSES + ), "discrete_numeric_targets": sorted(_DISCRETE_NUMERIC_TARGETS), "structural_target_policies": { _PREGNANCY_TARGET: _pregnancy_structural_policy_identity( @@ -339,6 +375,14 @@ def acs_transfer_execution_contract_identity( "tax_unit_link": _ADULT_CARE_UNIT, "mutable_rows": "newly_imputed_expense_cells_only", }, + "humanitarian_immigration": { + "enabled": set(_IMMIGRATION_STATUS_TARGETS).issubset(requested_targets), + "targets": list(_IMMIGRATION_STATUS_TARGETS), + "mutable_rows": "newly_imputed_paired_cells_only", + "target_basis": "manifest_residual_after_immutable_asec_mass", + "selection": "stable_person_hash_in_manifest_draw_order", + "candidate_shortfall": "error", + }, }, } payload["sha256"] = hashlib.sha256( @@ -493,7 +537,7 @@ class AcsImputedInput: derivation: str | None = None #: Post-fit structural reconciliation counts, when the column's surface #: was adjusted to a statute contract after prediction. - reconciliation: Mapping[str, int] | None = None + reconciliation: Mapping[str, object] | None = None #: Hard-domain and source-person fanout proof for structurally constrained #: transfer targets. This is receipt-only and never enters the frame. structural_receipt: Mapping[str, object] | None = None @@ -880,6 +924,37 @@ def acs_transfer_donor_requirements( if all(component in person.columns for component in components): required[donor.schema.person_entity].update(components) + person_targets = { + target + for targets in target_families.get(donor.schema.person_entity, {}).values() + for target in targets + } + if set(_IMMIGRATION_STATUS_TARGETS).issubset(person_targets): + evidence_triplets = ( + ("PRCITSHP", "PENATVTY", "PEINUSYR"), + ("CIT", "POBP", "YOEP"), + ) + active_triplets: list[tuple[str, str, str]] = [] + for triplet in evidence_triplets: + citizenship = triplet[0] + if citizenship not in person or not person[citizenship].notna().any(): + continue + missing = sorted(set(triplet) - set(person.columns)) + if missing: + raise ValueError( + "paired immigration targets have an incomplete donor " + f"evidence triplet {triplet!r}; missing={missing}." + ) + active_triplets.append(triplet) + if not active_triplets: + raise ValueError( + "paired immigration targets require a live ASEC " + "PRCITSHP/PENATVTY/PEINUSYR or ACS CIT/POBP/YOEP donor " + "evidence triplet." + ) + for triplet in active_triplets: + required[donor.schema.person_entity].update(triplet) + housing = bool(_HOUSING_TRANSFER_TARGETS.intersection(target_names)) head_source = next( ( @@ -1138,8 +1213,8 @@ def _prepare_pregnancy_structural_plan( "through 44." ) - source_codes, group_count, key_column, representatives = ( - _pregnancy_source_groups(table) + source_codes, group_count, key_column, representatives = _pregnancy_source_groups( + table ) eligible_min = np.ones(group_count, dtype=np.int8) eligible_max = np.zeros(group_count, dtype=np.int8) @@ -1449,9 +1524,7 @@ def transfer_acs_inputs( person, pregnancy_plan, ) - pregnancy_entity, pregnancy_family, pregnancy_targets = ( - pregnancy_request - ) + pregnancy_entity, pregnancy_family, pregnancy_targets = pregnancy_request if pregnancy_targets != (_PREGNANCY_TARGET,): # pragma: no cover raise AssertionError( "Pregnancy structural target was not isolated before receipt." @@ -1645,10 +1718,12 @@ def transfer_acs_inputs( _apply_post_transfer_structure( output_tables, provenance, + recipient=recipient, imputed_masks=imputed_masks, donor_spine=donor_spine, resolved_channel=resolved_channel, execution_contract=resolved_execution_contract, + seed=seed, ) tables: dict[str, pd.DataFrame] = dict(output_tables) @@ -1682,26 +1757,67 @@ def _apply_post_transfer_structure( output_tables: dict[str, pd.DataFrame], provenance: list[AcsImputedInput], *, + recipient: Frame, imputed_masks: Mapping[tuple[str, str], np.ndarray], donor_spine: str, resolved_channel: str | None, execution_contract: Mapping[str, object], + seed: int, ) -> None: """Apply the deterministic post-fit steps the base's construction implies. - Both steps key off cells THIS transfer filled, so custom test plans that + All steps key off cells THIS transfer filled, so custom test plans that never touch these families are unaffected: - The Schedule D CGD memo leg is derived from the two transferred capital-gain parents at the packaged share with route exclusivity. - The adult-care expense surface is reconciled to the statute structure (qualifying carriers only, at most one per tax unit). + - The paired immigration baseline is reconciled to manifest humanitarian + residual targets after the immutable ASEC contribution. """ person = output_tables.get("person") if person is None: return + immigration_masks = [ + imputed_masks.get(("person", target)) for target in _IMMIGRATION_STATUS_TARGETS + ] + if all(mask is not None for mask in immigration_masks): + assert immigration_masks[0] is not None + assert immigration_masks[1] is not None + if not np.array_equal(immigration_masks[0], immigration_masks[1]): + raise ValueError( + "ACS immigration transfer must impute its paired target cells " + "on exactly the same recipient rows." + ) + mutable_immigration = immigration_masks[0] + immigration_provenance = [ + (index, item) + for index, item in enumerate(provenance) + if item.entity == "person" and item.column in _IMMIGRATION_STATUS_TARGETS + ] + source_operator_family = any( + item.family.split("__batch_", 1)[0] == "source_operator_immigration" + for _, item in immigration_provenance + ) + if mutable_immigration.any() and source_operator_family: + reconciled, receipt = reconcile_us_immigration_humanitarian_transfer( + person, + weights=np.asarray( + recipient.resolve_weights("person").values, + dtype=np.float64, + ), + mutable_rows=mutable_immigration, + seed=seed, + time_period=ACS_2024_1YR_VINTAGE, + ) + for target in _IMMIGRATION_STATUS_TARGETS: + person[target] = reconciled[target] + for index, item in immigration_provenance: + provenance[index] = replace(item, reconciliation=receipt) + post_transfer_contract = execution_contract["post_transfer_structure"] assert isinstance(post_transfer_contract, Mapping) schedule_d_contract = post_transfer_contract[ @@ -2578,6 +2694,38 @@ def _transfer_feature_surface( else: surface = _group_feature_surface(donor, recipient, entity=entity) + if set(_IMMIGRATION_STATUS_TARGETS).issubset(targets): + if entity != donor.schema.person_entity: + raise ValueError( + "The paired immigration targets must be transferred on person." + ) + donor_evidence = us_immigration_evidence_features( + donor, + time_period=ACS_2024_1YR_VINTAGE, + ).rename( + columns={ + "is_us_citizen": _IMMIGRATION_CITIZENSHIP_FEATURE, + "birth_country_code": _IMMIGRATION_ORIGIN_FEATURE, + "arrival_year": _IMMIGRATION_ARRIVAL_FEATURE, + } + ) + recipient_evidence = us_immigration_evidence_features( + recipient, + time_period=ACS_2024_1YR_VINTAGE, + ).rename( + columns={ + "is_us_citizen": _IMMIGRATION_CITIZENSHIP_FEATURE, + "birth_country_code": _IMMIGRATION_ORIGIN_FEATURE, + "arrival_year": _IMMIGRATION_ARRIVAL_FEATURE, + } + ) + surface = _FeatureSurface( + donor=pd.concat([surface.donor, donor_evidence], axis=1), + recipient=pd.concat([surface.recipient, recipient_evidence], axis=1), + required=(*surface.required, *_IMMIGRATION_REQUIRED_FEATURES), + optional=surface.optional, + ) + housing = bool(_HOUSING_TRANSFER_TARGETS.intersection(targets)) if not housing: return surface @@ -3477,7 +3625,8 @@ def _target_encodings( result: dict[str, _TargetEncoding] = {} if immigration: pairs = list( - zip( + _baseline_immigration_pair(ssn, status) + for ssn, status in zip( table[_IMMIGRATION_STATUS_TARGETS[0]].to_numpy(dtype=object), table[_IMMIGRATION_STATUS_TARGETS[1]].to_numpy(dtype=object), strict=True, @@ -3524,6 +3673,18 @@ def _target_encodings( return result +def _baseline_immigration_pair(ssn: object, status: object) -> tuple[object, object]: + """Remove evidence-constrained labels from the unconstrained codec.""" + + if status not in _CONSTRAINED_IMMIGRATION_STATUSES: + return ssn, status + if ssn == "CITIZEN": + return "CITIZEN", "CITIZEN" + if ssn == "NONE": + return "NONE", "UNDOCUMENTED" + return ssn, "LEGAL_PERMANENT_RESIDENT" + + def _complete_case_target_encodings( table: pd.DataFrame, *, diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/asec_checkpoint.py b/packages/microcosm-build/src/microcosm/build/us_runtime/asec_checkpoint.py index 0c6ebe0f..63bc8559 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/asec_checkpoint.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/asec_checkpoint.py @@ -57,10 +57,10 @@ ASEC_RAW_STAGE_ARTIFACT_KIND = "populace_us_asec_raw_stage" ASEC_RAW_STAGE_CHECKPOINT_FILENAME = "asec_raw_stage.checkpoint.h5" ASEC_RAW_STAGE_OPERATOR_STATUS = "operator_untouched" -# Version 3 added the PAW_TYP restoration that gates TANF enrollment; older -# artifacts lack the gate column and must fail loudly rather than let -# PAW_VAL-only conflation back in (microcosm#591). -ASEC_RAW_STAGE_SCHEMA_VERSION = 3 +# Version 4 added measured A_LFSR for the Pew civilian-labor-force control; +# older artifacts lack the current labor-force status and must fail loudly +# rather than substitute prior-year earnings (microcosm#767). +ASEC_RAW_STAGE_SCHEMA_VERSION = 4 ASEC_RAW_STAGE_STAGE = "raw_source_mapping" _RAW_STAGE_BINDING_KEYS = frozenset( { @@ -75,9 +75,9 @@ "stage", } ) -_RAW_SOURCE_MAPPING_COLUMNS = frozenset({"ED_VAL", "LKWEEKS", "PAW_TYP"}) +_RAW_SOURCE_MAPPING_COLUMNS = frozenset({"A_LFSR", "ED_VAL", "LKWEEKS", "PAW_TYP"}) _RAW_STAGE_REQUIRED_PERSON_COLUMNS = frozenset( - {"ED_VAL", "LKWEEKS", "PAW_TYP", "PERIDNUM", "source_year"} + {"A_LFSR", "ED_VAL", "LKWEEKS", "PAW_TYP", "PERIDNUM", "source_year"} ) _RAW_SOURCE_MAPPING_KEYS = frozenset( { @@ -458,6 +458,19 @@ def _validate_raw_stage_source_columns(frame: Frame, *, path: Path) -> None: "in {0, 1, 2, 3}." ) + labor_force_status = pd.to_numeric(person["A_LFSR"], errors="coerce").to_numpy( + dtype=np.float64 + ) + valid_labor_force_status = np.isfinite(labor_force_status) & np.isin( + labor_force_status, + (0.0, 1.0, 2.0, 3.0, 4.0, 7.0), + ) + if not valid_labor_force_status.all(): + raise ValueError( + f"ASEC raw-stage checkpoint {path} A_LFSR must be complete integers " + "in {0, 1, 2, 3, 4, 7}." + ) + def _validate_asec_frame( frame: Frame, diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py b/packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py index c24a819c..5eccefa4 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/education_assistance_source.py @@ -1,4 +1,4 @@ -"""Measured CPS ASEC educational assistance for the education-inputs stage. +"""Pinned CPS ASEC person sidecar for measured ED_VAL and A_LFSR fields. The retired eCPS build carried person-level educational assistance directly from ASEC ``ED_VAL`` (archived ``datasets/cps/cps.py`` line 1493: @@ -6,8 +6,11 @@ recode, filter, or universe restriction), and the archived ``datasets/cps/census_cps.py`` listed ``ED_VAL`` among the raw ASEC columns it kept. The frozen ``census_cps_*.h5`` inputs microcosm builds from never -carried that column, so this module restores it from the official, immutable -Census ASEC public-use archives — the same repair class as +carried that column. Those frozen inputs also omitted current civilian labor- +force status ``A_LFSR``, which the immigration stage needs to reproduce Pew's +age-16+ working-or-looking-for-work definition without a prior-year earnings +proxy. This module restores both fields from the official, immutable Census +ASEC public-use archives — the same repair class as :mod:`.weeks_unemployed`'s ``LKWEEKS`` sidecar, extended to every pooled income year. @@ -36,11 +39,14 @@ "ASEC_EDUCATION_ASSISTANCE_ARCHIVES", "ASEC_EDUCATION_ASSISTANCE_INCOME_YEARS", "ASEC_EDUCATION_ASSISTANCE_SOURCE_COLUMNS", + "ASEC_LABOR_FORCE_STATUS_COLUMN", + "ASEC_LABOR_FORCE_STATUS_VALID_CODES", "EDUCATION_ASSISTANCE_ARCHIVED_DERIVATION_URL", "EDUCATION_ASSISTANCE_ARCHIVED_SOURCE_URL", "AsecEducationArchive", "fetch_asec_education_assistance_source", "fill_asec_education_assistance_source", + "fill_asec_labor_force_status_source", "load_asec_education_assistance_sources", ] @@ -59,12 +65,15 @@ ) _SOURCE = "ED_VAL" +ASEC_LABOR_FORCE_STATUS_COLUMN = "A_LFSR" +ASEC_LABOR_FORCE_STATUS_VALID_CODES = frozenset({0, 1, 2, 3, 4, 7}) ASEC_EDUCATION_ASSISTANCE_SOURCE_COLUMNS: tuple[str, ...] = ( "PH_SEQ", "P_SEQ", "A_LINENO", "PERIDNUM", _SOURCE, + ASEC_LABOR_FORCE_STATUS_COLUMN, ) _AUDIT_WEIGHT_COLUMN = "A_FNLWGT" @@ -458,6 +467,19 @@ def load_asec_education_assistance_sources( f"ASEC {pins.survey_year} ED_VAL must be finite and nonnegative " f"at row(s): {rows}." ) + labor_force_status = pd.to_numeric( + raw[ASEC_LABOR_FORCE_STATUS_COLUMN], errors="coerce" + ).to_numpy(dtype=np.float64) + valid_labor_force_status = np.isfinite(labor_force_status) & np.isin( + labor_force_status, + sorted(ASEC_LABOR_FORCE_STATUS_VALID_CODES), + ) + if not valid_labor_force_status.all(): + rows = np.flatnonzero(~valid_labor_force_status)[:5].tolist() + raise ValueError( + f"ASEC {pins.survey_year} A_LFSR must be a complete integer in " + f"{sorted(ASEC_LABOR_FORCE_STATUS_VALID_CODES)} at row(s): {rows}." + ) weights = pd.to_numeric(raw[_AUDIT_WEIGHT_COLUMN], errors="coerce").to_numpy( dtype=np.float64 ) @@ -509,6 +531,7 @@ def load_asec_education_assistance_sources( ) audits[income_year] = audit part = raw.loc[:, list(ASEC_EDUCATION_ASSISTANCE_SOURCE_COLUMNS)].copy() + part[ASEC_LABOR_FORCE_STATUS_COLUMN] = labor_force_status.astype("int64") part.insert(0, "source_year", np.int64(income_year)) parts.append(part) result = pd.concat(parts, ignore_index=True) @@ -620,3 +643,117 @@ def fill_asec_education_assistance_source( "source_audit", {} ) return result + + +def fill_asec_labor_force_status_source( + person: pd.DataFrame, + source: pd.DataFrame, +) -> pd.DataFrame: + """Fill measured ``A_LFSR`` via the pinned exact Census identity join.""" + + required_person = ("source_year", "PERIDNUM") + missing_person = [column for column in required_person if column not in person] + if missing_person: + raise ValueError( + "ASEC labor-force-status repair requires person column(s): " + f"{missing_person}." + ) + required_source = ("source_year", *ASEC_EDUCATION_ASSISTANCE_SOURCE_COLUMNS) + missing_source = [column for column in required_source if column not in source] + if missing_source: + raise ValueError( + f"ASEC labor-force-status sidecar missing column(s): {missing_source}." + ) + result = person.copy(deep=True) + person_years = pd.to_numeric(result["source_year"], errors="coerce") + if person_years.isna().any() or (person_years != np.floor(person_years)).any(): + raise ValueError("ASEC labor-force-status person source_year is invalid.") + needed_years = sorted(int(year) for year in person_years.unique()) + covered = set( + pd.to_numeric(source["source_year"], errors="coerce").astype(int).unique() + ) + uncovered = [year for year in needed_years if year not in covered] + if uncovered: + raise ValueError( + "ASEC labor-force-status sidecar does not cover pooled income " + f"year(s): {uncovered}." + ) + + donor = source.copy(deep=True) + donor["PERIDNUM"] = _fixed_width_peridnum( + donor["PERIDNUM"], label="ASEC labor-force-status sidecar" + ) + if ASEC_LABOR_FORCE_STATUS_COLUMN in result.columns: + existing = pd.to_numeric( + result[ASEC_LABOR_FORCE_STATUS_COLUMN], errors="coerce" + ) + if existing.notna().any(): + raise ValueError( + "ASEC labor-force-status fill found a preexisting A_LFSR " + "column with values; refusing to overwrite measured data." + ) + filled = np.zeros(len(result), dtype=np.int64) + + for year in needed_years: + year_mask = person_years.eq(year).to_numpy() + year_donor = donor.loc[ + pd.to_numeric(donor["source_year"], errors="coerce").eq(year) + ].set_index("PERIDNUM") + keys = _fixed_width_peridnum( + result.loc[year_mask, "PERIDNUM"], + label="ASEC labor-force-status frame", + ) + missing_keys = keys[~keys.isin(year_donor.index)].drop_duplicates() + if not missing_keys.empty: + raise ValueError( + "ASEC labor-force-status sidecar does not cover frame " + f"PERIDNUM key(s) for income year {year}: " + f"{missing_keys.tolist()[:5]}." + ) + aligned = year_donor.reindex(keys.to_numpy()) + aligned.index = result.index[year_mask] + identity_pairs = [ + ( + "PH_SEQ", + "source_household_id" if "source_household_id" in result else "PH_SEQ", + ), + ("P_SEQ", "P_SEQ"), + ("A_LINENO", "A_LINENO"), + ] + for donor_column, frame_column in identity_pairs: + if frame_column not in result: + continue + observed = pd.to_numeric( + result.loc[year_mask, frame_column], errors="coerce" + ).to_numpy(dtype=np.float64) + expected = pd.to_numeric(aligned[donor_column], errors="coerce").to_numpy( + dtype=np.float64 + ) + mismatch = ( + ~np.isfinite(observed) | ~np.isfinite(expected) | (observed != expected) + ) + if mismatch.any(): + rows = result.index[year_mask].to_numpy()[mismatch][:5].tolist() + raise ValueError( + "ASEC labor-force-status redundant identity mismatch for " + f"{frame_column} against sidecar {donor_column} in income " + f"year {year} at row(s): {rows}." + ) + values = pd.to_numeric( + aligned[ASEC_LABOR_FORCE_STATUS_COLUMN], errors="coerce" + ).to_numpy(dtype=np.float64) + valid = np.isfinite(values) & np.isin( + values, + sorted(ASEC_LABOR_FORCE_STATUS_VALID_CODES), + ) + if not valid.all(): + raise ValueError( + f"ASEC labor-force-status sidecar A_LFSR is invalid for " + f"income year {year}." + ) + filled[year_mask] = values.astype(np.int64) + result[ASEC_LABOR_FORCE_STATUS_COLUMN] = filled + result.attrs["labor_force_status_source_audit"] = source.attrs.get( + "source_audit", {} + ) + return result diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/h5_io.py b/packages/microcosm-build/src/microcosm/build/us_runtime/h5_io.py index 55de3427..7497ecd2 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/h5_io.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/h5_io.py @@ -77,6 +77,8 @@ US_MULTISPINE_AGREEMENT_DIAGNOSTICS_ARTIFACT_KIND = ( "populace_us_multispine_agreement_diagnostics" ) +# 10 binds the humanitarian-immigration composition gate into the stacked +# operator order and authenticated terminal-gate surface. # 9 binds the post-assembly household-geography assignment receipt and its # authenticated release-vintage authorities. # 8 binds the nullable-boolean-capable physical H5 materializer in both the @@ -87,7 +89,7 @@ # authority and restores its immutable Frame-metadata anchor on H5 load. # Schema 5 can authenticate the DAG receipt's structure, but cannot prove that # the published receipt is the one authorized by the generating transition. -US_MULTISPINE_POOL_MANIFEST_SCHEMA_VERSION = 9 +US_MULTISPINE_POOL_MANIFEST_SCHEMA_VERSION = 10 US_MULTISPINE_POOL_H5_MATERIALIZER_VERSION = 3 """Version 3 atomically binds release CD provenance attrs; v2 added BooleanDtype.""" _LEGACY_MULTISPINE_POOL_MANIFEST_SCHEMA_VERSION = 4 @@ -116,6 +118,14 @@ "materialize_multispine_agreement_outputs", "stacked_completeness_gate", "by_origin_battery", + "us_immigration_composition_gate", +) +_STACKED_TERMINAL_GATE_NAMES = frozenset( + { + "us_stacked_completeness", + "us_by_origin_battery", + "immigration_composition", + } ) _LEGACY_POOL_OPERATOR_ORDER = ( "assemble", @@ -303,14 +313,8 @@ def _validated_stacked_sampling_manifest_binding( ("sampling.sample_fraction", sampling_fraction), ("stack_manifest.sample_fraction", stack_fraction), ): - if ( - type(value) is not float - or not np.isfinite(value) - or not 0.0 < value <= 1.0 - ): - raise ValueError( - f"{label} {location} must be a finite float in (0, 1]." - ) + if type(value) is not float or not np.isfinite(value) or not 0.0 < value <= 1.0: + raise ValueError(f"{label} {location} must be a finite float in (0, 1].") if sampling_fraction != stack_fraction: raise ValueError( f"{label} sampling.sample_fraction differs from " @@ -617,10 +621,10 @@ def us_multispine_pool_release_receipt( ) gate_passed = gate.get("passed") gate_failures = gate.get("failures") - if type(gate_passed) is not bool or not isinstance( - gate_failures, list - ) or not all( - isinstance(failure, str) for failure in gate_failures + if ( + type(gate_passed) is not bool + or not isinstance(gate_failures, list) + or not all(isinstance(failure, str) for failure in gate_failures) ): raise ValueError( f"Authenticated US multispine pool gate {gate_name!r} has an " @@ -963,7 +967,7 @@ def _validate_stacked_late_dag_manifest_binding( *, manifest_path: Path, ) -> None: - """Make schema-9 consumers authenticate geography and late-DAG proofs.""" + """Make current-schema consumers authenticate geography and late-DAG proofs.""" if manifest.get("pipeline") != "us-stacked-pool": return @@ -1388,7 +1392,7 @@ def _validate_stacked_geography_h5_binding( manifest_path: Path, pool_path: Path, ) -> None: - """Bind schema-9 manifest geography claims to the authenticated H5.""" + """Bind current-schema manifest geography claims to the authenticated H5.""" if manifest.get("pipeline") != _STACKED_PIPELINE: return @@ -2229,6 +2233,24 @@ def _require_matching_terminal_gate_aliases( f"US stacked pool diagnostics {diagnostics_path} terminal_gates do " "not match agreement_gate." ) + manifest_gates = _mapping( + manifest_terminal_gates.get("gates"), + label=f"US stacked pool manifest {manifest_path}.terminal_gates.gates", + ) + diagnostics_gates = _mapping( + diagnostics_terminal_gates.get("gates"), + label=(f"US stacked pool diagnostics {diagnostics_path}.terminal_gates.gates"), + ) + for label, gates in ( + (f"US stacked pool manifest {manifest_path}", manifest_gates), + (f"US stacked pool diagnostics {diagnostics_path}", diagnostics_gates), + ): + if set(gates) != _STACKED_TERMINAL_GATE_NAMES: + raise ValueError( + f"{label} does not carry the canonical terminal gate set; " + f"expected={sorted(_STACKED_TERMINAL_GATE_NAMES)}, " + f"observed={sorted(gates)}." + ) def _publication_run_id(value: object, *, label: str) -> str: diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/immigration.py b/packages/microcosm-build/src/microcosm/build/us_runtime/immigration.py index 9c5bffbf..0bceea92 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/immigration.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/immigration.py @@ -12,7 +12,8 @@ ``OTHER_NON_CITIZEN`` / ``NONE`` (likely undocumented, i.e. ITIN-filer territory for tax purposes). - ``immigration_status_str``: ``CITIZEN`` / ``LEGAL_PERMANENT_RESIDENT`` / - ``CUBAN_HAITIAN_ENTRANT`` / ``DACA`` / ``UNDOCUMENTED``. + ``CUBAN_HAITIAN_ENTRANT`` / ``DACA`` / ``UNDOCUMENTED`` plus the supported + humanitarian and temporary-protection statuses listed below. Method — survey measurement first, published residual method second, and one narrowly-scoped control where the survey carries no signal at all: @@ -28,32 +29,51 @@ Residual Methods Robust?", SSRN 4662801). 3. **Work/study authorization split.** The CPS carries no work-authorization variable, so among residual non-citizens the split between EAD holders and - the unauthorized is unidentified from the survey alone. Workers and - students spill to ``NON_CITIZEN_VALID_EAD`` — in deterministic seeded - order — until the *remaining* undocumented worker/student counts match - their published control totals (Pew Research Center; Higher Ed - Immigration Portal). This is the only forced margin. -4. **The total undocumented population is emergent, not forced.** + the unauthorized is unidentified from the survey alone. Workers (measured + with ASEC ``A_LFSR`` at age 16+, not prior-year earnings) and students + spill to ``NON_CITIZEN_VALID_EAD`` in deterministic seeded order until the + broad reported universes match their controls. Pew retains DACA, parole, + TPS, and similar temporary protections in its unauthorized estimates, so + those statuses continue to count after receiving EADs. Residual + Cuban/Haitian rows that receive an EAD also stay in that universe because + the engine's broader ``CUBAN_HAITIAN_ENTRANT`` label carries the modeled + CHNV-parole/Haiti-TPS subset. +4. **The total Pew-defined unauthorized population is emergent, not forced.** Representation is calibration's job, not the label stage's: the release - gate checks the emergent total against a cited published anchor with a - coarse plausibility band, and a follow-up calibration target (Ledger - facts lane) can reconcile the level in the weights. On the 2024 ASEC the - emergent total is ≈13M — inside the range of published 2023–24 estimates - — with 26.6M non-citizens, 18.0M residual after the indicators, and 4.8M - spilled to EAD. -5. **Status tags carry only statutory tests the data can support.** + gate checks the emergent broad-universe total against Pew's cited 2023 + anchor with a coarse plausibility band. The engine's narrower + ``UNDOCUMENTED`` enum remains exactly paired with ``ssn_card_type=NONE``. +5. **Humanitarian/temporary-protection statuses are drawn to published + stocks, never blanketed (microcosm #767).** Blanket ``REFUGEE``/``TPS`` + labels for recent arrivals were originally refused because LPR treatment + gave near-identical means-tested eligibility; H.R.1 repealed exactly that + equivalence (Medicaid §71109 effective 2026-10-01, SNAP §10108 effective + 2025-07-01, ACA §71301/§71302 effective 2026/2027 — all keyed to these + enum values in policyengine-us parameters), so an LPR-only file silently + zeroes every one of those channels. Instead of blankets, the stage now + draws ``REFUGEE`` / ``ASYLEE`` / ``DEPORTATION_WITHHELD`` / + ``PAROLED_ONE_YEAR`` / ``TPS`` — in that order, mutually exclusive — + from candidates whose survey signature supports the status (origin + country x arrival window x legal-status indicators), each to a + manifest-cited weighted stock target. Cuba/Haiti-born persons are + excluded from every humanitarian draw (the ``CUBAN_HAITIAN_ENTRANT`` + class is the better statutory label and keeps H.R.1 eligibility), as is + the DACA statutory cohort. Draws from the residual pool (TPS and parole + only) flip ``ssn_card_type`` ``NONE`` to ``NON_CITIZEN_VALID_EAD`` — + both programs grant employment authorization — so the two output columns + never disagree. ``CONDITIONAL_ENTRANT`` is deliberately not emitted: the + INA 203(a)(7) class closed in 1980 and surviving holders are + indistinguishable from LPRs at this granularity. +6. **Status tags carry only statutory tests the data can support.** ``DACA`` applies the statutory cohort test (arrived by 2007 before age 16, aged 15+) to EAD holders; ``CUBAN_HAITIAN_ENTRANT`` applies the nationality-plus-arrival class to documented non-citizens; every other documented non-citizen stays ``LEGAL_PERMANENT_RESIDENT`` (the modal true - status). Blanket ``REFUGEE``/``TPS`` labels for recent arrivals or - leftover EAD holders are deliberately not emitted: they would mislabel - millions (true stocks are under a million each) and over-grant - refugee-class exemptions in benefit rules, while LPR treatment gives - near-identical means-tested eligibility for those populations. + status). Selection draws are seeded blake2b hashes keyed by the person's stable source -identity (``source_year``/``source_person_id`` when present), so +identity (``source_year``/``source_household_id``/``source_person_id`` when +present), so support-channel clones of one source person always receive the same status and reruns are bit-reproducible without global RNG state. @@ -89,16 +109,24 @@ from microcosm.frame.units import US_SCHEMA __all__ = [ + "HUMANITARIAN_STATUS_CATEGORIES", "IMMIGRATION_STATUS_VALUES", "SSN_CARD_TYPE_VALUES", "US_IMMIGRATION_NONCONSTANT_PERSON_COLUMNS", "US_IMMIGRATION_OUTPUT_COLUMNS", "US_IMMIGRATION_REQUIRED_SOURCE_COLUMNS", "US_IMMIGRATION_STAGE_NAME", + "HumanitarianDraw", + "ImmigrationControls", "UndocumentedControls", "derive_us_immigration_status_from_manifest", + "reconcile_us_immigration_humanitarian_transfer", "us_immigration_composition_gate", "us_immigration_composition_summary", + "us_immigration_controls", + "us_immigration_evidence_features", + "us_immigration_humanitarian_draw_mask", + "us_immigration_humanitarian_transfer_selection_masks", "us_immigration_stage_spec", "with_us_immigration_inputs", ] @@ -126,13 +154,18 @@ #: PolicyEngine-US ``ImmigrationStatus`` enum member names this stage emits #: (a deliberate subset of the engine's full enum domain; see module -#: docstring for why blanket REFUGEE/TPS labels are not fabricated). +#: docstring for why ``CONDITIONAL_ENTRANT`` is not fabricated). IMMIGRATION_STATUS_VALUES: tuple[str, ...] = ( "CITIZEN", "LEGAL_PERMANENT_RESIDENT", "CUBAN_HAITIAN_ENTRANT", "DACA", "UNDOCUMENTED", + "PAROLED_ONE_YEAR", + "REFUGEE", + "ASYLEE", + "DEPORTATION_WITHHELD", + "TPS", ) #: Raw CPS ASEC person columns the derivation reads. All of them ship in the @@ -146,8 +179,7 @@ "A_MARITL", "A_SPOUSE", "A_HSCOL", - "WSAL_VAL", - "SEMP_VAL", + "A_LFSR", "MCARE", "CAID", "IHSFLG", @@ -169,11 +201,13 @@ _FORCED_CONTROL_KEYS = ("undocumented_workers", "undocumented_students") _ANCHOR_KEY = "undocumented_population_anchor" +_HUMANITARIAN_KEY = "humanitarian_status_stocks" _DERIVE_IMMIGRATION_STATUS_PARAMETER_KEYS = frozenset( { *_FORCED_CONTROL_KEYS, _ANCHOR_KEY, + _HUMANITARIAN_KEY, "seed_from_build_config", "time_period_from_build_config", } @@ -212,7 +246,6 @@ 26: 2019, 27: 2021, 28: 2023, - 29: 2024, } #: PENATVTY codes for Cuba and Haiti; the Cuban/Haitian entrant class exists @@ -226,15 +259,194 @@ _DACA_MAX_AGE_AT_ENTRY = 16 _DACA_MIN_CURRENT_AGE = 15 +#: Humanitarian draw order: decreasing target precision (exact per-origin +#: program admissions first, national flow aggregates next, per-country TPS +#: registrations last). Draws are sequential and mutually exclusive. +HUMANITARIAN_STATUS_CATEGORIES: tuple[str, ...] = ( + "paroled_one_year", + "refugee", + "asylee", + "deportation_withheld", + "tps", +) + +_HUMANITARIAN_STATUS_BY_CATEGORY: Mapping[str, str] = { + "paroled_one_year": "PAROLED_ONE_YEAR", + "refugee": "REFUGEE", + "asylee": "ASYLEE", + "deportation_withheld": "DEPORTATION_WITHHELD", + "tps": "TPS", +} + +# Pew's residual estimates use a deliberately broader universe than the +# engine's ``UNDOCUMENTED`` enum. In particular, Pew retains DACA recipients +# and people with temporary protection (including parole and TPS) in its +# unauthorized-immigrant totals. Refugees and people already granted asylum +# are lawfully admitted statuses and therefore stay outside this universe. +# ``CUBAN_HAITIAN_ENTRANT`` is conditional: only EAD rows that came from the +# residual pool count, not assumed-documented entrants sharing that broad +# engine label. +_PEW_UNAUTHORIZED_STATUS_VALUES: tuple[str, ...] = ( + "UNDOCUMENTED", + "DACA", + "PAROLED_ONE_YEAR", + "DEPORTATION_WITHHELD", + "TPS", +) +_PEW_INCLUDED_HUMANITARIAN_STATUS_VALUES: tuple[str, ...] = ( + "PAROLED_ONE_YEAR", + "DEPORTATION_WITHHELD", + "TPS", +) + +# CPS ASEC A_LFSR: 1 working, 2 with a job/not at work, 3 unemployed/looking, +# 4 unemployed/on layoff. Code 0 covers children and Armed Forces; 7 is not +# in the labor force. Pew's labor-force total is for people age 16 or older. +_CPS_LABOR_FORCE_STATUS_CODES = (1, 2, 3, 4) +_CPS_LABOR_FORCE_STATUS_DOMAIN = (0, 1, 2, 3, 4, 7) + +#: PEINUSYR codes for 2020+ arrivals. The 2024 ASEC distinguishes 2020-2021 +#: (27) from 2022-2024 (28). Program-specific rules below do not treat those +#: windows as interchangeable: U4U and CHNV cannot draw from code 27, while +#: Operation Allies Welcome can. +_RECENT_ARRIVAL_CODES = (27, 28) +_PAROLE_ASEC_ARRIVAL_CODES: Mapping[str, tuple[int, ...]] = { + "afghanistan": _RECENT_ARRIVAL_CODES, + "ukraine": (28,), + "nicaragua": (28,), + "venezuela": (28,), +} +_PAROLE_ACS_MIN_ARRIVAL_YEAR: Mapping[str, int] = { + "afghanistan": 2021, + "ukraine": 2022, + "nicaragua": 2023, + "venezuela": 2022, +} +#: Asylum grants lag arrival by filing queues plus court backlog; grantees +#: in 2022-2024 overwhelmingly arrived 2016+ (OHSS asylee flow reports). +_ASYLEE_ARRIVAL_CODES = (25, 26, 27, 28) +#: Withholding of removal follows years of proceedings: settled arrivals only. +_WITHHELD_MAX_ARRIVAL_CODE = 26 + +#: PENATVTY birth-country codes, verified against Census CPS technical +#: documentation (cpsmar24.pdf) Appendix I "Countries and Areas of the +#: World". Cuba (327) and Haiti (332) stay out of every humanitarian draw: +#: the Cuban/Haitian-entrant class is the better statutory label. +#: +#: Parole per-origin keys: Operation Allies Welcome (Afghanistan), Uniting +#: for Ukraine, and the non-Cuban/Haitian CHNV nationalities. +_PAROLE_ORIGIN_CODES: Mapping[str, tuple[int, ...]] = { + "afghanistan": (200,), + "ukraine": (164,), + "nicaragua": (315,), + "venezuela": (373,), +} + +#: Top refugee-resettlement nationalities, OHSS Refugees Annual Flow Report +#: FY2023 Table 3 and the OHSS FY-24 report (DR Congo, Syria, Afghanistan, +#: Burma, Venezuela lead; Congo appears as both 412 Congo and 459 Zaire). +_REFUGEE_ORIGIN_CODES: tuple[int, ...] = ( + 412, # Congo + 459, # Zaire (Democratic Republic of the Congo) + 239, # Syria + 200, # Afghanistan + 205, # Myanmar (Burma) + 373, # Venezuela + 448, # Somalia + 451, # Sudan + 417, # Eritrea + 213, # Iraq + 313, # Guatemala + 164, # Ukraine +) + +#: Top asylum-grant nationalities, OHSS Asylees flow reports FY2022-FY2024 +#: (Afghanistan, China, Venezuela, Russia lead; Central/South America and +#: Egypt/Cameroon/Turkey round out the recurring top grant countries). +_ASYLEE_ORIGIN_CODES: tuple[int, ...] = ( + 200, # Afghanistan + 207, # China + 373, # Venezuela + 163, # Russia + 312, # El Salvador + 313, # Guatemala + 314, # Honduras + 315, # Nicaragua + 210, # India + 414, # Egypt + 407, # Cameroon + 364, # Colombia + 365, # Ecuador + 370, # Peru + 243, # Turkey + 239, # Syria +) + +#: TPS per-country keys: birth codes plus the latest PEINUSYR arrival code +#: compatible with the designation's required continuous-residence date +#: (CRS RS20844 Table 1). Legacy designations bind hard (El Salvador +#: 2001-02-13 -> code 17; Honduras/Nicaragua 1998-12-30 -> code 16; Nepal +#: 2015-06-24 -> code 24); 2021+ designations reach the top code, where the +#: 2022-2024 bin unavoidably includes some post-cutoff arrivals. +#: Ukraine and Afghanistan TPS registrants are carried by the parole draw +#: (same U4U/OAW people); Haiti TPS is carried by CUBAN_HAITIAN_ENTRANT. +_TPS_ORIGIN_CODES: Mapping[str, tuple[tuple[int, ...], int]] = { + "venezuela": ((373,), 28), + "el_salvador": ((312,), 17), + "honduras": ((314,), 16), + "nicaragua": ((315,), 16), + "nepal": ((229,), 24), + # Burma, Syria, Yemen, Lebanon, Cameroon, Ethiopia, Somalia, Sudan + # (South Sudan shares 451 in the CPS codebook). + "other_designated": ((205, 239, 248, 224, 407, 416, 448, 451), 28), +} + +#: ACS YOEP preserves an exact year, unlike the ASEC arrival bins. Use that +#: extra information for continuous-residence compatibility rather than +#: deliberately coarsening YOEP back to PEINUSYR. The per-country cutoffs are +#: calendar-year approximations of the designation dates; ASEC retains the +#: published coarse-bin contract above. +_TPS_ACS_MAX_ARRIVAL_YEAR_BY_BIRTH: Mapping[int, int] = { + 373: 2023, # Venezuela (2021 and 2023 designations) + 312: 2001, # El Salvador + 314: 1998, # Honduras + 315: 1998, # Nicaragua + 229: 2015, # Nepal + 205: 2024, # Burma + 239: 2024, # Syria + 248: 2024, # Yemen + 224: 2024, # Lebanon + 407: 2023, # Cameroon + 416: 2024, # Ethiopia + 448: 2024, # Somalia + 451: 2023, # Sudan / South Sudan (shared survey code) +} + +#: Categories whose flat manifest block carries one national target. +_FLAT_HUMANITARIAN_CATEGORIES = ("refugee", "asylee", "deportation_withheld") +#: Categories whose manifest block carries per-origin targets, and the +#: module table each block's keys must match exactly. +_PER_ORIGIN_HUMANITARIAN_CATEGORIES: Mapping[str, tuple[str, ...]] = { + "paroled_one_year": tuple(_PAROLE_ORIGIN_CODES), + "tps": tuple(_TPS_ORIGIN_CODES), +} + #: Weighted share of persons whose SSN card type is not ``CITIZEN`` must land #: in this band (Census counts ~22–27M non-citizens of ~336M residents; a #: share outside it means the imputation collapsed or exploded). _NON_CITIZEN_SHARE_BAND = (0.03, 0.12) -#: Emergent weighted ``NONE`` (undocumented) population relative to the cited -#: published anchor. Coarse by design — a release-blocking backstop against -#: collapse or explosion, not a calibration objective; the level belongs to -#: the calibration lane. +#: Emergent Pew-defined unauthorized population relative to the cited +#: published anchor. This includes DACA and specified temporary protections, +#: not only the engine's ``NONE``/``UNDOCUMENTED`` pair. Coarse by design — a +#: release-blocking backstop against collapse or explosion, not a calibration +#: objective; the level belongs to the calibration lane. _UNDOCUMENTED_ANCHOR_RELATIVE_BAND = (0.5, 1.6) +#: Emitted humanitarian mass per category relative to its cited target. +#: The draw forces the target when candidates suffice, so the band only +#: bites on pool exhaustion (the ASEC undercovers 2022-24 arrivals — the +#: Ukraine parole pool saturates below its admin count) or collapse. A +#: category with an explicit zero target must emit exactly zero. +_HUMANITARIAN_TARGET_RELATIVE_BAND = (0.35, 1.5) @dataclass(frozen=True) @@ -242,11 +454,12 @@ class UndocumentedControls: """Published totals the stage forces or gates against. Attributes: - workers: Undocumented-worker control the EAD spill enforces. - students: Undocumented-student control the EAD spill enforces. - population_anchor: Published total undocumented population the - composition gate checks the *emergent* total against (never - forced in the labels). + workers: Pew unauthorized-immigrant labor-force control the EAD spill + enforces, including DACA and temporary protections Pew retains. + students: Broad undocumented-student control the EAD spill enforces. + population_anchor: Published Pew unauthorized population the + composition gate checks the *emergent broad-universe* total + against (never forced in the labels). sources: Citation URL per key, straight from the manifest. """ @@ -273,6 +486,85 @@ def __post_init__(self) -> None: ) +@dataclass(frozen=True) +class HumanitarianDraw: + """One seeded humanitarian draw: a cited stock the stage selects toward. + + Attributes: + category: Manifest category key (``paroled_one_year`` … ``tps``). + origin: Per-origin key inside the category, or ``None`` for a + national draw. + status: ``ImmigrationStatus`` enum member name the draw emits. + target: Cited weighted stock. Zero is allowed and means the + category is explicitly not imputed (the citation documents why). + source: Citation URL or reference for the target. + """ + + category: str + origin: str | None + status: str + target: float + source: str + + def __post_init__(self) -> None: + if not np.isfinite(self.target) or self.target < 0: + raise ValueError( + f"Humanitarian target {self.label!r} must be non-negative, " + f"got {self.target!r}." + ) + if not self.source: + raise ValueError( + f"Humanitarian target {self.label!r} requires a source citation." + ) + + @property + def label(self) -> str: + return ( + self.category if self.origin is None else f"{self.category}:{self.origin}" + ) + + @property + def salt(self) -> str: + return f"immigration:{self.label}" + + +@dataclass(frozen=True) +class ImmigrationControls: + """Every manifest-sourced control the immigration stage consumes. + + Attributes: + undocumented: The legacy forced margins and the composition-gate + anchor for the residual pool. + humanitarian: The humanitarian draws in assignment order — the + per-origin draws of a category are adjacent, categories follow + ``HUMANITARIAN_STATUS_CATEGORIES``. + """ + + undocumented: UndocumentedControls + humanitarian: tuple[HumanitarianDraw, ...] + + def humanitarian_target(self, category: str) -> float: + """Summed cited target for one category across its origins.""" + + return float( + sum(draw.target for draw in self.humanitarian if draw.category == category) + ) + + +@dataclass(frozen=True) +class _ImmigrationEvidenceProfile: + """Canonical immigration evidence shared by ASEC and ACS rows.""" + + source_is_acs: np.ndarray + is_citizen: np.ndarray + birth_country: np.ndarray + arrival_code: np.ndarray + arrival_year: np.ndarray + age: np.ndarray + age_at_entry: np.ndarray + tps_acs_max_arrival_year: np.ndarray + + def us_immigration_stage_spec() -> SourceStageSpec: """Load the packaged ``immigration_status`` source-stage manifest entry.""" @@ -346,15 +638,17 @@ def derive_us_immigration_status_from_manifest( raise SourceRuntimeError( "US immigration derivation requires finite non-negative person weights." ) - ssn_codes = _assign_ssn_card_codes( + ssn_codes, humanitarian_marks = _assign_ssn_card_codes( result, weights.to_numpy(dtype=np.float64), seed=int(context.config.seed), controls=controls, + time_period=int(context.config.target_year), ) status = _derive_immigration_status( result, ssn_codes, + humanitarian_marks, time_period=int(context.config.target_year), ) code_to_name = { @@ -368,7 +662,118 @@ def derive_us_immigration_status_from_manifest( return result -def _controls_from_parameters(params: Mapping[str, object]) -> UndocumentedControls: +def _target_and_source( + block: object, + *, + label: str, + value_key: str, + allow_zero: bool, +) -> tuple[float, str]: + if not isinstance(block, Mapping): + raise SourceRuntimeError( + f"US immigration control {label!r} requires an object with " + f"{value_key!r} and 'source'." + ) + unexpected = sorted(set(block) - {value_key, "source"}) + if unexpected: + raise SourceRuntimeError( + f"US immigration control {label!r} has unsupported key(s): {unexpected}." + ) + value = block.get(value_key) + source = block.get("source") + valid_number = isinstance(value, int | float) and not isinstance(value, bool) + if not valid_number or (float(value) <= 0 and not (allow_zero and value == 0)): + requirement = "non-negative" if allow_zero else "positive" + raise SourceRuntimeError( + f"US immigration control {label!r} requires a {requirement} numeric " + f"{value_key!r}." + ) + if not isinstance(source, str) or not source: + raise SourceRuntimeError( + f"US immigration control {label!r} requires a source citation." + ) + return float(value), source + + +def _humanitarian_draws_from_parameters( + params: Mapping[str, object], +) -> tuple[HumanitarianDraw, ...]: + block = params.get(_HUMANITARIAN_KEY) + if not isinstance(block, Mapping): + raise SourceRuntimeError( + f"US immigration derivation requires a {_HUMANITARIAN_KEY!r} object " + f"with one block per category {list(HUMANITARIAN_STATUS_CATEGORIES)}." + ) + unexpected = sorted(set(block) - set(HUMANITARIAN_STATUS_CATEGORIES)) + if unexpected: + raise SourceRuntimeError( + f"US immigration {_HUMANITARIAN_KEY} has unsupported category(ies): " + f"{unexpected}." + ) + missing = sorted(set(HUMANITARIAN_STATUS_CATEGORIES) - set(block)) + if missing: + raise SourceRuntimeError( + f"US immigration {_HUMANITARIAN_KEY} is missing category(ies): {missing}." + ) + draws: list[HumanitarianDraw] = [] + for category in HUMANITARIAN_STATUS_CATEGORIES: + status = _HUMANITARIAN_STATUS_BY_CATEGORY[category] + category_block = block[category] + if category in _PER_ORIGIN_HUMANITARIAN_CATEGORIES: + origin_keys = _PER_ORIGIN_HUMANITARIAN_CATEGORIES[category] + if not isinstance(category_block, Mapping): + raise SourceRuntimeError( + f"US immigration control {category!r} requires per-origin " + f"blocks {list(origin_keys)}." + ) + unexpected_origins = sorted(set(category_block) - set(origin_keys)) + if unexpected_origins: + raise SourceRuntimeError( + f"US immigration control {category!r} has origin(s) outside " + f"the stage's codebook table: {unexpected_origins}." + ) + missing_origins = sorted(set(origin_keys) - set(category_block)) + if missing_origins: + raise SourceRuntimeError( + f"US immigration control {category!r} is missing origin(s): " + f"{missing_origins}." + ) + for origin in origin_keys: + target, source = _target_and_source( + category_block[origin], + label=f"{category}:{origin}", + value_key="target", + allow_zero=True, + ) + draws.append( + HumanitarianDraw( + category=category, + origin=origin, + status=status, + target=target, + source=source, + ) + ) + else: + target, source = _target_and_source( + category_block, + label=category, + value_key="target", + allow_zero=True, + ) + draws.append( + HumanitarianDraw( + category=category, + origin=None, + status=status, + target=target, + source=source, + ) + ) + return tuple(draws) + + +def _controls_from_parameters(params: Mapping[str, object]) -> ImmigrationControls: values: dict[str, float] = {} sources: dict[str, str] = {} for key, value_key in ( @@ -376,40 +781,37 @@ def _controls_from_parameters(params: Mapping[str, object]) -> UndocumentedContr ("undocumented_students", "target"), (_ANCHOR_KEY, "value"), ): - block = params.get(key) - if not isinstance(block, Mapping): - raise SourceRuntimeError( - f"US immigration derivation requires a {key!r} object with " - f"{value_key!r} and 'source'." - ) - unexpected = sorted(set(block) - {value_key, "source"}) - if unexpected: - raise SourceRuntimeError( - f"US immigration control {key!r} has unsupported key(s): {unexpected}." - ) - value = block.get(value_key) - source = block.get("source") - if ( - not isinstance(value, int | float) - or isinstance(value, bool) - or float(value) <= 0 - ): - raise SourceRuntimeError( - f"US immigration control {key!r} requires a positive numeric " - f"{value_key!r}." - ) - if not isinstance(source, str) or not source: - raise SourceRuntimeError( - f"US immigration control {key!r} requires a source citation." - ) - values[key] = float(value) + value, source = _target_and_source( + params.get(key), label=key, value_key=value_key, allow_zero=False + ) + values[key] = value sources[key] = source - return UndocumentedControls( + undocumented = UndocumentedControls( workers=values["undocumented_workers"], students=values["undocumented_students"], population_anchor=values[_ANCHOR_KEY], sources=sources, ) + return ImmigrationControls( + undocumented=undocumented, + humanitarian=_humanitarian_draws_from_parameters(params), + ) + + +def us_immigration_controls() -> ImmigrationControls: + """Return the controls bound to the packaged immigration manifest stage.""" + + derive = [ + operation + for operation in us_immigration_stage_spec().operations + if operation.kind == "derive_immigration_status" + ] + if len(derive) != 1: + raise ValueError( + "US immigration stage must declare exactly one " + "derive_immigration_status operation." + ) + return _controls_from_parameters(derive[0].parameters) def _integer_column(person: pd.DataFrame, column: str) -> np.ndarray: @@ -428,23 +830,199 @@ def _float_column(person: pd.DataFrame, column: str) -> np.ndarray: ) +def _numeric_evidence_column(person: pd.DataFrame, column: str) -> np.ndarray: + if column not in person.columns: + return np.full(len(person), np.nan, dtype=np.float64) + return pd.to_numeric(person[column], errors="coerce").to_numpy( + dtype=np.float64, + na_value=np.nan, + ) + + +def _source_aware_immigration_profile( + person: pd.DataFrame, + *, + time_period: int, +) -> _ImmigrationEvidenceProfile: + """Normalize ASEC PRCITSHP/PENATVTY/PEINUSYR and ACS CIT/POBP/YOEP. + + A stacked recipient carries the union of both raw survey schemas, with the + non-owning source null on each row. Exactly one citizenship source must be + present. Relevant POBP country codes share the Census country codebook with + PENATVTY, so the normalized origin is lossless; ACS YOEP retains its exact + year while ASEC necessarily uses the documented interval midpoint. + """ + + if isinstance(time_period, bool) or int(time_period) != time_period: + raise ValueError( + f"US immigration time_period must be an integer, got {time_period!r}." + ) + time_period = int(time_period) + asec_citizenship = _numeric_evidence_column(person, "PRCITSHP") + acs_citizenship = _numeric_evidence_column(person, "CIT") + has_asec = np.isfinite(asec_citizenship) + has_acs = np.isfinite(acs_citizenship) + ambiguous = has_asec & has_acs + missing = ~has_asec & ~has_acs + if ambiguous.any() or missing.any(): + raise SourceRuntimeError( + "US immigration evidence requires exactly one row-level citizenship " + "source (ASEC PRCITSHP or ACS CIT); " + f"ambiguous_rows={int(ambiguous.sum())}, missing_rows={int(missing.sum())}." + ) + + for label, values, present in ( + ("PRCITSHP", asec_citizenship, has_asec), + ("CIT", acs_citizenship, has_acs), + ): + invalid = present & ( + ~np.equal(values, np.rint(values)) | ~np.isin(values, [1, 2, 3, 4, 5]) + ) + if invalid.any(): + bad = sorted(set(values[invalid].tolist()))[:5] + raise SourceRuntimeError( + f"{label} carries value(s) outside the Census domain 1..5: {bad}." + ) + + citizenship = np.where(has_acs, acs_citizenship, asec_citizenship).astype(np.int64) + is_citizen = np.isin(citizenship, [1, 2, 3, 4]) + + asec_birth = _numeric_evidence_column(person, "PENATVTY") + acs_birth = _numeric_evidence_column(person, "POBP") + birth = np.where(has_acs, acs_birth, asec_birth) + invalid_birth = ~np.isfinite(birth) | ~np.equal(birth, np.rint(birth)) | (birth < 1) + if invalid_birth.any(): + raise SourceRuntimeError( + "US immigration evidence requires a positive integral PENATVTY/POBP " + f"country code on every row; invalid_rows={int(invalid_birth.sum())}." + ) + birth_country = birth.astype(np.int64) + + arrival_code_values = _numeric_evidence_column(person, "PEINUSYR") + invalid_asec_arrival = has_asec & ( + ~np.isfinite(arrival_code_values) + | ~np.equal(arrival_code_values, np.rint(arrival_code_values)) + | (arrival_code_values < 0) + | (arrival_code_values > max(_ARRIVAL_YEAR_MIDPOINTS)) + ) + if invalid_asec_arrival.any(): + bad = sorted(set(arrival_code_values[invalid_asec_arrival].tolist()))[:5] + raise SourceRuntimeError( + "PEINUSYR carries value(s) outside the 2024 ASEC domain " + f"0..{max(_ARRIVAL_YEAR_MIDPOINTS)}: {bad}." + ) + arrival_code = np.zeros(len(person), dtype=np.int64) + arrival_code[has_asec] = arrival_code_values[has_asec].astype(np.int64) + arrival_year = np.full(len(person), time_period, dtype=np.int64) + for code, midpoint in _ARRIVAL_YEAR_MIDPOINTS.items(): + arrival_year[has_asec & (arrival_code == code)] = midpoint + + acs_arrival = _numeric_evidence_column(person, "YOEP") + acs_foreign_born = has_acs & np.isin(citizenship, [4, 5]) + invalid_acs_arrival = acs_foreign_born & ( + ~np.isfinite(acs_arrival) + | ~np.equal(acs_arrival, np.rint(acs_arrival)) + | (acs_arrival < 1900) + | (acs_arrival > time_period) + ) + if invalid_acs_arrival.any(): + raise SourceRuntimeError( + "ACS YOEP must be an integral 1900..time_period year for every " + f"foreign-born CIT=4/5 row; invalid_rows={int(invalid_acs_arrival.sum())}." + ) + observed_acs_arrival = has_acs & np.isfinite(acs_arrival) + arrival_year[observed_acs_arrival] = acs_arrival[observed_acs_arrival].astype( + np.int64 + ) + + age_values = _numeric_evidence_column(person, "A_AGE") + if "age" in person.columns: + mapped_age = _numeric_evidence_column(person, "age") + age_values = np.where(np.isfinite(age_values), age_values, mapped_age) + invalid_age = ( + ~np.isfinite(age_values) + | ~np.equal(age_values, np.rint(age_values)) + | (age_values < 0) + ) + if invalid_age.any(): + raise SourceRuntimeError( + "US immigration evidence requires a finite non-negative integral " + f"A_AGE/age on every row; invalid_rows={int(invalid_age.sum())}." + ) + age = age_values.astype(np.int64) + age_at_entry = np.maximum(0, age - (time_period - arrival_year)) + tps_acs_max_arrival_year = np.full(len(person), -1, dtype=np.int64) + for birth_code, cutoff in _TPS_ACS_MAX_ARRIVAL_YEAR_BY_BIRTH.items(): + tps_acs_max_arrival_year[birth_country == birth_code] = cutoff + return _ImmigrationEvidenceProfile( + source_is_acs=has_acs, + is_citizen=is_citizen, + birth_country=birth_country, + arrival_code=arrival_code, + arrival_year=arrival_year, + age=age, + age_at_entry=age_at_entry, + tps_acs_max_arrival_year=tps_acs_max_arrival_year, + ) + + +def us_immigration_evidence_features( + frame: Frame, + *, + time_period: int = 2024, +) -> pd.DataFrame: + """Return finite, source-harmonized immigration predictors per person.""" + + person = frame.table(frame.schema.person_entity) + profile = _source_aware_immigration_profile(person, time_period=time_period) + return pd.DataFrame( + { + "is_us_citizen": profile.is_citizen.astype(np.float64), + "birth_country_code": profile.birth_country.astype(np.float64), + "arrival_year": profile.arrival_year.astype(np.float64), + }, + index=person.index, + ) + + def _stable_person_draws(person: pd.DataFrame, *, seed: int, salt: str) -> np.ndarray: """Deterministic uniform draws keyed by stable person identity. Support-channel clones carry their source person's ``source_year`` / - ``source_person_id``, so keying on those gives every clone of one source - person the same draw; frames without source ids fall back to - ``person_id``. + ``source_household_id`` / ``source_person_id``, so keying on those gives + every clone of one source person the same draw; frames without full source + ids fall back to the legacy source pair or ``person_id``. """ - if {"source_year", "source_person_id"}.issubset(person.columns): + def complete(columns: tuple[str, ...]) -> bool: + return set(columns).issubset(person.columns) and bool( + person.loc[:, list(columns)].notna().to_numpy(dtype=bool).all() + ) + + full_lineage = ("source_year", "source_household_id", "source_person_id") + legacy_lineage = ("source_year", "source_person_id") + if complete(full_lineage): keys = ( person["source_year"].astype(str) + ":" + + person["source_household_id"].astype(str) + + ":" + person["source_person_id"].astype(str) ) - else: + elif complete(legacy_lineage): + keys = ( + person["source_year"].astype(str) + + ":" + + person["source_person_id"].astype(str) + ) + elif complete(("person_id",)): keys = person["person_id"].astype(str) + else: + raise SourceRuntimeError( + "US immigration deterministic draws require one complete stable " + "person-lineage alternative: source_year/source_household_id/" + "source_person_id, source_year/source_person_id, or person_id." + ) denominator = float(2**64) return np.fromiter( ( @@ -500,13 +1078,770 @@ def _select_weight_to_target( return selected -def _assign_ssn_card_codes( +def _arrival_profile( + person: pd.DataFrame, *, time_period: int +) -> tuple[np.ndarray, np.ndarray, np.ndarray]: + """(PEINUSYR code, arrival-year midpoint, age at entry) per person.""" + + arrival_code = _integer_column(person, "PEINUSYR") + arrival_year = np.full(len(person), time_period, dtype=np.int64) + for code, midpoint in _ARRIVAL_YEAR_MIDPOINTS.items(): + arrival_year[arrival_code == code] = midpoint + years_in_us = time_period - arrival_year + age = _integer_column(person, "A_AGE") + age_at_entry = np.maximum(0, age - years_in_us) + return arrival_code, arrival_year, age_at_entry + + +def _daca_statutory_cohort( + arrival_year: np.ndarray, age_at_entry: np.ndarray, age: np.ndarray +) -> np.ndarray: + return ( + (arrival_year <= _DACA_LATEST_ARRIVAL_YEAR) + & (age_at_entry < _DACA_MAX_AGE_AT_ENTRY) + & (age >= _DACA_MIN_CURRENT_AGE) + ) + + +def _special_status_masks( + *, + ssn_codes: np.ndarray, + birth_country: np.ndarray, + arrival_year: np.ndarray, + age_at_entry: np.ndarray, + age: np.ndarray, +) -> tuple[np.ndarray, np.ndarray]: + """Return final Cuban/Haitian-entrant and DACA cohort masks. + + The source derivation applies DACA after the Cuban/Haitian entrant rule, + so a qualifying EAD holder in both cohorts retains DACA. Keeping that + precedence in one helper lets source derivation, recipient reconciliation, + and release validation share the same exact contract. + """ + + documented_noncitizen = np.isin(ssn_codes, [2, 3]) + daca = (ssn_codes == 2) & _daca_statutory_cohort( + arrival_year, + age_at_entry, + age, + ) + cuban_haitian = ( + documented_noncitizen + & np.isin(birth_country, _CUBAN_HAITIAN_BIRTH_CODES) + & (arrival_year >= _CUBAN_HAITIAN_ARRIVAL_CUTOFF) + & ~daca + ) + return cuban_haitian, daca + + +def _pew_unauthorized_projection_mask( + *, + ssn_codes: np.ndarray, + humanitarian_marks: np.ndarray, + retained_ead_cohort: np.ndarray, +) -> np.ndarray: + """Project final membership in Pew's broad unauthorized universe. + + This helper runs while EAD assignment is still in progress. Residual + ``NONE`` rows count directly; DACA and residual Cuban/Haitian cohort rows + continue to count if an EAD spill gives them a protected engine label; + and the temporary-protection statuses Pew explicitly retains count + regardless of their non-``NONE`` SSN code. + """ + + return ( + (ssn_codes == 0) + | ((ssn_codes == 2) & retained_ead_cohort) + | np.isin( + humanitarian_marks, + _PEW_INCLUDED_HUMANITARIAN_STATUS_VALUES, + ) + ) + + +def _spill_pew_unauthorized_excess( + person: pd.DataFrame, + ssn_codes: np.ndarray, + humanitarian_marks: np.ndarray, + weights: np.ndarray, + *, + noncitizens: np.ndarray, + scope: np.ndarray, + preserve_scope: np.ndarray, + retained_ead_cohort: np.ndarray, + target: float, + seed: int, + salt: str, +) -> None: + """Spill a broad Pew-universe margin to EAD in deterministic order. + + The first draw preserves the prior EAD/DACA allocation behavior by + selecting from the whole residual scope. A selected DACA or residual + Cuban/Haitian cohort row still belongs to Pew's universe, so a second pass + spills additional rows outside those retained cohorts when needed. Within + each pass, rows outside the other controlled margin are selected first; + this keeps the student and worker controls from disturbing one another + unless their overlap makes that unavoidable. + """ + + draws = _stable_person_draws(person, seed=seed, salt=salt) + + def current_excess() -> float: + included = _pew_unauthorized_projection_mask( + ssn_codes=ssn_codes, + humanitarian_marks=humanitarian_marks, + retained_ead_cohort=retained_ead_cohort, + ) + return float(weights[scope & included].sum()) - target + + # Preserve the existing seeded EAD allocation surface. DACA selections + # do not reduce the Pew count and are compensated by the corrective pass. + initial_candidates = (ssn_codes == 0) & noncitizens & scope + initial_excess = current_excess() + if initial_excess > 0: + for priority in (~preserve_scope, preserve_scope): + selected = _select_weight_to_target( + initial_candidates & priority, + weights, + draws, + current_excess(), + ) + ssn_codes[selected] = 2 + if current_excess() <= 0: + return + + # DACA and residual Cuban/Haitian cohort members retain protected engine + # labels after EAD assignment and therefore remain inside Pew's estimate. + # Only rows outside those cohorts can close any remaining gap. + for priority in (~preserve_scope, preserve_scope): + corrective_candidates = ( + (ssn_codes == 0) & noncitizens & scope & ~retained_ead_cohort & priority + ) + selected = _select_weight_to_target( + corrective_candidates, + weights, + draws, + current_excess(), + ) + ssn_codes[selected] = 2 + if current_excess() <= 0: + return + + +def _humanitarian_draw_candidates( + draw: HumanitarianDraw, + *, + ssn_codes: np.ndarray, + profile: _ImmigrationEvidenceProfile, +) -> np.ndarray: + """Candidate mask for one draw, before exclusions shared by all draws. + + ``REFUGEE``/``ASYLEE``/``DEPORTATION_WITHHELD`` draw from the + indicator-documented pool only (code 3: those statuses carry immediate + federal benefits access, the same signals the ASEC-UA method reads); + parole and TPS may also draw from the still-unspilled residual pool + (code 0), whose members then receive EAD-backed SSN cards. + """ + + noncitizen = ~profile.is_citizen + excluded = np.isin(profile.birth_country, _CUBAN_HAITIAN_BIRTH_CODES) + excluded |= _daca_statutory_cohort( + profile.arrival_year, + profile.age_at_entry, + profile.age, + ) + + if draw.category == "paroled_one_year": + origins = ( + (draw.origin,) if draw.origin is not None else tuple(_PAROLE_ORIGIN_CODES) + ) + cohort = np.zeros(len(ssn_codes), dtype=bool) + for origin in origins: + if origin not in _PAROLE_ORIGIN_CODES: + raise SourceRuntimeError( + f"US immigration has no parole origin rule for {origin!r}." + ) + origin_match = np.isin( + profile.birth_country, + _PAROLE_ORIGIN_CODES[origin], + ) + asec_window = (~profile.source_is_acs) & np.isin( + profile.arrival_code, + _PAROLE_ASEC_ARRIVAL_CODES[origin], + ) + acs_window = profile.source_is_acs & ( + profile.arrival_year >= _PAROLE_ACS_MIN_ARRIVAL_YEAR[origin] + ) + cohort |= origin_match & (asec_window | acs_window) + return noncitizen & np.isin(ssn_codes, (0, 3)) & cohort & ~excluded + if draw.category == "refugee": + return ( + noncitizen + & (ssn_codes == 3) + & np.isin(profile.birth_country, _REFUGEE_ORIGIN_CODES) + & ( + ((~profile.source_is_acs) & (profile.arrival_code == 28)) + | (profile.source_is_acs & (profile.arrival_year >= 2022)) + ) + & ~excluded + ) + if draw.category == "asylee": + return ( + noncitizen + & (ssn_codes == 3) + & np.isin(profile.birth_country, _ASYLEE_ORIGIN_CODES) + & ( + ( + (~profile.source_is_acs) + & np.isin(profile.arrival_code, _ASYLEE_ARRIVAL_CODES) + ) + | (profile.source_is_acs & (profile.arrival_year >= 2016)) + ) + & ~excluded + ) + if draw.category == "deportation_withheld": + return ( + noncitizen + & (ssn_codes == 3) + & ( + ( + (~profile.source_is_acs) + & (profile.arrival_code >= 1) + & (profile.arrival_code <= _WITHHELD_MAX_ARRIVAL_CODE) + ) + | (profile.source_is_acs & (profile.arrival_year <= 2019)) + ) + & ~excluded + ) + if draw.category == "tps": + origins = ( + (draw.origin,) if draw.origin is not None else tuple(_TPS_ORIGIN_CODES) + ) + cohort = np.zeros(len(ssn_codes), dtype=bool) + for origin in origins: + if origin not in _TPS_ORIGIN_CODES: + raise SourceRuntimeError( + f"US immigration has no TPS origin rule for {origin!r}." + ) + codes, max_arrival_code = _TPS_ORIGIN_CODES[origin] + origin_match = np.isin(profile.birth_country, codes) + asec_window = ( + (~profile.source_is_acs) + & (profile.arrival_code >= 1) + & (profile.arrival_code <= max_arrival_code) + ) + acs_window = ( + profile.source_is_acs + & (profile.tps_acs_max_arrival_year >= 0) + & (profile.arrival_year <= profile.tps_acs_max_arrival_year) + ) + cohort |= origin_match & (asec_window | acs_window) + return noncitizen & np.isin(ssn_codes, (0, 3)) & cohort & ~excluded + raise SourceRuntimeError( + f"US immigration derivation has no candidate rule for draw {draw.label!r}." + ) + + +def us_immigration_humanitarian_draw_mask( + frame: Frame, + draw: HumanitarianDraw, + *, + time_period: int = 2024, +) -> np.ndarray: + """Rows emitted as ``draw`` whose source evidence supports that label. + + This is the shared hard-cohort contract for transfer reconciliation, + calibration, and the release gate. It intentionally includes the emitted + status test: callers receive the achieved population for one manifest + draw, not the larger pool of possible candidates. + """ + + person = frame.table(frame.schema.person_entity) + missing = sorted(set(US_IMMIGRATION_OUTPUT_COLUMNS) - set(person.columns)) + if missing: + raise ValueError( + f"US humanitarian draw compatibility requires output column(s) {missing}." + ) + profile = _source_aware_immigration_profile(person, time_period=time_period) + return _humanitarian_emitted_mask(person, draw=draw, profile=profile) + + +def us_immigration_humanitarian_transfer_selection_masks( + frame: Frame, + *, + mutable_rows: np.ndarray, + seed: int, + time_period: int = 2024, + controls: ImmigrationControls | None = None, +) -> dict[str, np.ndarray]: + """Replay the exact deterministic mutable-row selection for every draw. + + The post-transfer frame retains everything needed to reconstruct the + reconciliation decision: source evidence, final SSN codes, stable person + lineage, resolved person weights, manifest controls, and draw order. For + parole and TPS, final EAD codes are mapped back to the residual-pool code + exactly as they are during reconciliation. Earlier expected selections, + rather than observed status labels, exclude rows from later draws, so a + downstream equal-mass status swap cannot redefine the replayed candidate + surface. + """ + + person = frame.table(frame.schema.person_entity) + missing = sorted(set(US_IMMIGRATION_OUTPUT_COLUMNS) - set(person.columns)) + if missing: + raise ValueError( + "Humanitarian transfer selection replay requires person column(s) " + f"{missing}." + ) + mutable = np.asarray(mutable_rows, dtype=bool) + if mutable.shape != (len(person),): + raise ValueError( + "Humanitarian transfer selection replay mutable_rows must align " + "one-to-one with the person table." + ) + weights = np.asarray( + frame.resolve_weights(frame.schema.person_entity).values, + dtype=np.float64, + ) + if ( + weights.shape != (len(person),) + or not np.isfinite(weights).all() + or (weights < 0).any() + ): + raise ValueError( + "Humanitarian transfer selection replay requires finite " + "non-negative person weights." + ) + controls = us_immigration_controls() if controls is None else controls + profile = _source_aware_immigration_profile(person, time_period=time_period) + ssn_codes = _ssn_name_codes(person) + immutable = ~mutable + selected_once = np.zeros(len(person), dtype=bool) + selections: dict[str, np.ndarray] = {} + for draw in controls.humanitarian: + immutable_draw = ( + _humanitarian_emitted_mask( + person, + draw=draw, + profile=profile, + ) + & immutable + ) + residual_target = max( + 0.0, + float(draw.target) - float(weights[immutable_draw].sum()), + ) + candidate_ssn = ssn_codes + if draw.category in {"paroled_one_year", "tps"}: + candidate_ssn = np.where(candidate_ssn == 2, 0, candidate_ssn) + candidates = ( + _humanitarian_draw_candidates( + draw, + ssn_codes=candidate_ssn, + profile=profile, + ) + & mutable + & ~selected_once + ) + stable_draws = _stable_person_draws( + person, + seed=seed, + salt=f"acs_transfer:{draw.salt}", + ) + selected = _select_weight_to_target( + candidates, + weights, + stable_draws, + residual_target, + ) + selections[draw.label] = selected + selected_once |= selected + return selections + + +def _ssn_name_codes(person: pd.DataFrame) -> np.ndarray: + ssn_lookup = { + "NONE": 0, + "CITIZEN": 1, + "NON_CITIZEN_VALID_EAD": 2, + "OTHER_NON_CITIZEN": 3, + } + return ( + person["ssn_card_type"] + .astype(str) + .map(ssn_lookup) + .fillna(-1) + .to_numpy(dtype=np.int64) + ) + + +def _humanitarian_emitted_mask( + person: pd.DataFrame, + *, + draw: HumanitarianDraw, + profile: _ImmigrationEvidenceProfile, +) -> np.ndarray: + ssn_codes = _ssn_name_codes(person) + candidates = _humanitarian_draw_candidates( + draw, + ssn_codes=ssn_codes, + profile=profile, + ) + # Parole and TPS selections from the residual pool are upgraded from NONE + # to EAD after candidacy is evaluated. Their emitted rows may therefore + # carry code 2 even though only code 0/3 was eligible before selection. + if draw.category in {"paroled_one_year", "tps"}: + candidates |= _humanitarian_draw_candidates( + draw, + ssn_codes=np.where(ssn_codes == 2, 0, ssn_codes), + profile=profile, + ) & (ssn_codes == 2) + status = person["immigration_status_str"].astype(str).to_numpy() + return np.asarray(candidates & (status == draw.status), dtype=bool) + + +def reconcile_us_immigration_humanitarian_transfer( person: pd.DataFrame, + *, + weights: np.ndarray, + mutable_rows: np.ndarray, + seed: int, + time_period: int = 2024, + controls: ImmigrationControls | None = None, +) -> tuple[pd.DataFrame, dict[str, object]]: + """Reconcile a transferred immigration pair to manifest draw targets. + + Immutable rows are the ASEC source contribution. Only rows whose paired + target cells were imputed may be repaired or selected. Each manifest draw + receives the residual of its national target after its compatible, + immutable contribution, in manifest order; candidate exhaustion is a hard + error rather than permission to ship a wrong-origin label. + """ + + result = person.copy() + missing = sorted(set(US_IMMIGRATION_OUTPUT_COLUMNS) - set(result.columns)) + if missing: + raise ValueError( + f"Humanitarian transfer reconciliation requires person column(s) {missing}." + ) + weights = np.asarray(weights, dtype=np.float64) + mutable = np.asarray(mutable_rows, dtype=bool) + if weights.shape != (len(result),) or mutable.shape != (len(result),): + raise ValueError( + "Humanitarian transfer weights and mutable_rows must align one-to-one " + "with the person table." + ) + if not np.isfinite(weights).all() or (weights < 0).any(): + raise ValueError( + "Humanitarian transfer reconciliation requires finite non-negative " + "person weights." + ) + controls = us_immigration_controls() if controls is None else controls + profile = _source_aware_immigration_profile(result, time_period=time_period) + original_ssn = result["ssn_card_type"].astype(str).to_numpy(copy=True) + original_status = result["immigration_status_str"].astype(str).to_numpy(copy=True) + ssn = original_ssn.copy() + status = original_status.copy() + valid_ssn = np.isin(ssn, SSN_CARD_TYPE_VALUES) + valid_status = np.isin(status, IMMIGRATION_STATUS_VALUES) + if (~valid_ssn).any() or (~valid_status).any(): + raise ValueError( + "Humanitarian transfer reconciliation received values outside the " + "PolicyEngine SSN/immigration enum domains." + ) + + humanitarian_statuses = set(_HUMANITARIAN_STATUS_BY_CATEGORY.values()) + evidence_derived_statuses = {"CUBAN_HAITIAN_ENTRANT", "DACA"} + constrained_statuses = humanitarian_statuses | evidence_derived_statuses + baseline_constrained = mutable & np.isin(status, list(constrained_statuses)) + if baseline_constrained.any(): + raise ValueError( + "ACS joint-QRF baseline emitted evidence-constrained labels before " + "reconciliation." + ) + + # Source citizenship is observed on both surveys. Normalize only mutable + # pairs; immutable ASEC disagreements remain an error below. + citizen_mutable = mutable & profile.is_citizen + ssn[citizen_mutable] = "CITIZEN" + status[citizen_mutable] = "CITIZEN" + noncitizen_mutable = mutable & ~profile.is_citizen + ssn[noncitizen_mutable & (ssn == "CITIZEN")] = "OTHER_NON_CITIZEN" + status[noncitizen_mutable & (status == "CITIZEN")] = "LEGAL_PERMANENT_RESIDENT" + status[noncitizen_mutable & (ssn == "NONE")] = "UNDOCUMENTED" + status[noncitizen_mutable & (ssn != "NONE") & (status == "UNDOCUMENTED")] = ( + "LEGAL_PERMANENT_RESIDENT" + ) + citizenship_repairs = int( + ( + mutable + & ((ssn != original_ssn) | (status != original_status)) + & ( + profile.is_citizen + | (original_ssn == "CITIZEN") + | (original_status == "CITIZEN") + ) + ).sum() + ) + pair_repairs = int( + (mutable & ((ssn != original_ssn) | (status != original_status))).sum() + ) + result["ssn_card_type"] = ssn + result["immigration_status_str"] = status + + # CHE and DACA have hard source-evidence rules just like the manifest + # draws, but no stock target. Rebuild them deterministically after the + # unconstrained codec and citizenship/pair repair rather than accepting a + # donor label on an incompatible ACS cohort. + special_ssn_codes = _ssn_name_codes(result) + cuban_haitian, daca = _special_status_masks( + ssn_codes=special_ssn_codes, + birth_country=profile.birth_country, + arrival_year=profile.arrival_year, + age_at_entry=profile.age_at_entry, + age=profile.age, + ) + special_before = status.copy() + status[mutable & cuban_haitian] = "CUBAN_HAITIAN_ENTRANT" + status[mutable & daca] = "DACA" + special_status_assignments = int((mutable & (status != special_before)).sum()) + result["immigration_status_str"] = status + + immutable = ~mutable + immutable_pair_invalid = immutable & ( + ((ssn == "CITIZEN") != (status == "CITIZEN")) + | ((ssn == "NONE") != (status == "UNDOCUMENTED")) + | ((ssn == "CITIZEN") != profile.is_citizen) + ) + if immutable_pair_invalid.any(): + raise ValueError( + "Immutable ASEC immigration rows violate citizenship/paired-status " + f"invariants; invalid_rows={int(immutable_pair_invalid.sum())}." + ) + immutable_special_invalid = immutable & ( + ((status == "CUBAN_HAITIAN_ENTRANT") != cuban_haitian) + | ((status == "DACA") != daca) + ) + if immutable_special_invalid.any(): + raise ValueError( + "Immutable ASEC immigration rows violate Cuban/Haitian entrant or " + f"DACA cohort invariants; invalid_rows={int(immutable_special_invalid.sum())}." + ) + + selected_once = np.zeros(len(result), dtype=bool) + compatible_humanitarian = np.zeros(len(result), dtype=bool) + draw_receipts: dict[str, object] = {} + largest_target = max((draw.target for draw in controls.humanitarian), default=0.0) + tolerance = max(1e-6, largest_target * 1e-12) + for draw in controls.humanitarian: + immutable_draw = ( + _humanitarian_emitted_mask( + result, + draw=draw, + profile=profile, + ) + & immutable + ) + compatible_humanitarian |= immutable_draw + immutable_population = float(weights[immutable_draw].sum()) + if draw.target <= 0 and immutable_population > tolerance: + raise ValueError( + f"Immutable ASEC rows emit {immutable_population:,.6f} weighted " + f"persons for explicit-zero humanitarian draw {draw.label!r}." + ) + residual_target = max(0.0, float(draw.target) - immutable_population) + + candidate_ssn = _ssn_name_codes(result) + if draw.category in {"paroled_one_year", "tps"}: + candidate_ssn = np.where(candidate_ssn == 2, 0, candidate_ssn) + candidates = ( + _humanitarian_draw_candidates( + draw, + ssn_codes=candidate_ssn, + profile=profile, + ) + & mutable + & ~selected_once + ) + available_population = float(weights[candidates].sum()) + if available_population + tolerance < residual_target: + raise ValueError( + "Humanitarian transfer candidate shortfall for " + f"{draw.label!r}: residual_target={residual_target:,.6f}, " + f"eligible_recipient_population={available_population:,.6f}, " + f"immutable_population={immutable_population:,.6f}." + ) + stable_draws = _stable_person_draws( + result, + seed=seed, + salt=f"acs_transfer:{draw.salt}", + ) + selected = _select_weight_to_target( + candidates, + weights, + stable_draws, + residual_target, + ) + selected_population = float(weights[selected].sum()) + if selected_population + tolerance < residual_target: + raise RuntimeError( + f"Humanitarian transfer selection underfilled {draw.label!r}." + ) + status[selected] = draw.status + if draw.category in {"paroled_one_year", "tps"}: + ssn[selected & (ssn == "NONE")] = "NON_CITIZEN_VALID_EAD" + selected_once |= selected + result["ssn_card_type"] = ssn + result["immigration_status_str"] = status + achieved_mask = _humanitarian_emitted_mask( + result, + draw=draw, + profile=profile, + ) + compatible_humanitarian |= achieved_mask + achieved_population = float(weights[achieved_mask].sum()) + threshold_tie_population = 0.0 + if selected.any(): + threshold = float(np.max(stable_draws[selected])) + threshold_tie_population = float( + weights[candidates & (stable_draws == threshold)].sum() + ) + residual_selection_error = abs(selected_population - residual_target) + if residual_selection_error > threshold_tie_population + tolerance: + raise RuntimeError( + f"Humanitarian transfer residual selection for {draw.label!r} " + "missed its target by more than the threshold tie mass." + ) + absolute_error = abs(achieved_population - float(draw.target)) + immutable_overshoot = max(0.0, immutable_population - float(draw.target)) + discrete_bound = immutable_overshoot + threshold_tie_population + tolerance + draw_receipts[draw.label] = { + "target": float(draw.target), + "immutable_population": immutable_population, + "residual_target": residual_target, + "eligible_recipient_population": available_population, + "selected_recipient_population": selected_population, + "achieved_population": achieved_population, + "absolute_error": absolute_error, + "selection_threshold_tie_population": threshold_tie_population, + "residual_selection_error": residual_selection_error, + "within_residual_discrete_weight_bound": True, + "immutable_overshoot": immutable_overshoot, + "within_discrete_weight_bound": absolute_error <= discrete_bound, + } + if absolute_error > discrete_bound: + raise RuntimeError( + f"Humanitarian transfer reconciliation for {draw.label!r} " + "missed its target by more than the discrete selection bound." + ) + + result["ssn_card_type"] = ssn + result["immigration_status_str"] = status + all_humanitarian = np.isin(status, list(humanitarian_statuses)) + incompatible_humanitarian = all_humanitarian & ~compatible_humanitarian + if incompatible_humanitarian.any(): + raise ValueError( + "Humanitarian transfer output contains status/origin/arrival/SSN " + f"incompatible rows: {int(incompatible_humanitarian.sum())}." + ) + final_ssn_codes = _ssn_name_codes(result) + final_cuban_haitian, final_daca = _special_status_masks( + ssn_codes=final_ssn_codes, + birth_country=profile.birth_country, + arrival_year=profile.arrival_year, + age_at_entry=profile.age_at_entry, + age=profile.age, + ) + special_status_invalid = ( + (status == "CUBAN_HAITIAN_ENTRANT") != final_cuban_haitian + ) | ((status == "DACA") != final_daca) + if special_status_invalid.any(): + raise RuntimeError( + "Humanitarian transfer reconciliation left Cuban/Haitian entrant " + "or DACA cohort invariants invalid on " + f"{int(special_status_invalid.sum())} row(s)." + ) + final_pair_invalid = ( + ((ssn == "CITIZEN") != (status == "CITIZEN")) + | ((ssn == "NONE") != (status == "UNDOCUMENTED")) + | ((ssn == "CITIZEN") != profile.is_citizen) + ) + if final_pair_invalid.any(): + raise RuntimeError( + "Humanitarian transfer reconciliation left citizenship/paired-status " + f"invariants invalid on {int(final_pair_invalid.sum())} row(s)." + ) + if not np.array_equal( + result.loc[immutable, "ssn_card_type"].astype(str).to_numpy(), + original_ssn[immutable], + ) or not np.array_equal( + result.loc[immutable, "immigration_status_str"].astype(str).to_numpy(), + original_status[immutable], + ): + raise RuntimeError( + "Humanitarian transfer reconciliation changed immutable ASEC rows." + ) + receipt: dict[str, object] = { + "kind": "deterministic_humanitarian_residual_target", + "seed": int(seed), + "time_period": int(time_period), + "mutable_rows": int(mutable.sum()), + "immutable_rows": int(immutable.sum()), + "citizenship_repairs": citizenship_repairs, + "pair_repairs": pair_repairs, + "special_status_assignments": special_status_assignments, + "floating_tolerance": tolerance, + "selection_order": [draw.label for draw in controls.humanitarian], + "draws": draw_receipts, + } + return result, receipt + + +def _assign_humanitarian_statuses( + person: pd.DataFrame, + ssn_codes: np.ndarray, weights: np.ndarray, *, seed: int, - controls: UndocumentedControls, + controls: ImmigrationControls, + time_period: int, ) -> np.ndarray: + """Mark humanitarian statuses and upgrade residual draws' SSN codes. + + Runs after the legal-status indicators and before the EAD worker/student + spill. Pew-included temporary protections remain inside the broad control + universe even after receiving EADs. Draws are sequential over + ``controls.humanitarian``; a person marked by an earlier draw is out of + every later candidate pool. Residual-pool selections (parole/TPS only) + move to ``NON_CITIZEN_VALID_EAD`` — both programs confer employment + authorization — keeping the ``NONE`` ⇔ ``UNDOCUMENTED`` invariant. + """ + + marks = np.full(len(person), "", dtype="U24") + profile = _source_aware_immigration_profile(person, time_period=time_period) + for draw in controls.humanitarian: + if draw.target <= 0: + continue + candidates = _humanitarian_draw_candidates( + draw, + ssn_codes=ssn_codes, + profile=profile, + ) & (marks == "") + draws = _stable_person_draws(person, seed=seed, salt=draw.salt) + selected = _select_weight_to_target(candidates, weights, draws, draw.target) + marks[selected] = draw.status + ssn_codes[selected & (ssn_codes == 0)] = 2 + return marks + + +def _assign_ssn_card_codes( + person: pd.DataFrame, + weights: np.ndarray, + *, + seed: int, + controls: ImmigrationControls, + time_period: int, +) -> tuple[np.ndarray, np.ndarray]: citizenship = _integer_column(person, "PRCITSHP") unknown = ~np.isin(citizenship, [1, 2, 3, 4, 5]) if unknown.any(): @@ -566,55 +1901,89 @@ def _assign_ssn_card_codes( ) ssn_codes[(ssn_codes == 0) & assumed_documented] = 3 - is_worker = (_float_column(person, "WSAL_VAL") > 0) | ( - _float_column(person, "SEMP_VAL") > 0 + # Humanitarian draws run between the indicators and the EAD spill. Pew + # retains temporary protection in its unauthorized estimate, so those + # rows contribute to the broad controls even though they receive EADs. + humanitarian_marks = _assign_humanitarian_statuses( + person, + ssn_codes, + weights, + seed=seed, + controls=controls, + time_period=time_period, ) - is_student = _integer_column(person, "A_HSCOL") == 2 - # The CPS has no work-authorization variable, so the EAD-vs-unauthorized - # split inside the residual pool is unidentified from the survey. Spill - # excess undocumented workers/students to EAD so the *remaining* counts - # match their published controls; counts already at or below a control - # spill nothing. The total undocumented population is emergent — the - # composition gate checks it against a cited anchor, and reconciling the - # level is the calibration lane's job, not this label stage's. - worker_candidates = (ssn_codes == 0) & noncitizens & is_worker - worker_excess = float(weights[worker_candidates].sum()) - controls.workers - worker_draws = _stable_person_draws( - person, seed=seed, salt="immigration:ead_workers" + labor_force_status = _integer_column(person, "A_LFSR") + invalid_labor_force_status = ~np.isin( + labor_force_status, + _CPS_LABOR_FORCE_STATUS_DOMAIN, ) - ssn_codes[ - _select_weight_to_target( - worker_candidates, weights, worker_draws, worker_excess + if invalid_labor_force_status.any(): + bad = sorted(set(labor_force_status[invalid_labor_force_status].tolist()))[:5] + raise SourceRuntimeError( + "A_LFSR carries value(s) outside the CPS ASEC domain " + f"{list(_CPS_LABOR_FORCE_STATUS_DOMAIN)}: {bad}." ) - ] = 2 + is_worker = (age >= 16) & np.isin( + labor_force_status, + _CPS_LABOR_FORCE_STATUS_CODES, + ) + is_student = _integer_column(person, "A_HSCOL") == 2 - student_candidates = (ssn_codes == 0) & noncitizens & is_student - student_excess = float(weights[student_candidates].sum()) - controls.students - student_draws = _stable_person_draws( - person, seed=seed, salt="immigration:ead_students" + _, arrival_year, age_at_entry = _arrival_profile(person, time_period=time_period) + daca_cohort = _daca_statutory_cohort(arrival_year, age_at_entry, age) + cuban_haitian_cohort = ( + np.isin(_integer_column(person, "PENATVTY"), _CUBAN_HAITIAN_BIRTH_CODES) + & (arrival_year >= _CUBAN_HAITIAN_ARRIVAL_CUTOFF) + & ~daca_cohort ) - ssn_codes[ - _select_weight_to_target( - student_candidates, weights, student_draws, student_excess - ) - ] = 2 - return ssn_codes + retained_ead_cohort = daca_cohort | cuban_haitian_cohort + + # The CPS has no work-authorization variable, so the EAD split inside the + # residual pool is unidentified. The two controls bind the same broad + # unauthorized universe their publishers report: residual undocumented + # people plus DACA and Pew-included temporary protections. Student spill + # runs first; the worker spill runs last and is therefore authoritative + # for Pew's published labor-force total. Each pass prefers rows outside + # the other margin so their overlap is disturbed only when unavoidable. + _spill_pew_unauthorized_excess( + person, + ssn_codes, + humanitarian_marks, + weights, + noncitizens=noncitizens, + scope=is_student, + preserve_scope=is_worker, + retained_ead_cohort=retained_ead_cohort, + target=controls.undocumented.students, + seed=seed, + salt="immigration:ead_students", + ) + _spill_pew_unauthorized_excess( + person, + ssn_codes, + humanitarian_marks, + weights, + noncitizens=noncitizens, + scope=is_worker, + preserve_scope=is_student, + retained_ead_cohort=retained_ead_cohort, + target=controls.undocumented.workers, + seed=seed, + salt="immigration:ead_workers", + ) + return ssn_codes, humanitarian_marks def _derive_immigration_status( person: pd.DataFrame, ssn_codes: np.ndarray, + humanitarian_marks: np.ndarray, *, time_period: int, ) -> np.ndarray: - arrival_code = _integer_column(person, "PEINUSYR") - arrival_year = np.full(len(person), time_period, dtype=np.int64) - for code, midpoint in _ARRIVAL_YEAR_MIDPOINTS.items(): - arrival_year[arrival_code == code] = midpoint - years_in_us = time_period - arrival_year + _, arrival_year, age_at_entry = _arrival_profile(person, time_period=time_period) age = _integer_column(person, "A_AGE") - age_at_entry = np.maximum(0, age - years_in_us) birth_country = _integer_column(person, "PENATVTY") # Documented non-citizens default to LPR — the modal true status — and @@ -625,21 +1994,21 @@ def _derive_immigration_status( status[ssn_codes == 1] = "CITIZEN" status[ssn_codes == 0] = "UNDOCUMENTED" - documented_noncitizen = np.isin(ssn_codes, [2, 3]) - cuban_haitian = ( - documented_noncitizen - & np.isin(birth_country, _CUBAN_HAITIAN_BIRTH_CODES) - & (arrival_year >= _CUBAN_HAITIAN_ARRIVAL_CUTOFF) + cuban_haitian, daca = _special_status_masks( + ssn_codes=ssn_codes, + birth_country=birth_country, + arrival_year=arrival_year, + age_at_entry=age_at_entry, + age=age, ) status[cuban_haitian] = "CUBAN_HAITIAN_ENTRANT" - - daca = ( - (ssn_codes == 2) - & (arrival_year <= _DACA_LATEST_ARRIVAL_YEAR) - & (age_at_entry < _DACA_MAX_AGE_AT_ENTRY) - & (age >= _DACA_MIN_CURRENT_AGE) - ) status[daca] = "DACA" + + # Humanitarian draws exclude Cuba/Haiti-born and the DACA cohort, so + # the marks are disjoint from both tags; every marked person carries a + # non-NONE SSN code, so the columns still agree about the undocumented. + marked = humanitarian_marks != "" + status[marked] = humanitarian_marks[marked] return status @@ -648,6 +2017,7 @@ def with_us_immigration_inputs( *, seed: int, time_period: int, + person_weight_scale: float = 1.0, ) -> Frame: """Run the ``immigration_status`` manifest stage over a US frame. @@ -659,6 +2029,11 @@ def with_us_immigration_inputs( CPS ASEC source columns. seed: Build-wide imputation seed. time_period: The dataset's time period (arrival-year arithmetic). + person_weight_scale: Stage-only multiplier for person design weights. + Pooled source projections use the inverse of their share of the + full pool's person mass so absolute national controls contribute + only that source's mass share. Standalone frames keep the default + multiplier of one. Returns: A new frame whose person table carries ``ssn_card_type`` and @@ -673,6 +2048,15 @@ def with_us_immigration_inputs( if frame.schema != US_SCHEMA: raise ValueError("US immigration inputs require the US schema.") + if ( + isinstance(person_weight_scale, bool) + or not np.isfinite(person_weight_scale) + or person_weight_scale <= 0 + ): + raise ValueError( + "US immigration person_weight_scale must be a finite positive " + f"number, got {person_weight_scale!r}." + ) person = frame.table("person") present = [ column for column in US_IMMIGRATION_OUTPUT_COLUMNS if column in person.columns @@ -687,7 +2071,9 @@ def with_us_immigration_inputs( ) stage_person = person.copy(deep=True) - stage_person[_PERSON_WEIGHT_COLUMN] = frame.resolve_weights("person").values + stage_person[_PERSON_WEIGHT_COLUMN] = np.asarray( + frame.resolve_weights("person").values, dtype=np.float64 + ) * float(person_weight_scale) output = run_source_stage( us_immigration_stage_spec(), tables={"person": stage_person}, @@ -746,7 +2132,7 @@ def us_immigration_composition_summary(frame: Frame) -> dict[str, object]: def us_immigration_composition_gate( frame: Frame, *, - controls: UndocumentedControls | None = None, + controls: ImmigrationControls | None = None, ) -> GateResult: """Release gate: the SSN/immigration surface exists and is plausible. @@ -754,23 +2140,16 @@ def us_immigration_composition_gate( mode: everyone a citizen with a valid SSN), when a value falls outside the engine enum domain, when the two columns disagree about citizenship or undocumented status, when the weighted non-citizen share leaves its - plausibility band, or when the emergent undocumented population strays - outside a coarse band around its cited published anchor. + plausibility band, when the emergent Pew-defined unauthorized population + strays outside a coarse band around its cited published anchor, or when a + humanitarian category's emitted mass leaves the coarse band around its + cited stock target (microcosm #767 — the H.R.1 §71109/§71301/§71302 and + SNAP §10108 eligibility channels all bind through these categories; an + explicit zero target must emit exactly zero). """ if controls is None: - stage = us_immigration_stage_spec() - derive = [ - operation - for operation in stage.operations - if operation.kind == "derive_immigration_status" - ] - if len(derive) != 1: - raise ValueError( - "US immigration stage must declare exactly one " - "derive_immigration_status operation." - ) - controls = _controls_from_parameters(derive[0].parameters) + controls = us_immigration_controls() person = frame.table("person") weights = np.asarray(frame.resolve_weights("person").values, dtype=np.float64) @@ -779,13 +2158,18 @@ def us_immigration_composition_gate( details: dict[str, object] = { "summary": us_immigration_composition_summary(frame), "controls": { - "undocumented_workers": controls.workers, - "undocumented_students": controls.students, - "undocumented_population_anchor": controls.population_anchor, - "sources": dict(controls.sources), + "undocumented_workers": controls.undocumented.workers, + "undocumented_students": controls.undocumented.students, + "undocumented_population_anchor": (controls.undocumented.population_anchor), + "sources": dict(controls.undocumented.sources), + "humanitarian_status_stocks": { + draw.label: {"target": draw.target, "source": draw.source} + for draw in controls.humanitarian + }, }, "non_citizen_share_band": list(_NON_CITIZEN_SHARE_BAND), "undocumented_anchor_relative_band": list(_UNDOCUMENTED_ANCHOR_RELATIVE_BAND), + "humanitarian_target_relative_band": list(_HUMANITARIAN_TARGET_RELATIVE_BAND), } missing = [ @@ -807,6 +2191,11 @@ def us_immigration_composition_gate( ssn = person["ssn_card_type"].astype(str) status = person["immigration_status_str"].astype(str) + try: + evidence = _source_aware_immigration_profile(person, time_period=2024) + except (SourceRuntimeError, ValueError) as exc: + evidence = None + failures.append(f"source-aware immigration evidence is invalid: {exc}") for column, values, domain in ( ("ssn_card_type", ssn, SSN_CARD_TYPE_VALUES), ("immigration_status_str", status, IMMIGRATION_STATUS_VALUES), @@ -838,6 +2227,33 @@ def us_immigration_composition_gate( f"{undocumented_disagreements} person(s) have ssn_card_type NONE " "and immigration_status_str UNDOCUMENTED disagreeing." ) + if evidence is not None: + evidence_citizen_disagreements = int((ssn_citizen != evidence.is_citizen).sum()) + if evidence_citizen_disagreements: + failures.append( + f"{evidence_citizen_disagreements} person(s) have emitted " + "citizenship disagreeing with their source PRCITSHP/CIT evidence." + ) + expected_cuban_haitian, expected_daca = _special_status_masks( + ssn_codes=_ssn_name_codes(person), + birth_country=evidence.birth_country, + arrival_year=evidence.arrival_year, + age_at_entry=evidence.age_at_entry, + age=evidence.age, + ) + special_status_invalid = ( + (status.to_numpy() == "CUBAN_HAITIAN_ENTRANT") != expected_cuban_haitian + ) | ((status.to_numpy() == "DACA") != expected_daca) + details["evidence_derived_status_compatibility"] = { + "cuban_haitian_entrant_rows": int(expected_cuban_haitian.sum()), + "daca_rows": int(expected_daca.sum()), + "invalid_rows": int(special_status_invalid.sum()), + } + if special_status_invalid.any(): + failures.append( + f"{int(special_status_invalid.sum())} person(s) violate the " + "source-aware Cuban/Haitian entrant or DACA cohort contract." + ) non_citizen_share = ( float(weights[~ssn_citizen].sum()) / total if total else float("nan") @@ -849,17 +2265,112 @@ def us_immigration_composition_gate( f"[{low}, {high}]." ) - undocumented = float(weights[ssn_none].sum()) - relative = undocumented / controls.population_anchor + status_values = status.to_numpy() + ssn_values = ssn.to_numpy() + pew_unauthorized_rows = np.isin( + status_values, + _PEW_UNAUTHORIZED_STATUS_VALUES, + ) | ( + (status_values == "CUBAN_HAITIAN_ENTRANT") + & (ssn_values == "NON_CITIZEN_VALID_EAD") + ) + pew_unauthorized = float(weights[pew_unauthorized_rows].sum()) + details["pew_unauthorized_population"] = pew_unauthorized + details["pew_unauthorized_statuses"] = list(_PEW_UNAUTHORIZED_STATUS_VALUES) + details["pew_unauthorized_paired_status"] = { + "immigration_status_str": "CUBAN_HAITIAN_ENTRANT", + "ssn_card_type": "NON_CITIZEN_VALID_EAD", + } + relative = pew_unauthorized / controls.undocumented.population_anchor rel_low, rel_high = _UNDOCUMENTED_ANCHOR_RELATIVE_BAND if not (rel_low <= relative <= rel_high): failures.append( - f"emergent undocumented population {undocumented:,.0f} is " + f"emergent Pew-defined unauthorized population " + f"{pew_unauthorized:,.0f} is " f"{relative:.2f}x the published anchor " - f"{controls.population_anchor:,.0f} (band [{rel_low}, {rel_high}], " - f"{controls.sources[_ANCHOR_KEY]})." + f"{controls.undocumented.population_anchor:,.0f} " + f"(band [{rel_low}, {rel_high}], " + f"{controls.undocumented.sources[_ANCHOR_KEY]})." ) + hum_low, hum_high = _HUMANITARIAN_TARGET_RELATIVE_BAND + humanitarian_statuses = set(_HUMANITARIAN_STATUS_BY_CATEGORY.values()) + humanitarian_rows = np.isin(status_values, list(humanitarian_statuses)) + draw_achieved: dict[str, dict[str, object]] = {} + compatible_humanitarian = np.zeros(len(person), dtype=bool) + if evidence is not None: + for draw in controls.humanitarian: + emitted_mask = _humanitarian_emitted_mask( + person, + draw=draw, + profile=evidence, + ) + compatible_humanitarian |= emitted_mask + emitted = float(weights[emitted_mask].sum()) + relative = (emitted / draw.target) if draw.target > 0 else None + draw_achieved[draw.label] = { + "category": draw.category, + "origin": draw.origin, + "status": draw.status, + "target": draw.target, + "population": emitted, + "relative": relative, + "source": draw.source, + } + if draw.target <= 0: + if emitted > 0: + failures.append( + f"{draw.label}: {emitted:,.0f} compatible weighted " + "persons emitted against an explicit zero target." + ) + continue + if relative is None or not (hum_low <= relative <= hum_high): + failures.append( + f"{draw.label}: compatible emitted population {emitted:,.0f} " + f"is {relative:.2f}x the cited stock target {draw.target:,.0f} " + f"(band [{hum_low}, {hum_high}])." + ) + incompatible = humanitarian_rows & ~compatible_humanitarian + if incompatible.any(): + incompatible_population = float(weights[incompatible].sum()) + examples = sorted(set(status_values[incompatible].tolist())) + failures.append( + f"{int(incompatible.sum())} humanitarian row(s) " + f"({incompatible_population:,.0f} weighted persons) violate " + "their source-aware status/origin/arrival/SSN cohort contract; " + f"statuses={examples}." + ) + details["humanitarian_draw_achieved"] = draw_achieved + + achieved_by_category: dict[str, dict[str, object]] = {} + for category in HUMANITARIAN_STATUS_CATEGORIES: + status_name = _HUMANITARIAN_STATUS_BY_CATEGORY[category] + target = controls.humanitarian_target(category) + emitted = float(weights[status_values == status_name].sum()) + achieved_by_category[category] = { + "status": status_name, + "target": target, + "population": emitted, + "relative": (emitted / target) if target > 0 else None, + } + if target <= 0: + if emitted > 0: + failures.append( + f"{status_name}: {emitted:,.0f} weighted persons emitted " + "against an explicit zero target — the manifest documents " + "this category as not imputed." + ) + continue + category_relative = emitted / target + if not (hum_low <= category_relative <= hum_high): + failures.append( + f"{status_name}: emitted population {emitted:,.0f} is " + f"{category_relative:.2f}x the cited stock target " + f"{target:,.0f} (band [{hum_low}, {hum_high}]) — the H.R.1 " + "eligibility channels through this category are degenerate." + ) + details["humanitarian_achieved"] = achieved_by_category + return GateResult( name="immigration_composition", passed=not failures, diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/multispine_pool.py b/packages/microcosm-build/src/microcosm/build/us_runtime/multispine_pool.py index 8f2ea06b..dbeeac6a 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/multispine_pool.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/multispine_pool.py @@ -469,7 +469,7 @@ def simulation_ready(self) -> bool: type PoolOperator = Callable[[Frame], PoolStageOutput] type AgreementGate = Callable[[Frame], GateResult] -type SourceFrameOperator = Callable[[Frame], Frame | PoolStageOutput] +type SourceFrameOperator = Callable[..., Frame | PoolStageOutput] @dataclass(frozen=True) @@ -861,9 +861,8 @@ def _resolve_take_up_program_bindings( for program in load_take_up_contract().programs ) for index, binding in enumerate(bindings): - if ( - len(binding) != 3 - or not all(isinstance(value, str) and value for value in binding) + if len(binding) != 3 or not all( + isinstance(value, str) and value for value in binding ): raise ValueError( "Take-up manifest program binding must contain three non-empty " @@ -1333,9 +1332,7 @@ def surface_provision(variable: str) -> str: variable, execution_scope="whole_pool", provision=provision, - available_by=( - "transferred" if variable in transfer_owned else "seeded" - ), + available_by=("transferred" if variable in transfer_owned else "seeded"), fallback=fallback, ) @@ -1941,10 +1938,13 @@ def _post_clone_source_operators() -> Mapping[str, SourceFrameOperator]: force_puf_imputation=True, ) ), - "with_us_immigration_inputs": lambda current: with_us_immigration_inputs( - current, - seed=POOL_RANDOM_SEED, - time_period=POOL_TIME_PERIOD, + "with_us_immigration_inputs": lambda current, *, person_weight_scale=1.0: ( + with_us_immigration_inputs( + current, + seed=POOL_RANDOM_SEED, + time_period=POOL_TIME_PERIOD, + person_weight_scale=person_weight_scale, + ) ), "with_us_education_inputs": lambda current: with_us_education_inputs( current, @@ -2150,8 +2150,42 @@ def _run_source_operator_chain( and contract.execution_scope == _CPS_SOURCE_EXECUTION_SCOPE ): declared_outputs = _persisted_source_outputs(declared_outputs) + person_design_weight_scaling: dict[str, float] | None = None if contract.execution_scope == _CPS_SOURCE_EXECUTION_SCOPE: available_mask = _cps_source_evidence_mask(current, phase=phase) + if operator_name == "with_us_immigration_inputs": + full_person_weights = np.asarray( + current.resolve_weights(current.schema.person_entity).values, + dtype=np.float64, + ) + cps_person_weights = full_person_weights[ + available_mask.to_numpy(dtype=bool) + ] + if ( + not np.isfinite(full_person_weights).all() + or (full_person_weights < 0).any() + or not np.isfinite(cps_person_weights).all() + or (cps_person_weights < 0).any() + ): + raise ValueError( + "Pooled immigration controls require finite non-negative " + "person design weights." + ) + full_person_mass = float(full_person_weights.sum()) + cps_person_mass = float(cps_person_weights.sum()) + if full_person_mass <= 0 or cps_person_mass <= 0: + raise ValueError( + "Pooled immigration controls require positive full-pool " + "and CPS-projection person design-weight mass; " + f"got full={full_person_mass!r}, cps={cps_person_mass!r}." + ) + person_weight_scale = full_person_mass / cps_person_mass + person_design_weight_scaling = { + "full_pool_person_design_weight_mass": full_person_mass, + "cps_projection_person_design_weight_mass": cps_person_mass, + "person_weight_scale": person_weight_scale, + "cps_person_mass_share": cps_person_mass / full_person_mass, + } available = _source_available_projection( current, available_mask, @@ -2170,13 +2204,29 @@ def _run_source_operator_chain( ) before_rows = _frame_row_counts(current) available_rows = _frame_row_counts(available) - kernel_outcome = operators[operator_name](available) + if person_design_weight_scaling is None: + kernel_outcome = operators[operator_name](available) + else: + kernel_outcome = operators[operator_name]( + available, + person_weight_scale=person_design_weight_scaling["person_weight_scale"], + ) kernel_receipt: Mapping[str, object] = {} if isinstance(kernel_outcome, PoolStageOutput): outcome = kernel_outcome.frame kernel_receipt = kernel_outcome.receipt else: outcome = kernel_outcome + if person_design_weight_scaling is not None: + if "person_design_weight_scaling" in kernel_receipt: + raise ValueError( + "Pooled immigration kernel receipt may not override the " + "orchestrator-owned person-design-weight scaling contract." + ) + kernel_receipt = { + **dict(kernel_receipt), + "person_design_weight_scaling": person_design_weight_scaling, + } if not isinstance(outcome, Frame): raise TypeError( f"Multispine source operator {operator_name!r} must return Frame, " diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/stacked_spine.py b/packages/microcosm-build/src/microcosm/build/us_runtime/stacked_spine.py index ae762b46..f4137d93 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/stacked_spine.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/stacked_spine.py @@ -99,6 +99,11 @@ TargetFamilies, transfer_acs_inputs, ) +from microcosm.build.us_runtime.immigration import ( + us_immigration_controls, + us_immigration_humanitarian_draw_mask, + us_immigration_humanitarian_transfer_selection_masks, +) from microcosm.build.us_runtime.late_producer_dag import ( ProducerContract, ProducerInput, @@ -4144,6 +4149,54 @@ def _validate_post_transfer_live_output_binding( "unmodeled_rows", "residual_null_rows", ) +_IMMIGRATION_TRANSFER_FAMILY = "source_operator_immigration" +_IMMIGRATION_TRANSFER_TARGETS = ("ssn_card_type", "immigration_status_str") +_IMMIGRATION_RECONCILIATION_KEY = "post_transfer_reconciliation" +_IMMIGRATION_RECONCILIATION_DRAW_FIELDS = frozenset( + { + "target", + "immutable_population", + "residual_target", + "eligible_recipient_population", + "selected_recipient_population", + "achieved_population", + "absolute_error", + "selection_threshold_tie_population", + "residual_selection_error", + "within_residual_discrete_weight_bound", + "immutable_overshoot", + "within_discrete_weight_bound", + } +) + + +def _is_immigration_transfer_target( + *, + entity: str, + family: str, + target: str, +) -> bool: + return ( + entity == "person" + and family.split("__batch_", 1)[0] == _IMMIGRATION_TRANSFER_FAMILY + and target in _IMMIGRATION_TRANSFER_TARGETS + ) + + +def _immigration_qrf_evidence_targets( + surface: TargetFamilies, +) -> tuple[tuple[str, str], ...]: + return tuple( + (entity, target) + for entity, families in surface.items() + for family, targets in families.items() + for target in targets + if _is_immigration_transfer_target( + entity=entity, + family=family, + target=target, + ) + ) def _validate_acs_transfer_row_counts( @@ -4179,6 +4232,260 @@ def _validate_acs_transfer_row_counts( return typed_counts +def _validate_immigration_post_transfer_reconciliation( + target_receipt: Mapping[str, object], + *, + boundary: str, + frame: Frame | None = None, +) -> Mapping[str, object]: + """Validate the persisted constrained-recipient reconciliation evidence.""" + + reconciliation = target_receipt.get(_IMMIGRATION_RECONCILIATION_KEY) + if not isinstance(reconciliation, Mapping): + raise ValueError( + f"{boundary}: immigration post-transfer reconciliation is absent." + ) + expected_keys = { + "kind", + "seed", + "time_period", + "mutable_rows", + "immutable_rows", + "citizenship_repairs", + "pair_repairs", + "special_status_assignments", + "floating_tolerance", + "selection_order", + "draws", + } + if set(reconciliation) != expected_keys or reconciliation.get("kind") != ( + "deterministic_humanitarian_residual_target" + ): + raise ValueError( + f"{boundary}: immigration post-transfer reconciliation schema is invalid." + ) + integer_fields = ( + "seed", + "time_period", + "mutable_rows", + "immutable_rows", + "citizenship_repairs", + "pair_repairs", + "special_status_assignments", + ) + if any( + not isinstance(reconciliation.get(field), int) + or isinstance(reconciliation[field], bool) + or reconciliation[field] < 0 + for field in integer_fields + ): + raise ValueError( + f"{boundary}: immigration reconciliation integer evidence is invalid." + ) + counts = _validate_acs_transfer_row_counts( + target_receipt, + boundary=boundary, + required=True, + ) + if ( + reconciliation["mutable_rows"] != counts["imputed_rows"] + or reconciliation["citizenship_repairs"] > reconciliation["mutable_rows"] + or reconciliation["pair_repairs"] > reconciliation["mutable_rows"] + or reconciliation["special_status_assignments"] > reconciliation["mutable_rows"] + or reconciliation["citizenship_repairs"] > reconciliation["pair_repairs"] + ): + raise ValueError( + f"{boundary}: immigration reconciliation row accounting is invalid." + ) + producer_rows = target_receipt.get("producer_rows") + if producer_rows is not None and ( + not isinstance(producer_rows, int) + or isinstance(producer_rows, bool) + or producer_rows < 0 + or reconciliation["immutable_rows"] != producer_rows + ): + raise ValueError( + f"{boundary}: immigration reconciliation immutable-row binding is invalid." + ) + controls = us_immigration_controls() + canonical_tolerance = max( + 1e-6, + max((draw.target for draw in controls.humanitarian), default=0.0) * 1e-12, + ) + tolerance = reconciliation.get("floating_tolerance") + if ( + not isinstance(tolerance, (int, float)) + or isinstance(tolerance, bool) + or not np.isfinite(float(tolerance)) + or tolerance <= 0 + or float(tolerance) != canonical_tolerance + ): + raise ValueError( + f"{boundary}: immigration reconciliation tolerance is non-canonical." + ) + tolerance = float(tolerance) + live_draw_populations: dict[str, tuple[float, float, float]] = {} + live_mutable_draw_masks: dict[str, np.ndarray] = {} + expected_mutable_draw_masks: dict[str, np.ndarray] = {} + if frame is not None: + person_entity = frame.schema.person_entity + person = frame.table(person_entity) + channel = person[support_channel_column(person_entity)].astype(str) + immutable_rows = channel.eq(BASE_ASEC_SUPPORT_CHANNEL).to_numpy(dtype=bool) + mutable_rows = channel.eq(ACS_STACKED_SUPPORT_CHANNEL).to_numpy(dtype=bool) + if ( + not np.logical_or(immutable_rows, mutable_rows).all() + or int(immutable_rows.sum()) != reconciliation["immutable_rows"] + or int(mutable_rows.sum()) != reconciliation["mutable_rows"] + ): + raise ValueError( + f"{boundary}: immigration reconciliation row counts differ " + "from the live ASEC/ACS frame." + ) + weights = np.asarray( + frame.resolve_weights(person_entity).values, + dtype=np.float64, + ) + if ( + weights.shape != (len(person),) + or not np.isfinite(weights).all() + or (weights < 0).any() + ): + raise ValueError( + f"{boundary}: immigration reconciliation live person weights " + "are invalid." + ) + for control in controls.humanitarian: + emitted = us_immigration_humanitarian_draw_mask( + frame, + control, + time_period=int(reconciliation["time_period"]), + ) + live_draw_populations[control.label] = ( + float(weights[emitted & immutable_rows].sum()), + float(weights[emitted & mutable_rows].sum()), + float(weights[emitted].sum()), + ) + live_mutable_draw_masks[control.label] = emitted & mutable_rows + expected_mutable_draw_masks = ( + us_immigration_humanitarian_transfer_selection_masks( + frame, + mutable_rows=mutable_rows, + seed=int(reconciliation["seed"]), + time_period=int(reconciliation["time_period"]), + controls=controls, + ) + ) + selection_order = reconciliation.get("selection_order") + draws = reconciliation.get("draws") + expected_order = [draw.label for draw in controls.humanitarian] + targets_by_label = { + draw.label: float(draw.target) for draw in controls.humanitarian + } + if ( + not isinstance(selection_order, list) + or selection_order != expected_order + or not isinstance(draws, Mapping) + or set(draws) != set(expected_order) + ): + raise ValueError( + f"{boundary}: immigration reconciliation draw order is non-canonical." + ) + numeric_fields = _IMMIGRATION_RECONCILIATION_DRAW_FIELDS - { + "within_residual_discrete_weight_bound", + "within_discrete_weight_bound", + } + for label in expected_order: + draw = draws[label] + if not isinstance(draw, Mapping) or set(draw) != ( + _IMMIGRATION_RECONCILIATION_DRAW_FIELDS + ): + raise ValueError( + f"{boundary}: immigration reconciliation draw {label!r} schema " + "is invalid." + ) + if any( + not isinstance(draw.get(field), (int, float)) + or isinstance(draw[field], bool) + or not np.isfinite(float(draw[field])) + or draw[field] < 0 + for field in numeric_fields + ) or any( + draw.get(field) is not True + for field in ( + "within_residual_discrete_weight_bound", + "within_discrete_weight_bound", + ) + ): + raise ValueError( + f"{boundary}: immigration reconciliation draw {label!r} values " + "are invalid." + ) + target = targets_by_label[label] + immutable = float(draw["immutable_population"]) + residual = float(draw["residual_target"]) + eligible = float(draw["eligible_recipient_population"]) + selected = float(draw["selected_recipient_population"]) + achieved = float(draw["achieved_population"]) + tie = float(draw["selection_threshold_tie_population"]) + expected_residual = max(0.0, target - immutable) + residual_error = abs(selected - residual) + absolute_error = abs(achieved - target) + immutable_overshoot = max(0.0, immutable - target) + if frame is not None: + live_immutable, live_selected, live_achieved = live_draw_populations[label] + if ( + abs(immutable - live_immutable) > tolerance + or abs(selected - live_selected) > tolerance + or abs(achieved - live_achieved) > tolerance + ): + raise ValueError( + f"{boundary}: immigration reconciliation draw {label!r} " + "differs from the live ASEC/ACS weighted population." + ) + if not np.array_equal( + live_mutable_draw_masks[label], + expected_mutable_draw_masks[label], + ): + raise ValueError( + f"{boundary}: immigration reconciliation draw {label!r} " + "selected mutable-row identities differ from the canonical " + "seeded selection." + ) + if target <= 0 and immutable > tolerance: + raise ValueError( + f"{boundary}: immigration reconciliation draw {label!r} has " + "immutable mass against an explicit-zero target." + ) + if eligible + tolerance < residual: + raise ValueError( + f"{boundary}: immigration reconciliation draw {label!r} " + "conceals a recipient candidate shortfall." + ) + if selected + tolerance < residual: + raise ValueError( + f"{boundary}: immigration reconciliation draw {label!r} " + "underfills its residual target." + ) + if ( + abs(float(draw["target"]) - target) > tolerance + or abs(residual - expected_residual) > tolerance + or selected > eligible + tolerance + or tie > eligible + tolerance + or abs(achieved - (immutable + selected)) > tolerance + or abs(float(draw["residual_selection_error"]) - residual_error) > tolerance + or residual_error > tie + tolerance + or abs(float(draw["absolute_error"]) - absolute_error) > tolerance + or abs(float(draw["immutable_overshoot"]) - immutable_overshoot) > tolerance + or absolute_error > immutable_overshoot + tie + tolerance + ): + raise ValueError( + f"{boundary}: immigration reconciliation draw {label!r} " + "accounting is invalid." + ) + return reconciliation + + _PREGNANCY_STRUCTURAL_COUNT_FIELDS = ( "source_persons_checked", "physical_rows_checked", @@ -4220,7 +4527,9 @@ def _validate_pregnancy_structural_receipt( or structural.get("status") != "verified" ): raise ValueError(f"{boundary}: pregnancy structural policy is invalid.") - counts = {field: structural.get(field) for field in _PREGNANCY_STRUCTURAL_COUNT_FIELDS} + counts = { + field: structural.get(field) for field in _PREGNANCY_STRUCTURAL_COUNT_FIELDS + } if any( not isinstance(value, int) or isinstance(value, bool) or value < 0 for value in counts.values() @@ -4250,9 +4559,7 @@ def _validate_pregnancy_structural_receipt( != row_counts["imputed_rows"] ) ): - raise ValueError( - f"{boundary}: pregnancy structural accounting is invalid." - ) + raise ValueError(f"{boundary}: pregnancy structural accounting is invalid.") def _acs_imputed_pattern_evidence(record: AcsImputedInput) -> dict[str, object]: @@ -4318,6 +4625,11 @@ def _acs_pattern_predictor_authority( mandatory = tuple(housing["mandatory_features"]) required = (*required, *mandatory) optional = tuple(item for item in optional if item not in mandatory) + immigration_targets = set(contract["immigration_status_targets"]) + if immigration_targets.issubset(family_targets): + mandatory = tuple(contract["immigration_required_predictors"]) + required = (*required, *mandatory) + optional = tuple(item for item in optional if item not in mandatory) return tuple(required), tuple(optional) optional_names = acs_transfer_runtime._GROUP_OPTIONAL_NAMES return ( @@ -4856,6 +5168,7 @@ def validate_stacked_post_puf_transfer_receipt( validated_calibration_keys.update( spec.key for spec in expected_calibrations.values() ) + group_immigration_reconciliations: list[Mapping[str, object]] = [] for target_key, target_receipt in group_targets.items(): if not isinstance(target_receipt, Mapping): raise ValueError( @@ -4882,8 +5195,36 @@ def validate_stacked_post_puf_transfer_receipt( ) owner_receipt = target_receipt.get("post_transfer_calibration") spec = expected_calibrations.get(target_key) + target = target_key.rsplit("/", 1)[1] + is_immigration = _is_immigration_transfer_target( + entity=group.entity, + family=group.family, + target=target, + ) + if is_immigration: + group_immigration_reconciliations.append( + _validate_immigration_post_transfer_reconciliation( + target_receipt, + boundary=f"{boundary} target {target_key}", + frame=frame, + ) + ) + _validate_acs_imputed_pattern_evidence( + target_receipt, + expected_entity=group.entity, + expected_family=group.family, + expected_target=target, + expected_family_targets=group.targets, + expected_regime_targets=group.targets, + boundary=f"{boundary} target {target_key}", + ) + elif _IMMIGRATION_RECONCILIATION_KEY in target_receipt: + raise ValueError( + f"{boundary}: undeclared immigration reconciliation is " + f"attached to {target_key!r}." + ) if spec is None: - if "qrf_pattern_evidence" in target_receipt: + if "qrf_pattern_evidence" in target_receipt and not is_immigration: raise ValueError( f"{boundary}: undeclared ACS QRF pattern evidence is " f"attached to {target_key!r}." @@ -4903,7 +5244,7 @@ def validate_stacked_post_puf_transfer_receipt( target_receipt, expected_entity=group.entity, expected_family=group.family, - expected_target=target_key.rsplit("/", 1)[1], + expected_target=target, expected_family_targets=group.targets, expected_regime_targets=expected_regime_targets, boundary=f"{boundary} target {target_key}", @@ -5007,6 +5348,17 @@ def validate_stacked_post_puf_transfer_receipt( spec=spec, boundary=f"{boundary} target {target_key}", ) + if group_immigration_reconciliations: + if len(group_immigration_reconciliations) != len( + _IMMIGRATION_TRANSFER_TARGETS + ) or any( + _json_ready(item) != _json_ready(group_immigration_reconciliations[0]) + for item in group_immigration_reconciliations[1:] + ): + raise ValueError( + f"{boundary}: paired immigration reconciliation evidence " + f"is incomplete or inconsistent for group {name!r}." + ) if validated_calibration_keys != set(late_calibration_specs): raise ValueError( f"{boundary}: post-transfer calibration coverage does not match " @@ -8067,6 +8419,17 @@ def validate_stacked_late_producer_transition_authority( """Validate the anchor after declared downstream operators have run.""" validate_stacked_late_producer_receipt(receipt, boundary=boundary) + transfer = receipt.get("post_puf_transfer") + assert isinstance(transfer, Mapping) + # Downstream fiscal and gate operators intentionally change the complete + # late-output digest, but they do not own source channels, resolved + # weights, or immigration outputs. Replay that reconstructible terminal + # evidence before authenticating the immutable transition carrier. + validate_stacked_post_puf_transfer_receipt( + transfer, + boundary=boundary, + frame=frame, + ) _validate_late_transition_authority( frame, receipt, @@ -10260,10 +10623,17 @@ def _transfer_stacked_post_puf_inputs_evaluate( derive_schedule_d=derive_schedule_d, execution_contract=execution_contract, regime_evidence_targets=tuple( - (spec.entity, spec.target) - for spec in _stacked_post_transfer_calibration_specs( - surface, - stage="late_transfer", + dict.fromkeys( + ( + *( + (spec.entity, spec.target) + for spec in _stacked_post_transfer_calibration_specs( + surface, + stage="late_transfer", + ) + ), + *_immigration_qrf_evidence_targets(surface), + ) ) ), ) @@ -10508,7 +10878,8 @@ def _verify_post_puf_transfer_outcome( target_families, stage="late_transfer", ) - } + } | set(_immigration_qrf_evidence_targets(target_families)) + immigration_reconciliations: dict[str, Mapping[str, object]] = {} for entity, families in target_families.items(): table = frame.table(entity) for family, family_targets in families.items(): @@ -10588,11 +10959,43 @@ def _verify_post_puf_transfer_outcome( target_receipt["qrf_pattern_evidence"] = ( _acs_imputed_pattern_evidence(record) ) + if _is_immigration_transfer_target( + entity=entity, + family=family, + target=target, + ): + if record is None or not isinstance(record.reconciliation, Mapping): + failures.append( + f"{label}: constrained immigration reconciliation " + "evidence is absent." + ) + else: + evidence = _json_ready(record.reconciliation) + assert isinstance(evidence, Mapping) + target_receipt[_IMMIGRATION_RECONCILIATION_KEY] = evidence + immigration_reconciliations[target] = evidence if record is not None and record.structural_receipt is not None: target_receipt["structural_policy"] = dict( record.structural_receipt ) target_receipts[target_receipt_key] = target_receipt + if set(immigration_reconciliations) not in ( + set(), + set(_IMMIGRATION_TRANSFER_TARGETS), + ): + failures.append( + "post_puf_transfer/person/source_operator_immigration: paired " + "reconciliation evidence is incomplete." + ) + elif immigration_reconciliations and any( + _json_ready(evidence) + != _json_ready(immigration_reconciliations[_IMMIGRATION_TRANSFER_TARGETS[0]]) + for evidence in immigration_reconciliations.values() + ): + failures.append( + "post_puf_transfer/person/source_operator_immigration: paired " + "reconciliation evidence disagrees." + ) if failures: raise ValueError( "Stacked post-PUF transfer outcome verification failed:\n " diff --git a/packages/microcosm-build/src/microcosm/build/us_runtime/us_late_producer_registry.py b/packages/microcosm-build/src/microcosm/build/us_runtime/us_late_producer_registry.py index 4b8206fa..6cfb5f75 100644 --- a/packages/microcosm-build/src/microcosm/build/us_runtime/us_late_producer_registry.py +++ b/packages/microcosm-build/src/microcosm/build/us_runtime/us_late_producer_registry.py @@ -99,7 +99,9 @@ "us_late_producer_schedule_receipt", ] -# v16 declares person.s_corp_income as a whole-pool primary-PUF output: its +# v17 binds the ASEC and ACS source-scoped evidence read by constrained +# immigration transfer and reconciliation. v16 declares person.s_corp_income +# as a whole-pool primary-PUF output: its # certified combined-source semantics are carried by partnership_income, while # the separate S-corporation leaf is an exact-zero universe. v15 content-binds # the complete late dual-producer ownership matrix and @@ -122,9 +124,11 @@ # implicit. Receipt v3 reconciles repeated physical evidence and scope # cardinalities across each execution row, binds source-receipt outputs to the # callback receipt, and requires the primary callback to report the exact -# resources it consumed. Receipt v2 introduced exact virtual-resource payloads. -US_LATE_PRODUCER_REGISTRY_SCHEMA_VERSION = 16 -US_LATE_PRODUCER_RECEIPT_SCHEMA_VERSION = 3 +# resources it consumed. Receipt v4 additionally persists and validates the +# paired constrained-immigration reconciliation and QRF evidence. Receipt v2 +# introduced exact virtual-resource payloads. +US_LATE_PRODUCER_REGISTRY_SCHEMA_VERSION = 17 +US_LATE_PRODUCER_RECEIPT_SCHEMA_VERSION = 4 US_LATE_PRODUCER_TRANSITION_AUTHORITY_VERSION = 1 US_LATE_PRODUCER_TRANSITION_AUTHORITY_KEY = "us_late_producer_transition_authority" US_LATE_PRODUCER_TRANSITION_AUTHORITY_ID = "us_stacked_late_producer_transition" @@ -954,8 +958,7 @@ def _inventory( "A_MARITL", "A_SPOUSE", "A_HSCOL", - "WSAL_VAL", - "SEMP_VAL", + "A_LFSR", "MCARE", "CAID", "IHSFLG", @@ -977,6 +980,11 @@ def _inventory( _single("resolved_person_weight", "person", "@resolved_weight"), _requirement( "stable_source_identity", + ( + _column("person", "source_year"), + _column("person", "source_household_id"), + _column("person", "source_person_id"), + ), ( _column("person", "source_year"), _column("person", "source_person_id"), @@ -1195,6 +1203,69 @@ def _transfer_input_inventory(group: TransferProducerGroup) -> SourceInputInvent _ADULT_CARE_ROLE_INPUT, ) ) + if group.name == transfer_producer_name("person", "source_operator_immigration"): + post_transfer_structure.extend( + ( + _requirement( + "immigration_stable_person_lineage", + ( + _column("person", "source_year"), + _column("person", "source_household_id"), + _column("person", "source_person_id"), + ), + ( + _column("person", "source_year"), + _column("person", "source_person_id"), + ), + (_column("person", "person_id"),), + ), + _single( + "immigration_asec_citizenship", + "person", + "PRCITSHP", + value_kind="finite_numeric", + required_scope=_ASEC_SOURCE_SCOPE, + ), + _single( + "immigration_asec_origin", + "person", + "PENATVTY", + value_kind="finite_numeric", + required_scope=_ASEC_SOURCE_SCOPE, + ), + _single( + "immigration_asec_arrival", + "person", + "PEINUSYR", + value_kind="finite_numeric", + required_scope=_ASEC_SOURCE_SCOPE, + ), + _single( + "immigration_acs_citizenship", + "person", + "CIT", + value_kind="finite_numeric", + required_scope=_ACS_SOURCE_SCOPE, + ), + _single( + "immigration_acs_origin", + "person", + "POBP", + value_kind="finite_numeric", + required_scope=_ACS_SOURCE_SCOPE, + ), + _single( + "immigration_acs_arrival", + "person", + "YOEP", + # Native-born ACS people carry a structural blank. The + # source-aware runtime requires a finite year only for + # foreign-born CIT=4/5 rows. + value_kind="column_present", + required_scope=_ACS_SOURCE_SCOPE, + ), + ) + ) return _inventory( group.name, *structural, @@ -2076,7 +2147,8 @@ def us_late_producer_schedule_payload() -> dict[str, object]: ), "top_binding": ( "entry_and_output_frame_sha256_execution_chain_source_" - "completion_and_nineteen_transfer_groups" + "completion_nineteen_transfer_groups_and_constrained_" + "immigration_reconciliation" ), "transition_authority": { "authority_id": US_LATE_PRODUCER_TRANSITION_AUTHORITY_ID, diff --git a/packages/microcosm-build/tests/test_spec_engine_compiler_ir.py b/packages/microcosm-build/tests/test_spec_engine_compiler_ir.py index fb141dec..acfd3be2 100644 --- a/packages/microcosm-build/tests/test_spec_engine_compiler_ir.py +++ b/packages/microcosm-build/tests/test_spec_engine_compiler_ir.py @@ -22,9 +22,7 @@ ) from microcosm.build.spec_engine.resolver import F0_CONTRACT_ONLY_KERNEL_IDS -US_SCHEDULE_SHA256 = ( - "e59c019d3d454eac99ac0ac209b6c5b6faaf9bdfcaeee18c36a25be19bf7da2f" -) +US_SCHEDULE_SHA256 = "88bc9243a3518982ae951c3de21bd55877e296ce4fcb183b9bee420d3a684b10" @pytest.fixture(scope="module") @@ -109,9 +107,9 @@ def test_us_seed_stream_map_is_complete_and_owner_typed( ) -> None: seed_map = compiled_us.seed_stream_map assert seed_map.protocol_id == "legacy-v1" - assert len(seed_map.sites) == 53 + assert len(seed_map.sites) == 66 assert len(seed_map.owners) == 54 - assert sum(len(site.owners) for site in seed_map.sites) == 112 + assert sum(len(site.owners) for site in seed_map.sites) == 125 assert {site.id for site in seed_map.sites} == { site.id for site in resolved_us.seed_protocol.sites } @@ -180,9 +178,7 @@ def test_normative_node_mutation_changes_its_slice_and_descendant_keys( def mutation(value: dict[str, Any]) -> None: node = next( - row - for row in value["producer_graph"]["nodes"] - if row["id"] == leaf_id + row for row in value["producer_graph"]["nodes"] if row["id"] == leaf_id ) node["capabilities"]["retry_safety"] = "nonretryable" @@ -196,9 +192,9 @@ def mutation(value: dict[str, Any]) -> None: def test_dangling_compiler_dependency_refuses(resolved_us: ResolvedSpec) -> None: def mutation(value: dict[str, Any]) -> None: - value["producer_graph"]["nodes"][0]["inputs"][0][ - "producing_stage" - ] = "missing_producer" + value["producer_graph"]["nodes"][0]["inputs"][0]["producing_stage"] = ( + "missing_producer" + ) mutated = _mutate_domain(resolved_us, ResourceKind.IMPUTATION, mutation) with pytest.raises(CompilerIRError, match="dangling producer"): @@ -211,9 +207,7 @@ def test_contract_only_kernel_cannot_back_a_producer( contract_only_kernel = min(F0_CONTRACT_ONLY_KERNEL_IDS) def mutation(value: dict[str, Any]) -> None: - value["producer_graph"]["nodes"][0]["kernel"] = ( - f"kernel:{contract_only_kernel}" - ) + value["producer_graph"]["nodes"][0]["kernel"] = f"kernel:{contract_only_kernel}" mutated = _mutate_domain(resolved_us, ResourceKind.IMPUTATION, mutation) with pytest.raises( @@ -250,5 +244,5 @@ def test_bundle_without_imputation_compiles_empty_graph( assert compiled.producer_graph.nodes == () assert compiled.stage_dag.nodes == () assert compiled.nodes == () - assert len(compiled.seed_stream_map.sites) == 53 + assert len(compiled.seed_stream_map.sites) == 66 assert compiled.seed_stream_map.owners == () diff --git a/packages/microcosm-build/tests/test_spec_engine_country_bundles.py b/packages/microcosm-build/tests/test_spec_engine_country_bundles.py index 700e8592..913b10d5 100644 --- a/packages/microcosm-build/tests/test_spec_engine_country_bundles.py +++ b/packages/microcosm-build/tests/test_spec_engine_country_bundles.py @@ -32,7 +32,7 @@ [ ( "be", - "7062e38f4d623553fb0604380a8dac0edacb6261c155b6e31fc38ef7c0f1c57c", + "16118de0f95b7821a7ede20104d1466235d6900c665d5220b2a68a0a52007515", { "household.household_id", "person.person_id", @@ -42,7 +42,7 @@ ), ( "uk", - "cce1c98ea40364a398ae361f4d15790c925d8379d8b9b427076d61059c7d6715", + "c4e77decba4be76d920af1c61c1244c365132bea0e77324c1a1f8f868475b990", { "benunit.benunit_id", "household.household_id", diff --git a/packages/microcosm-build/tests/test_spec_engine_coverage_tool.py b/packages/microcosm-build/tests/test_spec_engine_coverage_tool.py index 099e6881..dec9a4c6 100644 --- a/packages/microcosm-build/tests/test_spec_engine_coverage_tool.py +++ b/packages/microcosm-build/tests/test_spec_engine_coverage_tool.py @@ -52,22 +52,22 @@ def test_us_coverage_is_exact_complete_and_honest( assert_coverage_complete(coverage_report) assert coverage_report["status"] == "pass" fields = coverage_report["field_usage"] - assert fields["configuration_field_count"] == 42_154 - assert fields["authored_normative_field_count"] == 32_384 - assert fields["resolved_binding_field_count"] == 9_770 - assert fields["consumed_field_count"] == 42_154 + assert fields["configuration_field_count"] == 42_538 + assert fields["authored_normative_field_count"] == 32_521 + assert fields["resolved_binding_field_count"] == 10_017 + assert fields["consumed_field_count"] == 42_538 assert fields["unused_field_count"] == 0 assert fields["multiple_primary_use_field_count"] == 0 assert fields["claim_count"] == 49 assert fields["mode_counts"] == { - "legacy_behavior": 13_988, - "compiler_semantic": 27_715, + "legacy_behavior": 14_020, + "compiler_semantic": 28_067, "front_end_validation": 348, "identity_only": 103, } assert fields["generation0_effect_counts"] == { - "legacy_behavior": 38_476, - "no_generation0_effect": 3_678, + "legacy_behavior": 38_574, + "no_generation0_effect": 3_964, } inventory = coverage_report["inventory_coverage"] @@ -75,10 +75,10 @@ def test_us_coverage_is_exact_complete_and_honest( assert inventory["covered_item_count"] == 41 assert inventory["missing_item_count"] == 0 assert inventory["missing_items"] == [] - assert inventory["counts"]["producer_inputs"] == 2_744 + assert inventory["counts"]["producer_inputs"] == 2_750 assert inventory["counts"]["ownership_rows"] == 18 assert inventory["counts"]["tail_control_fields"] == 934 - assert inventory["counts"]["seed_owner_bindings"] == 112 + assert inventory["counts"]["seed_owner_bindings"] == 125 @pytest.mark.parametrize( @@ -89,9 +89,7 @@ def test_us_coverage_is_exact_complete_and_honest( "report schema version differs", ), ( - lambda report: report["spec_binding"].__setitem__( - "schema_version", 99 - ), + lambda report: report["spec_binding"].__setitem__("schema_version", 99), "spec_binding contract differs", ), ( diff --git a/packages/microcosm-build/tests/test_spec_engine_field_usage.py b/packages/microcosm-build/tests/test_spec_engine_field_usage.py index d33908b8..90eb846f 100644 --- a/packages/microcosm-build/tests/test_spec_engine_field_usage.py +++ b/packages/microcosm-build/tests/test_spec_engine_field_usage.py @@ -92,22 +92,22 @@ def _mutated_bundle( def test_exact_complete_ledger_has_one_primary_mode_per_pointer(field_ledger) -> None: - assert len(field_ledger.fields) == EXPECTED_CONFIGURATION_FIELD_COUNT == 42_154 + assert len(field_ledger.fields) == EXPECTED_CONFIGURATION_FIELD_COUNT == 42_538 assert field_ledger.source_counts == { - "authored": 32_384, - "resolved_bindings": 9_770, + "authored": 32_521, + "resolved_bindings": 10_017, } assert field_ledger.mode_counts == { - "legacy_behavior": 13_988, - "compiler_semantic": 27_715, + "legacy_behavior": 14_020, + "compiler_semantic": 28_067, "front_end_validation": 348, "identity_only": 103, } assert field_ledger.generation0_effect_counts == { - "legacy_behavior": 38_476, - "no_generation0_effect": 3_678, + "legacy_behavior": 38_574, + "no_generation0_effect": 3_964, } - assert len({field.pointer for field in field_ledger.fields}) == 42_154 + assert len({field.pointer for field in field_ledger.fields}) == 42_538 def test_eligibility_concepts_are_validation_not_generation0_behavior( @@ -162,15 +162,11 @@ def test_spine_assembly_mass_share_fields_name_exact_adapter_sinks( ) -> None: for channel in ("acs", "asec"): field = field_ledger.field( - "/authored/spec~1spine.yaml/assembly/household_mass_shares/" - f"{channel}" + f"/authored/spec~1spine.yaml/assembly/household_mass_shares/{channel}" ) assert field.mode is UsageMode.LEGACY_BEHAVIOR assert field.generation0_effect is Generation0Effect.LEGACY_BEHAVIOR - assert ( - f"/spine_assembly/household_mass_shares/{channel}" - in field.sink_pointers - ) + assert f"/spine_assembly/household_mass_shares/{channel}" in field.sink_pointers def test_copied_surfaces_cannot_rescue_a_missing_calibration_sink( @@ -358,14 +354,10 @@ def mutate(document: dict[str, object]) -> None: ledger = build_field_usage_ledger(mutated, legacy_payload=mutated_legacy) for channel in ("acs", "asec"): field = ledger.field( - "/authored/spec~1spine.yaml/assembly/household_mass_shares/" - f"{channel}" + f"/authored/spec~1spine.yaml/assembly/household_mass_shares/{channel}" ) assert field.claim_id == "spine_assembly_household_mass_shares" - assert ( - f"/spine_assembly/household_mass_shares/{channel}" - in field.sink_pointers - ) + assert f"/spine_assembly/household_mass_shares/{channel}" in field.sink_pointers def test_geography_declaration_mutation_changes_checkpoint_identity( diff --git a/packages/microcosm-build/tests/test_spec_engine_identity_contracts.py b/packages/microcosm-build/tests/test_spec_engine_identity_contracts.py index a2a682f6..8d1f13e2 100644 --- a/packages/microcosm-build/tests/test_spec_engine_identity_contracts.py +++ b/packages/microcosm-build/tests/test_spec_engine_identity_contracts.py @@ -63,7 +63,7 @@ def test_pipeline_contract_is_an_exact_generation_zero_projection() -> None: == { "artifact_kind": "populace_us_stacked_pool_checkpoint_identity", "schema_version": 1, - "materializer_version": 12, + "materializer_version": 13, "pipeline": "us-stacked-pool", } ) @@ -118,13 +118,13 @@ def test_identity_contract_objects_are_closed_world( load_schema_registry().validate(mutated, "spine.schema.json") -def test_all_53_seed_sites_resolve_to_typed_real_owners( +def test_all_seed_sites_resolve_to_typed_real_owners( identity_documents: tuple[dict[str, object], list[dict[str, object]]], ) -> None: spine, source_stages = identity_documents bindings = _resolve(spine, source_stages) - assert len(bindings) == len(LEGACY_V1_PROTOCOL.sites) == 53 + assert len(bindings) == len(LEGACY_V1_PROTOCOL.sites) == 66 assert [binding.site for binding in bindings] == [ site.id for site in LEGACY_V1_PROTOCOL.sites ] diff --git a/packages/microcosm-build/tests/test_spec_engine_imputation_semantics.py b/packages/microcosm-build/tests/test_spec_engine_imputation_semantics.py index c9ae7681..28a03e43 100644 --- a/packages/microcosm-build/tests/test_spec_engine_imputation_semantics.py +++ b/packages/microcosm-build/tests/test_spec_engine_imputation_semantics.py @@ -98,7 +98,7 @@ def test_imputation_projector_matches_live_plans_and_graph_receipts( for key, expected in live.items(): assert canonical_json_bytes(projected[key]) == canonical_json_bytes(expected) assert projected["late_producer_schedule_receipt"]["schedule_sha256"] == ( - "e59c019d3d454eac99ac0ac209b6c5b6faaf9bdfcaeee18c36a25be19bf7da2f" + "88bc9243a3518982ae951c3de21bd55877e296ce4fcb183b9bee420d3a684b10" ) assert projected["overlap_ownership"]["sha256"] == ( "5f64f0aac49e2313177564f71876bffc8c81b3ded4df701e70930e60e9c98356" diff --git a/packages/microcosm-build/tests/test_spec_engine_inventory_coverage.py b/packages/microcosm-build/tests/test_spec_engine_inventory_coverage.py index 922d2337..3b8a8cc3 100644 --- a/packages/microcosm-build/tests/test_spec_engine_inventory_coverage.py +++ b/packages/microcosm-build/tests/test_spec_engine_inventory_coverage.py @@ -79,14 +79,14 @@ "primary_targets": 65, "producer_authored_outputs": 92, "producer_compiled_outputs": 227, - "producer_inputs": 2_744, + "producer_inputs": 2_750, "producer_nodes": 38, "producer_virtual_resources": 75, "release_rungs": 5, "resolved_references": 334, - "seed_owner_bindings": 112, + "seed_owner_bindings": 125, "seed_owner_rows": 54, - "seed_sites": 53, + "seed_sites": 66, "seed_streams": 14, "source_operators": 16, "source_stages": 37, diff --git a/packages/microcosm-build/tests/test_spec_engine_legacy_adapter.py b/packages/microcosm-build/tests/test_spec_engine_legacy_adapter.py index 9edc4e02..5c72c776 100644 --- a/packages/microcosm-build/tests/test_spec_engine_legacy_adapter.py +++ b/packages/microcosm-build/tests/test_spec_engine_legacy_adapter.py @@ -195,10 +195,10 @@ def test_adapter_preserves_generation_zero_identity_components( imputation = legacy_payload["imputation"] assert isinstance(imputation, dict) assert legacy_payload["stacked_authority_receipt"]["sha256"] == ( - "e660a8ce42b69a39d29c5f0ec37264bc69d61b03f27adc386336ec8889531bb2" + "406b2cf93a7fb94dc63f1a24a52cfb20362ab83195eb8ebf61d4ef43072280ec" ) assert imputation["late_producer_schedule_receipt"]["schedule_sha256"] == ( - "e59c019d3d454eac99ac0ac209b6c5b6faaf9bdfcaeee18c36a25be19bf7da2f" + "88bc9243a3518982ae951c3de21bd55877e296ce4fcb183b9bee420d3a684b10" ) assert imputation["overlap_ownership"]["sha256"] == ( "5f64f0aac49e2313177564f71876bffc8c81b3ded4df701e70930e60e9c98356" diff --git a/packages/microcosm-build/tests/test_spec_engine_loader.py b/packages/microcosm-build/tests/test_spec_engine_loader.py index a77c90f6..4d66b251 100644 --- a/packages/microcosm-build/tests/test_spec_engine_loader.py +++ b/packages/microcosm-build/tests/test_spec_engine_loader.py @@ -236,7 +236,7 @@ def test_semantic_hash_has_golden_vector_and_surface_separation(tmp_path) -> Non # Pin the domain separator, normalization rules, schema-set receipt, and # exact normative projection as one reviewable golden vector. assert first.spec_sha256 == ( - "9f5b372796f2638378125d97ef5150be4c1f4cba9147b44973e4cd6a5f52f10a" + "8b72d260c755e46b4a07043ee81949ff79196e9b459cd9ae1096e99eeed2f4e4" ) second_root = _rich_minimal(tmp_path / "xy", note="second", store="local:b") diff --git a/packages/microcosm-build/tests/test_spec_engine_plan_lock.py b/packages/microcosm-build/tests/test_spec_engine_plan_lock.py index a0cd91c4..6764e243 100644 --- a/packages/microcosm-build/tests/test_spec_engine_plan_lock.py +++ b/packages/microcosm-build/tests/test_spec_engine_plan_lock.py @@ -42,7 +42,7 @@ def test_plan_lock_is_complete_and_closed_world(compiled_us: CompiledSpecIR) -> assert payload["spec_binding"]["attestation"] == "mirror-attested" assert len(payload["producer_graph"]["nodes"]) == 38 assert len(payload["producer_graph"]["authored"]["ownership_matrix"]) == 18 - assert len(payload["seed_stream_map"]["sites"]) == 53 + assert len(payload["seed_stream_map"]["sites"]) == 66 assert len(payload["nodes"]) == 38 mutated = copy.deepcopy(payload) diff --git a/packages/microcosm-build/tests/test_spec_engine_seeds.py b/packages/microcosm-build/tests/test_spec_engine_seeds.py index c0e181a9..8990163f 100644 --- a/packages/microcosm-build/tests/test_spec_engine_seeds.py +++ b/packages/microcosm-build/tests/test_spec_engine_seeds.py @@ -43,6 +43,19 @@ "housing_inputs_training_cap", "immigration_ead_students_assignment", "immigration_ead_workers_assignment", + "immigration_humanitarian_asylee_assignment", + "immigration_humanitarian_deportation_withheld_assignment", + "immigration_humanitarian_paroled_one_year_afghanistan_assignment", + "immigration_humanitarian_paroled_one_year_nicaragua_assignment", + "immigration_humanitarian_paroled_one_year_ukraine_assignment", + "immigration_humanitarian_paroled_one_year_venezuela_assignment", + "immigration_humanitarian_refugee_assignment", + "immigration_humanitarian_tps_el_salvador_assignment", + "immigration_humanitarian_tps_honduras_assignment", + "immigration_humanitarian_tps_nepal_assignment", + "immigration_humanitarian_tps_nicaragua_assignment", + "immigration_humanitarian_tps_other_designated_assignment", + "immigration_humanitarian_tps_venezuela_assignment", "legacy_congressional_district_assignment", "legacy_geography_ladder", "legacy_puma_ladder", @@ -140,6 +153,19 @@ ( "immigration_ead_workers_assignment", "immigration_ead_students_assignment", + "immigration_humanitarian_paroled_one_year_afghanistan_assignment", + "immigration_humanitarian_paroled_one_year_ukraine_assignment", + "immigration_humanitarian_paroled_one_year_nicaragua_assignment", + "immigration_humanitarian_paroled_one_year_venezuela_assignment", + "immigration_humanitarian_refugee_assignment", + "immigration_humanitarian_asylee_assignment", + "immigration_humanitarian_deportation_withheld_assignment", + "immigration_humanitarian_tps_venezuela_assignment", + "immigration_humanitarian_tps_el_salvador_assignment", + "immigration_humanitarian_tps_honduras_assignment", + "immigration_humanitarian_tps_nicaragua_assignment", + "immigration_humanitarian_tps_nepal_assignment", + "immigration_humanitarian_tps_other_designated_assignment", ), "packages/microcosm-build/src/microcosm/build/us_runtime/immigration.py", ), diff --git a/packages/microcosm-build/tests/test_spec_engine_stacked_authority_semantics.py b/packages/microcosm-build/tests/test_spec_engine_stacked_authority_semantics.py index 1096b409..96a870f3 100644 --- a/packages/microcosm-build/tests/test_spec_engine_stacked_authority_semantics.py +++ b/packages/microcosm-build/tests/test_spec_engine_stacked_authority_semantics.py @@ -73,7 +73,7 @@ def test_authority_projection_is_field_and_byte_identical_to_live_generation_zer assert projected == live assert stacked_identity_bytes(projected) == _canonical_bytes(live) assert projected["sha256"] == ( - "e660a8ce42b69a39d29c5f0ec37264bc69d61b03f27adc386336ec8889531bb2" + "406b2cf93a7fb94dc63f1a24a52cfb20362ab83195eb8ebf61d4ef43072280ec" ) assert { name: component["sha256"] for name, component in projected["components"].items() @@ -88,7 +88,7 @@ def test_authority_projection_is_field_and_byte_identical_to_live_generation_zer "cacc6c11e114dbae3aaa2761cc6b3fcb1191cd9b689b1c2bd096614c51ebff8b" ), "late_producer_schedule": ( - "1b81157b0e21e4763884620ec27b5c4e6e36cc28237273c24eeccdef05a7fbca" + "8f49aad23470d7b09b401cc0411f3199657cf9d77b26edfad2a435f271a8643d" ), "metric_registry": ( "d75cb9b29f8b0a9a085471a11f4c19c32ba04cbe5419053df94ea81cbe6125a9" @@ -191,7 +191,7 @@ def test_checkpoint_projection_is_field_and_byte_identical_to_live_oracle( ] assert ( projected["pool_code"]["late_producer_schedule"]["schedule_sha256"] - == "e59c019d3d454eac99ac0ac209b6c5b6faaf9bdfcaeee18c36a25be19bf7da2f" + == "88bc9243a3518982ae951c3de21bd55877e296ce4fcb183b9bee420d3a684b10" ) diff --git a/packages/microcosm-build/tests/test_us_acs_pums.py b/packages/microcosm-build/tests/test_us_acs_pums.py index 48b656d9..897a4b88 100644 --- a/packages/microcosm-build/tests/test_us_acs_pums.py +++ b/packages/microcosm-build/tests/test_us_acs_pums.py @@ -65,6 +65,9 @@ def _person( "SSIP": 0, "RETP": 0, "INTP": 0, + "CIT": 1, + "POBP": 6, + "YOEP": None, "PWGTP": 10, } row.update(overrides) @@ -94,7 +97,16 @@ def _source(tmp_path: Path) -> AcsPumsSource: person_zip, { "psam_pusa.csv": [ - _person("2024HU0000002", 1, 20, MAR=4, WAGP=30_000), + _person( + "2024HU0000002", + 1, + 20, + MAR=4, + WAGP=30_000, + CIT=5, + POBP=164, + YOEP=2021, + ), _person( "2024HU0000001", 1, @@ -193,6 +205,14 @@ def test_load_acs_pums_tables_streams_all_csv_members_and_keeps_native_blanks( "2024HU0000002", ] assert pd.isna(tables["person"].loc[2, "WAGP"]) + foreign_born = ( + tables["person"].loc[tables["person"]["SERIALNO"].eq("2024HU0000002")].iloc[0] + ) + assert (foreign_born["CIT"], foreign_born["POBP"], foreign_born["YOEP"]) == ( + 5, + 164, + 2021, + ) assert metadata["vacant_household_rows_dropped"] == 1 assert metadata["household_csv_members"] == ["psam_husa.csv", "psam_husb.csv"] assert metadata["person_csv_members"] == ["psam_pusa.csv", "psam_pusb.csv"] @@ -226,6 +246,9 @@ def test_build_acs_pums_unit_frame_preserves_lineage_geography_and_weights( assert person["source_household_id"].dtype == np.dtype(np.int64) assert pd.api.types.is_string_dtype(person["source_person_id"].dtype) assert person["source_row_id"].dtype == np.dtype(np.int64) + assert person.loc[person["source_household_id"].eq(2), "CIT"].item() == 5 + assert person.loc[person["source_household_id"].eq(2), "POBP"].item() == 164 + assert person.loc[person["source_household_id"].eq(2), "YOEP"].item() == 2021 assert metadata["weighted_household_population"] == pytest.approx(30.0) diff --git a/packages/microcosm-build/tests/test_us_acs_transfer.py b/packages/microcosm-build/tests/test_us_acs_transfer.py index e9e615be..7db570f6 100644 --- a/packages/microcosm-build/tests/test_us_acs_transfer.py +++ b/packages/microcosm-build/tests/test_us_acs_transfer.py @@ -11,6 +11,7 @@ import microcosm.build.us_runtime.acs_transfer as acs_transfer_module import microcosm.build.us_runtime.acs_transfer_bank as acs_transfer_bank_module +import microcosm.build.us_runtime.immigration as immigration_module from microcosm.build.frame_checkpoint import write_frame_checkpoint from microcosm.build.serialization_dtypes import CANONICAL_STRING_DTYPE from microcosm.build.us_runtime.acs_transfer import ( @@ -22,6 +23,7 @@ ACS_PERSON_TRANSFER_PREDICTORS, AcsTransferResult, acs_adult_care_qualifying_rows, + acs_transfer_donor_requirements, declared_acs_transfer_target_families, default_acs_transfer_target_families, transfer_acs_inputs, @@ -29,6 +31,11 @@ from microcosm.build.us_runtime.acs_transfer_bank import ( AcsTransferTargetBankStore, ) +from microcosm.build.us_runtime.immigration import ( + HumanitarianDraw, + ImmigrationControls, + UndocumentedControls, +) from microcosm.build.us_runtime.puf_support import clone_us_frame_for_puf_support from microcosm.build.us_runtime.spine_assembly import assemble_spines from microcosm.fit import Regime @@ -84,6 +91,9 @@ def _donor_frame() -> Frame: ), "age": [45.0, 43.0, 30.0, 8.0, 68.0, 66.0, 39.0, 17.0], "is_female": [False, True, True, False, False, True, True, False], + "PRCITSHP": [1, 1, 1, 1, 1, 1, 5, 5], + "PENATVTY": [57, 57, 57, 57, 57, 57, 164, 373], + "PEINUSYR": [0, 0, 0, 0, 0, 0, 28, 24], "is_household_head": [True, False, True, False, True, False, True, False], "employment_income_before_lsr": [ 80_000.0, @@ -208,6 +218,9 @@ def _recipient_frame() -> Frame: ), "age": [41.0, 40.0, 27.0, 72.0, 69.0, 15.0], "is_female": [False, True, True, False, True, False], + "CIT": [1, 1, 5, 1, 1, 5], + "POBP": [6, 6, 164, 36, 36, 373], + "YOEP": [np.nan, np.nan, 2023, np.nan, np.nan, 2015], "is_household_head": [True, False, True, True, False, False], "employment_income_before_lsr": [ 65_000.0, @@ -263,6 +276,90 @@ def _recipient_frame() -> Frame: ) +def _mixed_immigration_recipient() -> Frame: + """One immutable ASEC row followed by ACS rows with paired null targets.""" + + base = _recipient_frame() + tables = {entity: base.table(entity).copy() for entity in base.entities} + person = tables["person"] + for column in ("PRCITSHP", "PENATVTY", "PEINUSYR"): + person[column] = np.nan + immutable_index = person.index[0] + person.loc[immutable_index, ["CIT", "POBP", "YOEP"]] = np.nan + person.loc[immutable_index, ["PRCITSHP", "PENATVTY", "PEINUSYR"]] = [ + 5, + 200, + 27, + ] + person["ssn_card_type"] = pd.Series( + ["OTHER_NON_CITIZEN", *([np.nan] * (len(person) - 1))], + index=person.index, + dtype=object, + ) + person["immigration_status_str"] = pd.Series( + ["PAROLED_ONE_YEAR", *([np.nan] * (len(person) - 1))], + index=person.index, + dtype=object, + ) + return Frame( + tables, + base.schema, + {entity: base.weights_for(entity) for entity in base.weighted_entities}, + base.strata, + mass_log=base.mass_log, + ) + + +def _small_immigration_controls( + *, + ukraine_target: float = 60.0, +) -> ImmigrationControls: + source = "https://example.com/immigration-control" + return ImmigrationControls( + undocumented=UndocumentedControls( + workers=1.0, + students=1.0, + population_anchor=1.0, + sources={ + "undocumented_workers": source, + "undocumented_students": source, + "undocumented_population_anchor": source, + }, + ), + humanitarian=( + HumanitarianDraw( + category="paroled_one_year", + origin="afghanistan", + status="PAROLED_ONE_YEAR", + target=50.0, + source=source, + ), + HumanitarianDraw( + category="paroled_one_year", + origin="ukraine", + status="PAROLED_ONE_YEAR", + target=ukraine_target, + source=source, + ), + HumanitarianDraw( + category="tps", + origin="venezuela", + status="TPS", + target=70.0, + source=source, + ), + ), + ) + + +def _no_humanitarian_controls() -> ImmigrationControls: + controls = _small_immigration_controls() + return ImmigrationControls( + undocumented=controls.undocumented, + humanitarian=(), + ) + + def _replace_column( frame: Frame, entity: str, @@ -677,7 +774,6 @@ def test_explicit_declared_plan_preserves_deferred_geography( lambda: plan, ) monkeypatch.setattr(acs_transfer_module, "QRF", _MeanQRF) - result = transfer_acs_inputs( _recipient_frame(), _donor_frame(), @@ -745,6 +841,59 @@ def test_declared_families_are_independent_of_release_coverage_surface() -> None assert "receives_wic" in production_declared["person"]["model_required_boolean"] +def test_donor_requirements_include_immigration_source_evidence_triplet() -> None: + donor = _with_columns( + _donor_frame(), + "person", + { + "ssn_card_type": ["CITIZEN"] * 6 + ["NONE", "NONE"], + "immigration_status_str": ["CITIZEN"] * 6 + + ["UNDOCUMENTED", "UNDOCUMENTED"], + }, + ) + plan = { + "person": { + "source_operator_immigration": ( + "ssn_card_type", + "immigration_status_str", + ) + } + } + + requirements = acs_transfer_donor_requirements(donor, plan) + + assert {"PRCITSHP", "PENATVTY", "PEINUSYR"}.issubset(requirements["person"]) + + acs_tables = {entity: donor.table(entity).copy() for entity in donor.entities} + acs_person = acs_tables["person"].rename( + columns={"PRCITSHP": "CIT", "PENATVTY": "POBP"} + ) + acs_person = acs_person.drop(columns=["PEINUSYR"]) + acs_person["YOEP"] = [np.nan] * 6 + [2023, 2015] + acs_tables["person"] = acs_person + acs_donor = Frame( + acs_tables, + donor.schema, + {entity: donor.weights_for(entity) for entity in donor.weighted_entities}, + donor.strata, + mass_log=donor.mass_log, + ) + acs_requirements = acs_transfer_donor_requirements(acs_donor, plan) + assert {"CIT", "POBP", "YOEP"}.issubset(acs_requirements["person"]) + + tables = {entity: donor.table(entity).copy() for entity in donor.entities} + tables["person"] = tables["person"].drop(columns=["PEINUSYR"]) + incomplete = Frame( + tables, + donor.schema, + {entity: donor.weights_for(entity) for entity in donor.weighted_entities}, + donor.strata, + mass_log=donor.mass_log, + ) + with pytest.raises(ValueError, match="incomplete donor evidence triplet"): + acs_transfer_donor_requirements(incomplete, plan) + + def test_declared_plan_carries_the_23_stage_base_surface() -> None: """The M/N/O donor repairs must reach the ACS spine through the plan.""" @@ -871,6 +1020,244 @@ def test_explicit_transfer_adds_requested_model_inputs( and "immigration_status_str" not in call["targets"] for call in joint_calls ) + assert { + "__acs_transfer_is_us_citizen", + "__acs_transfer_birth_country_code", + "__acs_transfer_arrival_year", + }.issubset(joint_calls[0]["predictors"]) + + +def test_joint_immigration_codec_never_emits_constrained_baseline( + monkeypatch: pytest.MonkeyPatch, +) -> None: + constrained = [ + "PAROLED_ONE_YEAR", + "REFUGEE", + "ASYLEE", + "DEPORTATION_WITHHELD", + "TPS", + "CUBAN_HAITIAN_ENTRANT", + "DACA", + "REFUGEE", + ] + donor = _with_columns( + _donor_frame(), + "person", + { + "ssn_card_type": ["OTHER_NON_CITIZEN"] * 8, + "immigration_status_str": constrained, + }, + ) + monkeypatch.setattr(acs_transfer_module, "QRF", _MeanQRF) + monkeypatch.setattr( + immigration_module, + "us_immigration_controls", + _no_humanitarian_controls, + ) + + result = transfer_acs_inputs( + _recipient_frame(), + donor, + target_families={ + "person": { + "source_operator_immigration": ( + "ssn_card_type", + "immigration_status_str", + ) + } + }, + donor_channel=None, + n_estimators=1, + ) + + assert set(result.frame.person["immigration_status_str"]) <= { + "CITIZEN", + "LEGAL_PERMANENT_RESIDENT", + } + encoding = acs_transfer_module._target_encodings( + donor.person, + targets=("ssn_card_type", "immigration_status_str"), + )["immigration_status_str"] + assert not {pair[1] for pair in encoding.categories}.intersection(constrained) + + +def test_transfer_rederives_entrant_and_daca_only_on_compatible_acs_cohorts( + monkeypatch: pytest.MonkeyPatch, +) -> None: + donor = _with_columns( + _donor_frame(), + "person", + { + "ssn_card_type": ["NON_CITIZEN_VALID_EAD"] * 8, + "immigration_status_str": [ + "CUBAN_HAITIAN_ENTRANT", + "DACA", + ] + * 4, + }, + ) + recipient = _with_columns( + _recipient_frame(), + "person", + { + "age": [50.0, 50.0, 29.0, 50.0, 40.0, 70.0], + "CIT": [5, 5, 5, 5, 1, 5], + "POBP": [327, 164, 312, 312, 6, 332], + "YOEP": [2000, 2020, 2005, 2020, np.nan, 1970], + }, + ) + monkeypatch.setattr(acs_transfer_module, "QRF", _MeanQRF) + monkeypatch.setattr( + immigration_module, + "us_immigration_controls", + _no_humanitarian_controls, + ) + + result = transfer_acs_inputs( + recipient, + donor, + target_families={ + "person": { + "source_operator_immigration": ( + "ssn_card_type", + "immigration_status_str", + ) + } + }, + donor_channel=None, + n_estimators=1, + ) + + assert result.frame.person["immigration_status_str"].tolist() == [ + "CUBAN_HAITIAN_ENTRANT", + "LEGAL_PERMANENT_RESIDENT", + "DACA", + "LEGAL_PERMANENT_RESIDENT", + "CITIZEN", + "LEGAL_PERMANENT_RESIDENT", + ] + receipt = next( + item.reconciliation + for item in result.imputed_inputs + if item.column == "immigration_status_str" + ) + assert receipt is not None + assert receipt["special_status_assignments"] == 2 + + +def test_humanitarian_reconciliation_uses_residual_targets_and_records_receipt( + monkeypatch: pytest.MonkeyPatch, +) -> None: + donor = _with_columns( + _donor_frame(), + "person", + { + "ssn_card_type": ["OTHER_NON_CITIZEN"] * 8, + "immigration_status_str": ["LEGAL_PERMANENT_RESIDENT"] * 8, + }, + ) + monkeypatch.setattr(acs_transfer_module, "QRF", _MeanQRF) + monkeypatch.setattr( + immigration_module, + "us_immigration_controls", + _small_immigration_controls, + ) + + result = transfer_acs_inputs( + _mixed_immigration_recipient(), + donor, + target_families={ + "person": { + "source_operator_immigration": ( + "ssn_card_type", + "immigration_status_str", + ) + } + }, + donor_channel=None, + seed=19, + n_estimators=1, + ) + + person = result.frame.person + assert person["immigration_status_str"].tolist() == [ + "PAROLED_ONE_YEAR", # immutable Afghanistan contribution + "CITIZEN", # source CIT repairs the imputed baseline pair + "PAROLED_ONE_YEAR", # compatible Ukrainian residual + "CITIZEN", + "CITIZEN", + "TPS", # compatible Venezuelan residual + ] + assert person.loc[person.index[0], "ssn_card_type"] == "OTHER_NON_CITIZEN" + reconciled = [ + item + for item in result.imputed_inputs + if item.column in {"ssn_card_type", "immigration_status_str"} + ] + assert len(reconciled) == 2 + assert reconciled[0].reconciliation == reconciled[1].reconciliation + receipt = reconciled[0].reconciliation + assert receipt is not None + assert receipt["mutable_rows"] == 5 + assert receipt["immutable_rows"] == 1 + draws = receipt["draws"] + assert draws["paroled_one_year:afghanistan"] == { + "target": 50.0, + "immutable_population": 50.0, + "residual_target": 0.0, + "eligible_recipient_population": 0.0, + "selected_recipient_population": 0.0, + "achieved_population": 50.0, + "absolute_error": 0.0, + "selection_threshold_tie_population": 0.0, + "residual_selection_error": 0.0, + "within_residual_discrete_weight_bound": True, + "immutable_overshoot": 0.0, + "within_discrete_weight_bound": True, + } + assert draws["paroled_one_year:ukraine"]["residual_target"] == 60.0 + assert draws["paroled_one_year:ukraine"]["achieved_population"] == 60.0 + assert draws["tps:venezuela"]["achieved_population"] == 70.0 + assert all(draw["within_discrete_weight_bound"] for draw in draws.values()) + + +def test_humanitarian_reconciliation_fails_on_candidate_shortfall( + monkeypatch: pytest.MonkeyPatch, +) -> None: + donor = _with_columns( + _donor_frame(), + "person", + { + "ssn_card_type": ["OTHER_NON_CITIZEN"] * 8, + "immigration_status_str": ["LEGAL_PERMANENT_RESIDENT"] * 8, + }, + ) + monkeypatch.setattr(acs_transfer_module, "QRF", _MeanQRF) + monkeypatch.setattr( + immigration_module, + "us_immigration_controls", + lambda: _small_immigration_controls(ukraine_target=61.0), + ) + + with pytest.raises( + ValueError, + match="candidate shortfall.*paroled_one_year:ukraine", + ): + transfer_acs_inputs( + _mixed_immigration_recipient(), + donor, + target_families={ + "person": { + "source_operator_immigration": ( + "ssn_card_type", + "immigration_status_str", + ) + } + }, + donor_channel=None, + seed=19, + n_estimators=1, + ) def test_large_target_family_is_split_to_bound_retained_qrf_forests( @@ -1158,6 +1545,11 @@ def test_target_bank_resumes_joint_immigration_codec_as_one_model_target( tmp_path: Path, ) -> None: _lock_bank_fixture_threads(monkeypatch) + monkeypatch.setattr( + immigration_module, + "us_immigration_controls", + _no_humanitarian_controls, + ) donor = _with_columns( _donor_frame(), "person", @@ -1467,9 +1859,8 @@ def test_pregnancy_draws_once_per_eligible_source_person_and_fans_to_clones( ) person = result.frame.person - eligible = ( - person["is_female"].astype(bool) - & person["age"].between(15, 44, inclusive="both") + eligible = person["is_female"].astype(bool) & person["age"].between( + 15, 44, inclusive="both" ) assert person.loc[eligible, "is_pregnant"].all() assert not person.loc[~eligible, "is_pregnant"].any() @@ -1490,7 +1881,9 @@ def test_pregnancy_draws_once_per_eligible_source_person_and_fans_to_clones( .first() .sum() ) - record = next(item for item in result.imputed_inputs if item.column == "is_pregnant") + record = next( + item for item in result.imputed_inputs if item.column == "is_pregnant" + ) receipt = record.structural_receipt assert receipt is not None assert sum(pattern.recipient_rows for pattern in record.patterns) == ( @@ -1576,12 +1969,8 @@ def test_pregnancy_partial_clone_fanout_receipt_categories_are_disjoint( record = result.imputed_inputs[0] receipt = record.structural_receipt assert receipt is not None - assert receipt["preexisting_value_fanout_rows"] == int( - (missing & eligible).sum() - ) - assert receipt["ineligible_rows_assigned_false"] == int( - (missing & ~eligible).sum() - ) + assert receipt["preexisting_value_fanout_rows"] == int((missing & eligible).sum()) + assert receipt["ineligible_rows_assigned_false"] == int((missing & ~eligible).sum()) assert ( receipt["preexisting_value_fanout_rows"] + receipt["ineligible_rows_assigned_false"] diff --git a/packages/microcosm-build/tests/test_us_asec_checkpoint.py b/packages/microcosm-build/tests/test_us_asec_checkpoint.py index aa2a1338..e15ff5ec 100644 --- a/packages/microcosm-build/tests/test_us_asec_checkpoint.py +++ b/packages/microcosm-build/tests/test_us_asec_checkpoint.py @@ -134,7 +134,7 @@ def _raw_binding(frame: Frame) -> dict[str, object]: "operation": "exact_source_join", "source_pins": [pin], } - for column in ("ED_VAL", "LKWEEKS", "PAW_TYP") + for column in ("A_LFSR", "ED_VAL", "LKWEEKS", "PAW_TYP") }, "schema_version": ASEC_RAW_STAGE_SCHEMA_VERSION, "source_construction_identity": frame_identity(frame).to_payload(), @@ -166,6 +166,7 @@ def _raw_us_frame(*, id_offset: int = 0) -> Frame: tables["person"]["ED_VAL"] = [0.0, 500.0] tables["person"]["LKWEEKS"] = [-1, 12] tables["person"]["PAW_TYP"] = np.asarray([0, 1], dtype=np.int64) + tables["person"]["A_LFSR"] = np.asarray([1, 7], dtype=np.int64) return Frame( tables, source.schema, @@ -248,7 +249,7 @@ def test_loads_operator_untouched_raw_stage_checkpoint(tmp_path: Path) -> None: @pytest.mark.parametrize( "column", - ("ED_VAL", "LKWEEKS", "PAW_TYP", "PERIDNUM", "source_year"), + ("A_LFSR", "ED_VAL", "LKWEEKS", "PAW_TYP", "PERIDNUM", "source_year"), ) def test_raw_loader_rejects_missing_input_complete_source_column( tmp_path: Path, @@ -278,6 +279,7 @@ def test_raw_loader_rejects_missing_input_complete_source_column( ("ED_VAL", [0.0, np.nan], "ED_VAL must be complete"), ("LKWEEKS", [-1, 53], "LKWEEKS must be complete"), ("PAW_TYP", [0, 4], "PAW_TYP must be complete integers"), + ("A_LFSR", [1, 6], "A_LFSR must be complete integers"), ), ) def test_raw_loader_rejects_invalid_input_complete_source_values( diff --git a/packages/microcosm-build/tests/test_us_bundle_core_contracts.py b/packages/microcosm-build/tests/test_us_bundle_core_contracts.py index 1ab09588..d8b677c6 100644 --- a/packages/microcosm-build/tests/test_us_bundle_core_contracts.py +++ b/packages/microcosm-build/tests/test_us_bundle_core_contracts.py @@ -129,7 +129,7 @@ def test_source_surface_classification_is_complete() -> None: } assert normative["stage_asset"] == { "id": "source_stages", - "sha256": "dc58a0d700f0add7b658cec774df6e9587303beb58a1f432a35a18dcd1ac4097", + "sha256": "788b815f748abb2061c41efb0cec4cc4952435dbcb4448ecf21ccac85a5ccec2", } assert operational["stage_asset"] == { "path": "microcosm.build.us/source_stages.json" @@ -180,15 +180,12 @@ def test_cd_vintage_crosswalk_source_and_geography_authority_are_pinned() -> Non crosswalk_source = next( row for row in build_sources()["sources"] - if row["id"] - == "us_congressional_district_vintage_crosswalk_117_to_119" + if row["id"] == "us_congressional_district_vintage_crosswalk_117_to_119" ) assert crosswalk_source == { "id": "us_congressional_district_vintage_crosswalk_117_to_119", "role": "congressional_district_vintage_crosswalk", - "sha256": ( - "c7cb040b1f57ca2ea2adcbfe60cc2b250ca23acbc4b640cd421e766fa54c1aec" - ), + "sha256": ("c7cb040b1f57ca2ea2adcbfe60cc2b250ca23acbc4b640cd421e766fa54c1aec"), "byte_size": 77_935, "loader": "kernel:load_congressional_district_vintage_crosswalk", "vintages": ["vintage:cd_117", "vintage:cd_119"], @@ -204,9 +201,7 @@ def test_cd_vintage_crosswalk_source_and_geography_authority_are_pinned() -> Non assignment = build_geography()["assignment"] assert assignment["order"] == "before_gap_fill" assert assignment["congressional_district_vintage_crosswalk"] == { - "source_ref": ( - "source:us_congressional_district_vintage_crosswalk_117_to_119" - ), + "source_ref": ("source:us_congressional_district_vintage_crosswalk_117_to_119"), "source_vintage": "vintage:cd_117", "target_vintage": "vintage:cd_119", } @@ -217,9 +212,9 @@ def test_cd_vintage_crosswalk_source_reference_is_resolved() -> None: resources = {**_vintage_resources(), "geography": build_geography()} _resolve_vintages(resources) - resources["geography"]["assignment"][ - "congressional_district_vintage_crosswalk" - ]["source_ref"] = "source:missing_crosswalk" + resources["geography"]["assignment"]["congressional_district_vintage_crosswalk"][ + "source_ref" + ] = "source:missing_crosswalk" with pytest.raises(SpecResolutionError, match="dangling source reference"): _resolve_vintages(resources) diff --git a/packages/microcosm-build/tests/test_us_education_assistance_source.py b/packages/microcosm-build/tests/test_us_education_assistance_source.py index f821ea2c..9394a797 100644 --- a/packages/microcosm-build/tests/test_us_education_assistance_source.py +++ b/packages/microcosm-build/tests/test_us_education_assistance_source.py @@ -1,4 +1,4 @@ -"""The pinned ASEC education-assistance sidecar: load, verify, and fill. +"""The pinned ASEC ED_VAL/A_LFSR sidecar: load, verify, and fill. The frozen census_cps inputs never carried raw ASEC ``ED_VAL`` (microcosm#417's sibling gap), so the education-inputs stage restores it from @@ -21,8 +21,11 @@ from microcosm.build.us_runtime.education_assistance_source import ( ASEC_EDUCATION_ASSISTANCE_ARCHIVES, ASEC_EDUCATION_ASSISTANCE_INCOME_YEARS, + ASEC_LABOR_FORCE_STATUS_COLUMN, + ASEC_LABOR_FORCE_STATUS_VALID_CODES, AsecEducationArchive, fill_asec_education_assistance_source, + fill_asec_labor_force_status_source, load_asec_education_assistance_sources, ) @@ -41,6 +44,7 @@ def _source_frame(rows: int = 6) -> pd.DataFrame: "A_LINENO": np.ones(rows, dtype=np.int64), "PERIDNUM": [_peridnum(index) for index in range(rows)], "ED_VAL": [0.0, 0.0, 2_500.0, 0.0, 12_000.0, 0.0][:rows], + "A_LFSR": [0, 1, 2, 3, 4, 7][:rows], "A_FNLWGT": np.full(rows, 100.0), } ) @@ -112,9 +116,29 @@ def test_loader_reads_pinned_zip_and_audits(pinned_archive) -> None: ) assert list(source["source_year"].unique()) == [_YEAR] assert len(source) == len(frame) + assert source[ASEC_LABOR_FORCE_STATUS_COLUMN].tolist() == [0, 1, 2, 3, 4, 7] + assert source[ASEC_LABOR_FORCE_STATUS_COLUMN].dtype == np.dtype("int64") assert source.attrs["source_audit"][_YEAR]["positive_rows"] == 2 +def test_loader_rejects_invalid_labor_force_status( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + frame = _source_frame() + frame.loc[0, ASEC_LABOR_FORCE_STATUS_COLUMN] = 6 + member = f"pppub{str(_YEAR + 1)[2:]}.csv" + path = tmp_path / "invalid-a-lfsr.zip" + _write_archive(path, frame, member) + monkeypatch.setitem( + ASEC_EDUCATION_ASSISTANCE_ARCHIVES, + _YEAR, + _pins_for(path, frame, member), + ) + + with pytest.raises(ValueError, match="A_LFSR must be a complete integer"): + load_asec_education_assistance_sources({_YEAR: path}, income_years=(_YEAR,)) + + def test_loader_rejects_wrong_zip_bytes(pinned_archive, tmp_path) -> None: path, frame = pinned_archive tampered = tmp_path / "tampered.zip" @@ -155,6 +179,21 @@ def test_fill_joins_ed_val_by_identity(pinned_archive) -> None: assert row["ED_VAL"] == expected[row["PERIDNUM"]] +def test_fill_joins_labor_force_status_by_identity(pinned_archive) -> None: + path, frame = pinned_archive + source = load_asec_education_assistance_sources( + {_YEAR: path}, income_years=(_YEAR,) + ) + filled = fill_asec_labor_force_status_source(_person_frame(), source) + expected = frame.set_index("PERIDNUM")[ASEC_LABOR_FORCE_STATUS_COLUMN] + for _, row in filled.iterrows(): + assert row[ASEC_LABOR_FORCE_STATUS_COLUMN] == expected[row["PERIDNUM"]] + assert filled[ASEC_LABOR_FORCE_STATUS_COLUMN].dtype == np.dtype("int64") + assert set(filled[ASEC_LABOR_FORCE_STATUS_COLUMN]) <= ( + ASEC_LABOR_FORCE_STATUS_VALID_CODES + ) + + def test_fill_fails_closed_on_uncovered_key(pinned_archive) -> None: path, _ = pinned_archive source = load_asec_education_assistance_sources( @@ -199,6 +238,19 @@ def test_fill_refuses_to_overwrite_measured_values(pinned_archive) -> None: fill_asec_education_assistance_source(person, source) +def test_labor_force_fill_refuses_to_overwrite_measured_values( + pinned_archive, +) -> None: + path, _ = pinned_archive + source = load_asec_education_assistance_sources( + {_YEAR: path}, income_years=(_YEAR,) + ) + person = _person_frame() + person[ASEC_LABOR_FORCE_STATUS_COLUMN] = 7 + with pytest.raises(ValueError, match="refusing to overwrite"): + fill_asec_labor_force_status_source(person, source) + + def test_pins_reject_audit_drift(pinned_archive, monkeypatch) -> None: path, frame = pinned_archive pins = ASEC_EDUCATION_ASSISTANCE_ARCHIVES[_YEAR] diff --git a/packages/microcosm-build/tests/test_us_fiscal_refresh_builder.py b/packages/microcosm-build/tests/test_us_fiscal_refresh_builder.py index 0794037d..7e9f24f5 100644 --- a/packages/microcosm-build/tests/test_us_fiscal_refresh_builder.py +++ b/packages/microcosm-build/tests/test_us_fiscal_refresh_builder.py @@ -508,10 +508,10 @@ def test__given_target_frame_checkpoint__then_builder_round_trips_frame( ssi_take_up_assignment_sha256="ssi-flags-sha", selection_identities_sha256=None, ) - # 11 = target checkpoints preserve nullable booleans explicitly; schema 2 - # distinguishes the new values+mask codec from schema-1 checkpoints. + # 12 = target checkpoints include source-aware humanitarian-stock columns; + # schema 2 distinguishes the values+mask codec from schema-1 checkpoints. assert identity["schema_version"] == 2 - assert identity["materializer_version"] == 11 + assert identity["materializer_version"] == 12 # The SSI prior-weight basis is identity-bearing (microcosm#543 instance # 2): unflagged runs carry the key as None. assert identity["ssi_take_up_prior_weight_basis_sha256"] is None @@ -723,9 +723,9 @@ def test__given_stale_materializer_version_checkpoint__then_builder_rejects_it( ) -> None: """A checkpoint stored under a superseded materializer version must not load. - Version 11 adds the lossless nullable-boolean checkpoint codec. The version - constant participates in the identity comparison; this pins stored-10 - versus current-11 rejection directly. + Version 12 adds source-aware humanitarian-stock target columns. The version + constant participates in the identity comparison; this pins stored-11 + versus current-12 rejection directly. """ builder = _load_builder_module() monkeypatch.setattr(builder, "US_SCHEMA", small_frame.schema) @@ -759,10 +759,10 @@ def test__given_stale_materializer_version_checkpoint__then_builder_rejects_it( ssi_take_up_assignment_sha256="ssi-flags-sha", selection_identities_sha256=None, ) - # 10 = the pre-nullable-boolean-codec world; 9 = the still-older pre-#557 - # release-refit world. Both must miss against expected version 11. - stale_identity = {**dict(identity), "materializer_version": 10} - older_identity = {**dict(identity), "materializer_version": 9} + # 11 = the pre-humanitarian-target world; 10 = the still-older + # pre-nullable-boolean-codec world. Both must miss against version 12. + stale_identity = {**dict(identity), "materializer_version": 11} + older_identity = {**dict(identity), "materializer_version": 10} path = tmp_path / "target_frame_checkpoint.h5" builder._write_target_frame_checkpoint( path, @@ -1743,20 +1743,16 @@ def selected(path, *, allow_terminal_gate_failure): ) return - loaded_frame, receipt, loaded_identity = ( - builder._load_base_pool_if_identified( - pool_h5, - allow_gate_failed_base_pool=allow_gate_failed, - ) + loaded_frame, receipt, loaded_identity = builder._load_base_pool_if_identified( + pool_h5, + allow_gate_failed_base_pool=allow_gate_failed, ) assert loaded_frame is frame assert loaded_identity is authenticated assert receipt["status"] == status assert receipt["allow_gate_failed_base_pool"] is allow_gate_failed - assert receipt["agreement_gate_reference"]["failure_count"] == len( - gate_failures - ) + assert receipt["agreement_gate_reference"]["failure_count"] == len(gate_failures) def test_builder_refuses_actual_red_base_h5_pool_sidecar_without_opt_in( @@ -1777,7 +1773,9 @@ def test_builder_refuses_actual_red_base_h5_pool_sidecar_without_opt_in( ) out = tmp_path / "out" monkeypatch.setattr(builder, "_git_dirty", lambda: False) - monkeypatch.setattr(builder, "_refuse_certified_release_dir_reuse", lambda path: None) + monkeypatch.setattr( + builder, "_refuse_certified_release_dir_reuse", lambda path: None + ) monkeypatch.setattr( builder, "_load_frame", @@ -1826,7 +1824,9 @@ def test_builder_refuses_bare_stamped_pool_h5_before_generic_load( ) out = tmp_path / "out" monkeypatch.setattr(builder, "_git_dirty", lambda: False) - monkeypatch.setattr(builder, "_refuse_certified_release_dir_reuse", lambda path: None) + monkeypatch.setattr( + builder, "_refuse_certified_release_dir_reuse", lambda path: None + ) monkeypatch.setattr( builder, "_load_frame", @@ -4203,7 +4203,15 @@ def test_release_calibration_diagnostics_writes_nan_final_loss_as_null( @pytest.mark.parametrize( "terminal_mode", - ["merge", "integrity", "retirement", "crash", "telemetry", "puf_tail"], + [ + "merge", + "integrity", + "retirement", + "crash", + "telemetry", + "puf_tail", + "immigration_drift", + ], ) def test_main_writes_diagnostics_before_post_calibration_gate_failure( monkeypatch, tmp_path, terminal_mode @@ -4231,6 +4239,10 @@ def test_main_writes_diagnostics_before_post_calibration_gate_failure( ``puf_tail``: exact-k selection loses the original PUF capital-gains tail; the failure is batched while diagnostics and final-weight evidence remain, every later terminal group runs, and release artifacts stay suppressed. + ``immigration_drift``: the source-frame immigration composition passes, then + calibrated weights make the same selected status mix fail on the final + export-frame recheck; that final verdict reaches diagnostics and aborts + publication. """ builder = _load_builder_module() release_id = ( @@ -4285,9 +4297,23 @@ def n(self, entity): return 2 if terminal_mode == "puf_tail" else 4 def table(self, entity): - assert entity == "household" size = self.n("household") - return pd.DataFrame({"household_id": np.arange(1, size + 1, dtype="int64")}) + if entity == "household": + return pd.DataFrame( + {"household_id": np.arange(1, size + 1, dtype="int64")} + ) + assert entity == "person" + return pd.DataFrame( + { + "household_id": np.arange(1, size + 1, dtype="int64"), + "immigration_status_str": [ + "REFUGEE", + "LEGAL_PERMANENT_RESIDENT", + "ASYLEE", + "LEGAL_PERMANENT_RESIDENT", + ][:size], + } + ) def weights_for(self, entity): assert entity == "household" @@ -4309,8 +4335,20 @@ def weights_for(self, entity): ) def table(self, entity): - assert entity == "household" - return pd.DataFrame({"household_id": np.asarray([10, 20], dtype="int64")}) + if entity == "household": + return pd.DataFrame( + {"household_id": np.asarray([10, 20], dtype="int64")} + ) + assert entity == "person" + return pd.DataFrame( + { + "household_id": np.asarray([10, 20], dtype="int64"), + "immigration_status_str": [ + "REFUGEE", + "LEGAL_PERMANENT_RESIDENT", + ], + } + ) if terminal_mode == "puf_tail": loss_basis = builder._fiscal_target_loss_basis(registry, np.ones(1)) @@ -4458,7 +4496,13 @@ def complete(self): "_staging_telemetry", lambda *args, **kwargs: live_telemetry, ) - if terminal_mode in {"integrity", "retirement", "telemetry", "puf_tail"}: + if terminal_mode in { + "integrity", + "retirement", + "telemetry", + "puf_tail", + "immigration_drift", + }: monkeypatch.setattr( builder, "PolicyEngineUSEngine", @@ -4919,14 +4963,50 @@ def fake_energy_subsidy_signal_gate(frame): "with_us_immigration_inputs", lambda frame, *, seed, time_period: frame, ) + + def fake_immigration_composition_gate(frame): + statuses = frame.table("person")["immigration_status_str"].astype(str) + household_weights = frame.weights_for("household") + weights = np.asarray(household_weights.values, dtype=np.float64) + assert len(statuses) == len(weights) + humanitarian = statuses.isin( + {"REFUGEE", "ASYLEE", "DEPORTATION_WITHHELD", "CUBAN_HAITIAN_ENTRANT"} + ).to_numpy(dtype=bool) + humanitarian_share = float(np.average(humanitarian, weights=weights)) + calls = captured.setdefault("immigration_gate_calls", []) + is_final_drift = ( + terminal_mode == "immigration_drift" + and household_weights.kind == WeightKind.CALIBRATED + ) + calls.append( + { + "weight_kind": household_weights.kind.value, + "weights": weights.tolist(), + "statuses": statuses.tolist(), + "humanitarian_share": humanitarian_share, + "passed": not is_final_drift, + } + ) + failure = ( + "weighted humanitarian status share drifted from 0.500000 to " + f"{humanitarian_share:.6f} after calibration " + "[final-immigration-sentinel]" + ) + return builder.GateResult( + name="immigration_composition", + passed=not is_final_drift, + failures=(failure,) if is_final_drift else (), + details={ + "checked": True, + "weight_kind": household_weights.kind.value, + "humanitarian_share": humanitarian_share, + }, + ) + monkeypatch.setattr( builder, "us_immigration_composition_gate", - lambda frame: builder.GateResult( - name="immigration_composition", - passed=True, - details={"checked": True}, - ), + fake_immigration_composition_gate, ) monkeypatch.setattr( builder, @@ -5678,6 +5758,12 @@ def fake_other_health_insurance_signal_gate(frame): passed=True, details={"checked": True}, ) + if terminal_mode == "immigration_drift": + return builder.GateResult( + name="other_health_insurance_premiums_signal", + passed=True, + details={"checked": True}, + ) # Export-frame call fails deliberately: the microcosm#547 cofailure # regression proves a failing post-solve signal gate batches # alongside the SSI delivery failure instead of masking it with an @@ -5867,7 +5953,12 @@ def fake_ssi_delivery_gate(diagnostics, *, targets, enforcement_fences=None): # own early failure. Other modes retain the microcosm#547 delivery # cofailure and its written retry basis. captured.setdefault("ssi_event_order", []).append("delivery_gate") - passes = terminal_mode in {"integrity", "retirement", "puf_tail"} + passes = terminal_mode in { + "integrity", + "retirement", + "puf_tail", + "immigration_drift", + } return builder.GateResult( name="ssi_take_up_delivery", passed=passes, @@ -5905,6 +5996,14 @@ def recording_ssi_write(diagnostics, path): def fake_release_gate_failures(*args, **kwargs): if terminal_mode == "crash": raise RuntimeError("release-gate evaluation exploded [crash-sentinel]") + if terminal_mode == "immigration_drift": + final_immigration_gate = args[6] + assert final_immigration_gate.name == "immigration_composition" + assert not final_immigration_gate.passed + return [ + f"Immigration composition failed: {failure}" + for failure in final_immigration_gate.failures + ] if terminal_mode == "retirement": # The degraded pre-solve contract (PR #557 round 3): the failing # degenerate gate is NOT raised early — the gate object itself @@ -5973,16 +6072,23 @@ def fake_release_gate_failures(*args, **kwargs): "Bernoulli-law violation [final-integrity-sentinel]" ) assert "SSI take-up delivery failed:" not in message + elif terminal_mode == "immigration_drift": + assert message == ( + "Release gates failed: Immigration composition failed: weighted " + "humanitarian status share drifted from 0.500000 to 0.255319 " + "after calibration [final-immigration-sentinel]" + ) else: assert message.startswith( "Release gates failed: SSI take-up delivery failed: " "18_64 delivered over envelope [cofailure-sentinel]" ) - assert ( - "Other health insurance signal failed on the export frame: " - "premiums signal flattened [cofailure-sentinel]" in message - ) - if terminal_mode != "crash": + if terminal_mode != "immigration_drift": + assert ( + "Other health insurance signal failed on the export frame: " + "premiums signal flattened [cofailure-sentinel]" in message + ) + if terminal_mode not in {"crash", "immigration_drift"}: assert "ctc failed" in message if terminal_mode == "telemetry": assert ( @@ -5990,13 +6096,33 @@ def fake_release_gate_failures(*args, **kwargs): "attach_artifact('calibration_diagnostics') crashed" in message ) assert "telemetry-crash-sentinel" in message - else: + elif terminal_mode == "crash": assert "health-input exploded [crash-sentinel]" in message assert "release-gate evaluation exploded [crash-sentinel]" in message assert "ctc failed" not in message else: # pragma: no cover - defensive assertion raise AssertionError("Expected post-calibration gate failure.") + if terminal_mode == "immigration_drift": + immigration_calls = captured["immigration_gate_calls"] + assert len(immigration_calls) == 2 + source_call, final_call = immigration_calls + assert source_call["weight_kind"] == "importance" + assert source_call["weights"] == [1.0, 1.0, 1.0, 1.0] + assert source_call["statuses"] == [ + "REFUGEE", + "LEGAL_PERMANENT_RESIDENT", + "ASYLEE", + "LEGAL_PERMANENT_RESIDENT", + ] + assert source_call["humanitarian_share"] == pytest.approx(0.5) + assert source_call["passed"] is True + assert final_call["weight_kind"] == "calibrated" + assert final_call["weights"] == [12.0, 35.0] + assert final_call["statuses"] == ["REFUGEE", "LEGAL_PERMANENT_RESIDENT"] + assert final_call["humanitarian_share"] == pytest.approx(12.0 / 47.0) + assert final_call["passed"] is False + release_dir = out / "releases" / release_id written_diagnostics = json.loads( (release_dir / "calibration_diagnostics.json").read_text() @@ -6019,6 +6145,12 @@ def fake_release_gate_failures(*args, **kwargs): "Bernoulli-law violation [final-integrity-sentinel]" in written_diagnostics["build"]["release_gates"]["failures"] ) + elif terminal_mode == "immigration_drift": + assert written_diagnostics["build"]["release_gates"]["failures"] == [ + "Immigration composition failed: weighted humanitarian status share " + "drifted from 0.500000 to 0.255319 after calibration " + "[final-immigration-sentinel]" + ] # The SSI retry-basis artifact is written even though the run fails # terminally — it IS the remedy input for the next attempt. assert (release_dir / "us_ssi_take_up.json").exists() @@ -6043,6 +6175,15 @@ def fake_release_gate_failures(*args, **kwargs): cache_context = captured["materialize_kwargs"][ "target_materialization_cache_context" ] + active_specs = captured["ssi_band_targets_specs"] + assert any( + spec.metadata.get("target_role") == "humanitarian_immigration_stock" + for spec in active_specs + ) + assert ( + cache_context["target_registry_version"] + == TargetRegistry(active_specs, country="us").version + ) expected_evidence_identity = builder._target_frame_checkpoint_identity( base_dataset_sha256=cache_context["base_dataset_sha256"], policyengine_us_version=cache_context["policyengine_us_version"], @@ -6104,7 +6245,13 @@ def fake_release_gate_failures(*args, **kwargs): "source_sha256": builder.ASEC_2023_WEEKS_UNEMPLOYED_SOURCE_SHA256, "source_rows": 2, } - if terminal_mode in {"integrity", "retirement", "telemetry", "puf_tail"}: + if terminal_mode in { + "integrity", + "retirement", + "telemetry", + "puf_tail", + "immigration_drift", + }: assert captured["terminal_gate_events"] == [ "input_coverage", "input_mass_parity", @@ -6145,6 +6292,12 @@ def fake_release_gate_failures(*args, **kwargs): "premiums signal flattened [cofailure-sentinel]", "ctc failed", ] + elif terminal_mode == "immigration_drift": + expected_gate_failures = [ + "Immigration composition failed: weighted humanitarian status share " + "drifted from 0.500000 to 0.255319 after calibration " + "[final-immigration-sentinel]" + ] else: # The retry line carries the written artifact's sha256 — the # required --ssi-take-up-prior-weight-basis-sha256 pin, handed out @@ -6387,7 +6540,7 @@ def fake_release_gate_failures(*args, **kwargs): "integrity_gate", # persisted-flag recheck on the export frame "delivery_gate", # enforced-band delivery, after the artifact exists ] - if terminal_mode not in {"integrity", "retirement"}: + if terminal_mode not in {"integrity", "retirement", "immigration_drift"}: # A delivery miss rewrites the final measurement as the retry basis. expected_ssi_event_order.append("write:us_ssi_take_up.json") assert captured["ssi_event_order"] == expected_ssi_event_order @@ -8828,6 +8981,236 @@ def _invalidate_all_caches(self): assert compilation["dropped_target_names"] == [] +def test_humanitarian_stock_specs_follow_positive_manifest_draws( + monkeypatch, +) -> None: + builder = _load_builder_module() + draws = ( + SimpleNamespace( + label="paroled_one_year:afghanistan", + category="paroled_one_year", + origin="afghanistan", + status="PAROLED_ONE_YEAR", + target=75_000.0, + source="https://example.test/parole", + ), + SimpleNamespace( + label="refugee", + category="refugee", + origin=None, + status="REFUGEE", + target=160_000.0, + source="https://example.test/refugee", + ), + SimpleNamespace( + label="deportation_withheld", + category="deportation_withheld", + origin=None, + status="DEPORTATION_WITHHELD", + target=0.0, + source="https://example.test/withheld", + ), + SimpleNamespace( + label="tps:venezuela", + category="tps", + origin="venezuela", + status="TPS", + target=344_335.0, + source="https://example.test/tps", + ), + ) + monkeypatch.setattr( + builder, + "us_immigration_controls", + lambda: SimpleNamespace(humanitarian=draws), + ) + + specs = builder._humanitarian_immigration_stock_specs(time_period=2024) + + assert [spec.name for spec in specs] == [ + "humanitarian_immigration_stock.paroled_one_year.afghanistan", + "humanitarian_immigration_stock.refugee", + "humanitarian_immigration_stock.tps.venezuela", + ] + assert [spec.value for spec in specs] == [75_000.0, 160_000.0, 344_335.0] + assert [spec.source for spec in specs] == [ + "https://example.test/parole", + "https://example.test/refugee", + "https://example.test/tps", + ] + assert all(spec.period == 2024 for spec in specs) + assert all( + spec.metadata["materializer"] == "humanitarian_immigration_stock" + for spec in specs + ) + assert specs[0].metadata["humanitarian_origin"] == "afghanistan" + assert "humanitarian_origin" not in specs[1].metadata + # The active-registry content hash therefore binds the manifest controls; + # removing even one draw produces a different checkpoint registry id. + assert ( + TargetRegistry(specs, country="us").version + != TargetRegistry(specs[:-1], country="us").version + ) + + +def test_humanitarian_stock_materializer_collapses_source_aware_person_masks( + monkeypatch, + small_frame, +) -> None: + from microcosm.calibrate import build_constraint_matrix + + builder = _load_builder_module() + draws = ( + SimpleNamespace( + label="paroled_one_year:afghanistan", + category="paroled_one_year", + origin="afghanistan", + status="PAROLED_ONE_YEAR", + target=1.0, + source="fixture", + ), + SimpleNamespace( + label="refugee", + category="refugee", + origin=None, + status="REFUGEE", + target=2.0, + source="fixture", + ), + ) + controls = SimpleNamespace(humanitarian=draws) + monkeypatch.setattr(builder, "us_immigration_controls", lambda: controls) + observed: list[tuple[str, int]] = [] + + def draw_mask(frame, draw, *, time_period): + assert frame is small_frame + observed.append((draw.label, time_period)) + return { + "paroled_one_year:afghanistan": np.asarray([True, False, False, False]), + "refugee": np.asarray([False, True, True, False]), + }[draw.label] + + monkeypatch.setattr(builder, "us_immigration_humanitarian_draw_mask", draw_mask) + specs = builder._humanitarian_immigration_stock_specs(time_period=2024) + household = small_frame.table("household").copy() + + builder._materialize_humanitarian_immigration_stock_targets( + frame=small_frame, + household=household, + target_specs=specs, + time_period=2024, + ) + + np.testing.assert_array_equal( + household[ + "humanitarian_immigration_stock.paroled_one_year.afghanistan" + ].to_numpy(), + np.asarray([1.0, 0.0]), + ) + np.testing.assert_array_equal( + household["humanitarian_immigration_stock.refugee"].to_numpy(), + np.asarray([1.0, 1.0]), + ) + assert observed == [ + ("paroled_one_year:afghanistan", 2024), + ("refugee", 2024), + ] + + target_frame = Frame( + { + "person": small_frame.table("person"), + "household": household, + }, + small_frame.schema, + {"household": small_frame.weights_for("household")}, + small_frame.strata, + ) + registry, compilation = builder._compile_materialized_target_registry( + target_frame, + specs, + ) + target_set = registry.to_target_set() + + assert compilation["dropped_target_names"] == [] + assert tuple(target.row_name for target in target_set) == ( + "humanitarian_immigration_stock.paroled_one_year.afghanistan@2024", + "humanitarian_immigration_stock.refugee@2024", + ) + problem = build_constraint_matrix(target_frame, target_set) + assert problem.names == tuple(target.row_name for target in target_set) + assert problem.skipped == () + np.testing.assert_array_equal( + problem.matrix.toarray(), + np.asarray([[1.0, 0.0], [1.0, 1.0]]), + ) + np.testing.assert_array_equal( + problem.target_vector, + np.asarray([1.0, 2.0]), + ) + + +def test_dense_and_l0_solvers_consume_compiled_registry_target_set() -> None: + import ast + + builder = _load_builder_module() + tree = ast.parse(Path(builder.__file__).read_text()) + main_fn = next( + node + for node in ast.walk(tree) + if isinstance(node, ast.FunctionDef) and node.name == "_main" + ) + + for solver_name in ("calibrate", "calibrate_l0_refit"): + calls = [ + node + for node in ast.walk(main_fn) + if isinstance(node, ast.Call) + and isinstance(node.func, ast.Name) + and node.func.id == solver_name + ] + assert len(calls) == 1 + assert ast.unparse(calls[0].args[0]) == "target_frame" + assert ast.unparse(calls[0].args[1]) == "registry.to_target_set()" + + +def test_main_rechecks_immigration_composition_on_export_for_terminal_gates() -> None: + import ast + + builder = _load_builder_module() + tree = ast.parse(Path(builder.__file__).read_text()) + main_fn = next( + node + for node in ast.walk(tree) + if isinstance(node, ast.FunctionDef) and node.name == "_main" + ) + final_assignments = [ + node + for node in ast.walk(main_fn) + if isinstance(node, ast.Assign) + and any( + isinstance(target, ast.Name) and target.id == "final_immigration_gate" + for target in node.targets + ) + ] + assert len(final_assignments) == 1 + final_assignment = final_assignments[0] + assert isinstance(final_assignment.value, ast.Call) + assert getattr(final_assignment.value.func, "id", None) == ( + "us_immigration_composition_gate" + ) + assert ast.unparse(final_assignment.value.args[0]) == "export_frame" + + terminal_calls = [ + node + for node in ast.walk(main_fn) + if isinstance(node, ast.Call) + and getattr(node.func, "id", None) == "_release_gate_failures" + and node.lineno > final_assignment.lineno + ] + assert len(terminal_calls) == 1 + assert ast.unparse(terminal_calls[0].args[6]) == "final_immigration_gate" + + def test_unknown_ledger_filter_metadata_fails_closed() -> None: builder = _load_builder_module() target = TargetSpec( @@ -9160,9 +9543,7 @@ def test_exact_k_receipt_stays_strict_even_when_base_h5_opt_in_is_present() -> N builder = _load_builder_module() with pytest.raises(RuntimeError, match="lost its passing agreement gate"): - builder._exact_k_ladder_manifest_payload( - **_gate_failed_exact_k_inputs(builder) - ) + builder._exact_k_ladder_manifest_payload(**_gate_failed_exact_k_inputs(builder)) def _gate_failed_base_pool_receipt() -> dict[str, object]: diff --git a/packages/microcosm-build/tests/test_us_hr1_immigration_channels.py b/packages/microcosm-build/tests/test_us_hr1_immigration_channels.py new file mode 100644 index 00000000..ef3c2dfa --- /dev/null +++ b/packages/microcosm-build/tests/test_us_hr1_immigration_channels.py @@ -0,0 +1,172 @@ +"""The four H.R.1 immigration probes bind through the engine (microcosm #767). + +The release-blocking reform-coverage smoke gate scores the pinned probe +reforms on the full export; these tests prove the same channels move, +nonzero and sign-correct, on synthetic households whose +``immigration_status_str`` carries the humanitarian values the source stage +now imputes. A probe whose channel dies here would score $0 on every +release, so this is the PR-CI early warning for the engine side of the +silent-zero failure — without needing restricted microdata. + +Each situation pins the take-up flag its program reads (the same +data-seeded surface the release carries) and, for the ACA channels, an +``slcsp`` override so no rating-area geography is required. +""" + +from __future__ import annotations + +import pytest + +pytest.importorskip("policyengine_us") + +from policyengine_us import Simulation # noqa: E402 + +from microcosm.build.us_runtime.reform_coverage_smoke import ( # noqa: E402 + _build_reform, +) +from microcosm.build.us_runtime.release_input_coverage import ( # noqa: E402 + us_release_reform_coverage_probes, +) + +_HR1_PROBE_IDS = ( + "hr1_medicaid_humanitarian_eligibility_restoration", + "hr1_aca_below_fpl_exception_restoration", + "hr1_aca_lawful_presence_restoration", + "hr1_snap_humanitarian_eligibility_restoration", +) + + +@pytest.fixture(scope="module") +def hr1_probes(): + probes = { + probe.id: probe + for probe in us_release_reform_coverage_probes() + if probe.id.startswith("hr1_") + } + assert set(probes) == set(_HR1_PROBE_IDS) + return probes + + +def _household( + status: str, + income: float, + year: int, + *, + person: dict | None = None, + tax_unit: dict | None = None, + spm_unit: dict | None = None, +) -> dict: + people = { + "parent": { + "age": {year: 35}, + "employment_income": {year: income}, + "immigration_status_str": {year: status}, + **(person or {}), + }, + "child": { + "age": {year: 8}, + "immigration_status_str": {year: status}, + **(person or {}), + }, + } + return { + "people": people, + "families": {"f": {"members": ["parent", "child"]}}, + "marital_units": {"m": {"members": ["parent"]}}, + "tax_units": {"t": {"members": ["parent", "child"], **(tax_unit or {})}}, + "spm_units": {"s": {"members": ["parent", "child"], **(spm_unit or {})}}, + "households": { + "h": {"members": ["parent", "child"], "state_name": {year: "TX"}} + }, + } + + +def _effect(probe, situation: dict, year: int) -> tuple[float, float]: + baseline = Simulation(situation=situation) + reformed = Simulation(situation=situation, reform=_build_reform(probe)) + base = float(baseline.calculate(probe.budget_measure, year).sum()) + reform = float(reformed.calculate(probe.budget_measure, year).sum()) + return base, reform + + +def test_medicaid_restoration_reenrolls_refugee_family(hr1_probes) -> None: + # At 2027 law refugees are outside the §71109-narrowed qualified list; + # restoring the pre-H.R.1 list must re-enroll the anchored taker. + probe = hr1_probes["hr1_medicaid_humanitarian_eligibility_restoration"] + situation = _household( + "REFUGEE", + 25_000.0, + 2027, + person={"takes_up_medicaid_if_eligible": {2027: True}}, + ) + base, reform = _effect(probe, situation, 2027) + assert base == 0.0 + assert reform > 0.0 + + +def test_below_fpl_exception_restoration_pays_ptc_to_tps_unit(hr1_probes) -> None: + # A below-poverty TPS unit is lawfully present for the ACA at 2026 law + # yet Medicaid-ineligible by status — exactly the §71302 population. + probe = hr1_probes["hr1_aca_below_fpl_exception_restoration"] + situation = _household( + "TPS", + 12_000.0, + 2026, + tax_unit={ + "takes_up_aca_if_eligible": {2026: True}, + "slcsp": {2026: 8_000.0}, + }, + ) + base, reform = _effect(probe, situation, 2026) + assert base == 0.0 + assert reform > 0.0 + + +def test_lawful_presence_restoration_requalifies_asylee_unit(hr1_probes) -> None: + # At 2027 law §71301 adds asylees to the ACA ineligible list; restoring + # the pre-H.R.1 list (DACA/UNDOCUMENTED only) must re-qualify the unit. + probe = hr1_probes["hr1_aca_lawful_presence_restoration"] + situation = _household( + "ASYLEE", + 35_000.0, + 2027, + tax_unit={ + "takes_up_aca_if_eligible": {2027: True}, + "slcsp": {2027: 8_000.0}, + }, + ) + base, reform = _effect(probe, situation, 2027) + assert base == 0.0 + assert reform > 0.0 + + +def test_snap_restoration_reincludes_refugee_members(hr1_probes) -> None: + # At 2026 law §10108 makes refugees excluded SNAP members; restoring the + # pre-H.R.1 list must raise the household allotment. + probe = hr1_probes["hr1_snap_humanitarian_eligibility_restoration"] + situation = _household( + "REFUGEE", + 20_000.0, + 2026, + spm_unit={"takes_up_snap_if_eligible": {2026: True}}, + ) + base, reform = _effect(probe, situation, 2026) + assert reform > base + + +def test_probe_magnitude_floors_sit_far_below_plausible_scale(hr1_probes) -> None: + # The floors are release-scale guards: far above numerical noise, far + # below the plausible aggregate effect of the imputed stocks (about + # 1.6M weighted persons across the humanitarian categories). + floors = { + "hr1_medicaid_humanitarian_eligibility_restoration": 50_000_000.0, + "hr1_aca_below_fpl_exception_restoration": 5_000_000.0, + "hr1_aca_lawful_presence_restoration": 10_000_000.0, + "hr1_snap_humanitarian_eligibility_restoration": 25_000_000.0, + } + for probe_id, floor in floors.items(): + probe = hr1_probes[probe_id] + assert probe.min_abs_effect == floor + assert probe.expected_sign == "positive" + assert probe.binding_inputs == ("immigration_status_str",) + assert probe.issue == "PolicyEngine/microcosm#767" diff --git a/packages/microcosm-build/tests/test_us_immigration.py b/packages/microcosm-build/tests/test_us_immigration.py index 23a1cb19..259f1dda 100644 --- a/packages/microcosm-build/tests/test_us_immigration.py +++ b/packages/microcosm-build/tests/test_us_immigration.py @@ -6,6 +6,7 @@ import pandas as pd import pytest +import microcosm.build.us_runtime.immigration as immigration_module from microcosm.build.source_manifest import SourceStageSpec from microcosm.build.source_runtime import ( SourceRuntimeConfig, @@ -13,6 +14,7 @@ run_source_stage, ) from microcosm.build.us_runtime import ( + HUMANITARIAN_STATUS_CATEGORIES, IMMIGRATION_STATUS_VALUES, SSN_CARD_TYPE_VALUES, US_DONORS, @@ -20,6 +22,8 @@ US_IMMIGRATION_STAGE_NAME, US_SOURCE_MANIFEST, US_STAGE_NAMES, + HumanitarianDraw, + ImmigrationControls, UndocumentedControls, derive_us_immigration_status_from_manifest, us_immigration_composition_gate, @@ -27,18 +31,66 @@ us_immigration_stage_spec, with_us_immigration_inputs, ) +from microcosm.build.us_runtime.immigration import ( + us_immigration_controls, + us_immigration_humanitarian_draw_mask, +) from microcosm.frame import US_SCHEMA, Frame, WeightKind, Weights _HANDLERS = {"derive_immigration_status": derive_us_immigration_status_from_manifest} TIME_PERIOD = 2024 +_PAROLE_ORIGIN_KEYS = ("afghanistan", "ukraine", "nicaragua", "venezuela") +_TPS_ORIGIN_KEYS = ( + "venezuela", + "el_salvador", + "honduras", + "nicaragua", + "nepal", + "other_designated", +) + + +def _humanitarian_block( + *, + paroled_one_year: dict[str, float] | None = None, + refugee: float = 0.0, + asylee: float = 0.0, + deportation_withheld: float = 0.0, + tps: dict[str, float] | None = None, +) -> dict: + """Manifest-shaped humanitarian block; unnamed targets default to zero.""" + + parole_targets = dict.fromkeys(_PAROLE_ORIGIN_KEYS, 0.0) | (paroled_one_year or {}) + tps_targets = dict.fromkeys(_TPS_ORIGIN_KEYS, 0.0) | (tps or {}) + return { + "paroled_one_year": { + origin: { + "target": target, + "source": f"https://example.com/parole/{origin}", + } + for origin, target in parole_targets.items() + }, + "refugee": {"target": refugee, "source": "https://example.com/refugee"}, + "asylee": {"target": asylee, "source": "https://example.com/asylee"}, + "deportation_withheld": { + "target": deportation_withheld, + "source": "https://example.com/withheld", + }, + "tps": { + origin: {"target": target, "source": f"https://example.com/tps/{origin}"} + for origin, target in tps_targets.items() + }, + } + def _stage_spec( *, workers: float, students: float, anchor: float, + humanitarian: dict | None = None, ) -> SourceStageSpec: return SourceStageSpec.from_mapping( { @@ -64,6 +116,11 @@ def _stage_spec( "value": anchor, "source": "https://example.com/population", }, + "humanitarian_status_stocks": ( + humanitarian + if humanitarian is not None + else _humanitarian_block() + ), }, ], "outputs": list(US_IMMIGRATION_OUTPUT_COLUMNS), @@ -82,6 +139,7 @@ def _person_table(rows: list[dict]) -> pd.DataFrame: "A_MARITL": 7, "A_SPOUSE": 0, "A_HSCOL": 0, + "A_LFSR": 7, "WSAL_VAL": 0.0, "SEMP_VAL": 0.0, "MCARE": 2, @@ -126,10 +184,16 @@ def _run( workers: float = 100.0, students: float = 100.0, anchor: float = 10.0, + humanitarian: dict | None = None, seed: int = 0, ) -> pd.DataFrame: return run_source_stage( - _stage_spec(workers=workers, students=students, anchor=anchor), + _stage_spec( + workers=workers, + students=students, + anchor=anchor, + humanitarian=humanitarian, + ), tables={"person": person}, operation_handlers=_HANDLERS, config=SourceRuntimeConfig(seed=seed, target_year=TIME_PERIOD), @@ -180,13 +244,15 @@ def test_noncitizen_without_indicators_is_undocumented(self) -> None: assert output.loc[0, "ssn_card_type"] == "NONE" assert output.loc[0, "immigration_status_str"] == "UNDOCUMENTED" - def test_worker_spill_leaves_undocumented_workers_at_control(self) -> None: + def test_worker_spill_leaves_pew_unauthorized_labor_force_at_control( + self, + ) -> None: person = _person_table( - [_noncitizen(WSAL_VAL=10_000.0) for _ in range(10)] + [_noncitizen(A_LFSR=1, WSAL_VAL=10_000.0) for _ in range(10)] + [_noncitizen(), _noncitizen()] ) output = _run(person, workers=4.0) - workers = output["WSAL_VAL"] > 0 + workers = output["A_LFSR"].isin([1, 2, 3, 4]) undocumented_workers = ((output["ssn_card_type"] == "NONE") & workers).sum() ead_workers = ( (output["ssn_card_type"] == "NON_CITIZEN_VALID_EAD") & workers @@ -196,8 +262,89 @@ def test_worker_spill_leaves_undocumented_workers_at_control(self) -> None: # Non-workers are untouched by the worker spill. assert (output.loc[~workers, "ssn_card_type"] == "NONE").all() + def test_worker_spill_uses_labor_force_status_not_prior_year_earnings( + self, + ) -> None: + person = _person_table( + [ + # Looking for work now, despite having no prior-year earnings. + _noncitizen(A_LFSR=3), + # Earned wages last year, but is now outside the labor force. + _noncitizen(A_LFSR=7, WSAL_VAL=10_000.0), + ] + ) + output = _run(person, workers=0.001) + assert output.loc[0, "ssn_card_type"] == "NON_CITIZEN_VALID_EAD" + assert output.loc[1, "ssn_card_type"] == "NONE" + + def test_worker_control_retains_residual_cuban_haitian_ead_rows(self) -> None: + person = _person_table( + [ + _noncitizen(A_LFSR=1, PENATVTY=327, PEINUSYR=24), + _noncitizen(A_LFSR=1, PENATVTY=332, PEINUSYR=24), + *[_noncitizen(A_LFSR=1) for _ in range(8)], + ] + ) + output = _run(person, workers=4.0) + labor_force = output["A_LFSR"].isin([1, 2, 3, 4]) + pew_unauthorized = output["immigration_status_str"].isin( + [ + "UNDOCUMENTED", + "DACA", + "PAROLED_ONE_YEAR", + "DEPORTATION_WITHHELD", + "TPS", + ] + ) | ( + output["immigration_status_str"].eq("CUBAN_HAITIAN_ENTRANT") + & output["ssn_card_type"].eq("NON_CITIZEN_VALID_EAD") + ) + assert int((labor_force & pew_unauthorized).sum()) == 4 + + def test_worker_control_uses_pew_age_16_labor_force_universe(self) -> None: + person = _person_table( + [ + _noncitizen(A_AGE=15, A_LFSR=1), + _noncitizen(A_AGE=16, A_LFSR=3), + ] + ) + output = _run(person, workers=0.001) + assert output.loc[0, "ssn_card_type"] == "NONE" + assert output.loc[1, "ssn_card_type"] == "NON_CITIZEN_VALID_EAD" + + def test_invalid_labor_force_status_is_refused(self) -> None: + with pytest.raises(SourceRuntimeError, match="A_LFSR"): + _run(_person_table([_noncitizen(A_LFSR=6)])) + + def test_worker_control_counts_tps_before_spilling_residual_workers( + self, + ) -> None: + person = _person_table( + [ + _noncitizen( + PENATVTY=373, + PEINUSYR=24, + A_LFSR=1, + ), + _noncitizen(A_LFSR=1), + ] + ) + output = _run( + person, + workers=1.0, + humanitarian=_humanitarian_block(tps={"venezuela": 1.0}), + ) + assert output.loc[0, "immigration_status_str"] == "TPS" + # Pew includes TPS holders in its unauthorized estimate. The TPS row + # therefore exhausts the one-person labor-force control, so the other + # residual worker must spill to EAD rather than remain undocumented. + assert output.loc[1, "ssn_card_type"] == "NON_CITIZEN_VALID_EAD" + assert output.loc[1, "immigration_status_str"] == ("LEGAL_PERMANENT_RESIDENT") + def test_below_control_counts_spill_nothing(self) -> None: - person = _person_table([_noncitizen(WSAL_VAL=10_000.0), _noncitizen(A_HSCOL=2)]) + person = _person_table( + [_noncitizen(A_LFSR=1, WSAL_VAL=10_000.0), _noncitizen(A_HSCOL=2)] + ) output = _run(person, workers=50.0, students=50.0) assert (output["ssn_card_type"] == "NONE").all() @@ -210,16 +357,16 @@ def test_student_spill_leaves_undocumented_students_at_control(self) -> None: def test_weights_drive_the_spill_amounts(self) -> None: person = _person_table( [ - _noncitizen(WSAL_VAL=10_000.0, person_weight=6.0), - _noncitizen(WSAL_VAL=10_000.0, person_weight=6.0), + _noncitizen(A_LFSR=1, WSAL_VAL=10_000.0, person_weight=6.0), + _noncitizen(A_LFSR=1, WSAL_VAL=10_000.0, person_weight=6.0), ] ) output = _run(person, workers=6.0) assert set(output["ssn_card_type"]) == {"NON_CITIZEN_VALID_EAD", "NONE"} def test_indicator_holders_never_flip_to_undocumented(self) -> None: - # The total undocumented population is emergent: a short count is - # never topped up from people with legal-status indicators. + # The total Pew-defined unauthorized population is emergent: a short + # count is never topped up from people with legal-status indicators. person = _person_table( [_noncitizen(), _noncitizen(CAID=1, person_household_id=1)] ) @@ -230,6 +377,10 @@ def test_prcitshp_outside_domain_raises(self) -> None: with pytest.raises(SourceRuntimeError, match="PRCITSHP"): _run(_person_table([{"PRCITSHP": 7}])) + def test_2024_asec_rejects_nonexistent_peinusyr_29(self) -> None: + with pytest.raises(SourceRuntimeError, match=r"PEINUSYR.*0\.\.28"): + _run(_person_table([_noncitizen(PEINUSYR=29)])) + def test_missing_required_column_raises(self) -> None: person = _person_table([_noncitizen()]).drop(columns=["PEINUSYR"]) with pytest.raises(SourceRuntimeError, match="PEINUSYR"): @@ -245,14 +396,18 @@ class TestImmigrationStatusTags: def test_daca_statutory_cohort_among_ead_holders(self) -> None: # Arrived 2005 (code 19) aged 10 → age at entry < 16, now 29, EAD via # worker spill with a zero control. - person = _person_table([_noncitizen(PEINUSYR=19, A_AGE=29, WSAL_VAL=20_000.0)]) + person = _person_table( + [_noncitizen(PEINUSYR=19, A_AGE=29, A_LFSR=1, WSAL_VAL=20_000.0)] + ) output = _run(person, workers=0.001) assert output.loc[0, "ssn_card_type"] == "NON_CITIZEN_VALID_EAD" assert output.loc[0, "immigration_status_str"] == "DACA" def test_ead_outside_daca_cohort_is_lpr(self) -> None: # Arrived 2015 as an adult: fails the DACA arrival test. - person = _person_table([_noncitizen(PEINUSYR=24, A_AGE=40, WSAL_VAL=20_000.0)]) + person = _person_table( + [_noncitizen(PEINUSYR=24, A_AGE=40, A_LFSR=1, WSAL_VAL=20_000.0)] + ) output = _run(person, workers=0.001) assert output.loc[0, "ssn_card_type"] == "NON_CITIZEN_VALID_EAD" assert output.loc[0, "immigration_status_str"] == "LEGAL_PERMANENT_RESIDENT" @@ -279,7 +434,7 @@ def test_undocumented_tag_matches_none_ssn_exactly(self) -> None: [ _noncitizen(), _noncitizen(CAID=1), - _noncitizen(WSAL_VAL=10_000.0), + _noncitizen(A_LFSR=1, WSAL_VAL=10_000.0), {"PRCITSHP": 1}, ] ) @@ -295,10 +450,15 @@ def test_emitted_values_stay_inside_engine_enum_domains(self) -> None: for overrides in ( {}, {"CAID": 1}, - {"WSAL_VAL": 10_000.0}, + {"A_LFSR": 1, "WSAL_VAL": 10_000.0}, {"A_HSCOL": 2}, {"PENATVTY": 327}, - {"PEINUSYR": 19, "A_AGE": 25, "WSAL_VAL": 5_000.0}, + { + "PEINUSYR": 19, + "A_AGE": 25, + "A_LFSR": 1, + "WSAL_VAL": 5_000.0, + }, ) ] + [{"PRCITSHP": 1}] @@ -308,9 +468,573 @@ def test_emitted_values_stay_inside_engine_enum_domains(self) -> None: assert set(output["immigration_status_str"]) <= set(IMMIGRATION_STATUS_VALUES) +class TestHumanitarianDraws: + def test_parole_draw_takes_documented_origin_cohort(self) -> None: + person = _person_table( + [ + # 2022-arrived Ukrainian reporting Medicaid: the U4U signature. + _noncitizen(PENATVTY=164, PEINUSYR=28, CAID=1), + # Same profile from a non-parole origin stays LPR. + _noncitizen(PENATVTY=303, PEINUSYR=28, CAID=1), + ] + ) + output = _run( + person, + humanitarian=_humanitarian_block(paroled_one_year={"ukraine": 1.0}), + ) + assert output.loc[0, "immigration_status_str"] == "PAROLED_ONE_YEAR" + assert output.loc[0, "ssn_card_type"] == "OTHER_NON_CITIZEN" + assert output.loc[1, "immigration_status_str"] == "LEGAL_PERMANENT_RESIDENT" + + def test_parole_origin_targets_do_not_cross(self) -> None: + person = _person_table( + [ + _noncitizen(PENATVTY=315, PEINUSYR=28, CAID=1), # Nicaraguan + ] + ) + output = _run( + person, + humanitarian=_humanitarian_block(paroled_one_year={"ukraine": 5.0}), + ) + assert output.loc[0, "immigration_status_str"] == "LEGAL_PERMANENT_RESIDENT" + + def test_parole_needs_recent_arrival(self) -> None: + # A 2014-2015-arrived Ukrainian predates every parole program. + person = _person_table([_noncitizen(PENATVTY=164, PEINUSYR=24, CAID=1)]) + output = _run( + person, + humanitarian=_humanitarian_block(paroled_one_year={"ukraine": 5.0}), + ) + assert output.loc[0, "immigration_status_str"] == "LEGAL_PERMANENT_RESIDENT" + + def test_u4u_rejects_2020_2021_ukrainian_arrivals(self) -> None: + person = _person_table( + [ + _noncitizen(PENATVTY=164, PEINUSYR=27, CAID=1), + _noncitizen(PENATVTY=164, PEINUSYR=28, CAID=1), + ] + ) + output = _run( + person, + humanitarian=_humanitarian_block(paroled_one_year={"ukraine": 2.0}), + ) + assert output["immigration_status_str"].tolist() == [ + "LEGAL_PERMANENT_RESIDENT", + "PAROLED_ONE_YEAR", + ] + + compatibility = _us_frame( + [ + _noncitizen( + PENATVTY=164, + PEINUSYR=27, + ssn_card_type="OTHER_NON_CITIZEN", + immigration_status_str="PAROLED_ONE_YEAR", + ), + _noncitizen( + PENATVTY=164, + PEINUSYR=28, + ssn_card_type="OTHER_NON_CITIZEN", + immigration_status_str="PAROLED_ONE_YEAR", + ), + ] + ) + draw = HumanitarianDraw( + category="paroled_one_year", + origin="ukraine", + status="PAROLED_ONE_YEAR", + target=1.0, + source="https://example.com/u4u", + ) + assert us_immigration_humanitarian_draw_mask( + compatibility, + draw, + time_period=TIME_PERIOD, + ).tolist() == [False, True] + + tables = { + entity: compatibility.table(entity).copy() + for entity in compatibility.entities + } + tables["person"] = tables["person"].drop( + columns=["PRCITSHP", "PENATVTY", "PEINUSYR"] + ) + tables["person"]["CIT"] = [5, 5] + tables["person"]["POBP"] = [164, 164] + tables["person"]["YOEP"] = [2021, 2022] + acs_compatibility = Frame( + tables, + compatibility.schema, + { + entity: compatibility.weights_for(entity) + for entity in compatibility.weighted_entities + }, + ) + assert us_immigration_humanitarian_draw_mask( + acs_compatibility, + draw, + time_period=TIME_PERIOD, + ).tolist() == [False, True] + + def test_acs_venezuela_tps_includes_2023_designation_cohort(self) -> None: + compatibility = _us_frame( + [ + _noncitizen( + PENATVTY=373, + PEINUSYR=28, + ssn_card_type="NON_CITIZEN_VALID_EAD", + immigration_status_str=status, + ) + for status in ("TPS", "TPS", "TPS", "PAROLED_ONE_YEAR") + ] + ) + tables = { + entity: compatibility.table(entity).copy() + for entity in compatibility.entities + } + person = tables["person"].drop(columns=["PRCITSHP", "PENATVTY", "PEINUSYR"]) + person["CIT"] = [5, 5, 5, 5] + person["POBP"] = [373, 373, 373, 373] + person["YOEP"] = [2021, 2023, 2024, 2023] + tables["person"] = person + frame = Frame( + tables, + compatibility.schema, + { + entity: compatibility.weights_for(entity) + for entity in compatibility.weighted_entities + }, + ) + tps = HumanitarianDraw( + category="tps", + origin="venezuela", + status="TPS", + target=1.0, + source="https://example.com/venezuela-tps", + ) + parole = HumanitarianDraw( + category="paroled_one_year", + origin="venezuela", + status="PAROLED_ONE_YEAR", + target=1.0, + source="https://example.com/chnv", + ) + + assert us_immigration_humanitarian_draw_mask( + frame, + tps, + time_period=TIME_PERIOD, + ).tolist() == [True, True, False, False] + assert us_immigration_humanitarian_draw_mask( + frame, + parole, + time_period=TIME_PERIOD, + ).tolist() == [False, False, False, True] + + @pytest.mark.parametrize( + ("citizenship", "arrival_year"), + ( + (4, np.nan), + (5, np.nan), + (4, 1899), + (5, TIME_PERIOD + 1), + (5, 2020.5), + ), + ) + def test_acs_foreign_born_arrival_evidence_fails_closed( + self, + citizenship: int, + arrival_year: float, + ) -> None: + compatibility = _us_frame( + [ + _noncitizen( + PENATVTY=373, + PEINUSYR=28, + ssn_card_type="NON_CITIZEN_VALID_EAD", + immigration_status_str="TPS", + ) + ] + ) + tables = { + entity: compatibility.table(entity).copy() + for entity in compatibility.entities + } + person = tables["person"].drop(columns=["PRCITSHP", "PENATVTY", "PEINUSYR"]) + person["CIT"] = [citizenship] + person["POBP"] = [373] + person["YOEP"] = [arrival_year] + tables["person"] = person + frame = Frame( + tables, + compatibility.schema, + { + entity: compatibility.weights_for(entity) + for entity in compatibility.weighted_entities + }, + ) + draw = HumanitarianDraw( + category="tps", + origin="venezuela", + status="TPS", + target=1.0, + source="https://example.com/venezuela-tps", + ) + + with pytest.raises( + SourceRuntimeError, + match="ACS YOEP must be an integral 1900..time_period year", + ): + us_immigration_humanitarian_draw_mask( + frame, + draw, + time_period=TIME_PERIOD, + ) + + @pytest.mark.parametrize( + ("birth_country", "origin", "cutoff"), + ( + (373, "venezuela", 2023), + (312, "el_salvador", 2001), + (314, "honduras", 1998), + (315, "nicaragua", 1998), + (229, "nepal", 2015), + (205, "other_designated", 2024), + (239, "other_designated", 2024), + (248, "other_designated", 2024), + (224, "other_designated", 2024), + (407, "other_designated", 2023), + (416, "other_designated", 2024), + (448, "other_designated", 2024), + (451, "other_designated", 2023), + ), + ) + def test_acs_tps_continuous_residence_cutoff( + self, + birth_country: int, + origin: str, + cutoff: int, + ) -> None: + compatibility = _us_frame( + [ + _noncitizen( + PENATVTY=birth_country, + PEINUSYR=28, + A_AGE=70, + ssn_card_type="NON_CITIZEN_VALID_EAD", + immigration_status_str="TPS", + ), + _noncitizen( + PENATVTY=birth_country, + PEINUSYR=28, + A_AGE=70, + ssn_card_type="NON_CITIZEN_VALID_EAD", + immigration_status_str="TPS", + ), + ] + ) + tables = { + entity: compatibility.table(entity).copy() + for entity in compatibility.entities + } + person = tables["person"].drop(columns=["PRCITSHP", "PENATVTY", "PEINUSYR"]) + person["CIT"] = [5, 5] + person["POBP"] = [birth_country, birth_country] + person["YOEP"] = [cutoff, cutoff + 1] + tables["person"] = person + frame = Frame( + tables, + compatibility.schema, + { + entity: compatibility.weights_for(entity) + for entity in compatibility.weighted_entities + }, + ) + draw = HumanitarianDraw( + category="tps", + origin=origin, + status="TPS", + target=1.0, + source="https://example.com/tps-cutoff", + ) + + assert us_immigration_humanitarian_draw_mask( + frame, + draw, + time_period=max(TIME_PERIOD, cutoff + 1), + ).tolist() == [True, False] + + def test_refugee_draw_excludes_residual_pool(self) -> None: + # A 2022-arrived Congolese person with no legal-status indicator is + # not refugee-consistent (refugees carry immediate benefits access). + person = _person_table( + [ + _noncitizen(PENATVTY=412, PEINUSYR=28), + _noncitizen(PENATVTY=412, PEINUSYR=28, CAID=1), + ] + ) + output = _run(person, humanitarian=_humanitarian_block(refugee=1.0)) + assert output.loc[0, "immigration_status_str"] == "UNDOCUMENTED" + assert output.loc[1, "immigration_status_str"] == "REFUGEE" + + def test_asylee_draw_uses_backlog_window(self) -> None: + person = _person_table( + [ + # 2018-2019 arrival from China with Medicaid: grant-consistent. + _noncitizen(PENATVTY=207, PEINUSYR=26, CAID=1), + # 2004-2005 arrival predates the grant backlog window. + _noncitizen(PENATVTY=207, PEINUSYR=19, CAID=1, A_AGE=50), + ] + ) + output = _run(person, humanitarian=_humanitarian_block(asylee=1.0)) + assert output.loc[0, "immigration_status_str"] == "ASYLEE" + assert output.loc[1, "immigration_status_str"] == "LEGAL_PERMANENT_RESIDENT" + + def test_tps_residual_draw_flips_ssn_to_ead(self) -> None: + person = _person_table([_noncitizen(PENATVTY=373, PEINUSYR=28)]) + output = _run(person, humanitarian=_humanitarian_block(tps={"venezuela": 1.0})) + assert output.loc[0, "immigration_status_str"] == "TPS" + assert output.loc[0, "ssn_card_type"] == "NON_CITIZEN_VALID_EAD" + none_ssn = output["ssn_card_type"] == "NONE" + undocumented = output["immigration_status_str"] == "UNDOCUMENTED" + assert (none_ssn == undocumented).all() + + def test_tps_legacy_continuous_residence_window_binds(self) -> None: + person = _person_table( + [ + # Arrived 1996-1997: inside the El Salvador 2001 cutoff. + _noncitizen(PENATVTY=312, PEINUSYR=15, CAID=1, A_AGE=55), + # Arrived 2014-2015: cannot hold El Salvador TPS. + _noncitizen(PENATVTY=312, PEINUSYR=24, CAID=1, A_AGE=40), + ] + ) + output = _run( + person, humanitarian=_humanitarian_block(tps={"el_salvador": 5.0}) + ) + assert output.loc[0, "immigration_status_str"] == "TPS" + assert output.loc[1, "immigration_status_str"] == "LEGAL_PERMANENT_RESIDENT" + + def test_daca_cohort_is_excluded_from_humanitarian_draws(self) -> None: + # Arrived 2000-2001 aged ~7: the DACA statutory cohort. The worker + # spill gives the EAD card and the DACA tag survives TPS targeting. + person = _person_table( + [ + _noncitizen( + PENATVTY=312, + PEINUSYR=17, + A_AGE=30, + A_LFSR=1, + WSAL_VAL=20_000.0, + ) + ] + ) + output = _run( + person, + workers=0.001, + humanitarian=_humanitarian_block(tps={"el_salvador": 5.0}), + ) + assert output.loc[0, "immigration_status_str"] == "DACA" + + def test_cuban_haitian_entrants_survive_humanitarian_targets(self) -> None: + person = _person_table([_noncitizen(PENATVTY=327, PEINUSYR=28, CAID=1)]) + output = _run( + person, + humanitarian=_humanitarian_block( + paroled_one_year={"venezuela": 5.0}, tps={"venezuela": 5.0} + ), + ) + assert output.loc[0, "immigration_status_str"] == "CUBAN_HAITIAN_ENTRANT" + + def test_draw_order_gives_parole_precedence_over_tps(self) -> None: + person = _person_table([_noncitizen(PENATVTY=373, PEINUSYR=28, CAID=1)]) + both = _run( + person, + humanitarian=_humanitarian_block( + paroled_one_year={"venezuela": 1.0}, tps={"venezuela": 1.0} + ), + ) + tps_only = _run( + person, humanitarian=_humanitarian_block(tps={"venezuela": 1.0}) + ) + assert both.loc[0, "immigration_status_str"] == "PAROLED_ONE_YEAR" + assert tps_only.loc[0, "immigration_status_str"] == "TPS" + + def test_draws_are_weight_targeted(self) -> None: + person = _person_table( + [ + _noncitizen(PENATVTY=164, PEINUSYR=28, CAID=1, person_weight=3.0), + _noncitizen(PENATVTY=164, PEINUSYR=28, CAID=1, person_weight=3.0), + ] + ) + output = _run( + person, + humanitarian=_humanitarian_block(paroled_one_year={"ukraine": 3.0}), + ) + statuses = sorted(output["immigration_status_str"]) + assert statuses == ["LEGAL_PERMANENT_RESIDENT", "PAROLED_ONE_YEAR"] + + def test_zero_targets_draw_nothing(self) -> None: + person = _person_table( + [ + _noncitizen(PENATVTY=164, PEINUSYR=28, CAID=1), + _noncitizen(PENATVTY=373, PEINUSYR=28), + _noncitizen(PENATVTY=412, PEINUSYR=28, CAID=1), + ] + ) + output = _run(person, humanitarian=_humanitarian_block()) + assert set(output["immigration_status_str"]) <= { + "LEGAL_PERMANENT_RESIDENT", + "UNDOCUMENTED", + } + + def test_humanitarian_spill_interaction_preserves_worker_control(self) -> None: + # TPS extraction removes a residual worker before the spill, so the + # remaining undocumented workers still land on the control exactly. + person = _person_table( + [ + _noncitizen( + PENATVTY=373, + PEINUSYR=28, + A_LFSR=1, + WSAL_VAL=10_000.0, + ) + ] + + [_noncitizen(A_LFSR=1, WSAL_VAL=10_000.0) for _ in range(10)] + ) + output = _run( + person, + workers=4.0, + humanitarian=_humanitarian_block(tps={"venezuela": 1.0}), + ) + workers = output["A_LFSR"].isin([1, 2, 3, 4]) + pew_workers = workers & output["immigration_status_str"].isin( + [ + "UNDOCUMENTED", + "DACA", + "PAROLED_ONE_YEAR", + "DEPORTATION_WITHHELD", + "TPS", + ] + ) + assert output.loc[0, "immigration_status_str"] == "TPS" + assert pew_workers.sum() == 4 + + def test_missing_humanitarian_block_is_refused(self) -> None: + spec_mapping = { + "stage": US_IMMIGRATION_STAGE_NAME, + "survey": "test", + "source": "https://example.com", + "grain": "person", + "operations": [ + {"kind": "read_table", "table": "person"}, + { + "kind": "derive_immigration_status", + "seed_from_build_config": True, + "time_period_from_build_config": True, + "undocumented_workers": { + "target": 1, + "source": "https://example.com", + }, + "undocumented_students": { + "target": 1, + "source": "https://example.com", + }, + "undocumented_population_anchor": { + "value": 1, + "source": "https://example.com", + }, + }, + ], + "outputs": list(US_IMMIGRATION_OUTPUT_COLUMNS), + } + with pytest.raises(SourceRuntimeError, match="humanitarian_status_stocks"): + run_source_stage( + SourceStageSpec.from_mapping(spec_mapping), + tables={"person": _person_table([_noncitizen()])}, + operation_handlers=_HANDLERS, + config=SourceRuntimeConfig(seed=0, target_year=TIME_PERIOD), + ) + + @pytest.mark.parametrize( + "mutate, match", + [ + (lambda block: block.pop("refugee"), "missing category"), + ( + lambda block: block.__setitem__( + "mystery", {"target": 1, "source": "https://example.com"} + ), + "unsupported category", + ), + ( + lambda block: block["paroled_one_year"].pop("ukraine"), + "missing origin", + ), + ( + lambda block: block["tps"].__setitem__( + "atlantis", {"target": 1, "source": "https://example.com"} + ), + "outside the stage's codebook table", + ), + ( + lambda block: block["refugee"].pop("source"), + "source citation", + ), + ( + lambda block: block["refugee"].__setitem__("target", -1), + "non-negative", + ), + ], + ) + def test_malformed_humanitarian_blocks_are_refused(self, mutate, match) -> None: + block = _humanitarian_block(refugee=1.0) + mutate(block) + with pytest.raises(SourceRuntimeError, match=match): + _run(_person_table([_noncitizen()]), humanitarian=block) + + def test_humanitarian_clones_stay_consistent(self) -> None: + rows = [ + _noncitizen( + PENATVTY=164, + PEINUSYR=28, + CAID=1, + person_id=index + 1, + source_year=2024, + source_person_id=f"P{index % 5}", + ) + for index in range(10) + ] + output = _run( + _person_table(rows), + humanitarian=_humanitarian_block(paroled_one_year={"ukraine": 5.0}), + ) + by_source = output.groupby("source_person_id")["immigration_status_str"] + assert (by_source.nunique() == 1).all() + + class TestDeterminism: def _worker_pool(self) -> pd.DataFrame: - return _person_table([_noncitizen(WSAL_VAL=10_000.0) for _ in range(20)]) + return _person_table( + [_noncitizen(A_LFSR=1, WSAL_VAL=10_000.0) for _ in range(20)] + ) + + def test_partial_household_lineage_uses_complete_legacy_alternative(self) -> None: + person = _person_table([_noncitizen(), _noncitizen()]) + person["source_year"] = [2024, 2024] + person["source_household_id"] = [100, np.nan] + person["source_person_id"] = [1, 2] + + partial_full = immigration_module._stable_person_draws( + person, + seed=17, + salt="lineage-regression", + ) + legacy_only = immigration_module._stable_person_draws( + person.drop(columns="source_household_id"), + seed=17, + salt="lineage-regression", + ) + + np.testing.assert_array_equal(partial_full, legacy_only) def test_same_seed_is_bit_reproducible(self) -> None: first = _run(self._worker_pool(), workers=10.0, seed=7) @@ -325,6 +1049,7 @@ def test_different_seeds_select_different_ead_holders(self) -> None: def test_source_identity_keys_make_clones_consistent(self) -> None: rows = [ _noncitizen( + A_LFSR=1, WSAL_VAL=10_000.0, person_id=index + 1, source_year=2024, @@ -363,6 +1088,52 @@ def test_manifest_controls_carry_citations(self) -> None: assert float(block[value_key]) > 0 assert str(block["source"]).startswith("https://") + def test_manifest_humanitarian_stocks_carry_citations(self) -> None: + humanitarian = ( + us_immigration_stage_spec() + .operations[1] + .parameters["humanitarian_status_stocks"] + ) + assert set(humanitarian) == set(HUMANITARIAN_STATUS_CATEGORIES) + flat_blocks = { + "refugee": humanitarian["refugee"], + "asylee": humanitarian["asylee"], + "deportation_withheld": humanitarian["deportation_withheld"], + **{ + f"paroled_one_year:{origin}": block + for origin, block in humanitarian["paroled_one_year"].items() + }, + **{f"tps:{origin}": block for origin, block in humanitarian["tps"].items()}, + } + assert set(humanitarian["paroled_one_year"]) == set(_PAROLE_ORIGIN_KEYS) + assert set(humanitarian["tps"]) == set(_TPS_ORIGIN_KEYS) + for label, block in flat_blocks.items(): + assert float(block["target"]) >= 0, label + assert str(block["source"]).startswith("https://"), label + assert humanitarian["refugee"] == { + "target": 160_000, + "source": ( + "https://ohss.dhs.gov/topics/immigration/refugees/" + "annual-flow-report/fy-24-refugees-flow-report" + ), + } + assert humanitarian["asylee"] == { + "target": 155_000, + "source": ( + "https://ohss.dhs.gov/topics/immigration/asylees/" + "annual-flow-report/fy24-asylees-flow-report" + ), + } + # The withheld stock is an explicit cited zero (no published stock); + # every other category carries a positive stock. + assert float(humanitarian["deportation_withheld"]["target"]) == 0 + positive = [ + label + for label, block in flat_blocks.items() + if label != "deportation_withheld" + ] + assert all(float(flat_blocks[label]["target"]) > 0 for label in positive) + def test_unexpected_parameter_is_refused(self) -> None: spec = SourceStageSpec.from_mapping( { @@ -481,7 +1252,7 @@ def test_with_us_immigration_inputs_writes_both_columns(self) -> None: rows = ( [{"PRCITSHP": 1} for _ in range(93)] + [_noncitizen(CAID=1) for _ in range(2)] - + [_noncitizen(WSAL_VAL=10_000.0) for _ in range(12)] + + [_noncitizen(A_LFSR=1, WSAL_VAL=10_000.0) for _ in range(12)] + [_noncitizen() for _ in range(5)] ) frame = _us_frame(rows, household_weights=[1e6] * len(rows)) @@ -491,7 +1262,7 @@ def test_with_us_immigration_inputs_writes_both_columns(self) -> None: assert column in person.columns assert set(person["ssn_card_type"]) <= set(SSN_CARD_TYPE_VALUES) assert (person.loc[person["PRCITSHP"] == 1, "ssn_card_type"] == "CITIZEN").all() - # 12M weighted undocumented workers against the 8.3M Pew control: + # 12M weighted unauthorized workers against the 9.7M Pew control: # some spill to EAD, the rest stay undocumented. assert (person["ssn_card_type"] == "NON_CITIZEN_VALID_EAD").any() assert (person["ssn_card_type"] == "NONE").any() @@ -521,16 +1292,41 @@ def test_missing_raw_columns_raise_loudly(self) -> None: with_us_immigration_inputs(stripped, seed=0, time_period=TIME_PERIOD) -def _plausible_controls() -> UndocumentedControls: - return UndocumentedControls( - workers=20.0, - students=5.0, - population_anchor=30.0, - sources={ - "undocumented_workers": "https://example.com/workers", - "undocumented_students": "https://example.com/students", - "undocumented_population_anchor": "https://example.com/population", - }, +def _humanitarian_draws(**targets: float) -> tuple[HumanitarianDraw, ...]: + """Zero-target draw per category, overridden by keyword (national sums).""" + + status_by_category = { + "paroled_one_year": "PAROLED_ONE_YEAR", + "refugee": "REFUGEE", + "asylee": "ASYLEE", + "deportation_withheld": "DEPORTATION_WITHHELD", + "tps": "TPS", + } + return tuple( + HumanitarianDraw( + category=category, + origin=None, + status=status_by_category[category], + target=float(targets.get(category, 0.0)), + source=f"https://example.com/{category}", + ) + for category in HUMANITARIAN_STATUS_CATEGORIES + ) + + +def _plausible_controls(**humanitarian_targets: float) -> ImmigrationControls: + return ImmigrationControls( + undocumented=UndocumentedControls( + workers=20.0, + students=5.0, + population_anchor=30.0, + sources={ + "undocumented_workers": "https://example.com/workers", + "undocumented_students": "https://example.com/students", + "undocumented_population_anchor": "https://example.com/population", + }, + ), + humanitarian=_humanitarian_draws(**humanitarian_targets), ) @@ -540,6 +1336,7 @@ def _composition_frame( other: int = 30, ead: int = 10, none: int = 30, + humanitarian: dict[str, int] | None = None, ) -> Frame: rows: list[dict] = [] values: list[tuple[str, str]] = ( @@ -547,13 +1344,30 @@ def _composition_frame( + [("OTHER_NON_CITIZEN", "LEGAL_PERMANENT_RESIDENT")] * other + [("NON_CITIZEN_VALID_EAD", "LEGAL_PERMANENT_RESIDENT")] * ead + [("NONE", "UNDOCUMENTED")] * none + + [ + ( + ("NON_CITIZEN_VALID_EAD" if status == "DACA" else "OTHER_NON_CITIZEN"), + status, + ) + for status, count in (humanitarian or {}).items() + for _ in range(count) + ] ) + evidence_by_status = { + "PAROLED_ONE_YEAR": {"PENATVTY": 164, "PEINUSYR": 28}, + "REFUGEE": {"PENATVTY": 412, "PEINUSYR": 28}, + "ASYLEE": {"PENATVTY": 207, "PEINUSYR": 26}, + "DEPORTATION_WITHHELD": {"PENATVTY": 303, "PEINUSYR": 24}, + "TPS": {"PENATVTY": 373, "PEINUSYR": 24}, + "DACA": {"PENATVTY": 303, "PEINUSYR": 19, "A_AGE": 29}, + } for ssn, status in values: rows.append( { "PRCITSHP": 1 if ssn == "CITIZEN" else 5, "ssn_card_type": ssn, "immigration_status_str": status, + **evidence_by_status.get(status, {}), } ) return _us_frame(rows) @@ -616,22 +1430,172 @@ def test_fails_when_undocumented_far_from_anchor(self) -> None: def test_fails_when_non_citizen_share_implausible(self) -> None: gate = us_immigration_composition_gate( _composition_frame(citizens=40, other=20, ead=20, none=20), - controls=UndocumentedControls( - workers=8.0, - students=1.0, - population_anchor=20.0, - sources=_plausible_controls().sources, + controls=ImmigrationControls( + undocumented=UndocumentedControls( + workers=8.0, + students=1.0, + population_anchor=20.0, + sources=_plausible_controls().undocumented.sources, + ), + humanitarian=_humanitarian_draws(), ), ) assert not gate.passed assert any("non-citizen weighted share" in failure for failure in gate.failures) def test_gate_reads_packaged_controls_by_default(self) -> None: + packaged = us_immigration_controls() + assert packaged.undocumented.workers == 9_700_000 + assert packaged.humanitarian_target("refugee") == 160_000 gate = us_immigration_composition_gate(_us_frame([{"PRCITSHP": 1}])) assert not gate.passed controls = gate.details["controls"] - assert controls["undocumented_workers"] == 8_300_000 - assert controls["undocumented_population_anchor"] == 11_000_000 + assert controls["undocumented_workers"] == 9_700_000 + assert controls["undocumented_population_anchor"] == 14_000_000 + stocks = controls["humanitarian_status_stocks"] + assert stocks["paroled_one_year:afghanistan"]["target"] == 73_566 + assert stocks["refugee"] == { + "target": 160_000.0, + "source": ( + "https://ohss.dhs.gov/topics/immigration/refugees/" + "annual-flow-report/fy-24-refugees-flow-report" + ), + } + assert stocks["asylee"] == { + "target": 155_000.0, + "source": ( + "https://ohss.dhs.gov/topics/immigration/asylees/" + "annual-flow-report/fy24-asylees-flow-report" + ), + } + assert stocks["tps:venezuela"]["target"] == 605_015 + assert stocks["deportation_withheld"]["target"] == 0 + + def test_gate_passes_with_in_band_humanitarian_masses(self) -> None: + gate = us_immigration_composition_gate( + _composition_frame(humanitarian={"REFUGEE": 4, "TPS": 6}), + controls=_plausible_controls(refugee=4.0, tps=6.0), + ) + assert gate.passed, gate.failures + achieved = gate.details["humanitarian_achieved"] + assert achieved["refugee"]["population"] == 4.0 + assert achieved["refugee"]["relative"] == 1.0 + assert achieved["tps"]["target"] == 6.0 + + def test_gate_counts_temporary_protections_in_pew_population(self) -> None: + gate = us_immigration_composition_gate( + _composition_frame( + none=10, + humanitarian={"DACA": 5, "PAROLED_ONE_YEAR": 10, "TPS": 10}, + ), + controls=_plausible_controls(paroled_one_year=10.0, tps=10.0), + ) + assert gate.passed, gate.failures + assert gate.details["pew_unauthorized_population"] == 35.0 + + def test_gate_counts_only_residual_ead_cuban_haitian_entrants(self) -> None: + frame = _composition_frame() + person = frame.table("person") + ead_row = person.index[person["ssn_card_type"].eq("NON_CITIZEN_VALID_EAD")][0] + documented_row = person.index[person["ssn_card_type"].eq("OTHER_NON_CITIZEN")][ + 0 + ] + for row in (ead_row, documented_row): + person.loc[row, "immigration_status_str"] = "CUBAN_HAITIAN_ENTRANT" + person.loc[row, "PENATVTY"] = 327 + person.loc[row, "PEINUSYR"] = 24 + + gate = us_immigration_composition_gate( + frame, + controls=_plausible_controls(), + ) + assert gate.passed, gate.failures + assert gate.details["pew_unauthorized_population"] == 31.0 + assert gate.details["pew_unauthorized_paired_status"] == { + "immigration_status_str": "CUBAN_HAITIAN_ENTRANT", + "ssn_card_type": "NON_CITIZEN_VALID_EAD", + } + + def test_gate_excludes_refugees_and_asylees_from_pew_population(self) -> None: + controls = ImmigrationControls( + undocumented=UndocumentedControls( + workers=20.0, + students=5.0, + population_anchor=10.0, + sources=_plausible_controls().undocumented.sources, + ), + humanitarian=_humanitarian_draws(refugee=10.0, asylee=10.0), + ) + gate = us_immigration_composition_gate( + _composition_frame( + none=10, + humanitarian={"REFUGEE": 10, "ASYLEE": 10}, + ), + controls=controls, + ) + assert gate.passed, gate.failures + assert gate.details["pew_unauthorized_population"] == 10.0 + + def test_gate_fails_when_humanitarian_category_collapses(self) -> None: + gate = us_immigration_composition_gate( + _composition_frame(), + controls=_plausible_controls(refugee=4.0), + ) + assert not gate.passed + assert any( + "REFUGEE" in failure and "degenerate" in failure + for failure in gate.failures + ) + + def test_gate_fails_when_explicit_zero_category_emits(self) -> None: + gate = us_immigration_composition_gate( + _composition_frame(humanitarian={"DEPORTATION_WITHHELD": 2}), + controls=_plausible_controls(), + ) + assert not gate.passed + assert any("explicit zero target" in failure for failure in gate.failures) + + def test_gate_accepts_saturated_draws_inside_band(self) -> None: + # The ASEC undercovers 2022-24 arrivals; a draw that saturates at + # roughly three-quarters of its admin target stays inside the band. + gate = us_immigration_composition_gate( + _composition_frame(humanitarian={"PAROLED_ONE_YEAR": 3}), + controls=_plausible_controls(paroled_one_year=4.0), + ) + assert gate.passed, gate.failures + + def test_gate_checks_per_origin_draws_not_only_category_total(self) -> None: + source = "https://example.com/parole" + controls = ImmigrationControls( + undocumented=_plausible_controls().undocumented, + humanitarian=( + HumanitarianDraw( + category="paroled_one_year", + origin="afghanistan", + status="PAROLED_ONE_YEAR", + target=1.0, + source=source, + ), + HumanitarianDraw( + category="paroled_one_year", + origin="ukraine", + status="PAROLED_ONE_YEAR", + target=1.0, + source=source, + ), + ), + ) + # Both rows carry Ukraine evidence. The category total is exactly two, + # but the per-origin targets are a zero/two redistribution. + gate = us_immigration_composition_gate( + _composition_frame(humanitarian={"PAROLED_ONE_YEAR": 2}), + controls=controls, + ) + assert not gate.passed + assert any( + "paroled_one_year:afghanistan" in failure for failure in gate.failures + ) + assert any("paroled_one_year:ukraine" in failure for failure in gate.failures) def test_summary_reports_weighted_composition(self) -> None: summary = us_immigration_composition_summary( diff --git a/packages/microcosm-build/tests/test_us_immigration_pool_scaling.py b/packages/microcosm-build/tests/test_us_immigration_pool_scaling.py new file mode 100644 index 00000000..45820b87 --- /dev/null +++ b/packages/microcosm-build/tests/test_us_immigration_pool_scaling.py @@ -0,0 +1,329 @@ +"""Pooled immigration absolute-control mass regressions (microcosm #767).""" + +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from microcosm.build.source_manifest import SourceStageSpec +from microcosm.build.us_runtime import immigration as immigration_module +from microcosm.build.us_runtime import multispine_pool as multispine_pool_module +from microcosm.build.us_runtime.immigration import ( + US_IMMIGRATION_OUTPUT_COLUMNS, + US_IMMIGRATION_STAGE_NAME, + with_us_immigration_inputs, +) +from microcosm.build.us_runtime.puf_support import clone_us_frame_for_puf_support +from microcosm.build.us_runtime.spine_assembly import assemble_spines +from microcosm.frame import US_SCHEMA, Frame, WeightKind, Weights + + +def _humanitarian_controls(*, ukraine: float) -> dict[str, object]: + parole_origins = ("afghanistan", "ukraine", "nicaragua", "venezuela") + tps_origins = ( + "venezuela", + "el_salvador", + "honduras", + "nicaragua", + "nepal", + "other_designated", + ) + return { + "paroled_one_year": { + origin: { + "target": ukraine if origin == "ukraine" else 0.0, + "source": f"https://example.com/parole/{origin}", + } + for origin in parole_origins + }, + "refugee": {"target": 0.0, "source": "https://example.com/refugee"}, + "asylee": {"target": 0.0, "source": "https://example.com/asylee"}, + "deportation_withheld": { + "target": 0.0, + "source": "https://example.com/withheld", + }, + "tps": { + origin: { + "target": 0.0, + "source": f"https://example.com/tps/{origin}", + } + for origin in tps_origins + }, + } + + +def _stage_spec() -> SourceStageSpec: + return SourceStageSpec.from_mapping( + { + "stage": US_IMMIGRATION_STAGE_NAME, + "survey": "test ASEC", + "source": "https://example.com", + "grain": "person", + "operations": [ + {"kind": "read_table", "table": "person"}, + { + "kind": "derive_immigration_status", + "seed_from_build_config": True, + "time_period_from_build_config": True, + "undocumented_workers": { + "target": 4.0, + "source": "https://example.com/workers", + }, + "undocumented_students": { + "target": 2.0, + "source": "https://example.com/students", + }, + "undocumented_population_anchor": { + "value": 10.0, + "source": "https://example.com/population", + }, + "humanitarian_status_stocks": _humanitarian_controls(ukraine=4.0), + }, + ], + "outputs": list(US_IMMIGRATION_OUTPUT_COLUMNS), + } + ) + + +def _immigration_frame() -> Frame: + rows: list[dict[str, object]] = [] + rows.extend({"PRCITSHP": 5, "A_LFSR": 1, "WSAL_VAL": 10_000.0} for _ in range(4)) + rows.extend({"PRCITSHP": 5, "A_HSCOL": 2} for _ in range(3)) + rows.extend( + { + "PRCITSHP": 5, + "PENATVTY": 164, + "PEINUSYR": 28, + "CAID": 1, + } + for _ in range(8) + ) + baseline: dict[str, object] = { + "PRCITSHP": 1, + "PEINUSYR": 0, + "PENATVTY": 57, + "A_AGE": 40, + "A_MARITL": 7, + "A_SPOUSE": 0, + "A_HSCOL": 0, + "A_LFSR": 7, + "WSAL_VAL": 0.0, + "SEMP_VAL": 0.0, + "MCARE": 2, + "CAID": 2, + "IHSFLG": 2, + "CHAMPVA": 2, + "MIL": 2, + "PEN_SC1": 0, + "PEN_SC2": 0, + "RESNSS1": 0, + "RESNSS2": 0, + "SS_YN": 2, + "SSI_YN": 2, + "PEIO1COW": 0, + "A_MJOCC": 0, + "PEAFEVER": 2, + "SPM_CAPHOUSESUB": 0.0, + } + records = [] + for position, overrides in enumerate(rows, start=1): + record = dict(baseline) + record.update(overrides) + record.update( + { + "person_id": position, + "person_household_id": position, + "person_tax_unit_id": position, + "person_spm_unit_id": position, + "person_family_id": position, + "person_marital_unit_id": position, + } + ) + records.append(record) + person = pd.DataFrame(records) + ids = np.arange(1, len(person) + 1, dtype=np.int64) + return Frame( + { + "person": person, + **{ + entity: pd.DataFrame({f"{entity}_id": ids}) + for entity in US_SCHEMA.group_entities + }, + }, + US_SCHEMA, + { + "household": Weights( + np.ones(len(person), dtype=np.float64), + WeightKind.DESIGN, + ) + }, + ) + + +def _weighted_count(frame: Frame, mask: pd.Series) -> float: + weights = np.asarray(frame.resolve_weights("person").values, dtype=np.float64) + return float(weights[mask.to_numpy(dtype=bool)].sum()) + + +def test_stage_weight_scale_allocates_every_absolute_control_to_source_share( + monkeypatch: pytest.MonkeyPatch, +) -> None: + monkeypatch.setattr(immigration_module, "us_immigration_stage_spec", _stage_spec) + + standalone = with_us_immigration_inputs( + _immigration_frame(), seed=0, time_period=2024 + ) + pooled_projection = with_us_immigration_inputs( + _immigration_frame(), + seed=0, + time_period=2024, + person_weight_scale=2.0, + ) + + standalone_person = standalone.table("person") + pooled_person = pooled_projection.table("person") + assert ( + _weighted_count( + standalone, + standalone_person["immigration_status_str"].eq("PAROLED_ONE_YEAR"), + ) + == 4.0 + ) + assert ( + _weighted_count( + pooled_projection, + pooled_person["immigration_status_str"].eq("PAROLED_ONE_YEAR"), + ) + == 2.0 + ) + + standalone_workers = standalone_person["A_LFSR"].isin([1, 2, 3, 4]) + pooled_workers = pooled_person["A_LFSR"].isin([1, 2, 3, 4]) + assert ( + _weighted_count( + standalone, + standalone_workers & standalone_person["ssn_card_type"].eq("NONE"), + ) + == 4.0 + ) + assert ( + _weighted_count( + pooled_projection, + pooled_workers & pooled_person["ssn_card_type"].eq("NONE"), + ) + == 2.0 + ) + + standalone_students = standalone_person["A_HSCOL"].eq(2) + pooled_students = pooled_person["A_HSCOL"].eq(2) + assert ( + _weighted_count( + standalone, + standalone_students & standalone_person["ssn_card_type"].eq("NONE"), + ) + == 2.0 + ) + assert ( + _weighted_count( + pooled_projection, + pooled_students & pooled_person["ssn_card_type"].eq("NONE"), + ) + == 1.0 + ) + + +def _spine(*, cps: bool, offset: int = 0) -> Frame: + ids = np.asarray([1, 2], dtype=np.int64) + person = pd.DataFrame( + { + "person_id": ids, + "person_household_id": ids, + "person_tax_unit_id": ids, + "person_spm_unit_id": ids, + "person_family_id": ids, + "person_marital_unit_id": ids, + "age": [30.0, 50.0], + **({"PERIDNUM": [f"p{offset + 1}", f"p{offset + 2}"]} if cps else {}), + } + ) + return Frame( + { + "person": person, + **{ + entity: pd.DataFrame({f"{entity}_id": ids}) + for entity in US_SCHEMA.group_entities + }, + }, + US_SCHEMA, + { + "household": Weights( + np.asarray([2.0, 2.0]), + WeightKind.DESIGN, + ) + }, + ) + + +def _replace_person(frame: Frame, person: pd.DataFrame) -> Frame: + return Frame( + { + "person": person, + **{ + entity: frame.table(entity) + for entity in frame.entities + if entity != "person" + }, + }, + frame.schema, + {entity: frame.weights_for(entity) for entity in frame.weighted_entities}, + frame.strata, + mass_log=frame.mass_log, + metadata=frame.metadata, + ) + + +def test_postclone_chain_derives_and_receipts_inverse_cps_person_mass_share() -> None: + assembled = assemble_spines( + {"asec": _spine(cps=True), "acs": _spine(cps=False, offset=100)}, + household_mass_shares={"asec": 0.5, "acs": 0.5}, + ) + cloned = clone_us_frame_for_puf_support(assembled) + observed_scale: list[float] = [] + + def immigration( + available: Frame, + *, + person_weight_scale: float, + ) -> Frame: + observed_scale.append(person_weight_scale) + person = available.table("person").copy() + person["ssn_card_type"] = "CITIZEN" + person["immigration_status_str"] = "CITIZEN" + return _replace_person(available, person) + + completed = multispine_pool_module._run_source_operator_chain( + cloned, + phase="post_clone", + operator_names=("with_us_immigration_inputs",), + operators={"with_us_immigration_inputs": immigration}, + ) + + person = cloned.table("person") + weights = np.asarray(cloned.resolve_weights("person").values, dtype=np.float64) + cps = person["PERIDNUM"].notna().to_numpy(dtype=bool) + full_mass = float(weights.sum()) + cps_mass = float(weights[cps].sum()) + expected_scale = full_mass / cps_mass + assert cps_mass / full_mass == pytest.approx(0.5) + assert observed_scale == [pytest.approx(expected_scale)] + + scaling = completed.receipt["suboperators"][0]["kernel_receipt"][ + "person_design_weight_scaling" + ] + assert scaling == { + "full_pool_person_design_weight_mass": pytest.approx(full_mass), + "cps_projection_person_design_weight_mass": pytest.approx(cps_mass), + "person_weight_scale": pytest.approx(expected_scale), + "cps_person_mass_share": pytest.approx(cps_mass / full_mass), + } diff --git a/packages/microcosm-build/tests/test_us_late_producer_dag.py b/packages/microcosm-build/tests/test_us_late_producer_dag.py index d1857176..babf5988 100644 --- a/packages/microcosm-build/tests/test_us_late_producer_dag.py +++ b/packages/microcosm-build/tests/test_us_late_producer_dag.py @@ -784,9 +784,9 @@ def test_canonical_us_late_schedule_is_import_validated_and_byte_stable() -> Non assert reconstructed == CANONICAL_US_LATE_PRODUCER_SCHEDULE receipt = us_late_producer_schedule_receipt() - assert receipt["schema_version"] == 16 + assert receipt["schema_version"] == 17 assert receipt["execution_receipt_contract"] == { - "version": 3, + "version": 4, "row_binding": ( "declared_globally_reconciled_input_and_scope_exact_output_source_" "and_primary_callback_resource_receipt_and_previous_execution_sha256" @@ -794,7 +794,8 @@ def test_canonical_us_late_schedule_is_import_validated_and_byte_stable() -> Non "virtual_resource_binding": ("exact_kind_specific_semantic_payload_and_sha256"), "top_binding": ( "entry_and_output_frame_sha256_execution_chain_source_" - "completion_and_nineteen_transfer_groups" + "completion_nineteen_transfer_groups_and_constrained_" + "immigration_reconciliation" ), "transition_authority": { "authority_id": "us_stacked_late_producer_transition", @@ -910,6 +911,72 @@ def test_every_transfer_declares_predictors_and_optional_absence_receipts() -> N ) +def test_immigration_transfer_declares_source_scoped_raw_evidence() -> None: + producer = transfer_producer_name("person", "source_operator_immigration") + inventory = US_LATE_TRANSFER_INPUT_INVENTORIES[producer] + by_label = { + requirement.label: requirement for requirement in inventory.requirements + } + expected = { + "immigration_asec_citizenship": ("PRCITSHP", "asec_source"), + "immigration_asec_origin": ("PENATVTY", "asec_source"), + "immigration_asec_arrival": ("PEINUSYR", "asec_source"), + "immigration_acs_citizenship": ("CIT", "acs_source"), + "immigration_acs_origin": ("POBP", "acs_source"), + "immigration_acs_arrival": ("YOEP", "acs_source"), + } + + assert set(expected) <= set(by_label) + for label, (column, scope) in expected.items(): + requirement = by_label[label] + assert requirement.required_scope == scope + assert requirement.optional is False + value_kind = "column_present" if column == "YOEP" else "finite_numeric" + assert requirement.alternatives == ( + (ProducerInputColumn("person", column, value_kind),), + ) + + expected_lineage = { + ("person_id",), + ("source_person_id", "source_year"), + ("source_household_id", "source_person_id", "source_year"), + } + transfer_lineage = by_label["immigration_stable_person_lineage"] + assert transfer_lineage.required_scope is None + assert { + tuple(column.column for column in alternative) + for alternative in transfer_lineage.alternatives + } == expected_lineage + + source_inventory = US_LATE_SOURCE_INPUT_INVENTORIES["with_us_immigration_inputs"] + source_lineage = next( + requirement + for requirement in source_inventory.requirements + if requirement.label == "stable_source_identity" + ) + assert { + tuple(column.column for column in alternative) + for alternative in source_lineage.alternatives + } == expected_lineage + + contract = CANONICAL_US_LATE_PRODUCER_REGISTRY[producer] + effective = { + item.column: item + for item in contract.inputs + if item.column.startswith("@effective:immigration_") + } + assert set(effective) == { + *(f"@effective:{label}" for label in expected), + "@effective:immigration_stable_person_lineage", + } + for label, (_column, scope) in expected.items(): + assert effective[f"@effective:{label}"].required_scope == scope + assert ( + effective["@effective:immigration_stable_person_lineage"].required_scope + == "whole_pool" + ) + + def test_every_transfer_declares_complete_cross_grain_validation_surface() -> None: entities = ( "person", @@ -966,6 +1033,7 @@ def test_every_origin_exclusive_raw_input_has_its_native_scope() -> None: } asec_person_raw_columns = { "A_HSCOL", + "A_LFSR", "A_MJOCC", "CAID", "CHAMPVA", @@ -1034,6 +1102,12 @@ def test_every_origin_exclusive_raw_input_has_its_native_scope() -> None: "PERIDNUM", } }, + ("person", "PRCITSHP"): 2, + ("person", "PENATVTY"): 2, + ("person", "PEINUSYR"): 2, + ("person", "CIT"): 1, + ("person", "POBP"): 1, + ("person", "YOEP"): 1, } observed: Counter[tuple[str, str, str]] = Counter() @@ -1055,15 +1129,15 @@ def test_every_origin_exclusive_raw_input_has_its_native_scope() -> None: assert observed == Counter( {(*key, required_scope[key]): count for key, count in expected_counts.items()} ) - assert sum(observed.values()) == 101 - assert sum(count for key, count in observed.items() if key[2] == "acs_source") == 43 + assert sum(observed.values()) == 108 + assert sum(count for key, count in observed.items() if key[2] == "acs_source") == 46 assert ( - sum(count for key, count in observed.items() if key[2] == "asec_source") == 58 + sum(count for key, count in observed.items() if key[2] == "asec_source") == 62 ) assert receipts[("household", "TYPEHUGQ")] == set() execution_identity = acs_transfer_module.acs_transfer_execution_contract_identity() - assert execution_identity["schema_version"] == 2 + assert execution_identity["schema_version"] == 3 assert execution_identity["housing"]["head_source_precedence"] == [ {"source": "is_household_head", "head_codes": [True]}, {"source": "A_EXPRRP", "head_codes": [1, 2]}, @@ -1268,8 +1342,7 @@ def test_source_numeric_input_audit_is_fully_executable() -> None: "A_MARITL", "A_SPOUSE", "A_HSCOL", - "WSAL_VAL", - "SEMP_VAL", + "A_LFSR", "MCARE", "CAID", "IHSFLG", diff --git a/packages/microcosm-build/tests/test_us_multispine_pool.py b/packages/microcosm-build/tests/test_us_multispine_pool.py index 8f4f3448..9afc509b 100644 --- a/packages/microcosm-build/tests/test_us_multispine_pool.py +++ b/packages/microcosm-build/tests/test_us_multispine_pool.py @@ -1847,6 +1847,7 @@ def _producer_dtype_source_frame() -> Frame: person["PRCITSHP"] = [1, 5, 1, 5] person["PEINUSYR"] = [0, 24, 0, 24] person["PENATVTY"] = [57, 303, 57, 303] + person["A_LFSR"] = [1, 0, 1, 0] person["A_SPOUSE"] = 0 person["CAID"] = 2 person["IHSFLG"] = 2 @@ -1879,6 +1880,9 @@ def _producer_dtype_acs_source_frame() -> Frame: person["acs_social_security_income"] = [0.0, 12_000.0] person["acs_retirement_income"] = [0.0, 8_000.0] person["acs_interest_dividend_rental_income"] = [100.0, 2_000.0] + person["CIT"] = [1, 5] + person["POBP"] = [6, 373] + person["YOEP"] = [np.nan, 2022] household = tables["household"] household["state_fips"] = [6, 36] household["tenure_type"] = ["RENTED", "OWNED_WITH_MORTGAGE"] @@ -2165,7 +2169,7 @@ def test_every_pool_transfer_family_accepts_its_produced_physical_dtype( ) assert len(targets) == 118 - assert len(predictors) == 32 + assert len(predictors) == 35 assert len(primary_predictor_sets) == 65 primary_targets = tuple( ( @@ -3535,8 +3539,13 @@ def apply( name: str = operator_name, column: str = output, value: float = float(index + 1), + person_weight_scale: float = 1.0, ) -> Frame: calls.append(name) + if name == "with_us_immigration_inputs": + assert person_weight_scale > 1.0 + else: + assert person_weight_scale == 1.0 assert "us_spine_assembly_manifest" not in available.metadata assert not available.mass_log person = available.table("person") @@ -3805,9 +3814,10 @@ def test_pool_seed_stage_preserves_inputs_and_receipts_disclosed_defaults() -> N after_person.loc[measured_person, "takes_up_medicare_if_eligible"].tolist() == before_person.loc[measured_person, "takes_up_medicare_if_eligible"].tolist() ) - assert after_person["takes_up_wic_if_eligible"].tolist() == before_person[ - "takes_up_wic_if_eligible" - ].tolist() + assert ( + after_person["takes_up_wic_if_eligible"].tolist() + == before_person["takes_up_wic_if_eligible"].tolist() + ) assert ( after_spm.loc[measured_spm, "takes_up_tanf_if_eligible"].tolist() == before_spm.loc[measured_spm, "takes_up_tanf_if_eligible"].tolist() diff --git a/packages/microcosm-build/tests/test_us_multispine_pool_h5_io.py b/packages/microcosm-build/tests/test_us_multispine_pool_h5_io.py index 4bdf237e..2669b153 100644 --- a/packages/microcosm-build/tests/test_us_multispine_pool_h5_io.py +++ b/packages/microcosm-build/tests/test_us_multispine_pool_h5_io.py @@ -8,8 +8,9 @@ import pandas as pd import pytest -import microcosm.build.us_runtime.acs_transfer as acs_transfer_module +import microcosm.build.us_runtime.acs_transfer as acs_transfer_runtime import microcosm.build.us_runtime.h5_io as h5_io +import microcosm.build.us_runtime.immigration as immigration_runtime import microcosm.build.us_runtime.post_transfer_calibration as post_transfer_calibration_runtime import microcosm.build.us_runtime.stacked_spine as stacked_spine_module from microcosm.build.frame_checkpoint import ( @@ -45,6 +46,10 @@ write_nullable_us_h5, ) from microcosm.build.us_runtime.support_provenance import ( + BASE_ASEC_SUPPORT_CHANNEL, + spine_assembly_manifest, + spine_provenance_counts, + support_channel_column, support_clone_index_column, support_source_id_column, ) @@ -97,10 +102,7 @@ def _pool_frame() -> Frame: ) -def _pool_frame_with_object_strings_on_every_entity( - *, - stacked: bool = False, -) -> Frame: +def _pool_frame_with_object_strings_on_every_entity() -> Frame: """Match the assembled pool's object-backed source-string shape.""" frame = _pool_frame() @@ -118,21 +120,6 @@ def _pool_frame_with_object_strings_on_every_entity( ), ) tables[entity] = table - if stacked: - household = tables["household"].copy() - household["puma"] = ["0600101", "0600102", "0600102"] - household["congressional_district_geoid"] = np.asarray( - [601, 601, 601], - dtype=np.int64, - ) - household["county_fips"] = ["06001", "06001", "06001"] - household[support_source_id_column("household")] = np.asarray( - [10, 20, 20], dtype=np.int64 - ) - household[support_clone_index_column("household")] = np.asarray( - [0, 0, 1], dtype=np.int64 - ) - tables["household"] = household return Frame( tables, frame.schema, @@ -143,6 +130,181 @@ def _pool_frame_with_object_strings_on_every_entity( ) +def _stacked_pool_frame_with_live_immigration( + *, + include_zero_weight_clone: bool = False, +) -> tuple[Frame, dict[str, object]]: + """Build a persisted-shape final stack and its real immigration receipt.""" + + controls = immigration_runtime.us_immigration_controls() + positive_draws = tuple(draw for draw in controls.humanitarian if draw.target > 0) + acs_evidence = { + "paroled_one_year:afghanistan": (200, 2021, "NONE", "UNDOCUMENTED"), + "paroled_one_year:ukraine": (164, 2022, "NONE", "UNDOCUMENTED"), + "paroled_one_year:nicaragua": (315, 2023, "NONE", "UNDOCUMENTED"), + "paroled_one_year:venezuela": (373, 2022, "NONE", "UNDOCUMENTED"), + "refugee": ( + 412, + 2022, + "OTHER_NON_CITIZEN", + "LEGAL_PERMANENT_RESIDENT", + ), + "asylee": ( + 207, + 2020, + "OTHER_NON_CITIZEN", + "LEGAL_PERMANENT_RESIDENT", + ), + "tps:venezuela": (373, 2015, "NONE", "UNDOCUMENTED"), + "tps:el_salvador": (312, 2001, "NONE", "UNDOCUMENTED"), + "tps:honduras": (314, 1998, "NONE", "UNDOCUMENTED"), + "tps:nicaragua": (315, 1998, "NONE", "UNDOCUMENTED"), + "tps:nepal": (229, 2015, "NONE", "UNDOCUMENTED"), + "tps:other_designated": (248, 2020, "NONE", "UNDOCUMENTED"), + } + expected_labels = {draw.label for draw in positive_draws} + if set(acs_evidence) != expected_labels: + raise AssertionError( + "Stacked H5 immigration evidence must exactly cover the positive " + "canonical humanitarian draws." + ) + + row_count = 1 + len(positive_draws) + int(include_zero_weight_clone) + ids = np.arange(1, row_count + 1, dtype=np.int64) + channels = np.asarray( + [BASE_ASEC_SUPPORT_CHANNEL] + + [stacked_spine_module.ACS_STACKED_SUPPORT_CHANNEL] * len(positive_draws) + + ([BASE_ASEC_SUPPORT_CHANNEL] if include_zero_weight_clone else []), + dtype=object, + ) + clone_indices = np.zeros(row_count, dtype=np.int64) + source_ids = ids.copy() + if include_zero_weight_clone: + clone_indices[-1] = 1 + source_ids[-1] = ids[0] + + person_rows: list[dict[str, object]] = [ + { + "PRCITSHP": 1, + "PENATVTY": 57, + "PEINUSYR": 0, + "CIT": np.nan, + "POBP": np.nan, + "YOEP": np.nan, + "A_AGE": 50, + "age": 50, + "ssn_card_type": "CITIZEN", + "immigration_status_str": "CITIZEN", + } + ] + for draw in positive_draws: + birth_country, arrival_year, ssn_card_type, immigration_status = acs_evidence[ + draw.label + ] + person_rows.append( + { + "PRCITSHP": np.nan, + "PENATVTY": np.nan, + "PEINUSYR": np.nan, + "CIT": 5, + "POBP": birth_country, + "YOEP": arrival_year, + "A_AGE": np.nan, + "age": 50, + "ssn_card_type": ssn_card_type, + "immigration_status_str": immigration_status, + } + ) + if include_zero_weight_clone: + person_rows.append(dict(person_rows[0])) + + person = pd.DataFrame(person_rows) + person.insert(0, "PERIDNUM", pd.Series([f"person-{value}" for value in ids])) + person.insert(1, "person_id", ids) + for entity in US_SCHEMA.group_entities: + person[US_SCHEMA.membership_column(entity)] = ids + person["nullable_input"] = np.resize( + np.asarray([True, None, False], dtype=object), + row_count, + ) + person["is_incapable_of_self_care"] = True + person["tax_unit_role_input"] = "DEPENDENT" + person[support_channel_column("person")] = channels + person[support_source_id_column("person")] = source_ids + person[support_clone_index_column("person")] = clone_indices + + household_weights = np.asarray( + [1.0] + + [float(draw.target) for draw in positive_draws] + + ([0.0] if include_zero_weight_clone else []), + dtype=np.float64, + ) + mutable = channels == stacked_spine_module.ACS_STACKED_SUPPORT_CHANNEL + reconciled_person, reconciliation = ( + immigration_runtime.reconcile_us_immigration_humanitarian_transfer( + person, + weights=household_weights, + mutable_rows=mutable, + seed=0, + time_period=2024, + controls=controls, + ) + ) + + tables = {"person": reconciled_person} + for entity in US_SCHEMA.group_entities: + table = pd.DataFrame({US_SCHEMA.id_column(entity): ids}) + table.insert( + 0, + f"{entity}_source_label", + pd.Series([f"{entity}-{value}" for value in ids], dtype=object), + ) + table[support_channel_column(entity)] = channels + table[support_source_id_column(entity)] = source_ids + table[support_clone_index_column(entity)] = clone_indices + tables[entity] = table + + household = tables["household"] + household["puma"] = "0600101" + household["congressional_district_geoid"] = np.full( + row_count, + 601, + dtype=np.int64, + ) + household["county_fips"] = "06001" + + for index, spec in enumerate( + post_transfer_calibration_runtime.POST_TRANSFER_CALIBRATION_SPECS.values() + ): + tables[spec.entity][spec.target] = ( + 10.0 + index + np.arange(row_count, dtype=np.float64) + ) + + assembly_tables = { + entity: table.loc[table[support_clone_index_column(entity)].eq(0)] + for entity, table in tables.items() + } + metadata = spine_assembly_manifest( + assembly_tables, + channels=( + BASE_ASEC_SUPPORT_CHANNEL, + stacked_spine_module.ACS_STACKED_SUPPORT_CHANNEL, + ), + ) + frame = Frame( + tables, + US_SCHEMA, + { + "household": Weights( + household_weights, + WeightKind.IMPORTANCE, + ) + }, + metadata=metadata, + ) + return frame, reconciliation + + def _semantic_string_columns(table: pd.DataFrame) -> tuple[str, ...]: return tuple( column @@ -630,24 +792,54 @@ def _write_ready_pool( tmp_path: Path, *, stacked: bool = False, + include_zero_weight_clone: bool = False, sample_fraction: float = 1.0, ) -> Path: run_id = "fixture-publication" pool_path = tmp_path / "pool.h5" diagnostics_path = tmp_path / "pool.agreement.json" manifest_path = tmp_path / "pool.manifest.json" + gate_names = ( + ( + "us_stacked_completeness", + "us_by_origin_battery", + "immigration_composition", + ) + if stacked + else ("us_spine_agreement",) + ) agreement_gate = { "passed": True, "gates": { - "us_spine_agreement": { + name: { "passed": True, "failures": [], "details": {"fixture": True}, } + for name in gate_names }, } + if include_zero_weight_clone and not stacked: + raise ValueError("Only the stacked H5 fixture may include a clone row.") schema_version = US_MULTISPINE_POOL_MANIFEST_SCHEMA_VERSION if stacked else 4 - pool_frame = _pool_frame_with_object_strings_on_every_entity(stacked=stacked) + if stacked: + pool_frame, immigration_reconciliation = ( + _stacked_pool_frame_with_live_immigration( + include_zero_weight_clone=include_zero_weight_clone, + ) + ) + dag = _canonical_stacked_late_dag_receipt( + pool_frame, + immigration_reconciliation=immigration_reconciliation, + ) + provenance_counts = spine_provenance_counts( + pool_frame, + boundary="stacked H5 fixture provenance counts", + ) + else: + pool_frame = _pool_frame_with_object_strings_on_every_entity() + dag = None + provenance_counts = {"household": {"rows": pool_frame.n("household")}} geography_assignment = ( _fixture_geography_assignment(pool_frame.table("household")) if stacked @@ -725,7 +917,7 @@ def _write_ready_pool( }, }, "agreement_gate": agreement_gate, - "provenance_counts": {"household": {"rows": 3}}, + "provenance_counts": provenance_counts, "pool_h5": { "path": str(pool_path.resolve()), "sha256": _sha256(pool_path), @@ -740,7 +932,7 @@ def _write_ready_pool( }, } if stacked: - dag = _canonical_stacked_late_dag_receipt() + assert dag is not None assert geography_assignment is not None sampling, stack_manifest = _fixture_stacked_sampling(sample_fraction) transition_authority = ( @@ -792,28 +984,44 @@ def _write_ready_pool( def _canonical_late_calibration_owner_receipt( + frame: Frame, spec: post_transfer_calibration_runtime.PostTransferCalibrationSpec, ) -> dict[str, object]: - values = np.asarray( - [10.0, 20.0, 30.0, 40.0, 50.0, 100.0, 200.0, 300.0, 400.0, 500.0] + table = frame.table(spec.entity) + channel = table[support_channel_column(spec.entity)].astype(str) + clone_index = pd.to_numeric( + table[support_clone_index_column(spec.entity)], + errors="raise", + ) + reference = (channel.eq(BASE_ASEC_SUPPORT_CHANNEL) & clone_index.eq(0)).to_numpy( + dtype=bool ) - weights = np.asarray([2.0, 3.0, 5.0, 5.0, 5.0, 4.0, 4.0, 4.0, 4.0, 4.0]) - entity_ids = np.arange(1, len(values) + 1) - reference = np.asarray([True] * 5 + [False] * 5) - recipient = ~reference + recipient = ( + channel.eq(stacked_spine_module.ACS_STACKED_SUPPORT_CHANNEL) & clone_index.eq(0) + ).to_numpy(dtype=bool) + weights = np.asarray( + frame.resolve_weights(spec.entity).values, + dtype=np.float64, + ) + entity_ids = table[frame.schema.entity_id_column(spec.entity)].to_numpy(copy=False) constrained = spec.special_constraint != "none" - result = post_transfer_calibration_runtime.calibrate_post_transfer_values( - values, - weights, - entity_ids, - spec=spec, + application = post_transfer_calibration_runtime.apply_post_transfer_calibration( + frame, + entity=spec.entity, + family=spec.family, + target=spec.target, reference_rows=reference, recipient_rows=recipient, mutable_rows=recipient, allowed_carrier_rows=recipient if constrained else None, addition_candidate_rows=recipient if constrained else None, ) - calibration = result.receipt + result_values = application.frame.table(spec.entity)[spec.target].to_numpy( + dtype=np.float64, + copy=True, + ) + table[spec.target] = result_values + calibration = application.receipt scope = calibration["scope"] constraint: dict[str, object] = {"constraint": spec.special_constraint} if spec.special_constraint == "adult_care_qualifying_one_per_tax_unit": @@ -854,13 +1062,13 @@ def _canonical_late_calibration_owner_receipt( ), "reference_output_values_sha256": ( stacked_spine_module._post_transfer_float64_sha256( - result.values[reference], + result_values[reference], boundary="synthetic reference calibration output", ) ), "recipient_output_values_sha256": ( stacked_spine_module._post_transfer_float64_sha256( - result.values[recipient], + result_values[recipient], boundary="synthetic recipient calibration output", ) ), @@ -886,7 +1094,7 @@ def _canonical_late_calibration_owner_receipt( def _canonical_pregnancy_structural_receipt() -> dict[str, object]: - policy = acs_transfer_module.acs_transfer_execution_contract_identity( + policy = acs_transfer_runtime.acs_transfer_execution_contract_identity( targets=("is_pregnant",), derive_schedule_d=False, )["structural_target_policies"]["is_pregnant"] @@ -914,7 +1122,11 @@ def _canonical_pregnancy_structural_receipt() -> dict[str, object]: } -def _canonical_stacked_late_dag_receipt() -> dict[str, object]: +def _canonical_stacked_late_dag_receipt( + frame: Frame, + *, + immigration_reconciliation: dict[str, object], +) -> dict[str, object]: """Build a signed fixture receipt over the live canonical contracts.""" schedule = stacked_spine_module.CANONICAL_US_LATE_PRODUCER_SCHEDULE @@ -964,17 +1176,80 @@ def _canonical_stacked_late_dag_receipt() -> dict[str, object]: "sha256" ] ) + immigration_reconciliation = stacked_spine_module._json_ready( + immigration_reconciliation + ) + immigration_recipient_rows = int(immigration_reconciliation["mutable_rows"]) + immigration_producer_rows = int(immigration_reconciliation["immutable_rows"]) group_receipts: dict[str, object] = {} for group in stacked_spine_module.CANONICAL_US_LATE_TRANSFER_GROUPS: + is_immigration_group = ( + group.entity == "person" and group.family == "source_operator_immigration" + ) group_targets = { f"{group.entity}/{group.family}/{target}": { - "authorized_null_rows": 0, - "imputed_rows": 0, + **( + { + "producer_roles": ["asec_source"], + "producer_rows": immigration_producer_rows, + } + if is_immigration_group + else {} + ), + "authorized_null_rows": ( + immigration_recipient_rows if is_immigration_group else 0 + ), + "imputed_rows": ( + immigration_recipient_rows if is_immigration_group else 0 + ), "unmodeled_rows": 0, "residual_null_rows": 0, } for target in group.targets } + if is_immigration_group: + required_predictors, _optional_predictors = ( + stacked_spine_module._acs_pattern_predictor_authority( + entity=group.entity, + family_targets=group.targets, + ) + ) + pattern = acs_transfer_runtime.AcsTransferPattern( + name=acs_transfer_runtime._pattern_name(0, ()), + observed_optional_predictors=(), + predictors=required_predictors, + seed=0, + weight_kind="design", + donor_rows=immigration_producer_rows, + recipient_rows=immigration_recipient_rows, + target_regimes=tuple( + (model_target, "positive_only") + for model_target in acs_transfer_runtime._model_target_names( + group.targets + ) + ), + ) + for target in group.targets: + key = f"{group.entity}/{group.family}/{target}" + record = acs_transfer_runtime.AcsImputedInput( + column=target, + entity=group.entity, + family=group.family, + donor_spine="synthetic_stacked_h5_fixture", + donor_channel="asec", + predictors=pattern.predictors, + seed=pattern.seed, + weight_kind=pattern.weight_kind, + patterns=(pattern,), + imputed_recipient_rows=immigration_recipient_rows, + reconciliation=immigration_reconciliation, + ) + group_targets[key]["qrf_pattern_evidence"] = ( + stacked_spine_module._acs_imputed_pattern_evidence(record) + ) + group_targets[key]["post_transfer_reconciliation"] = dict( + immigration_reconciliation + ) pregnancy_key = f"{group.entity}/{group.family}/is_pregnant" if pregnancy_key in group_targets: group_targets[pregnancy_key]["structural_policy"] = ( @@ -983,7 +1258,7 @@ def _canonical_stacked_late_dag_receipt() -> dict[str, object]: calibrated_keys = sorted(set(group_targets) & set(late_specs)) for key in calibrated_keys: group_targets[key]["post_transfer_calibration"] = ( - _canonical_late_calibration_owner_receipt(late_specs[key]) + _canonical_late_calibration_owner_receipt(frame, late_specs[key]) ) group_receipts[group.name] = { "producer": group.name, @@ -1427,8 +1702,58 @@ def test_ready_stacked_pool_loader_binds_terminal_gate_aliases( frame, manifest, _ = load_simulation_ready_us_multispine_pool(manifest_path) + positive_draws = tuple( + draw + for draw in immigration_runtime.us_immigration_controls().humanitarian + if draw.target > 0 + ) + expected_rows = 1 + len(positive_draws) + expected_weights = np.asarray( + [1.0, *(float(draw.target) for draw in positive_draws)], + dtype=np.float64, + ) + assert all(frame.n(entity) == expected_rows for entity in US_SCHEMA.entities) + np.testing.assert_array_equal( + frame.weights_for("household").values, + expected_weights, + ) + person = frame.table("person") + assert person.loc[0, "PRCITSHP"] == 1 + assert pd.isna(person.loc[0, "CIT"]) + assert person.loc[0, "immigration_status_str"] == "CITIZEN" + assert person.loc[1:, "CIT"].eq(5).all() + assert person.loc[1:, ["POBP", "YOEP"]].notna().all().all() + for entity in US_SCHEMA.group_entities: + np.testing.assert_array_equal( + person[US_SCHEMA.membership_column(entity)], + frame.table(entity)[US_SCHEMA.id_column(entity)], + ) + for draw in positive_draws: + emitted = immigration_runtime.us_immigration_humanitarian_draw_mask( + frame, + draw, + ) + assert int(emitted.sum()) == 1 + assert float(frame.resolve_weights("person").values[emitted].sum()) == float( + draw.target + ) + for entity in US_SCHEMA.entities: + assert manifest["provenance_counts"][entity] == { + "rows": expected_rows, + "by_source_channel": { + BASE_ASEC_SUPPORT_CHANNEL: 1, + stacked_spine_module.ACS_STACKED_SUPPORT_CHANNEL: len(positive_draws), + }, + "by_clone_index": {"0": expected_rows}, + "by_source_channel_and_clone_index": { + BASE_ASEC_SUPPORT_CHANNEL: {"0": 1}, + stacked_spine_module.ACS_STACKED_SUPPORT_CHANNEL: { + "0": len(positive_draws) + }, + }, + } assert manifest["terminal_gates"] == manifest["agreement_gate"] - assert manifest["schema_version"] == 9 + assert manifest["schema_version"] == 10 assert ( manifest["pool_h5"]["materializer_version"] == US_MULTISPINE_POOL_H5_MATERIALIZER_VERSION @@ -1446,10 +1771,11 @@ def test_ready_stacked_pool_loader_binds_terminal_gate_aliases( transition_authority["sha256"] == manifest["late_producer_transition_authority_sha256"] ) - assert stacked_spine_module._json_ready( - frame.metadata[stacked_spine_module.STACKED_SPINE_MANIFEST_KEY] - ) == ( - manifest["stack_manifest"] + assert ( + stacked_spine_module._json_ready( + frame.metadata[stacked_spine_module.STACKED_SPINE_MANIFEST_KEY] + ) + == (manifest["stack_manifest"]) ) @@ -1465,12 +1791,10 @@ def test_ready_stacked_pool_loader_restores_sampled_rung_manifest( frame, manifest, _ = load_simulation_ready_us_multispine_pool(manifest_path) - stack_manifest = frame.metadata[ - stacked_spine_module.STACKED_SPINE_MANIFEST_KEY - ] - assert stacked_spine_module._json_ready(stack_manifest) == manifest[ - "stack_manifest" - ] + stack_manifest = frame.metadata[stacked_spine_module.STACKED_SPINE_MANIFEST_KEY] + assert ( + stacked_spine_module._json_ready(stack_manifest) == manifest["stack_manifest"] + ) assert stack_manifest["version"] == 4 assert stack_manifest["sample_fraction"] == 0.25 @@ -1575,6 +1899,61 @@ def test_ready_stacked_pool_loader_rejects_boolean_sampling_aliases( load_simulation_ready_us_multispine_pool(manifest_path) +def test_ready_stacked_pool_loader_replays_immigration_status_against_h5( + tmp_path: Path, +) -> None: + """An ordinary H5 re-hash cannot forge the sealed live immigration proof.""" + + pytest.importorskip("tables") + manifest_path = _write_ready_pool(tmp_path, stacked=True) + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + pool_path = Path(manifest["pool_h5"]["path"]) + with pd.HDFStore(pool_path, mode="a") as store: + person = h5_io.read_frame_table(store, "person") + parole_rows = person["immigration_status_str"].eq("PAROLED_ONE_YEAR") + assert parole_rows.any() + person.loc[parole_rows.idxmax(), "immigration_status_str"] = ( + "LEGAL_PERMANENT_RESIDENT" + ) + h5_io.put_frame_table( + store, + "person", + person, + preferred_format="fixed", + ) + + # Re-sign only the ordinary byte-level artifact envelope. The immutable + # late-transition receipt still describes the untampered live population. + manifest["pool_h5"]["sha256"] = _sha256(pool_path) + manifest["pool_h5"]["size_bytes"] = pool_path.stat().st_size + manifest_path.write_text(json.dumps(manifest), encoding="utf-8") + + with pytest.raises( + ValueError, + match="differs from the live ASEC/ACS weighted population", + ): + load_simulation_ready_us_multispine_pool(manifest_path) + + +def test_ready_stacked_pool_loader_requires_immigration_terminal_gate( + tmp_path: Path, +) -> None: + pytest.importorskip("tables") + manifest_path = _write_ready_pool(tmp_path, stacked=True) + manifest = json.loads(manifest_path.read_text(encoding="utf-8")) + diagnostics_path = Path(manifest["agreement_diagnostics"]["path"]) + diagnostics = json.loads(diagnostics_path.read_text(encoding="utf-8")) + for payload in (manifest, diagnostics): + payload["terminal_gates"]["gates"].pop("immigration_composition") + payload["agreement_gate"]["gates"].pop("immigration_composition") + diagnostics_path.write_text(json.dumps(diagnostics), encoding="utf-8") + manifest["agreement_diagnostics"]["sha256"] = _sha256(diagnostics_path) + manifest_path.write_text(json.dumps(manifest), encoding="utf-8") + + with pytest.raises(ValueError, match="canonical terminal gate set"): + load_simulation_ready_us_multispine_pool(manifest_path) + + def test_ready_stacked_pool_loader_binds_h5_cd_vintage_attrs( tmp_path: Path, ) -> None: @@ -1643,7 +2022,11 @@ def test_ready_stacked_pool_loader_rejects_divergent_clone_geography( tmp_path: Path, ) -> None: pytest.importorskip("tables") - manifest_path = _write_ready_pool(tmp_path, stacked=True) + manifest_path = _write_ready_pool( + tmp_path, + stacked=True, + include_zero_weight_clone=True, + ) manifest = json.loads(manifest_path.read_text(encoding="utf-8")) pool_path = Path(manifest["pool_h5"]["path"]) clone_column = support_clone_index_column("household") @@ -1669,15 +2052,12 @@ def test_scoring_pool_loader_authenticates_failed_stacked_terminal_receipt( manifest = json.loads(manifest_path.read_text(encoding="utf-8")) diagnostics_path = Path(manifest["agreement_diagnostics"]["path"]) diagnostics = json.loads(diagnostics_path.read_text(encoding="utf-8")) - failed_gate = { + failed_gate = json.loads(json.dumps(manifest["terminal_gates"])) + failed_gate["passed"] = False + failed_gate["gates"]["us_stacked_completeness"] = { "passed": False, - "gates": { - "us_spine_agreement": { - "passed": False, - "failures": ["fixture terminal failure"], - "details": {"fixture": False}, - } - }, + "failures": ["fixture terminal failure"], + "details": {"fixture": False}, } manifest.update( { @@ -1716,7 +2096,11 @@ def test_scoring_pool_loader_authenticates_failed_stacked_terminal_receipt( ) ) - assert frame.n("household") == 3 + expected_rows = 1 + sum( + draw.target > 0 + for draw in immigration_runtime.us_immigration_controls().humanitarian + ) + assert frame.n("household") == expected_rows assert loaded_manifest["status"] == "gate_failed" assert loaded_manifest["simulation_ready"] is False assert loaded_manifest["terminal_gates"]["passed"] is False @@ -1740,7 +2124,7 @@ def test_scoring_pool_loader_authenticates_failed_stacked_terminal_receipt( "failure_count": 1, "failures": [ { - "gate": "us_spine_agreement", + "gate": "us_stacked_completeness", "message": "fixture terminal failure", } ], diff --git a/packages/microcosm-build/tests/test_us_multispine_pool_tool.py b/packages/microcosm-build/tests/test_us_multispine_pool_tool.py index f4ccd147..7080f985 100644 --- a/packages/microcosm-build/tests/test_us_multispine_pool_tool.py +++ b/packages/microcosm-build/tests/test_us_multispine_pool_tool.py @@ -22,6 +22,7 @@ import pytest import microcosm.build.us_runtime.acs_transfer as acs_transfer_module +import microcosm.build.us_runtime.immigration as immigration_module import microcosm.build.us_runtime.multispine_pool as multispine_pool_module import microcosm.build.us_runtime.post_transfer_calibration as post_transfer_calibration_runtime import microcosm.build.us_runtime.stacked_spine as stacked_spine_module @@ -185,6 +186,86 @@ def _many_household_source_frame( ) +def _with_household_design_weights(frame: Frame, values: np.ndarray) -> Frame: + """Return a source fixture with replacement household design weights.""" + + weights = np.asarray(values, dtype=np.float64) + assert weights.shape == (len(frame.table("household")),) + return Frame( + {entity: frame.table(entity).copy(deep=True) for entity in frame.entities}, + frame.schema, + {"household": Weights(weights, frame.weights_for("household").kind)}, + frame.strata.copy(deep=True), + mass_log=frame.mass_log, + metadata=frame.metadata, + ) + + +def _humanitarian_weighted_source_frames( + *, + asec_count: int, + acs_count: int, + sample_fraction: float, + sample_seed: int, + state_fips: str | None = None, + puma: str | None = None, +) -> tuple[Frame, Frame]: + """Shape sampled survey weights to the live humanitarian controls. + + The stacked assembly normalizes each sampled arm back to its source mass, + then splits the ASEC anchor mass equally across ASEC and ACS. Setting the + ASEC source mass to twice the humanitarian target total and the selected + ACS weights to the individual targets therefore makes both assembled arms + exact without changing the frozen assembly manifest later in the fixture + pipeline. + """ + + controls = immigration_module.us_immigration_controls() + positive_draws = tuple(draw for draw in controls.humanitarian if draw.target > 0) + target_weights = np.asarray( + [draw.target for draw in positive_draws], + dtype=np.float64, + ) + target_mass = float(target_weights.sum()) + asec = _many_household_source_frame( + count=asec_count, + state_fips=state_fips, + ) + acs = _many_household_source_frame( + count=acs_count, + measured_offset=1_000.0, + state_fips=state_fips, + puma=puma, + ) + sampled_acs, _receipt = stacked_spine_module.sample_acs_households( + acs, + fraction=sample_fraction, + seed=sample_seed, + ) + selected_ids = sampled_acs.table("household")["household_id"].to_numpy( + dtype=np.int64 + ) + assert len(selected_ids) == len(target_weights) + + asec = _with_household_design_weights( + asec, + np.full(asec_count, 2.0 * target_mass / asec_count, dtype=np.float64), + ) + acs_household_ids = acs.table("household")["household_id"].to_numpy(dtype=np.int64) + acs_weights = np.ones(acs_count, dtype=np.float64) + id_to_position = { + int(household_id): position + for position, household_id in enumerate(acs_household_ids) + } + for household_id, target_weight in zip( + selected_ids, + target_weights, + strict=True, + ): + acs_weights[id_to_position[int(household_id)]] = target_weight + return asec, _with_household_design_weights(acs, acs_weights) + + def _replace_person( frame: Frame, person: pd.DataFrame, @@ -1222,6 +1303,283 @@ def _canonical_late_calibration_owner_receipt( return owner +_HUMANITARIAN_ACS_EVIDENCE = { + "paroled_one_year:afghanistan": (200, 2021), + "paroled_one_year:ukraine": (164, 2022), + "paroled_one_year:nicaragua": (315, 2023), + "paroled_one_year:venezuela": (373, 2022), + "refugee": (412, 2022), + "asylee": (207, 2016), + "deportation_withheld": (501, 2015), + "tps:venezuela": (373, 2015), + "tps:el_salvador": (312, 2001), + "tps:honduras": (314, 1998), + "tps:nicaragua": (315, 1998), + "tps:nepal": (229, 2015), + "tps:other_designated": (205, 2015), +} + + +def _with_reconciled_immigration_fixture( + frame: Frame, +) -> tuple[Frame, dict[str, object]]: + """Build and reconcile one live ACS household per positive manifest draw.""" + + controls = immigration_module.us_immigration_controls() + positive_draws = tuple(draw for draw in controls.humanitarian if draw.target > 0) + expected_labels = {draw.label for draw in controls.humanitarian} + assert set(_HUMANITARIAN_ACS_EVIDENCE) == expected_labels + row_count = 1 + len(positive_draws) + ids = np.arange(1, row_count + 1, dtype=np.int64) + expected_channels = {"asec": 1, "acs": len(positive_draws)} + preserve_live_lineage = all( + len(frame.table(entity)) == row_count + and support_channel_column(entity) in frame.table(entity) + and frame.table(entity)[support_channel_column(entity)] + .astype(str) + .value_counts() + .to_dict() + == expected_channels + for entity in frame.entities + ) + + tables: dict[str, pd.DataFrame] = {} + for entity in frame.entities: + source = frame.table(entity) + assert not source.empty + if preserve_live_lineage: + table = source.copy(deep=True) + else: + table = pd.concat([source.iloc[[0]]] * row_count, ignore_index=True) + table[f"{entity}_id"] = ids + table[support_channel_column(entity)] = np.asarray( + ["asec", *(["acs"] * len(positive_draws))], + dtype=object, + ) + table[support_clone_index_column(entity)] = np.zeros( + row_count, + dtype=np.int64, + ) + tables[entity] = table + + person = tables[frame.schema.person_entity] + if not preserve_live_lineage: + for entity in frame.schema.group_entities: + person[f"person_{entity}_id"] = ids + person["person_id"] = ids + person_channels = person[support_channel_column(frame.schema.person_entity)].astype( + str + ) + asec_rows = person_channels.eq("asec").to_numpy(dtype=bool) + acs_rows = person_channels.eq("acs").to_numpy(dtype=bool) + assert int(asec_rows.sum()) == 1 + assert int(acs_rows.sum()) == len(positive_draws) + # Keep legacy TPS arrivals outside the DACA age-at-entry cohort so each + # row remains eligible for the single humanitarian draw it represents. + person["A_AGE"] = np.full(row_count, 60.0) + if "age" in person: + person["age"] = np.full(row_count, 60.0) + if "source_year" in person and not preserve_live_lineage: + person["source_year"] = np.full(row_count, 2024, dtype=np.int64) + if "source_household_id" in person and not preserve_live_lineage: + person["source_household_id"] = ids + if "source_person_id" in person and not preserve_live_lineage: + person["source_person_id"] = ids + if "PERIDNUM" in person and not preserve_live_lineage: + person["PERIDNUM"] = pd.Series(ids.astype(str), dtype="string") + + acs_evidence = [_HUMANITARIAN_ACS_EVIDENCE[draw.label] for draw in positive_draws] + for column in ("PRCITSHP", "PENATVTY", "PEINUSYR", "CIT", "POBP", "YOEP"): + person[column] = np.full(row_count, np.nan, dtype=np.float64) + person.loc[asec_rows, ["PRCITSHP", "PENATVTY", "PEINUSYR"]] = ( + 1.0, + 57.0, + 0.0, + ) + person.loc[acs_rows, "CIT"] = 5.0 + person.loc[acs_rows, "POBP"] = [item[0] for item in acs_evidence] + person.loc[acs_rows, "YOEP"] = [item[1] for item in acs_evidence] + person["ssn_card_type"] = pd.Series( + np.where(asec_rows, "CITIZEN", "OTHER_NON_CITIZEN"), + index=person.index, + dtype="string", + ) + person["immigration_status_str"] = pd.Series( + np.where(asec_rows, "CITIZEN", "LEGAL_PERMANENT_RESIDENT"), + index=person.index, + dtype="string", + ) + + household = tables["household"] + existing_household_weights = pd.Series( + np.asarray(frame.weights_for("household").values, dtype=np.float64), + index=frame.table("household")["household_id"].to_numpy(dtype=np.int64), + ) + asec_household_id = int(person.loc[asec_rows, "person_household_id"].iloc[0]) + household_id_to_weight = { + asec_household_id: ( + float(existing_household_weights.loc[asec_household_id]) + if preserve_live_lineage + else 1.0 + ), + **{ + int(household_id): float(draw.target) + for household_id, draw in zip( + person.loc[acs_rows, "person_household_id"], + positive_draws, + strict=True, + ) + }, + } + household_weights = household["household_id"].map(household_id_to_weight) + assert household_weights.notna().all() + assert tuple(frame.weighted_entities) == ("household",) + weights = { + "household": Weights( + household_weights.to_numpy(dtype=np.float64), + frame.weights_for("household").kind, + ) + } + source_strata = np.asarray(frame.strata, dtype=object) + strata = ( + frame.strata + if preserve_live_lineage + else pd.Series(np.resize(source_strata, row_count), dtype=object) + ) + metadata = dict(frame.metadata) + assembly_key = stacked_spine_module.SPINE_ASSEMBLY_MANIFEST_KEY + assembly = metadata.get(assembly_key) + if not preserve_live_lineage and isinstance(assembly, Mapping): + assembly = copy.deepcopy(dict(assembly)) + declared_channels = tuple(assembly["channels"]) + assembly["native_row_counts"] = { + entity: { + channel: int(table[support_channel_column(entity)].eq(channel).sum()) + for channel in declared_channels + } + for entity, table in tables.items() + } + metadata[assembly_key] = assembly + stacked_key = stacked_spine_module.STACKED_SPINE_MANIFEST_KEY + stacked = metadata.get(stacked_key) + if not preserve_live_lineage and isinstance(stacked, Mapping): + stacked = copy.deepcopy(dict(stacked)) + household_channels = household[support_channel_column("household")].astype(str) + live_weights = household_weights.to_numpy(dtype=np.float64) + live_masses = { + channel: float( + live_weights[household_channels.eq(channel).to_numpy()].sum() + ) + for channel in ("asec", "acs") + } + total_mass = float(live_weights.sum()) + shares = { + channel: live_masses[channel] / total_mass for channel in ("asec", "acs") + } + sample_receipts = { + channel: dict(sample) + for channel, sample in stacked["survey_samples"].items() + } + mass_anchor = str(stacked["mass_anchor_channel"]) + normalized_masses = { + channel: total_mass if channel == mass_anchor else live_masses[channel] + for channel in ("asec", "acs") + } + for channel, sample in sample_receipts.items(): + sampled_mass = float(sample["sampled_household_mass"]) + normalized_mass = normalized_masses[channel] + sample.update( + { + "incoming_household_mass": normalized_mass, + "normalization_factor": normalized_mass / sampled_mass, + "normalized_household_mass": normalized_mass, + } + ) + stacked["survey_samples"] = sample_receipts + stacked["household_mass_shares"] = shares + harmonization = { + channel: dict(arm) + for channel, arm in stacked["weight_harmonization"].items() + } + for channel, arm in harmonization.items(): + incoming_mass = normalized_masses[channel] + allocated_mass = live_masses[channel] + arm.update( + { + "share": shares[channel], + "incoming_mass": incoming_mass, + "allocated_mass": allocated_mass, + "declared_allocation": shares[channel] * total_mass, + "scale_factor": allocated_mass / incoming_mass, + } + ) + stacked["weight_harmonization"] = harmonization + metadata[stacked_key] = stacked + reconciled_frame = Frame( + tables, + frame.schema, + weights, + strata, + mass_log=frame.mass_log, + metadata=metadata, + ) + reconciled_person, receipt = ( + immigration_module.reconcile_us_immigration_humanitarian_transfer( + reconciled_frame.table(frame.schema.person_entity), + weights=np.asarray( + reconciled_frame.resolve_weights(frame.schema.person_entity).values, + dtype=np.float64, + ), + mutable_rows=acs_rows, + seed=0, + ) + ) + return _replace_person(reconciled_frame, reconciled_person), receipt + + +def _immigration_fixture_qrf_evidence( + *, + target: str, + family_targets: tuple[str, ...], + donor_rows: int, + recipient_rows: int, +) -> dict[str, object]: + """Return canonical joint-codec QRF evidence for one paired output leaf.""" + + required_predictors, _optional_predictors = ( + stacked_spine_module._acs_pattern_predictor_authority( + entity="person", + family_targets=family_targets, + ) + ) + pattern = acs_transfer_module.AcsTransferPattern( + name=acs_transfer_module._pattern_name(0, ()), + observed_optional_predictors=(), + predictors=required_predictors, + seed=0, + weight_kind=WeightKind.DESIGN.value, + donor_rows=donor_rows, + recipient_rows=recipient_rows, + target_regimes=tuple( + (model_target, "positive_only") + for model_target in acs_transfer_module._model_target_names(family_targets) + ), + ) + record = acs_transfer_module.AcsImputedInput( + column=target, + entity="person", + family="source_operator_immigration", + donor_spine="synthetic_pool_tool_fixture", + donor_channel="asec", + predictors=required_predictors, + seed=0, + weight_kind=WeightKind.DESIGN.value, + patterns=(pattern,), + imputed_recipient_rows=recipient_rows, + ) + return stacked_spine_module._acs_imputed_pattern_evidence(record) + + def _canonical_pregnancy_structural_receipt() -> dict[str, object]: policy = acs_transfer_module.acs_transfer_execution_contract_identity( targets=("is_pregnant",), @@ -1256,7 +1614,12 @@ def _canonical_late_transfer_receipt( *, authority: Mapping[str, object] | None = None, frame: Frame | None = None, + immigration_reconciliation: Mapping[str, object] | None = None, ) -> dict[str, object]: + if immigration_reconciliation is None: + _fixture_frame, immigration_reconciliation = ( + _with_reconciled_immigration_fixture(_source_frame()) + ) canonical_family = { (entity, target): family for entity, families in ( @@ -1278,12 +1641,36 @@ def _canonical_late_transfer_receipt( ] ) for group in pool_tool.CANONICAL_US_LATE_TRANSFER_GROUPS: + is_immigration_group = ( + group.entity == "person" and group.family == "source_operator_immigration" + ) + immigration_mutable_rows = int(immigration_reconciliation["mutable_rows"]) + immigration_immutable_rows = int(immigration_reconciliation["immutable_rows"]) group_targets = { f"{group.entity}/{group.family}/{target}": { - "authorized_null_rows": 0, - "imputed_rows": 0, + "authorized_null_rows": immigration_mutable_rows + if is_immigration_group + else 0, + "imputed_rows": immigration_mutable_rows if is_immigration_group else 0, "unmodeled_rows": 0, "residual_null_rows": 0, + **( + { + "qrf_pattern_evidence": ( + _immigration_fixture_qrf_evidence( + target=target, + family_targets=group.targets, + donor_rows=immigration_immutable_rows, + recipient_rows=immigration_mutable_rows, + ) + ), + "post_transfer_reconciliation": copy.deepcopy( + immigration_reconciliation + ), + } + if is_immigration_group + else {} + ), } for target in group.targets } @@ -1346,6 +1733,7 @@ def _canonical_late_dag_receipt( authority: Mapping[str, object] | None = None, output_frame_sha256: str = "f" * 64, frame: Frame | None = None, + immigration_reconciliation: Mapping[str, object] | None = None, ) -> dict[str, object]: schedule = stacked_spine_module.CANONICAL_US_LATE_PRODUCER_SCHEDULE schedule_receipt = pool_tool._json_ready( @@ -1389,6 +1777,7 @@ def _canonical_late_dag_receipt( pool_tool, authority=authority, frame=frame, + immigration_reconciliation=immigration_reconciliation, ) input_frame_sha256 = "e" * 64 previous_sha256 = stacked_spine_module._late_execution_genesis_sha256( @@ -1610,8 +1999,9 @@ def _authorized_late_impute_fixture( *, authority: Mapping[str, object] | None = None, ) -> tuple[Frame, dict[str, object], str]: - """Bind one structurally signed synthetic DAG proof to a live fixture frame.""" + """Bind one structurally signed fixture DAG proof to its live frame.""" + frame, immigration_reconciliation = _with_reconciled_immigration_fixture(frame) tables = {entity: frame.table(entity).copy(deep=True) for entity in frame.entities} for entity, table in tables.items(): if support_channel_column(entity) not in table: @@ -1656,6 +2046,7 @@ def _authorized_late_impute_fixture( authority=authority, output_frame_sha256=stacked_spine_module._late_frame_content_sha256(frame), frame=frame, + immigration_reconciliation=immigration_reconciliation, ) authorized, transition_authority_sha256 = ( stacked_spine_module._bind_late_producer_transition_authority(frame, dag) @@ -1686,15 +2077,17 @@ def _install_stacked_entrypoint_stubs( verified = _verified_inputs_fixture(pool_tool, tmp_path / "pins") source_manifest = pool_tool.load_acs_source_manifest() puf_donor = pd.DataFrame({"fixture": np.arange(7)}) + asec_source, acs_source = _humanitarian_weighted_source_frames( + asec_count=100, + acs_count=1_200, + sample_fraction=0.01, + sample_seed=578, + state_fips="06" if real_geography_assignment else None, + puma="0600100" if real_geography_assignment else None, + ) loaded = pool_tool._LoadedInputs( - asec=_many_household_source_frame( - state_fips="06" if real_geography_assignment else None, - ), - acs=_many_household_source_frame( - measured_offset=1_000.0, - state_fips="06" if real_geography_assignment else None, - puma="0600100" if real_geography_assignment else None, - ), + asec=asec_source, + acs=acs_source, acs_rent_donor=pd.DataFrame({"fixture": [1.0]}), puf_donor=puf_donor, asec_raw_stage_checkpoint={"artifact": "fixture-raw-stage"}, @@ -1835,8 +2228,8 @@ def gap_fill(frame: Frame, **kwargs): ) assert counts == { ("person", "strike_benefits"): { - "authorized_null_rows": 1, - "recipient_rows": 1, + "authorized_null_rows": 12, + "recipient_rows": 12, "donor_rows": 1, } } @@ -1947,9 +2340,12 @@ def late_producer_dag(frame: Frame, **kwargs: object): "family": group.family, "ordered_targets": list(group.targets), } + late_base, immigration_reconciliation = _with_reconciled_immigration_fixture( + primary_puf_result.frame + ) late_tables = { - entity: primary_puf_result.frame.table(entity).copy(deep=True) - for entity in primary_puf_result.frame.entities + entity: late_base.table(entity).copy(deep=True) + for entity in late_base.entities } for ( spec @@ -1977,22 +2373,17 @@ def late_producer_dag(frame: Frame, **kwargs: object): index=late_person.index, dtype="string", ) - late_tables.update( - { - name: primary_puf_result.frame.link(name) - for name in primary_puf_result.frame.links - } - ) + late_tables.update({name: late_base.link(name) for name in late_base.links}) late_frame = Frame( late_tables, - primary_puf_result.frame.schema, + late_base.schema, { - entity: primary_puf_result.frame.weights_for(entity) - for entity in primary_puf_result.frame.weighted_entities + entity: late_base.weights_for(entity) + for entity in late_base.weighted_entities }, - primary_puf_result.frame.strata, - mass_log=primary_puf_result.frame.mass_log, - metadata=primary_puf_result.frame.metadata, + late_base.strata, + mass_log=late_base.mass_log, + metadata=late_base.metadata, ) dag_receipt = _canonical_late_dag_receipt( pool_tool, @@ -2001,6 +2392,7 @@ def late_producer_dag(frame: Frame, **kwargs: object): late_frame ), frame=late_frame, + immigration_reconciliation=immigration_reconciliation, ) authorized_frame, transition_authority_sha256 = ( stacked_spine_module._bind_late_producer_transition_authority( @@ -2092,6 +2484,18 @@ def battery( monkeypatch.setattr(pool_tool, "stacked_completeness_gate", completeness) monkeypatch.setattr(pool_tool, "by_origin_battery", battery) + + def immigration(_frame: Frame) -> GateResult: + order.append("immigration") + if terminal == "immigration_red": + return GateResult( + name="fixture_immigration", + passed=False, + failures=("fixture immigration terminal failure",), + ) + return GateResult(name="fixture_immigration", passed=True) + + monkeypatch.setattr(pool_tool, "us_immigration_composition_gate", immigration) real_publish = pool_tool._write_stacked_outputs def publish(*args, **kwargs): @@ -2152,6 +2556,7 @@ def test_stacked_operator_target_requires_preparation_before_activation_authorit [ ("success", 0, "iterating"), ("red", 1, "failed"), + ("immigration_red", 1, "failed"), ("error", None, "failed"), ], ) @@ -2214,6 +2619,7 @@ def test_stacked_tool_entrypoint_fixture_e2e_emits_one_logbook_row_at_every_term "simulate", "completeness", "battery", + "immigration", "publish", ] # Exported rows must never embed host-absolute paths; pytest tmp @@ -2232,6 +2638,28 @@ def test_stacked_tool_entrypoint_fixture_e2e_emits_one_logbook_row_at_every_term manifest = json.loads( (tmp_path / "stacked-pool.manifest.json").read_text(encoding="utf-8") ) + if terminal == "immigration_red": + assert manifest["status"] == "gate_failed" + assert manifest["simulation_ready"] is False + assert manifest["terminal_gates"]["passed"] is False + assert ( + manifest["terminal_gates"]["gates"]["fixture_battery"]["passed"] is True + ) + assert manifest["terminal_gates"]["gates"]["fixture_immigration"] == { + "passed": False, + "failures": ["fixture immigration terminal failure"], + "details": {}, + } + assert row.gate_verdicts["fixture_battery"]["verdict"] == "passed" + assert row.gate_verdicts["fixture_immigration"]["verdict"] == "failed" + immigration_receipt = json.loads( + _receipt_file_from_reference( + row.gate_verdicts["fixture_immigration"]["receipt"] + ).read_text(encoding="utf-8") + ) + assert immigration_receipt["terminal_gates"]["gates"][ + "fixture_immigration" + ]["failures"] == ["fixture immigration terminal failure"] assert manifest["schema_version"] == pool_tool.POOL_MANIFEST_SCHEMA_VERSION assert manifest["pipeline"] == "us-stacked-pool" assert manifest["pool_h5"]["materializer_version"] == ( @@ -2300,13 +2728,14 @@ def test_stacked_tool_entrypoint_fixture_e2e_emits_one_logbook_row_at_every_term "materialize_multispine_agreement_outputs", "stacked_completeness_gate", "by_origin_battery", + "us_immigration_composition_gate", ] assert manifest["sampling"] == { **manifest["sampling"], "sample_fraction": 0.01, "fraction_token": "f001", "sample_seed": 578, - "realized_households": {"asec": 1, "acs": 1}, + "realized_households": {"asec": 1, "acs": 12}, } @@ -2499,7 +2928,7 @@ def capture_equality(expected: object, actual: object) -> None: "country": "us", "schema_id": "country_spec", "schema_version": 1, - "spec_sha256": "835b4d61de3b153b13c536e371dca6a16fad6d41e003d070f9804c7c251f3590", + "spec_sha256": "ae83c7a32a4a2970070f71b95aa7de771a56ab435f71004c11ee2e6b181fe07b", }, } @@ -2702,7 +3131,7 @@ def test_constants_adapter_fixture_checkpoints_are_byte_identical_and_only_recei "country": "us", "schema_id": "country_spec", "schema_version": 1, - "spec_sha256": "5378bb9189aec96f50da22aac71e5bd2c3d919e9795f6ef2147e0bc9c739dd8e", + "spec_sha256": "ea202eebdfe30a68323af5cdce6abcc16b94c9f3f18f44ae5369fe7c1d9a5702", } def run_fixture(root: Path, *, config_authority: str) -> dict[str, object]: @@ -3246,6 +3675,7 @@ def fail_publication(*_args, **_kwargs) -> None: assert set(row.gate_verdicts) == { "fixture_completeness", "fixture_battery", + "fixture_immigration", "pipeline_error", } terminal_path = _receipt_file_from_reference( @@ -3518,7 +3948,7 @@ def test_legacy_checkpoint_identity_excludes_stacked_late_producer_schedule( assert changed == current -def test_stacked_checkpoint_identity_binds_v12_semantic_contracts( +def test_stacked_checkpoint_identity_binds_v13_semantic_contracts( pool_tool: ModuleType, monkeypatch: pytest.MonkeyPatch, tmp_path: Path, @@ -3551,7 +3981,7 @@ def identity() -> dict[str, object]: current = identity() pool_code = current["pool_code"] - assert current["materializer_version"] == 12 + assert current["materializer_version"] == 13 assert current["stacked_authority"]["version"] == 11 assert current["geography_assignment"] == ( pool_tool._stacked_geography_assignment_contract() @@ -3574,6 +4004,7 @@ def identity() -> dict[str, object]: "materialize_multispine_agreement_outputs", "stacked_completeness_gate", "by_origin_battery", + "us_immigration_composition_gate", ] assert pool_code["late_producer_schedule"] == pool_tool._json_ready( pool_tool.us_late_producer_schedule_receipt() @@ -3779,7 +4210,7 @@ def changed_source_stage_binding( ) ) - assert current["materializer_version"] == stale_qrf["materializer_version"] == 12 + assert current["materializer_version"] == stale_qrf["materializer_version"] == 13 assert stale_qrf["pool_code"]["primary_qrf_checkpoint_schema_version"] == 5 assert ( pool_tool._discover_stacked_checkpoint_identity( @@ -3827,7 +4258,7 @@ def changed_source_stage_binding( assert "checkpoint base identity is stale" in capsys.readouterr().out -def test_pool_envelope_v7_preserves_stacked_bank_identity_but_rejects_v6( +def test_pool_envelope_v8_preserves_stacked_bank_identity_but_rejects_v7( pool_tool: ModuleType, monkeypatch: pytest.MonkeyPatch, tmp_path: Path, @@ -3860,7 +4291,7 @@ def identity() -> dict[str, object]: legacy.setattr( pool_tool, "POOL_STAGE_CHECKPOINT_MATERIALIZER_VERSION", - 6, + 7, ) assert identity() == current_identity legacy_store = pool_tool._PoolStageCheckpointStore( @@ -3881,7 +4312,7 @@ def identity() -> dict[str, object]: ) capsys.readouterr() - assert pool_tool.POOL_STAGE_CHECKPOINT_MATERIALIZER_VERSION == 7 + assert pool_tool.POOL_STAGE_CHECKPOINT_MATERIALIZER_VERSION == 8 assert identity() == current_identity current_store = pool_tool._PoolStageCheckpointStore( checkpoint_root, @@ -4034,7 +4465,7 @@ def test_qbi_receipt_route_resolution_rejects_wrong_or_ambiguous_paths( ) -@pytest.mark.parametrize("legacy_version", (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11)) +@pytest.mark.parametrize("legacy_version", (1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12)) def test_legacy_stacked_materializer_checkpoint_is_not_discovered( pool_tool: ModuleType, monkeypatch: pytest.MonkeyPatch, @@ -4088,7 +4519,7 @@ def test_legacy_stacked_materializer_checkpoint_is_not_discovered( ) ) - assert pool_tool._STACKED_CHECKPOINT_MATERIALIZER_VERSION == 12 + assert pool_tool._STACKED_CHECKPOINT_MATERIALIZER_VERSION == 13 assert ( pool_tool._discover_stacked_checkpoint_identity( checkpoint_root, @@ -4107,9 +4538,15 @@ def test_stacked_resume_rejects_noncanonical_post_puf_transfer_receipt( pool_tool: ModuleType, tmp_path: Path, ) -> None: + asec_source, acs_source = _humanitarian_weighted_source_frames( + asec_count=10, + acs_count=120, + sample_fraction=0.10, + sample_seed=578, + ) stack = pool_tool.assemble_stacked_spine( - _many_household_source_frame(), - _many_household_source_frame(measured_offset=1_000.0), + asec_source, + acs_source, sample_fraction=0.10, sample_seed=578, ) @@ -4230,12 +4667,14 @@ def test_stacked_entrypoint_resumes_each_checkpoint_boundary( "simulate", "completeness", "battery", + "immigration", "publish", ] assert simulated_resume_order == [ "build_stacked_pool", "completeness", "battery", + "immigration", "publish", ] assert transferred_resume_order == [ @@ -4246,6 +4685,7 @@ def test_stacked_entrypoint_resumes_each_checkpoint_boundary( "simulate", "completeness", "battery", + "immigration", "publish", ] assert assembled_resume_order == [ @@ -4260,6 +4700,7 @@ def test_stacked_entrypoint_resumes_each_checkpoint_boundary( "simulate", "completeness", "battery", + "immigration", "publish", ] final_rows = [ @@ -4492,8 +4933,8 @@ def deterministic_fixture_h5( outputs = pool_tool._output_paths(output, checkpoint_root=checkpoint_root) manifest = pool_tool._read_json_object(outputs.manifest) diagnostics = pool_tool._read_json_object(outputs.agreement_diagnostics) - assert pool_tool.POOL_MANIFEST_SCHEMA_VERSION == 9 - assert pool_tool.POOL_STAGE_CHECKPOINT_MATERIALIZER_VERSION == 7 + assert pool_tool.POOL_MANIFEST_SCHEMA_VERSION == 10 + assert pool_tool.POOL_STAGE_CHECKPOINT_MATERIALIZER_VERSION == 8 assert manifest["schema_version"] == 4 assert diagnostics["schema_version"] == 4 assert "materializer_version" not in manifest["pool_h5"] @@ -5802,7 +6243,7 @@ def test_pool_checkpoint_store_round_trips_nullable_boolean_families( manifest = pool_tool._read_json_object( cold_store.checkpoint_manifest_path(stage) ) - assert manifest["materializer_version"] == 7 + assert manifest["materializer_version"] == 8 loaded = pool_tool.load_frame_checkpoint(path).frame if stage == "assembled": assert "fixture_declared_boolean" not in loaded.person @@ -5823,11 +6264,11 @@ def test_pool_checkpoint_store_round_trips_nullable_boolean_families( assert resumed.frame.person["fixture_declared_boolean"].isna().sum() == 1 -def test_simulated_v7_checkpoint_accepts_both_string_encodings_without_rewrite( +def test_simulated_v8_checkpoint_accepts_both_string_encodings_without_rewrite( pool_tool: ModuleType, tmp_path: Path, ) -> None: - """V7 authenticates both physical string encodings as one logical frame.""" + """V8 authenticates both physical string encodings as one logical frame.""" pytest.importorskip("h5py") checkpoint_root = tmp_path / "checkpoints" @@ -5839,7 +6280,7 @@ def test_simulated_v7_checkpoint_accepts_both_string_encodings_without_rewrite( loaded = pool_tool.load_frame_checkpoint(checkpoint_path) canonical_v2_bytes = checkpoint_path.read_bytes() canonical_identity = loaded.metadata["identity"] - assert loaded.metadata["materializer_version"] == 7 + assert loaded.metadata["materializer_version"] == 8 assert any( column["dtype"] == str(CANONICAL_STRING_DTYPE) for columns in loaded.metadata["frame_schema"]["entities"].values() @@ -5870,7 +6311,7 @@ def test_simulated_v7_checkpoint_accepts_both_string_encodings_without_rewrite( banked_v2_bytes = checkpoint_path.read_bytes() assert banked_v2_bytes != canonical_v2_bytes assert legacy_metadata["identity"] == canonical_identity - assert legacy_metadata["materializer_version"] == 7 + assert legacy_metadata["materializer_version"] == 8 assert any( column["dtype"] == "object" for columns in legacy_metadata["frame_schema"]["entities"].values() @@ -6245,7 +6686,7 @@ def test_tail_support_contract_identity_mutation_rebuilds_pool_checkpoints( assert changed_store.load_deepest() is None -@pytest.mark.parametrize("legacy_version", (1, 2, 3, 4, 5, 6)) +@pytest.mark.parametrize("legacy_version", (1, 2, 3, 4, 5, 6, 7)) def test_legacy_pool_materializer_artifacts_fail_closed_with_named_receipts( pool_tool: ModuleType, monkeypatch: pytest.MonkeyPatch, @@ -6278,9 +6719,9 @@ def test_legacy_pool_materializer_artifacts_fail_closed_with_named_receipts( assert manifest["identity"]["materializer_version"] == legacy_version capsys.readouterr() - assert pool_tool.POOL_STAGE_CHECKPOINT_MATERIALIZER_VERSION == 7 + assert pool_tool.POOL_STAGE_CHECKPOINT_MATERIALIZER_VERSION == 8 current_store = _checkpoint_fixture_store(pool_tool, checkpoint_root) - assert current_store.base_identity["materializer_version"] == 7 + assert current_store.base_identity["materializer_version"] == 8 assert current_store.load_deepest() is None output = capsys.readouterr().out diff --git a/packages/microcosm-build/tests/test_us_plan.py b/packages/microcosm-build/tests/test_us_plan.py index b893f8cb..309ab0e8 100644 --- a/packages/microcosm-build/tests/test_us_plan.py +++ b/packages/microcosm-build/tests/test_us_plan.py @@ -1255,8 +1255,8 @@ class TestBaseStageSourceClosure: #: Raw columns restored from pinned official sidecars because the frozen #: census_cps inputs never carried them (LKWEEKS only for income year - #: 2022; ED_VAL and PAW_TYP for every pooled year). - SIDECAR_RESTORED_COLUMNS = frozenset({"LKWEEKS", "ED_VAL", "PAW_TYP"}) + #: 2022; ED_VAL, A_LFSR, and PAW_TYP for every pooled year). + SIDECAR_RESTORED_COLUMNS = frozenset({"LKWEEKS", "ED_VAL", "A_LFSR", "PAW_TYP"}) #: Release-time stage constants whose inputs are produced inside the #: fiscal-refresh release tool, not the base builder (org_wages consumes diff --git a/packages/microcosm-build/tests/test_us_puf_support_base_builder.py b/packages/microcosm-build/tests/test_us_puf_support_base_builder.py index 7f2d44dd..0884dd0e 100644 --- a/packages/microcosm-build/tests/test_us_puf_support_base_builder.py +++ b/packages/microcosm-build/tests/test_us_puf_support_base_builder.py @@ -231,6 +231,7 @@ def _education_source() -> pd.DataFrame: "A_LINENO": [1, 2, 1], "PERIDNUM": [f"{value:022d}" for value in (1, 2, 3)], "ED_VAL": [0.0, 500.0, 1_000.0], + "A_LFSR": [1, 3, 7], } ) source.attrs["source_audit"] = {2022: {"rows": 3}} @@ -830,11 +831,13 @@ def test_raw_stage_copy_adds_only_exact_source_mappings( assert frame_identity(source) == before assert "LKWEEKS" not in source.table("person") assert "ED_VAL" not in source.table("person") + assert "A_LFSR" not in source.table("person") assert "PAW_TYP" not in source.table("person") assert raw.table("person")["LKWEEKS"].tolist() == [7.0, -1.0, 12.0] assert raw.table("person")["ED_VAL"].tolist() == [0.0, 500.0, 1_000.0] + assert raw.table("person")["A_LFSR"].tolist() == [1, 3, 7] assert raw.table("person")["PAW_TYP"].tolist() == [0, 1, 2] - assert set(mappings) == {"ED_VAL", "LKWEEKS", "PAW_TYP"} + assert set(mappings) == {"A_LFSR", "ED_VAL", "LKWEEKS", "PAW_TYP"} assert all( mapping["operation"] == "exact_source_join" and mapping["join_keys"] == ["source_year", "PERIDNUM"] @@ -843,6 +846,49 @@ def test_raw_stage_copy_adds_only_exact_source_mappings( builder.assert_operator_free_source_frame(raw, label="raw-stage fixture") +def test_direct_builder_attaches_measured_labor_force_status_without_mutation() -> None: + builder = _load_support_builder_module() + source = _raw_asec_frame() + before = frame_identity(source) + + attached = builder._with_asec_labor_force_status_source( + source, + _education_source(), + ) + + assert frame_identity(source) == before + assert "A_LFSR" not in source.table("person") + assert attached.table("person")["A_LFSR"].tolist() == [1, 3, 7] + assert np.array_equal( + attached.weights_for("household").values, + source.weights_for("household").values, + ) + + +def test_direct_immigration_path_uses_labor_force_status_ephemerally( + monkeypatch: pytest.MonkeyPatch, +) -> None: + builder = _load_support_builder_module() + source = _raw_asec_frame() + observed: list[list[int]] = [] + + def immigration(frame: Frame, *, seed: int, time_period: int) -> Frame: + observed.append(frame.table("person")["A_LFSR"].tolist()) + return frame + + monkeypatch.setattr(builder, "with_us_immigration_inputs", immigration) + result = builder._with_us_immigration_inputs_from_asec_source( + source, + _education_source(), + seed=7, + time_period=2024, + ) + + assert observed == [[1, 3, 7]] + assert "A_LFSR" not in source.table("person") + assert "A_LFSR" not in result.table("person") + + def test_pooled_source_stage_dual_exports_without_changing_legacy_checkpoints( monkeypatch: pytest.MonkeyPatch, tmp_path: Path, @@ -891,10 +937,14 @@ def add_age(frame: Frame, **_kwargs) -> Frame: "with_us_energy_subsidy_input", "with_us_retirement_contribution_inputs", "with_us_retirement_distribution_inputs", - "with_us_immigration_inputs", ) for name in identity_transforms: monkeypatch.setattr(builder, name, lambda value, **_kwargs: value) + monkeypatch.setattr( + builder, + "_with_us_immigration_inputs_from_asec_source", + lambda value, _source, **_kwargs: value, + ) passing_gate = SimpleNamespace(passed=True, failures=(), details={}) for name in ( "us_relationship_inputs_signal_gate", @@ -917,6 +967,7 @@ def add_age(frame: Frame, **_kwargs) -> Frame: assert source_checkpoint.path.read_bytes() == baseline_source.path.read_bytes() assert "ED_VAL" not in source_checkpoint.frame.table("person") assert "LKWEEKS" not in source_checkpoint.frame.table("person") + assert "A_LFSR" not in source_checkpoint.frame.table("person") assert "PAW_TYP" not in source_checkpoint.frame.table("person") raw_path = args.checkpoint_dir / builder.ASEC_RAW_STAGE_CHECKPOINT_FILENAME @@ -924,6 +975,7 @@ def add_age(frame: Frame, **_kwargs) -> Frame: assert raw_metadata["stage"] == "raw_source_mapping" assert raw.table("person")["LKWEEKS"].tolist() == [7.0, -1.0, 12.0] assert raw.table("person")["ED_VAL"].tolist() == [0.0, 500.0, 1_000.0] + assert raw.table("person")["A_LFSR"].tolist() == [1, 3, 7] assert raw.table("person")["PAW_TYP"].tolist() == [0, 1, 2] assert "age" not in raw.table("person") assert [path.name for path in args.checkpoint_dir.glob("*.frame.h5")] == [ @@ -938,6 +990,7 @@ def add_age(frame: Frame, **_kwargs) -> Frame: assert enriched.table("person")["age"].tolist() == [31, 29, 50] assert "ED_VAL" not in enriched.table("person") assert "LKWEEKS" not in enriched.table("person") + assert "A_LFSR" not in enriched.table("person") assert "PAW_TYP" not in enriched.table("person") assert sorted(path.name for path in args.checkpoint_dir.glob("*.frame.h5")) == [ "000_source_construction.frame.h5", @@ -975,6 +1028,7 @@ def test_completed_source_stage_repairs_raw_auxiliary_without_rewriting_legacy( assert context_path.read_bytes() == expected_context repaired, _metadata = builder.load_asec_raw_stage_checkpoint(raw_path) assert repaired.table("person")["ED_VAL"].tolist() == [0.0, 500.0, 1_000.0] + assert repaired.table("person")["A_LFSR"].tolist() == [1, 3, 7] assert repaired.table("person")["PAW_TYP"].tolist() == [0, 1, 2] @@ -1030,10 +1084,14 @@ def test_source_and_preclone_stages_round_trip_design_weight_kind( "with_us_energy_subsidy_input", "with_us_retirement_contribution_inputs", "with_us_retirement_distribution_inputs", - "with_us_immigration_inputs", ) for name in identity_transforms: monkeypatch.setattr(builder, name, lambda value, **_kwargs: value) + monkeypatch.setattr( + builder, + "_with_us_immigration_inputs_from_asec_source", + lambda value, _source, **_kwargs: value, + ) passing_gate = SimpleNamespace(passed=True, failures=(), details={}) for name in ( "us_relationship_inputs_signal_gate", @@ -2185,8 +2243,8 @@ def fake_retirement_distributions( ) monkeypatch.setattr( builder, - "with_us_immigration_inputs", - lambda frame, *, seed, time_period: frame, + "_with_us_immigration_inputs_from_asec_source", + lambda frame, _source, *, seed, time_period: frame, ) monkeypatch.setattr( builder, diff --git a/packages/microcosm-build/tests/test_us_spec_bundle.py b/packages/microcosm-build/tests/test_us_spec_bundle.py index 892ece52..915a7e21 100644 --- a/packages/microcosm-build/tests/test_us_spec_bundle.py +++ b/packages/microcosm-build/tests/test_us_spec_bundle.py @@ -190,7 +190,7 @@ def _load_generator_module(): LEGACY_COMPATIBILITY_SHA256 = { "source_stages.json": ( - "dc58a0d700f0add7b658cec774df6e9587303beb58a1f432a35a18dcd1ac4097" + "788b815f748abb2061c41efb0cec4cc4952435dbcb4448ecf21ccac85a5ccec2" ), "support_spine.json": ( "68f37dc6ae6e0cde7ebccb53f88dd4a800e63456f838fa214ff98d1db8d815be" @@ -382,11 +382,11 @@ def test_constant_derived_domain_counts_are_complete( assert len(compiled_schedule["waves"]) == 6 assert ( compiled_schedule["schedule_sha256"] - == "e59c019d3d454eac99ac0ac209b6c5b6faaf9bdfcaeee18c36a25be19bf7da2f" + == "88bc9243a3518982ae951c3de21bd55877e296ce4fcb183b9bee420d3a684b10" ) assert ( compiled_schedule["payload_sha256"] - == "02e618cc656eb39990ed99dca2b30a52794e01e2b06a3c2df87ca4a7d85ab086" + == "324b8e495920af091cb7c14f78b461b2a10fb3fedeb4ce706d68ae6b4c2e3b6b" ) assert len(take_up["programs"]) == 17 @@ -930,7 +930,7 @@ def test_legacy_seed_vintage_and_publication_grammars_are_pinned( "identity_generation": 1, "seed_protocol": LEGACY_V1_PROTOCOL.id, } - assert len(LEGACY_V1_PROTOCOL.sites) == 53 + assert len(LEGACY_V1_PROTOCOL.sites) == 66 assert len(LEGACY_V1_PROTOCOL.streams) == 14 assert LEGACY_V1_PROTOCOL.site("survey_sample_asec").default == 578 assert LEGACY_V1_PROTOCOL.site("puf_live_aggregate_disaggregation").default == 0 @@ -949,12 +949,8 @@ def test_legacy_seed_vintage_and_publication_grammars_are_pinned( assert geography["assignment"]["anchor"] == "puma" assert geography["assignment"]["order"] == "before_gap_fill" assert geography["assignment"]["assign_tract"] is False - assert geography["assignment"][ - "congressional_district_vintage_crosswalk" - ] == { - "source_ref": ( - "source:us_congressional_district_vintage_crosswalk_117_to_119" - ), + assert geography["assignment"]["congressional_district_vintage_crosswalk"] == { + "source_ref": ("source:us_congressional_district_vintage_crosswalk_117_to_119"), "source_vintage": "vintage:cd_117", "target_vintage": "vintage:cd_119", } @@ -984,9 +980,12 @@ def test_legacy_seed_vintage_and_publication_grammars_are_pinned( assert engine_version_occurrences == { kind: int(kind == "take_up") for kind in TYPED_DOMAIN_KINDS } - assert take_up["legacy_contract_metadata"]["asserted_engine"][ - "inventory_built_against" - ] == engine_version + assert ( + take_up["legacy_contract_metadata"]["asserted_engine"][ + "inventory_built_against" + ] + == engine_version + ) assert _count_scalar(engine_lock, engine_version) == 1 resolved_vintages = thaw_json(resolved_us_spec.vintage_authorities) assert resolved_vintages["records"]["policyengine_us_surface"]["value"] == ( diff --git a/packages/microcosm-build/tests/test_us_stacked_spine.py b/packages/microcosm-build/tests/test_us_stacked_spine.py index 0d455681..f4ba918c 100644 --- a/packages/microcosm-build/tests/test_us_stacked_spine.py +++ b/packages/microcosm-build/tests/test_us_stacked_spine.py @@ -26,6 +26,7 @@ import microcosm.build.us_runtime.acs_income_universe as universe_module import microcosm.build.us_runtime.acs_transfer as acs_transfer_module +import microcosm.build.us_runtime.immigration as immigration_module import microcosm.build.us_runtime.multispine_pool as multispine_pool_module import microcosm.build.us_runtime.post_transfer_calibration as post_transfer_calibration_runtime import microcosm.build.us_runtime.puf_capital_gains_tail as tail_module @@ -2750,6 +2751,24 @@ def _canonical_gap_fill_receipt_with_pattern_evidence() -> tuple[ return receipt, direction_name, key, entity, family, family_targets +def test_immigration_pattern_authority_requires_harmonized_evidence() -> None: + required, optional = stacked_spine_module._acs_pattern_predictor_authority( + entity="person", + family_targets=("ssn_card_type", "immigration_status_str"), + ) + immigration = ( + "__acs_transfer_is_us_citizen", + "__acs_transfer_birth_country_code", + "__acs_transfer_arrival_year", + ) + + assert required == ( + *acs_transfer_module.ACS_PERSON_TRANSFER_PREDICTORS, + *immigration, + ) + assert not set(immigration).intersection(optional) + + def test_gap_fill_validator_accepts_canonical_calibration_evidence() -> None: stacked_spine_module.validate_stacked_gap_fill_receipt( _canonical_gap_fill_calibration_receipt(), @@ -3661,6 +3680,59 @@ def test_late_readiness_rejects_object_typed_nonfinite_numeric_input() -> None: ) +def test_immigration_arrival_readiness_accepts_native_acs_yoep_blanks() -> None: + producer = next( + group.name + for group in stacked_spine_module.CANONICAL_US_LATE_TRANSFER_GROUPS + if group.family == "source_operator_immigration" + ) + full_contract = stacked_spine_module.CANONICAL_US_LATE_PRODUCER_REGISTRY[producer] + requirement = next( + item + for item in full_contract.inputs + if item.column == "@effective:immigration_acs_arrival" + ) + contract = replace(full_contract, inputs=(requirement,), outputs=()) + frame = _post_puf_transfer_fixture() + person = frame.table("person").copy() + acs_rows = person[support_channel_column("person")].astype(str).eq("acs") + assert acs_rows.any() + person["CIT"] = np.where(acs_rows, 1.0, np.nan) + person["YOEP"] = np.nan + tables = {entity: frame.table(entity) for entity in frame.entities} + tables["person"] = person + native_blank = Frame( + tables, + frame.schema, + {entity: frame.weights_for(entity) for entity in frame.weighted_entities}, + frame.strata, + mass_log=frame.mass_log, + metadata=frame.metadata, + ) + + unfilled, invalid = stacked_spine_module._late_input_readiness_rows( + native_blank, + contract, + ) + + assert requirement.required_scope == "acs_source" + assert requirement.alternatives == ( + (ProducerInputColumn("person", "YOEP", "column_present"),), + ) + assert unfilled == {requirement: 0} + assert invalid == {requirement: 0} + assert ( + stacked_spine_module.run_producer_when_ready( + contract, + lambda: "ran", + unfilled_rows=unfilled, + invalid_rows=invalid, + absence_receipts={}, + ) + == "ran" + ) + + def _typehugq_cross_origin_readiness_fixture() -> tuple[ Frame, ProducerContract, @@ -5207,6 +5279,112 @@ def test_universe_raw_authority_binds_present_column_with_structural_nulls() -> ) +def _humanitarian_late_executor_entry_fixture() -> Frame: + """Build one exact weighted ACS carrier for every positive manifest draw.""" + + controls = immigration_module.us_immigration_controls() + positive_draws = tuple(draw for draw in controls.humanitarian if draw.target > 0) + profiles = { + "paroled_one_year:afghanistan": (200, 2021), + "paroled_one_year:ukraine": (164, 2022), + "paroled_one_year:nicaragua": (315, 2023), + "paroled_one_year:venezuela": (373, 2022), + "refugee": (412, 2022), + "asylee": (207, 2016), + "tps:venezuela": (373, 2021), + "tps:el_salvador": (312, 2001), + "tps:honduras": (314, 1998), + "tps:nicaragua": (315, 1998), + "tps:nepal": (229, 2015), + "tps:other_designated": (224, 2024), + } + assert {draw.label for draw in positive_draws} == set(profiles) + positive_target = float(sum(draw.target for draw in positive_draws)) + + # Stacked assembly assigns half of the ASEC anchor mass to each source. + # Make the ACS incoming total equal that allocation so every live ACS + # household retains its manifest target as its final resolved weight. + asec = _source_frame( + household_ids=[11], + weights=[2.0 * positive_target], + stratum="asec_2024", + ) + asec_person = asec.table("person") + asec_person["is_female"] = False + asec_person["is_household_head"] = True + asec_person["employment_income_before_lsr"] = 10_000.0 + asec_person["self_employment_income_before_lsr"] = 0.0 + asec_person["pre_subsidy_rent"] = 12_000.0 + asec_person["unemployment_compensation"] = 1.0 + asec_person["is_disabled"] = False + for column, value in ( + ("taxable_interest_income", 100.0), + ("tax_exempt_interest_income", 0.0), + ("qualified_dividend_income", 50.0), + ("non_qualified_dividend_income", 25.0), + ("rental_income", 0.0), + ("estate_income", 0.0), + ): + asec_person[column] = value + asec_person["PRCITSHP"] = 1.0 + asec_person["PENATVTY"] = 57.0 + asec_person["PEINUSYR"] = 0.0 + asec_person["source_year"] = 2024 + asec_person["source_household_id"] = asec_person["person_household_id"] + asec_person["source_person_id"] = asec_person["person_id"] + asec.table("household")["tenure_type"] = "RENTED" + + acs = _source_frame( + household_ids=list(range(101, 101 + len(positive_draws))), + weights=[float(draw.target) for draw in positive_draws], + stratum="acs_2024_1yr", + ) + acs_person = acs.table("person") + acs_person["is_female"] = False + acs_person["is_household_head"] = True + acs_person["employment_income_before_lsr"] = 10_000.0 + acs_person["WAGP"] = 10_000.0 + acs_person["self_employment_income_before_lsr"] = 0.0 + acs_person["SEMP"] = 0.0 + acs_person["acs_interest_dividend_rental_income"] = 0.0 + acs_person["CIT"] = 5.0 + acs_person["POBP"] = [profiles[draw.label][0] for draw in positive_draws] + acs_person["YOEP"] = [profiles[draw.label][1] for draw in positive_draws] + acs_person["age"] = [12.0, *([60.0] * (len(positive_draws) - 1))] + acs_person["source_year"] = 2024 + acs_person["source_household_id"] = acs_person["person_household_id"] + acs_person["source_person_id"] = acs_person["person_id"] + acs.table("household")["TYPEHUGQ"] = 1 + acs.table("household")["tenure_type"] = "RENTED" + + registry = stacked_spine_module.CANONICAL_US_LATE_PRODUCER_REGISTRY + primary_contract = registry[stacked_spine_module.US_LATE_PRIMARY_PUF_STAGE] + initial = _fill_late_contract_surface( + assemble_stacked_spine( + asec, + acs, + acs_sample_fraction=1.0, + acs_sample_seed=578, + ).frame, + contracts=(primary_contract,), + include_outputs=False, + ) + initial_person = initial.table("person") + structural_row = initial_person.index[ + initial_person[support_channel_column("person")].eq("acs") + ][0] + initial_person.loc[ + structural_row, + [ + "WAGP", + "SEMP", + "employment_income_before_lsr", + "self_employment_income_before_lsr", + ], + ] = np.nan + return initial + + def _run_real_late_executor_fixture( monkeypatch: pytest.MonkeyPatch, *, @@ -5215,7 +5393,7 @@ def _run_real_late_executor_fixture( asec_earnings_delta: float = 0.0, ) -> tuple[stacked_spine_module.StackedLateProducerResult, tuple[str, ...], int]: registry = stacked_spine_module.CANONICAL_US_LATE_PRODUCER_REGISTRY - initial = _late_universe_entry_fixture() + initial = _humanitarian_late_executor_entry_fixture() initial_person = initial.table("person") if asec_earnings_delta: asec_row = initial_person.index[ @@ -5281,6 +5459,33 @@ def primary(frame: Frame): include_outputs=True, ) completed_person = completed.table("person") + asec_rows = completed_person[support_channel_column("person")].eq("asec") + completed_person.loc[asec_rows, ["CIT", "POBP", "YOEP"]] = np.nan + completed_person.loc[ + ~asec_rows, + ["PRCITSHP", "PENATVTY", "PEINUSYR"], + ] = np.nan + completed_person["A_AGE"] = completed_person["age"] + completed_person["ssn_card_type"] = np.where( + asec_rows, + "CITIZEN", + "NONE", + ) + completed_person["immigration_status_str"] = np.where( + asec_rows, + "CITIZEN", + "UNDOCUMENTED", + ) + indicator_documented = ~asec_rows & pd.to_numeric( + completed_person["POBP"], + errors="raise", + ).isin((207, 412)) + completed_person.loc[indicator_documented, "ssn_card_type"] = ( + "OTHER_NON_CITIZEN" + ) + completed_person.loc[indicator_documented, "immigration_status_str"] = ( + "LEGAL_PERMANENT_RESIDENT" + ) completed_person["unemployment_compensation"] = np.ones( len(completed_person), dtype=np.float64, @@ -5390,10 +5595,55 @@ def transfer( for spec in post_transfer_calibration_runtime.POST_TRANSFER_CALIBRATION_SPECS.values() if spec.stage == "late_transfer" } + is_immigration_group = ( + group.entity == "person" and group.family == "source_operator_immigration" + ) + immigration_reconciliation: Mapping[str, object] | None = None + imputed_recipient_rows = 1 + if is_immigration_group: + mutable_rows = ( + frame.table("person")[support_channel_column("person")] + .astype(str) + .eq("acs") + .to_numpy(dtype=bool) + ) + reconciled_person, immigration_reconciliation = ( + immigration_module.reconcile_us_immigration_humanitarian_transfer( + frame.table("person"), + weights=np.asarray( + frame.resolve_weights("person").values, + dtype=np.float64, + ), + mutable_rows=mutable_rows, + seed=0, + time_period=2024, + ) + ) + tables = {entity: frame.table(entity) for entity in frame.entities} + tables["person"] = reconciled_person + frame = Frame( + tables, + frame.schema, + { + entity: frame.weights_for(entity) + for entity in frame.weighted_entities + }, + frame.strata, + mass_log=frame.mass_log, + metadata=frame.metadata, + ) + imputed_recipient_rows = int(mutable_rows.sum()) evidence_targets = tuple( - target - for target in group.targets - if f"{group.entity}/{group.family}/{target}" in late_specs + dict.fromkeys( + ( + *( + target + for target in group.targets + if f"{group.entity}/{group.family}/{target}" in late_specs + ), + *(group.targets if is_immigration_group else ()), + ) + ) ) model_targets = acs_transfer_module._model_target_names(evidence_targets) pattern = AcsTransferPattern( @@ -5403,7 +5653,7 @@ def transfer( seed=0, weight_kind="design", donor_rows=1, - recipient_rows=1, + recipient_rows=imputed_recipient_rows, target_regimes=tuple((target, "positive_only") for target in model_targets), ) plain_pattern = replace(pattern, target_regimes=()) @@ -5418,7 +5668,8 @@ def transfer( seed=pattern.seed, weight_kind=pattern.weight_kind, patterns=(pattern if target in evidence_targets else plain_pattern,), - imputed_recipient_rows=1, + imputed_recipient_rows=imputed_recipient_rows, + reconciliation=immigration_reconciliation, ) for target in group.targets ) @@ -5440,15 +5691,15 @@ def transfer( ): key = f"{group.entity}/{group.family}/{target}" target_receipt: dict[str, object] = { - "authorized_null_rows": 1, - "imputed_rows": 1, + "authorized_null_rows": imputed_recipient_rows, + "imputed_rows": imputed_recipient_rows, "unmodeled_rows": 0, "residual_null_rows": 0, } if target == "is_pregnant": - pregnancy_policy = execution_contract[ - "structural_target_policies" - ]["is_pregnant"] + pregnancy_policy = execution_contract["structural_target_policies"][ + "is_pregnant" + ] target_receipt["structural_policy"] = { "policy_sha256": pregnancy_policy["sha256"], "source_person_key": "person_source_id", @@ -5471,10 +5722,14 @@ def transfer( "final_clone_disagreement_source_persons": 0, "status": "verified", } - if key in late_specs: + if key in late_specs or is_immigration_group: target_receipt["qrf_pattern_evidence"] = ( stacked_spine_module._acs_imputed_pattern_evidence(record) ) + if is_immigration_group: + target_receipt["post_transfer_reconciliation"] = dict( + record.reconciliation or {} + ) target_receipts[key] = target_receipt calibrated_keys = sorted(set(target_receipts) & set(late_specs)) for key in calibrated_keys: @@ -5684,6 +5939,32 @@ def test_real_late_executor_follows_canonical_order_and_finalizes_sources_once( stacked_spine_module.US_LATE_PRODUCER_TRANSITION_AUTHORITY_KEY ]["sha256"] ) + immigration_group = result.receipt["post_puf_transfer"]["groups"][ + "transfer:person/source_operator_immigration" + ] + immigration_targets = immigration_group["targets"] + pair_receipts = [ + immigration_targets[f"person/source_operator_immigration/{target}"] + for target in ("ssn_card_type", "immigration_status_str") + ] + assert ( + pair_receipts[0]["post_transfer_reconciliation"] + == pair_receipts[1]["post_transfer_reconciliation"] + ) + assert pair_receipts[0]["post_transfer_reconciliation"]["kind"] == ( + "deterministic_humanitarian_residual_target" + ) + expected_required, _optional = ( + stacked_spine_module._acs_pattern_predictor_authority( + entity="person", + family_targets=("ssn_card_type", "immigration_status_str"), + ) + ) + assert all( + tuple(receipt["qrf_pattern_evidence"]["patterns"][0]["predictors"]) + == expected_required + for receipt in pair_receipts + ) stacked_spine_module.validate_stacked_late_producer_receipt( result.receipt, boundary="executor regression", @@ -5692,6 +5973,306 @@ def test_real_late_executor_follows_canonical_order_and_finalizes_sources_once( ) +def test_immigration_group_receipt_rejects_reordered_harmonized_predictors( + monkeypatch: pytest.MonkeyPatch, +) -> None: + result, _events, _finalizer_calls = _run_real_late_executor_fixture(monkeypatch) + transfer = deepcopy(dict(result.receipt["post_puf_transfer"])) + target = transfer["groups"]["transfer:person/source_operator_immigration"][ + "targets" + ]["person/source_operator_immigration/immigration_status_str"] + evidence = target["qrf_pattern_evidence"] + predictors = evidence["patterns"][0]["predictors"] + predictors[-1], predictors[-2] = predictors[-2], predictors[-1] + unsigned = dict(evidence) + unsigned.pop("sha256") + evidence["sha256"] = stacked_spine_module._canonical_sha256(unsigned) + + with pytest.raises(ValueError, match="predictor order is outside canonical"): + stacked_spine_module.validate_stacked_post_puf_transfer_receipt( + transfer, + boundary="reordered immigration predictor regression", + ) + + +@pytest.mark.parametrize( + ("mutation", "message"), + ( + ("remove", "immigration post-transfer reconciliation is absent"), + ("diverge", "paired immigration reconciliation evidence is incomplete"), + ("inflated_tolerance", "reconciliation tolerance is non-canonical"), + ("zero_target_immutable", "mass against an explicit-zero target"), + ), +) +def test_immigration_group_receipt_rejects_missing_or_divergent_reconciliation( + monkeypatch: pytest.MonkeyPatch, + mutation: str, + message: str, +) -> None: + result, _events, _finalizer_calls = _run_real_late_executor_fixture(monkeypatch) + transfer = deepcopy(dict(result.receipt["post_puf_transfer"])) + targets = transfer["groups"]["transfer:person/source_operator_immigration"][ + "targets" + ] + status = targets["person/source_operator_immigration/immigration_status_str"] + reconciliation = status["post_transfer_reconciliation"] + if mutation == "remove": + status.pop("post_transfer_reconciliation") + elif mutation == "diverge": + reconciliation["seed"] += 1 + elif mutation == "inflated_tolerance": + reconciliation["floating_tolerance"] *= 1_000_000 + else: + zero_draw = next( + draw for draw in reconciliation["draws"].values() if draw["target"] == 0 + ) + zero_draw.update( + { + "immutable_population": 1.0, + "achieved_population": 1.0, + "absolute_error": 1.0, + "immutable_overshoot": 1.0, + } + ) + + with pytest.raises(ValueError, match=message): + stacked_spine_module.validate_stacked_post_puf_transfer_receipt( + transfer, + boundary=f"{mutation} immigration reconciliation regression", + ) + + +def test_immigration_reconciliation_replays_receipt_and_downstream_live_frame( + monkeypatch: pytest.MonkeyPatch, +) -> None: + result, _events, _finalizer_calls = _run_real_late_executor_fixture(monkeypatch) + transfer = deepcopy(dict(result.receipt["post_puf_transfer"])) + immigration_group = transfer["groups"][ + "transfer:person/source_operator_immigration" + ] + immigration_targets = immigration_group["targets"] + source_receipt = immigration_targets[ + "person/source_operator_immigration/ssn_card_type" + ]["post_transfer_reconciliation"] + fabricated_reconciliation = deepcopy(source_receipt) + draw = next( + item + for item in fabricated_reconciliation["draws"].values() + if item["target"] > 0 + ) + draw["eligible_recipient_population"] += 1.0 + draw["selected_recipient_population"] += 1.0 + draw["achieved_population"] += 1.0 + draw["residual_selection_error"] = 1.0 + draw["absolute_error"] = 1.0 + for target in ("ssn_card_type", "immigration_status_str"): + key = f"person/source_operator_immigration/{target}" + immigration_targets[key]["post_transfer_reconciliation"] = deepcopy( + fabricated_reconciliation + ) + transfer["targets"][key] = deepcopy(immigration_targets[key]) + + # The forged values remain internally self-consistent; only live replay can + # distinguish them from the actually emitted weighted population. + stacked_spine_module.validate_stacked_post_puf_transfer_receipt( + transfer, + boundary="fabricated immigration mass receipt-only control", + ) + with pytest.raises( + ValueError, + match="differs from the live ASEC/ACS weighted population", + ): + stacked_spine_module.validate_stacked_post_puf_transfer_receipt( + transfer, + boundary="fabricated immigration mass live replay", + frame=result.frame, + ) + + tables = {entity: result.frame.table(entity) for entity in result.frame.entities} + person = tables["person"].copy() + first_positive_draw = next( + draw + for draw in immigration_module.us_immigration_controls().humanitarian + if draw.target > 0 + ) + emitted = immigration_module.us_immigration_humanitarian_draw_mask( + result.frame, + first_positive_draw, + time_period=2024, + ) + person.loc[person.index[np.flatnonzero(emitted)[0]], "immigration_status_str"] = ( + "LEGAL_PERMANENT_RESIDENT" + ) + tables["person"] = person + drifted = Frame( + tables, + result.frame.schema, + { + entity: result.frame.weights_for(entity) + for entity in result.frame.weighted_entities + }, + result.frame.strata, + mass_log=result.frame.mass_log, + metadata=result.frame.metadata, + ) + with pytest.raises( + ValueError, + match="differs from the live ASEC/ACS weighted population", + ): + stacked_spine_module.validate_stacked_late_producer_transition_authority( + drifted, + result.receipt, + boundary="downstream immigration label live replay", + expected_transition_authority_sha256=(result.transition_authority_sha256), + ) + + +def test_immigration_live_replay_rejects_equal_weight_selection_identity_swap( + monkeypatch: pytest.MonkeyPatch, +) -> None: + asec = _source_frame( + household_ids=[11], + weights=[2.0], + stratum="asec_2024", + ) + acs = _source_frame( + household_ids=[101, 102], + weights=[1.0, 1.0], + extra_household_columns={"TYPEHUGQ": 1}, + stratum="acs_2024_1yr", + ) + base = assemble_stacked_spine( + asec, + acs, + acs_sample_fraction=1.0, + acs_sample_seed=578, + ).frame + person = base.table("person").copy() + channel = person[support_channel_column("person")].astype(str) + immutable = channel.eq("asec").to_numpy(dtype=bool) + mutable = channel.eq("acs").to_numpy(dtype=bool) + person["PRCITSHP"] = np.where(immutable, 1.0, np.nan) + person["PENATVTY"] = np.where(immutable, 57.0, np.nan) + person["PEINUSYR"] = np.where(immutable, 0.0, np.nan) + person["CIT"] = np.where(mutable, 5.0, np.nan) + person["POBP"] = np.where(mutable, 164.0, np.nan) + person["YOEP"] = np.where(mutable, 2022.0, np.nan) + person["source_year"] = 2024 + person["source_household_id"] = person["person_household_id"] + person["source_person_id"] = person["person_id"] + person["ssn_card_type"] = np.where( + immutable, + "CITIZEN", + "OTHER_NON_CITIZEN", + ) + person["immigration_status_str"] = np.where( + immutable, + "CITIZEN", + "LEGAL_PERMANENT_RESIDENT", + ) + weights = np.asarray(base.resolve_weights("person").values, dtype=np.float64) + mutable_weight = float(weights[mutable][0]) + assert weights[mutable].tolist() == [mutable_weight, mutable_weight] + draw = immigration_module.HumanitarianDraw( + category="paroled_one_year", + origin="ukraine", + status="PAROLED_ONE_YEAR", + target=mutable_weight, + source="https://example.com/u4u", + ) + controls = immigration_module.ImmigrationControls( + undocumented=immigration_module.UndocumentedControls( + workers=1.0, + students=1.0, + population_anchor=1.0, + sources={ + "undocumented_workers": "https://example.com/workers", + "undocumented_students": "https://example.com/students", + "undocumented_population_anchor": "https://example.com/population", + }, + ), + humanitarian=(draw,), + ) + reconciled_person, reconciliation = ( + immigration_module.reconcile_us_immigration_humanitarian_transfer( + person, + weights=weights, + mutable_rows=mutable, + seed=0, + time_period=2024, + controls=controls, + ) + ) + + def with_person(frame: Frame, replacement: pd.DataFrame) -> Frame: + tables = {entity: frame.table(entity) for entity in frame.entities} + tables["person"] = replacement + return Frame( + tables, + frame.schema, + {entity: frame.weights_for(entity) for entity in frame.weighted_entities}, + frame.strata, + mass_log=frame.mass_log, + metadata=frame.metadata, + ) + + reconciled = with_person(base, reconciled_person) + target_receipt = { + "producer_rows": int(immutable.sum()), + "authorized_null_rows": int(mutable.sum()), + "imputed_rows": int(mutable.sum()), + "unmodeled_rows": 0, + "residual_null_rows": 0, + "post_transfer_reconciliation": reconciliation, + } + monkeypatch.setattr( + stacked_spine_module, + "us_immigration_controls", + lambda: controls, + ) + stacked_spine_module._validate_immigration_post_transfer_reconciliation( + target_receipt, + boundary="canonical equal-weight selection", + frame=reconciled, + ) + + swapped_person = reconciled_person.copy() + selected_index = swapped_person.index[ + mutable + & swapped_person["immigration_status_str"].eq("PAROLED_ONE_YEAR").to_numpy() + ][0] + unselected_index = swapped_person.index[ + mutable + & swapped_person["immigration_status_str"] + .eq("LEGAL_PERMANENT_RESIDENT") + .to_numpy() + ][0] + swapped_person.loc[selected_index, "immigration_status_str"] = ( + "LEGAL_PERMANENT_RESIDENT" + ) + swapped_person.loc[unselected_index, "immigration_status_str"] = "PAROLED_ONE_YEAR" + swapped = with_person(reconciled, swapped_person) + assert float( + swapped.resolve_weights("person") + .values[ + swapped_person["immigration_status_str"] + .eq("PAROLED_ONE_YEAR") + .to_numpy(dtype=bool) + ] + .sum() + ) == pytest.approx(mutable_weight) + + with pytest.raises( + ValueError, + match="selected mutable-row identities differ from the canonical seeded selection", + ): + stacked_spine_module._validate_immigration_post_transfer_reconciliation( + target_receipt, + boundary="equal-weight selection identity swap", + frame=swapped, + ) + + def test_late_executor_signature_rejects_qrf_regime_evidence_tampering( monkeypatch: pytest.MonkeyPatch, ) -> None: @@ -6360,7 +6941,9 @@ def test_post_puf_transfer_preserves_complete_asec_source_producers() -> None: ) -def test_complete_pregnancy_surface_retains_zero_imputation_structural_receipt() -> None: +def test_complete_pregnancy_surface_retains_zero_imputation_structural_receipt() -> ( + None +): frame = _post_puf_transfer_fixture() person = frame.table("person").copy() recipient_rows = person[support_channel_column("person")].astype(str).eq("acs") @@ -6389,9 +6972,7 @@ def test_complete_pregnancy_surface_retains_zero_imputation_structural_receipt() n_estimators=10, ) - receipt = result.receipt["targets"][ - "person/model_required_boolean/is_pregnant" - ] + receipt = result.receipt["targets"]["person/model_required_boolean/is_pregnant"] assert receipt["authorized_null_rows"] == 0 assert receipt["imputed_rows"] == 0 assert receipt["structural_policy"]["status"] == "verified" @@ -6412,9 +6993,7 @@ def test_post_puf_transfer_refuses_invalid_pregnancy_source_producer() -> None: person["is_female"].astype(bool) & person["age"].between(15, 44, inclusive="both") ) - donor_ineligible = ineligible & person[ - support_clone_index_column("person") - ].eq(1) + donor_ineligible = ineligible & person[support_clone_index_column("person")].eq(1) person.loc[person.index[donor_ineligible][0], "is_pregnant"] = True tables = {entity: frame.table(entity) for entity in frame.entities} tables["person"] = person diff --git a/tools/build_us_fiscal_refresh_release.py b/tools/build_us_fiscal_refresh_release.py index 8c1ed2c7..cd64d005 100644 --- a/tools/build_us_fiscal_refresh_release.py +++ b/tools/build_us_fiscal_refresh_release.py @@ -237,6 +237,10 @@ require_authenticated_us_multispine_pool_h5, us_multispine_pool_release_receipt, ) +from microcosm.build.us_runtime.immigration import ( + us_immigration_controls, + us_immigration_humanitarian_draw_mask, +) from microcosm.build.us_runtime.input_mass import us_input_mass_totals from microcosm.build.us_runtime.l0_refit_export import ( attach_l0_refit_entity_weights, @@ -439,7 +443,10 @@ class PoolReleaseIdentityMismatchError(ValueError): # 11: target-frame checkpoint columns now preserve nullable booleans as # canonical bool values plus an explicit uint8 null mask. Older checkpoints # cannot attest this lossless physical representation. -TARGET_FRAME_CHECKPOINT_MATERIALIZER_VERSION = 11 +# 12: source-aware humanitarian-stock targets now materialize from the paired +# immigration labels and native ASEC/ACS origin/arrival evidence. Version-11 +# checkpoints cannot attest those per-draw calibration columns or mask semantics. +TARGET_FRAME_CHECKPOINT_MATERIALIZER_VERSION = 12 DEFAULT_MAXIMUM_MICROSIM_BATCH_SIZE = 5_000 DEFAULT_L0_REFIT_LAMBDA_SHARE = 0.8 DEFAULT_US_FISCAL_CALIBRATION_EPOCHS = 1_500 @@ -1605,9 +1612,7 @@ def _parse_args(argv: Sequence[str] | None = None) -> argparse.Namespace: len(value) != 64 or any(character not in "0123456789abcdef" for character in value) ): - parser.error( - f"{flag} must be exactly 64 lowercase hexadecimal characters." - ) + parser.error(f"{flag} must be exactly 64 lowercase hexadecimal characters.") if args.evidence_release and args.exact_k is not None: parser.error( "--evidence-release is incompatible with --exact-k: ladder " @@ -2176,6 +2181,92 @@ def _selection_mass_protection_specs( return tuple(specs) +def _humanitarian_immigration_stock_specs( + *, + time_period: int = PERIOD, +) -> tuple[TargetSpec, ...]: + """Return one positive calibration target per manifest humanitarian draw. + + These are builder-owned preservation targets, not Ledger fiscal facts, so + the caller adds them only after the fiscal target-parity gate has checked + the feed registry. Values and citations come directly from the same + immigration source-stage manifest that assigns the person labels. + """ + + specs: list[TargetSpec] = [] + for draw in us_immigration_controls().humanitarian: + if draw.target <= 0: + continue + measure = f"humanitarian_immigration_stock.{draw.label.replace(':', '.')}" + metadata = { + "materializer": "humanitarian_immigration_stock", + "measure_mode": "indicator_sum", + "target_role": "humanitarian_immigration_stock", + "humanitarian_draw": draw.label, + "humanitarian_category": draw.category, + "humanitarian_status": draw.status, + "issue": "PolicyEngine/microcosm#767", + } + if draw.origin is not None: + metadata["humanitarian_origin"] = draw.origin + specs.append( + TargetSpec( + name=measure, + entity="household", + value=draw.target, + measure=measure, + period=time_period, + source=draw.source, + family="humanitarian_immigration_stock", + metadata=metadata, + ) + ) + return tuple(specs) + + +def _materialize_humanitarian_immigration_stock_targets( + *, + frame: Frame, + household: pd.DataFrame, + target_specs: Sequence[TargetSpec], + time_period: int = PERIOD, +) -> None: + """Collapse source-aware humanitarian person indicators to households.""" + + humanitarian_specs = [ + spec + for spec in target_specs + if spec.metadata.get("materializer") == "humanitarian_immigration_stock" + ] + if not humanitarian_specs: + return + controls = us_immigration_controls() + draws = {draw.label: draw for draw in controls.humanitarian} + if len(draws) != len(controls.humanitarian): + raise RuntimeError("Humanitarian immigration controls have duplicate draws.") + for spec in humanitarian_specs: + draw_label = spec.metadata.get("humanitarian_draw") + if draw_label not in draws: + raise RuntimeError( + f"Humanitarian target {spec.name!r} refers to unknown manifest " + f"draw {draw_label!r}." + ) + mask = np.asarray( + us_immigration_humanitarian_draw_mask( + frame, + draws[draw_label], + time_period=time_period, + ), + dtype=bool, + ) + if mask.shape != (frame.n("person"),): + raise RuntimeError( + f"Humanitarian draw {draw_label!r} returned mask shape " + f"{mask.shape}; expected {(frame.n('person'),)}." + ) + household[spec.measure] = _collapse_person(frame, mask.astype(np.float64)) + + def _target_frame_checkpoint_identity( *, base_dataset_sha256: str, @@ -4668,6 +4759,12 @@ def _materialize_target_frame( variable="state_income_tax", tax_unit_positions=tax_unit_positions, ) + _materialize_humanitarian_immigration_stock_targets( + frame=base_frame, + household=hh, + target_specs=target_specs, + time_period=PERIOD, + ) population_age_target_specs = [ spec for spec in target_specs @@ -8840,7 +8937,10 @@ def _main(argv: Sequence[str] | None = None) -> None: for failure in target_parity_gate.failures ) ) - target_specs = target_registry.specs + target_specs = ( + *target_registry.specs, + *_humanitarian_immigration_stock_specs(time_period=PERIOD), + ) active_target_registry = TargetRegistry(target_specs, country="us") # SSI take-up wiring resolves as soon as the registry exists (fail-fast, # microcosm#507/#508): the band targets come from the same ledger-fed @@ -10870,6 +10970,11 @@ def _main(argv: Sequence[str] | None = None) -> None: ) else: export_frame = _with_l0_refit_weights(base_frame, result) + # Calibration can redistribute person stocks even though the persisted + # labels remain structurally valid. Re-evaluate the composition contract + # at delivered weights and use this final verdict in every terminal gate + # and release artifact below. + final_immigration_gate = us_immigration_composition_gate(export_frame) compilation = dict(compilation) final_uncapped_ssi = _ssi_person_uncapped_amount( export_frame, @@ -11157,7 +11262,7 @@ def _main(argv: Sequence[str] | None = None) -> None: health_input_gate, base_population_gate, incumbent_diagnostics, - immigration_gate, + final_immigration_gate, enforced_input_mass_reference_gate, degenerate_input_gate, ecps_parity_gate=enforced_ecps_parity_gate, @@ -11206,7 +11311,7 @@ def _main(argv: Sequence[str] | None = None) -> None: target_profile_gate=target_profile_gate, health_input_gate=health_input_gate, base_population_gate=base_population_gate, - immigration_gate=immigration_gate, + immigration_gate=final_immigration_gate, input_mass_reference_gate=input_mass_reference_gate, hours_worked_gate=hours_worked_gate, snap_take_up_gate=snap_take_up_gate, @@ -11852,7 +11957,7 @@ def _main(argv: Sequence[str] | None = None) -> None: health_input_gate=health_input_gate, base_population_gate=base_population_gate, incumbent_diagnostics=incumbent_diagnostics, - immigration_gate=immigration_gate, + immigration_gate=final_immigration_gate, input_mass_reference_gate=enforced_input_mass_reference_gate, degenerate_input_gate=degenerate_input_gate, ecps_parity_gate=enforced_ecps_parity_gate, diff --git a/tools/build_us_multispine_pool.py b/tools/build_us_multispine_pool.py index 23e0de84..306ea41b 100644 --- a/tools/build_us_multispine_pool.py +++ b/tools/build_us_multispine_pool.py @@ -6,9 +6,10 @@ ``stack -> geography -> gap-fill -> PUF pass + tail -> late DAG -> derive -> seed -> simulate -> gates``. Both survey arms use one composition-preserving ``--sample-fraction``; PUF -donors always remain full. The terminal completeness gate plus by-origin -battery replace two-spine agreement. ``--legacy-two-spine`` retains the -retiring pipeline byte-for-byte for reproducibility. +donors always remain full. The terminal completeness, by-origin, and +humanitarian-immigration gates replace two-spine agreement. +``--legacy-two-spine`` retains the retiring pipeline byte-for-byte for +reproducibility. Every input is local and explicitly SHA-pinned; this tool never downloads data. It writes a nullable input-only H5 plus a manifest and terminal gate @@ -130,6 +131,7 @@ ACS_2022_RENT_ARTIFACT_SHA256, load_acs_2022_rent_donor, ) +from microcosm.build.us_runtime.immigration import us_immigration_composition_gate from microcosm.build.us_runtime.multispine_pool import ( POOL_CHECKPOINT_STAGE_ORDER, POOL_DERIVE_OPERATOR_ORDER, @@ -284,6 +286,9 @@ # 7: Stacked primary-PUF output universes are explicit. Earlier envelopes can # contain nulls outside the PUF clone for an output declared over the whole # pool and therefore cannot resume safely even when their bank is reusable. +# 8: ACS native citizenship/origin/arrival inputs, pooled immigration control +# scaling, and hard post-transfer humanitarian reconciliation change the +# assembled and transferred outputs. Earlier checkpoints cannot resume. # # Bump this version whenever any producer above changes a stage output without # changing one of the explicit identity fields below. In particular, adding, @@ -298,7 +303,7 @@ # normalizes that logical view in memory. Moving between those encodings does # not change a producer's scalar output and therefore does not advance this # ledger; changing string values or the canonical logical dtype policy does. -POOL_STAGE_CHECKPOINT_MATERIALIZER_VERSION = 7 +POOL_STAGE_CHECKPOINT_MATERIALIZER_VERSION = 8 _PRIMARY_QRF_N_ESTIMATORS = 100 _ACS_TRANSFER_N_ESTIMATORS = 100 @@ -329,6 +334,9 @@ _STACKED_CHECKPOINT_IDENTITY_ARTIFACT_KIND = ( "populace_us_stacked_pool_checkpoint_identity" ) +# Version 13 binds native ACS humanitarian eligibility, pooled immigration +# control scaling, reconciliation, and the final composition gate. Earlier +# stacked checkpoints predate the source-aware status surface. # Version 12 binds the post-assembly household geography assignment authority, # target vintage, algorithm, operator order, and seed. Earlier checkpoints # predate the congressional-district support required by release preflight. @@ -336,7 +344,7 @@ # Earlier checkpoints must rebuild rather than resume with a nullable # s_corp_income leaf. Version 10 bound the complete late-resource semantics and # corrected outer order (the primary PUF callback is nested inside the DAG). -_STACKED_CHECKPOINT_MATERIALIZER_VERSION = 12 +_STACKED_CHECKPOINT_MATERIALIZER_VERSION = 13 _STACKED_RELEASE_ID_PATTERN = re.compile( r"^populace-us-2024-stacked-f(?:001|004|010|025|100)-s[0-9]+-" r"asec[0-9]+-acs[0-9]+-[0-9]{8}T[0-9]{6}Z-[0-9a-f]{8}$" @@ -391,14 +399,14 @@ class PoolBuildOutputs: @dataclass(frozen=True) class StackedPoolBuildResult: - """Input-only stacked pool and its two fresh terminal gate verdicts.""" + """Input-only stacked pool and its fresh terminal gate verdicts.""" frame: Frame stack_receipt: Mapping[str, object] assembly_receipt: Mapping[str, object] provenance_counts: Mapping[str, Mapping[str, object]] stage_receipts: Mapping[str, Mapping[str, object]] - terminal_gates: tuple[GateResult, GateResult] + terminal_gates: tuple[GateResult, ...] release_id: str qbi_transition_authority_sha256: str | None = None late_producer_transition_authority_sha256: str | None = None @@ -1648,9 +1656,7 @@ def _ordered_household_geography_receipt( dtype=np.float64, na_value=np.nan, ) - valid_ids = np.isfinite(household_ids) & ( - household_ids == np.floor(household_ids) - ) + valid_ids = np.isfinite(household_ids) & (household_ids == np.floor(household_ids)) if not valid_ids.all(): raise ValueError("Stacked geography output household IDs must be integral.") @@ -1718,9 +1724,11 @@ def _validate_stacked_geography_assignment_receipt( ) -> None: """Bind a live assembled-or-later Frame to its assignment receipt.""" - if receipt.get("artifact_kind") != ( - "populace_us_stacked_household_geography_assignment" - ) or receipt.get("schema_version") != 1: + if ( + receipt.get("artifact_kind") + != ("populace_us_stacked_household_geography_assignment") + or receipt.get("schema_version") != 1 + ): raise ValueError(f"{boundary}: geography assignment receipt is unsupported.") expected_contract = _stacked_geography_assignment_contract() if _json_ready(receipt.get("contract")) != _json_ready(expected_contract): @@ -1729,9 +1737,7 @@ def _validate_stacked_geography_assignment_receipt( target_districts ) if receipt.get("target_universe") != expected_universe: - raise ValueError( - f"{boundary}: congressional-district target universe changed." - ) + raise ValueError(f"{boundary}: congressional-district target universe changed.") output = receipt.get("output") if not isinstance(output, Mapping): raise ValueError(f"{boundary}: geography assignment output is missing.") @@ -1759,9 +1765,10 @@ def _validate_stacked_geography_assignment_receipt( clone_index = pd.to_numeric(household[clone_column], errors="coerce") native_household = household.loc[clone_index.eq(0)] expected_order = receipt.get("pre_assignment_household_order") - if ( - len(native_household) != assigned_rows - or expected_order != _ordered_household_id_receipt(native_household) + if len( + native_household + ) != assigned_rows or expected_order != _ordered_household_id_receipt( + native_household ): raise ValueError( f"{boundary}: ordered native household IDs differ from the seeded " @@ -1901,9 +1908,7 @@ def _assign_stacked_household_geography( "schema_version": 1, "contract": _stacked_geography_assignment_contract(), "pre_assignment_household_order": pre_assignment_order, - "assigned_household_geography": _ordered_household_geography_receipt( - household - ), + "assigned_household_geography": _ordered_household_geography_receipt(household), "target_universe": _target_congressional_district_universe_receipt( target_districts ), @@ -3994,9 +3999,10 @@ def primary_puf_producer(primary_input: Frame): simulation_frame, tail_manifest=tail_manifest, ) + immigration = us_immigration_composition_gate(current) # Manifest conversion is itself the final canonical-authority check and # deliberately happens before publication or readiness is asserted. - GateReport((completeness, battery)).to_manifest() + GateReport((completeness, battery, immigration)).to_manifest() mark_phase("terminal_gates") counts = spine_provenance_counts( current, @@ -4018,7 +4024,7 @@ def primary_puf_producer(primary_input: Frame): assembly_receipt=dict(assembly_receipt), provenance_counts=counts, stage_receipts=receipts, - terminal_gates=(completeness, battery), + terminal_gates=(completeness, battery, immigration), release_id=release_id, qbi_transition_authority_sha256=qbi_transition_authority_sha256, late_producer_transition_authority_sha256=( @@ -4489,9 +4495,7 @@ def _write_stacked_outputs( materializer_version=US_MULTISPINE_POOL_H5_MATERIALIZER_VERSION, root_attributes={ CONGRESSIONAL_DISTRICT_VINTAGE_CROSSWALK_SHA256_ATTR: ( - verified_inputs[ - _STACKED_CD_CROSSWALK_INPUT_ROLE - ].actual_sha256 + verified_inputs[_STACKED_CD_CROSSWALK_INPUT_ROLE].actual_sha256 ), CONGRESSIONAL_DISTRICT_VINTAGE_TARGET_ATTR: ( CURRENT_CONGRESSIONAL_DISTRICT_VINTAGE diff --git a/tools/build_us_puf_support_base.py b/tools/build_us_puf_support_base.py index 5bdd07fe..a5632ac7 100644 --- a/tools/build_us_puf_support_base.py +++ b/tools/build_us_puf_support_base.py @@ -46,6 +46,7 @@ ASEC_2023_WEEKS_UNEMPLOYED_SOURCE_YEAR, ASEC_2023_WEEKS_UNEMPLOYED_ZIP_URL, ASEC_EDUCATION_ASSISTANCE_ARCHIVES, + ASEC_LABOR_FORCE_STATUS_COLUMN, ASEC_RAW_STAGE_ARTIFACT_KIND, ASEC_RAW_STAGE_CHECKPOINT_FILENAME, ASEC_RAW_STAGE_OPERATOR_STATUS, @@ -61,6 +62,7 @@ PUF_TAX_DETAIL_DEFAULT_PERSON_OUTPUTS, PUF_TAX_DETAIL_DEFAULT_TAX_UNIT_OUTPUTS, PUF_TAX_DETAIL_SUPPORT_CHANNEL, + US_IMMIGRATION_OUTPUT_COLUMNS, US_PUF_SUPPORT_FIT_NAME, US_SOURCE_MANIFEST, US_SUPPORT_SPINE_SPEC, @@ -75,6 +77,7 @@ fetch_asec_2023_weeks_unemployed_source, fill_asec_2022_weeks_unemployed_source, fill_asec_education_assistance_source, + fill_asec_labor_force_status_source, fill_asec_public_assistance_type_source, finalize_puf_e01000_reconciliation, impute_us_housing_assistance_to_puf_support, @@ -322,8 +325,8 @@ def _parse_args(argv: list[str] | None = None) -> argparse.Namespace: help=( "Optional INCOME_YEAR=PATH mapping to a local copy of the " "SHA-pinned official ASEC survey archive (zip or extracted " - "pppub member) restoring that pooled income year's ED_VAL and " - "PAW_TYP (income year YYYY maps to the survey-year YYYY+1 " + "pppub member) restoring that pooled income year's ED_VAL, " + "A_LFSR, and PAW_TYP (income year YYYY maps to the survey-year YYYY+1 " "archive). Years without a mapping are fetched from the " "official Census archive and verified against the same pins." ), @@ -870,6 +873,80 @@ def _write_policyengine_dataset( ) +def _with_asec_labor_force_status_source( + frame: Frame, + source: pd.DataFrame, +) -> Frame: + """Attach measured A_LFSR without mutating the source-construction frame.""" + + tables = {entity: frame.table(entity).copy(deep=True) for entity in frame.entities} + tables["person"] = fill_asec_labor_force_status_source(tables["person"], source) + return Frame( + tables, + frame.schema, + { + entity: Weights( + frame.weights_for(entity).values.copy(), + frame.weights_for(entity).kind, + ) + for entity in frame.weighted_entities + }, + frame.strata.copy(deep=True), + mass_log=frame.mass_log, + metadata=frame.metadata, + ) + + +def _with_us_immigration_inputs_from_asec_source( + frame: Frame, + source: pd.DataFrame, + *, + seed: int, + time_period: int, +) -> Frame: + """Run immigration with measured A_LFSR as an ephemeral raw input.""" + + person = frame.table("person") + if set(US_IMMIGRATION_OUTPUT_COLUMNS).issubset(person.columns): + return with_us_immigration_inputs( + frame, + seed=seed, + time_period=time_period, + ) + carried_labor_force_status = ASEC_LABOR_FORCE_STATUS_COLUMN in person + staged = ( + frame + if carried_labor_force_status + else _with_asec_labor_force_status_source(frame, source) + ) + result = with_us_immigration_inputs( + staged, + seed=seed, + time_period=time_period, + ) + if carried_labor_force_status: + return result + + tables = { + entity: result.table(entity).copy(deep=True) for entity in result.entities + } + tables["person"] = tables["person"].drop(columns=[ASEC_LABOR_FORCE_STATUS_COLUMN]) + return Frame( + tables, + result.schema, + { + entity: Weights( + result.weights_for(entity).values.copy(), + result.weights_for(entity).kind, + ) + for entity in result.weighted_entities + }, + result.strata.copy(deep=True), + mass_log=result.mass_log, + metadata=result.metadata, + ) + + def _run_all( args: argparse.Namespace, *, @@ -1027,8 +1104,9 @@ def _run_all( seed=args.seed, time_period=args.target_year, ) - base = with_us_immigration_inputs( + base = _with_us_immigration_inputs_from_asec_source( base, + education_assistance_source, seed=args.seed, time_period=args.target_year, ) @@ -1918,6 +1996,7 @@ def _asec_raw_source_mapping_frame( weeks_source, ) person = fill_asec_education_assistance_source(person, education_source) + person = fill_asec_labor_force_status_source(person, education_source) person = fill_asec_public_assistance_type_source( person, public_assistance_type_source, @@ -1962,6 +2041,14 @@ def _asec_raw_source_mapping_frame( "operation": "exact_source_join", "source_pins": education_pins, }, + "A_LFSR": { + "audit": dict(education_source.attrs.get("source_audit", {})), + "column": "A_LFSR", + "entity": "person", + "join_keys": ["source_year", "PERIDNUM"], + "operation": "exact_source_join", + "source_pins": education_pins, + }, "LKWEEKS": { "audit": dict(weeks_source.attrs.get("source_audit", {})), "column": "LKWEEKS", @@ -1978,8 +2065,8 @@ def _asec_raw_source_mapping_frame( } ], }, - # PAW_TYP lives in the same pinned survey-year person members as - # ED_VAL, so the mapping reuses those archive pins (microcosm#591). + # PAW_TYP and A_LFSR live in the same pinned survey-year person members + # as ED_VAL, so their mappings reuse those archive pins. "PAW_TYP": { "audit": dict(public_assistance_type_source.attrs.get("source_audit", {})), "column": "PAW_TYP", @@ -2002,6 +2089,10 @@ def _pre_clone_enrichment_stage( _asec_education_source_paths(args), income_years=_pooled_income_years(args), ) + education_assistance_source = load_asec_education_assistance_sources( + _asec_education_source_paths(args), + income_years=_pooled_income_years(args), + ) base = derive_us_cps_carried_inputs( raw_base, public_assistance_type_source=public_assistance_type_source, @@ -2120,8 +2211,9 @@ def _pre_clone_enrichment_stage( seed=args.seed, time_period=args.target_year, ) - base = with_us_immigration_inputs( + base = _with_us_immigration_inputs_from_asec_source( base, + education_assistance_source, seed=args.seed, time_period=args.target_year, ) diff --git a/tools/build_us_release_input_coverage_manifest.py b/tools/build_us_release_input_coverage_manifest.py index fbb5b03c..33d50444 100644 --- a/tools/build_us_release_input_coverage_manifest.py +++ b/tools/build_us_release_input_coverage_manifest.py @@ -1406,6 +1406,137 @@ ), "issue": "PolicyEngine/microcosm#312", }, + { + "id": "hr1_medicaid_humanitarian_eligibility_restoration", + "name": "H.R.1 SS71109 Medicaid humanitarian-status restoration", + "parameter_changes": { + "gov.hhs.medicaid.eligibility.eligible_immigration_statuses": { + "2026-10-01.2100-12-31": [ + "CITIZEN", + "LEGAL_PERMANENT_RESIDENT", + "REFUGEE", + "ASYLEE", + "DEPORTATION_WITHHELD", + "CUBAN_HAITIAN_ENTRANT", + "CONDITIONAL_ENTRANT", + "PAROLED_ONE_YEAR", + ] + } + }, + "budget_measure": "medicaid", + "period": 2027, + "effect_direction": "reform_minus_baseline", + "expected_sign": "positive", + "binding_inputs": ["immigration_status_str"], + "min_abs_effect": 50_000_000.0, + "reason": ( + "H.R.1 SS71109 narrows the Medicaid qualified-immigrant list to " + "citizens, LPRs, and Cuban/Haitian entrants from 2026-10-01; the " + "probe restores the pre-narrowing list at 2027 law, which binds " + "only through the REFUGEE/ASYLEE/DEPORTATION_WITHHELD/" + "PAROLED_ONE_YEAR values of immigration_status_str. The source " + "stage draws roughly 160k refugees (74% reporting Medicaid on the " + "2024 ASEC), 155k asylees, and 405k parolees to cited DHS/OHSS " + "stocks, so restoring their eligibility must re-enroll anchored " + "takers and move person-level medicaid dollars by far more than " + "the floor. A ~$0 score means the humanitarian statuses regressed " + "to zero records (microcosm #767's silent-zero failure) or the " + "engine channel broke." + ), + "issue": "PolicyEngine/microcosm#767", + }, + { + "id": "hr1_aca_below_fpl_exception_restoration", + "name": "H.R.1 SS71302 ACA below-FPL immigrant exception restoration", + "parameter_changes": { + "gov.aca.below_fpl_immigration_exception_in_effect": { + "2026-01-01.2026-12-31": True + } + }, + "budget_measure": "aca_ptc", + "period": 2026, + "effect_direction": "reform_minus_baseline", + "expected_sign": "positive", + "binding_inputs": ["immigration_status_str"], + "min_abs_effect": 5_000_000.0, + "reason": ( + "H.R.1 SS71302 repeals the 26 USC 36B(c)(1)(B) below-FPL " + "lawfully-present exception from 2026. Restoring it for 2026 " + "binds, at 2026 law, only through tax units containing a " + "non-citizen who is ACA-lawfully-present yet " + "Medicaid-ineligible by status - on this file exactly the TPS " + "population the source stage draws to the CRS RS20844 per-country " + "stocks (865k weighted persons; the other humanitarian classes " + "remain Medicaid-qualified at 2026 annual parameters). " + "Below-poverty TPS units with marketplace take-up must regain " + "premium tax credits above the floor; ~$0 means the TPS records " + "vanished or the exception channel broke." + ), + "issue": "PolicyEngine/microcosm#767", + }, + { + "id": "hr1_aca_lawful_presence_restoration", + "name": "H.R.1 SS71301 ACA lawful-presence list restoration", + "parameter_changes": { + "gov.aca.ineligible_immigration_statuses": { + "2027-01-01.2100-12-31": ["DACA", "UNDOCUMENTED"] + } + }, + "budget_measure": "aca_ptc", + "period": 2027, + "effect_direction": "reform_minus_baseline", + "expected_sign": "positive", + "binding_inputs": ["immigration_status_str"], + "min_abs_effect": 10_000_000.0, + "reason": ( + "H.R.1 SS71301 adds TPS/REFUGEE/ASYLEE/DEPORTATION_WITHHELD/" + "PAROLED_ONE_YEAR to the ACA ineligible-status list from " + "2027-01-01. The probe restores the pre-H.R.1 list (DACA and " + "UNDOCUMENTED only) at 2027, re-qualifying every humanitarian " + "status the source stage draws (~1.6M weighted persons across " + "parole, refugee, asylee, and TPS). Units among them with " + "marketplace take-up must regain premium tax credits above the " + "floor; ~$0 means the humanitarian statuses regressed or the " + "lawful-presence channel broke." + ), + "issue": "PolicyEngine/microcosm#767", + }, + { + "id": "hr1_snap_humanitarian_eligibility_restoration", + "name": "H.R.1 SS10108 SNAP humanitarian-status restoration", + "parameter_changes": { + "gov.usda.snap.eligibility.eligible_immigration_statuses": { + "2025-07-01.2100-12-31": [ + "CITIZEN", + "LEGAL_PERMANENT_RESIDENT", + "REFUGEE", + "ASYLEE", + "DEPORTATION_WITHHELD", + "CUBAN_HAITIAN_ENTRANT", + "CONDITIONAL_ENTRANT", + "PAROLED_ONE_YEAR", + ] + } + }, + "budget_measure": "snap", + "period": 2026, + "effect_direction": "reform_minus_baseline", + "expected_sign": "positive", + "binding_inputs": ["immigration_status_str"], + "min_abs_effect": 25_000_000.0, + "reason": ( + "H.R.1 SS10108 narrows SNAP alien eligibility to citizens, LPRs, " + "Cuban/Haitian entrants, and COFA citizens from 2025-07-01 " + "(modeled 2025-07). Restoring the pre-H.R.1 list at 2026 " + "re-includes the REFUGEE/ASYLEE/DEPORTATION_WITHHELD/" + "PAROLED_ONE_YEAR members the source stage draws to cited " + "DHS/OHSS stocks; is_snap_excluded_member stops excluding them, " + "so SNAP allotments for their households must rise above the " + "floor. ~$0 means the humanitarian statuses regressed to zero " + "records or the SNAP immigration channel broke." + ), + "issue": "PolicyEngine/microcosm#767", + }, ] diff --git a/tools/generate_us_bundle_from_constants.py b/tools/generate_us_bundle_from_constants.py index 465bd4a4..31c4de9b 100644 --- a/tools/generate_us_bundle_from_constants.py +++ b/tools/generate_us_bundle_from_constants.py @@ -122,7 +122,7 @@ def _domain_builders() -> dict[str, Callable[[], object]]: # frozen files, is the forward YAML -> legacy-payload path. FROZEN_LEGACY_RESOURCE_SHA256 = { "source_stages.json": ( - "dc58a0d700f0add7b658cec774df6e9587303beb58a1f432a35a18dcd1ac4097" + "788b815f748abb2061c41efb0cec4cc4952435dbcb4448ecf21ccac85a5ccec2" ), "support_spine.json": ( "68f37dc6ae6e0cde7ebccb53f88dd4a800e63456f838fa214ff98d1db8d815be" diff --git a/tools/spec_engine_coverage.py b/tools/spec_engine_coverage.py index 3f5ae7e7..b2eb3317 100644 --- a/tools/spec_engine_coverage.py +++ b/tools/spec_engine_coverage.py @@ -41,7 +41,7 @@ REPORT_SCHEMA_VERSION = 3 EXPECTED_POINTER_INVENTORY_SHA256 = ( - "3fc6b9480ea81b9635bd0db56e180c2daf32a5cd2006a70d350586c570f96754" + "faef18fd543699ba65a91f3ffaf6fdf38a807ffdc88bc3debfeb970881050d62" ) DEFAULT_REPORT_PATH = ( Path(__file__).resolve().parents[1] @@ -202,9 +202,10 @@ def valid_sha256(value: object) -> bool: if not isinstance(compiler_abi, Mapping): failures.append("compiler_ir_abi is missing") else: - if set(compiler_abi) != {"version", "sha256"} or compiler_abi.get( - "version" - ) != COMPILER_IR_ABI_VERSION: + if ( + set(compiler_abi) != {"version", "sha256"} + or compiler_abi.get("version") != COMPILER_IR_ABI_VERSION + ): failures.append("compiler_ir_abi contract differs") if not valid_sha256(compiler_abi.get("sha256")): failures.append("compiler_ir_abi/sha256 is invalid") @@ -249,9 +250,9 @@ def valid_sha256(value: object) -> bool: if not isinstance(claims, list): failures.append("field_usage claim receipts are incomplete") claims = [] - if field_usage.get("claim_count") != len(definitions) or len( - claims - ) != len(definitions): + if field_usage.get("claim_count") != len(definitions) or len(claims) != len( + definitions + ): failures.append( "field_usage claim count differs from the closed claim registry" ) diff --git a/tools/us_bundle_generation/identity_contracts.py b/tools/us_bundle_generation/identity_contracts.py index 06b2bf04..8173dabf 100644 --- a/tools/us_bundle_generation/identity_contracts.py +++ b/tools/us_bundle_generation/identity_contracts.py @@ -102,7 +102,7 @@ def _stages_with_operation( def build_seed_site_bindings( source_document: Mapping[str, Any], ) -> list[dict[str, object]]: - """Bind all 53 legacy-v1 sites to their concrete execution owners.""" + """Bind all 66 legacy-v1 sites to their concrete execution owners.""" source_ids = _source_stage_ids(source_document) transfer_nodes = tuple(group.name for group in CANONICAL_US_LATE_TRANSFER_GROUPS) @@ -173,6 +173,24 @@ def build_seed_site_bindings( "snap_discretionary_exemption_assignment": source( "snap_abawd_discretionary_exemption" ), + **{ + f"immigration_humanitarian_{label}_assignment": source("immigration_status") + for label in ( + "paroled_one_year_afghanistan", + "paroled_one_year_ukraine", + "paroled_one_year_nicaragua", + "paroled_one_year_venezuela", + "refugee", + "asylee", + "deportation_withheld", + "tps_venezuela", + "tps_el_salvador", + "tps_honduras", + "tps_nicaragua", + "tps_nepal", + "tps_other_designated", + ) + }, "immigration_ead_workers_assignment": source("immigration_status"), "immigration_ead_students_assignment": source("immigration_status"), "ssi_take_up_assignment": source("ssi_take_up"), @@ -203,9 +221,9 @@ def build_seed_site_bindings( } protocol_site_ids = tuple(site.id for site in LEGACY_V1_PROTOCOL.sites) - if len(protocol_site_ids) != 53 or set(owners) != set(protocol_site_ids): + if len(protocol_site_ids) != 66 or set(owners) != set(protocol_site_ids): raise RuntimeError( - "legacy-v1 seed owner ledger must cover exactly 53 protocol sites; " + "legacy-v1 seed owner ledger must cover exactly 66 protocol sites; " f"missing={sorted(set(protocol_site_ids) - owners.keys())!r}, " f"extra={sorted(owners.keys() - set(protocol_site_ids))!r}" ) diff --git a/tools/us_bundle_generation/imputation.py b/tools/us_bundle_generation/imputation.py index 42a11953..5f569cdb 100644 --- a/tools/us_bundle_generation/imputation.py +++ b/tools/us_bundle_generation/imputation.py @@ -871,6 +871,15 @@ def _normalise_transfer_execution( ) ) adult_care.pop("enabled") + humanitarian_immigration = deepcopy( + dict( + _mapping_like( + post_transfer["humanitarian_immigration"], + "humanitarian-immigration post-transfer contract", + ) + ) + ) + humanitarian_immigration.pop("enabled") result["predictor_bindings"] = { "person_required": "acs_person_required", "person_optional": [ @@ -890,6 +899,17 @@ def _normalise_transfer_execution( }, "contract": adult_care, }, + "humanitarian_immigration": { + "activation": { + "all_targets": list( + _array_like( + result["immigration_status_targets"], + "immigration status targets", + ) + ) + }, + "contract": humanitarian_immigration, + }, "schedule_d_capital_gain_distributions": { "activation": { "derive_schedule_d": True, @@ -2152,7 +2172,7 @@ def _assert_invariants( late_authored_output_count, len(tolerated_receipts), ) - expected_graph_counts = (2744, 92, 227, 35, 0, 213) + expected_graph_counts = (2750, 92, 227, 35, 0, 213) if graph_counts != expected_graph_counts: raise RuntimeError( "US producer graph input/output/absence counts changed: "