diff --git a/changelog.d/bea-nipa-private-group-health-insurance.added.md b/changelog.d/bea-nipa-private-group-health-insurance.added.md new file mode 100644 index 00000000..6165618f --- /dev/null +++ b/changelog.d/bea-nipa-private-group-health-insurance.added.md @@ -0,0 +1 @@ +Add BEA NIPA Table 7.8 line 17 (series B4923C), employer contributions for private group health insurance, for calendar years 2023 to 2025 from the Section 7 workbook of BEA's 30 September 2026 annual update. diff --git a/chronicle/source_package.py b/chronicle/source_package.py index f8066b9c..c45fde35 100644 --- a/chronicle/source_package.py +++ b/chronicle/source_package.py @@ -189,6 +189,9 @@ "bea/nipa_personal_income_disposition" ), "bea-nipa-pension-contributions": Path("bea/nipa_pension_contributions"), + "bea-nipa-private-group-health-insurance": Path( + "bea/nipa_private_group_health_insurance" + ), "bea-nipa-total-wages-salaries": Path("bea/nipa_total_wages_salaries"), "bea-regional-state-personal-income-components-2024": Path( "bea/regional_personal_income_state" diff --git a/db/data/bea/nipa_private_group_health_insurance/Section7All_xls.xlsx b/db/data/bea/nipa_private_group_health_insurance/Section7All_xls.xlsx new file mode 100644 index 00000000..82605e59 Binary files /dev/null and b/db/data/bea/nipa_private_group_health_insurance/Section7All_xls.xlsx differ diff --git a/db/data/bea/nipa_private_group_health_insurance/manifest.yaml b/db/data/bea/nipa_private_group_health_insurance/manifest.yaml new file mode 100644 index 00000000..ce7d5792 --- /dev/null +++ b/db/data/bea/nipa_private_group_health_insurance/manifest.yaml @@ -0,0 +1,18 @@ +source_id: bea +package_id: bea-nipa-private-group-health-insurance +dataset: bea_nipa_private_group_health_insurance +source_page: https://apps.bea.gov/iTable/?isuri=1&reqid=19&step=4&categories=flatfiles&nipa_table_list=1 +table: NIPA Table 7.8. Supplements to Wages and Salaries by Type +files: + 2025: + filename: Section7All_xls.xlsx + source_url: https://apps.bea.gov/national/Release/XLS/Survey/Section7All_xls.xlsx + sha256: de1c34e37da9b8b8d765efc0611cc653102373fe522fe9218c7689067b29b7fd + size_bytes: 944273 + fetched_at: '2026-10-10T17:56:28+00:00' + storage: + r2: + provider: r2 + bucket: ledger-raw + key: raw/bea/bea-nipa-private-group-health-insurance/2025/de1c34e37da9b8b8d765efc0611cc653102373fe522fe9218c7689067b29b7fd/Section7All_xls.xlsx + uri: r2://ledger-raw/raw/bea/bea-nipa-private-group-health-insurance/2025/de1c34e37da9b8b8d765efc0611cc653102373fe522fe9218c7689067b29b7fd/Section7All_xls.xlsx diff --git a/packages/bea/nipa_private_group_health_insurance/source_package.yaml b/packages/bea/nipa_private_group_health_insurance/source_package.yaml new file mode 100644 index 00000000..b80adefb --- /dev/null +++ b/packages/bea/nipa_private_group_health_insurance/source_package.yaml @@ -0,0 +1,150 @@ +schema_version: ledger.source_package.v1 +package_id: bea-nipa-private-group-health-insurance +label: BEA NIPA employer contributions for private group health insurance +# A cell-parsed package has no source row to read the series label from, so +# the label is declared: it is the line 17 stub of Table 7.8 (cell B26, also +# guarded below) without its indentation and footnote marker. +dimension_value_labels: + bea_nipa.series_code: + B4923C: Private group health insurance +artifact: + source_name: bea + source_table: NIPA Table 7.8. Supplements to Wages and Salaries by Type + resource_package: db + resource_directory: data/bea/nipa_private_group_health_insurance + manifest: manifest.yaml + vintage: bea_nipa_annual_update_2026_09_30 + extracted_at: "2026-10-10" + extraction_method: xlsx selected-worksheet used-range cell parse + parser: xlsx_used_range + # Section7All_xls.xlsx carries 46 worksheets; Table 7.8 annual is the only + # one this package reads. + sheets: + - T70800-A + artifact_year: 2025 +record_sets: + # NIPA Table 7.8 line 17, series B4923C: employer contributions for private + # group health insurance, a supplement to wages and salaries. The same + # series is NIPA Table 6.11D line 32, "Group health insurance". National + # accounts are a constructed estimate, so the facts are model output (as the + # other bea_nipa packages). The series is not an active-employee amount: + # see the measure's concept evidence. + - record_set_id: bea_nipa.cy{year}.private_group_health_insurance + provenance_class: model_output + record_set_spec_id: bea_nipa.private_group_health_insurance.v1 + source_record_id_prefix: bea_nipa.cy{year}.private_group_health_insurance + sheet_name: T70800-A + period_type: calendar_year + period: "{year}" + geography_id: 0100000US + geography_level: country + geography_name: United States + geography_vintage: current + entity: person + entity_role: employee + domain: compensation_of_employees + groupby_dimension: bea_nipa.series_code + rows: + - value_id: b4923c + label: Private group health insurance + ordinal: 0 + row_number: 26 + expected_row_header_column: C + expected_row_header: B4923C + filters: + bea_nipa.series_code: B4923C + constraints: + - variable: bea_nipa.series_code + operator: "==" + value: B4923C + label: BEA NIPA series code + guard_cells: + - column: A + row: 1 + expected_value: Table 7.8. Supplements to Wages and Salaries by Type + label: table title + - column: A + row: 2 + expected_value: "[Millions of dollars]" + label: unit banner + - column: A + row: 5 + expected_value: Data published September 30, 2026 + label: publication date + - column: A + row: 8 + expected_value: Line + label: line-number column header + - column: A + row: start + expected_value: 17 + label: table line number + - column: B + row: start + expected_value: ' Private group health insurance\3\' + label: line label + # Restates expected_row_header as a guard cell so the series-code + # cell is in the fact's source-cell lineage. + - column: C + row: start + expected_value: B4923C + label: series code + - column: C + row: 21 + expected_value: A2210C + label: parent health insurance series code + - column: B + row: 21 + expected_value: Health insurance + label: parent health insurance line label + - column: C + row: 11 + expected_value: B040RC + label: employer contributions for employee pension and insurance funds series code + - column: A + row: 43 + expected_value: >- + 3. Government contributions to privately administered health, + life, and workers' compensation insurance for government + employees are classified as employer contributions for employee + pension and insurance funds. + label: footnote 3 + measures: + - measure_id: amount + label: Employer contributions for private group health insurance + ordinal: 0 + column_by_year: + 2023: CA + 2024: CB + 2025: CC + source_column_id: value + expected_column_header_row: 8 + expected_column_header_by_year: + 2023: 2023 + 2024: 2024 + 2025: 2025 + concept: bea_nipa.private_group_health_insurance + source_concept: bea_nipa.b4923c_private_group_health_insurance + concept_relation: source_label + concept_authority: bea + concept_evidence_url: https://www.bea.gov/sites/default/files/methodologies/SPI-Methodology.pdf + concept_evidence_notes: >- + BEA's series register lists B4923C, "Group health insurance", at + NIPA Table 7.8 line 17 (stub "Private group health insurance") and + Table 6.11D line 32. BEA's State Personal Income and Employment: + Concepts and Methods (December 2025), paragraphs 3.18 and 3.19, says + the national estimates of employer contributions for group health + insurance plans rest mainly on Medical Expenditure Panel Survey data + covering "health insurance purchased by employers for their active + and retired employees" and self-insured plans, and, for "active and + retired federal civilian employees", on Office of Personnel + Management reports for the Federal Employee Health Benefits Program. + The NIPA Handbook (chapter 10, Table 10.B, line 32) names the same + sources and does not say whether retired employees are covered. + Chronicle's reading of the 2025 text: the series includes employer + contributions on behalf of retired employees and is not an + active-employee amount. + unit: usd + aggregation: sum + value_scale: 1000000 + expected_cell_type: number diff --git a/tests/test_chronicle_artifact_year_restamp.py b/tests/test_chronicle_artifact_year_restamp.py index 4e759b1a..b03fb52d 100644 --- a/tests/test_chronicle_artifact_year_restamp.py +++ b/tests/test_chronicle_artifact_year_restamp.py @@ -828,6 +828,7 @@ def test_no_pinned_package_restamps_at_neighbouring_or_default_years(package_pat "irs_soi/historic_table_2", "ici/fact_book_table_30", "bea/regional_personal_income_state", + "bea/nipa_private_group_health_insurance", "cbo/revenue_projections_income_by_source_2026_02", "cbo/individual_income_tax_receipts_2026_02", "cms_nhe/table_24", diff --git a/tests/test_chronicle_bea_private_group_health_insurance.py b/tests/test_chronicle_bea_private_group_health_insurance.py new file mode 100644 index 00000000..0bd4db56 --- /dev/null +++ b/tests/test_chronicle_bea_private_group_health_insurance.py @@ -0,0 +1,358 @@ +"""BEA NIPA Table 7.8 line 17: employer contributions for private group health insurance. + +The package reads series B4923C from the Section 7 workbook BEA published with +its 2026 annual update (30 September 2026). The tests pin each published value, +the guards that tie the selected cell to line 17, the row against BEA's own +flat file, and the coverage statement the facts carry: BEA's estimates include +contributions for retired employees, so the series is not an active-employee +amount (consumer context: PolicyEngine/microcosm#454). +""" + +from __future__ import annotations + +from copy import deepcopy +from pathlib import Path + +import pytest +import yaml + +from chronicle.consumer_contract import ( + consumer_fact_rows, + validate_consumer_fact_contract, +) +from chronicle.core import validate_facts +from chronicle.source_package import ( + SOURCE_PACKAGE_ALIASES, + artifact_year_restamp_issues, + load_source_package, +) +from chronicle.sources.cells import validate_source_cells +from policyengine_chronicle.schema import validate_consumer_fact_row + +REPO_ROOT = Path(__file__).resolve().parents[1] +ALIAS = "bea-nipa-private-group-health-insurance" +PACKAGE_DIR = REPO_ROOT / "packages" / "bea" / "nipa_private_group_health_insurance" +WORKBOOK_SHA256 = "de1c34e37da9b8b8d765efc0611cc653102373fe522fe9218c7689067b29b7fd" +# Table 7.8 line 17, millions of dollars, and the column each year sits in. +PUBLISHED_MILLIONS = {2023: 923_195, 2024: 977_034, 2025: 1_027_929} +YEAR_COLUMN = {2023: "CA", 2024: "CB", 2025: "CC"} +LINE_17_ROW = 26 +HEADER_ROW = 8 +# The annual flat file the other bea_nipa packages pin carries the same series +# at an earlier vintage (its manifest records the fetch on 11 May 2026). +FLAT_FILE = ( + REPO_ROOT / "db" / "data" / "bea" / "nipa_total_wages_salaries" / "NipaDataA.txt" +) + + +def _record_id(year: int) -> str: + return f"bea_nipa.cy{year}.private_group_health_insurance.b4923c.amount" + + +def _column_letters(number: int) -> str: + letters = "" + while number: + number, remainder = divmod(number - 1, 26) + letters = chr(ord("A") + remainder) + letters + return letters + + +def _variant(tmp_path: Path, edit) -> Path: + """Copy the package with ``edit`` applied; it still reads the pinned workbook.""" + payload = yaml.safe_load((PACKAGE_DIR / "source_package.yaml").read_text()) + edit(payload) + package_dir = tmp_path / "package" + package_dir.mkdir() + (package_dir / "source_package.yaml").write_text( + yaml.safe_dump(payload, sort_keys=False), encoding="utf-8" + ) + return package_dir + + +def _row(payload: dict) -> dict: + return payload["record_sets"][0]["rows"][0] + + +def _measure(payload: dict) -> dict: + return payload["record_sets"][0]["measures"][0] + + +def test_alias_resolves_to_the_package(): + assert SOURCE_PACKAGE_ALIASES[ALIAS] == Path( + "bea/nipa_private_group_health_insurance" + ) + package = load_source_package(ALIAS) + assert package.package_id == ALIAS + assert package.artifact.artifact_year == 2025 + assert package.artifact.sheets == ("T70800-A",) + + +@pytest.mark.parametrize("year", sorted(PUBLISHED_MILLIONS)) +def test_each_published_year_is_the_table_7_8_line_17_cell(year): + package = load_source_package(ALIAS) + cells = package.build_source_cells(year) + records = package.build_source_records(year, cells=cells) + facts = package.build_facts(year, cells=cells) + + assert validate_source_cells(cells).valid + assert validate_facts(facts).valid + # Only the Table 7.8 annual sheet is parsed: 43 rows by 81 columns, sparse. + assert {cell.sheet_name for cell in cells} == {"T70800-A"} + assert len(cells) == 3_483 + assert len(facts) == 1 + (fact,) = facts + (record,) = records + + assert fact.source_record_id == _record_id(year) + assert fact.value == PUBLISHED_MILLIONS[year] * 1_000_000 + assert isinstance(fact.value, int) + # The value cell first, then its column header, then the eleven guard cells. + assert record.source_cell_addresses[:2] == ( + f"{YEAR_COLUMN[year]}{LINE_17_ROW}", + f"{YEAR_COLUMN[year]}{HEADER_ROW}", + ) + assert set(record.source_cell_addresses[2:]) == { + "A1", + "A2", + "A5", + "A8", + f"A{LINE_17_ROW}", + f"B{LINE_17_ROW}", + f"C{LINE_17_ROW}", + "B21", + "C21", + "C11", + "A43", + } + + assert fact.period.type == "calendar_year" + assert fact.period.value == year + assert fact.geography.id == "0100000US" + assert fact.geography.level == "country" + assert fact.entity.name == "person" + assert fact.entity.role == "employee" + assert fact.provenance_class == "model_output" + assert fact.measure.unit == "usd" + assert fact.measure.concept == "bea_nipa.private_group_health_insurance" + assert fact.measure.source_concept == ( + "bea_nipa.b4923c_private_group_health_insurance" + ) + assert fact.measure.concept_relation == "source_label" + assert fact.measure.concept_authority == "bea" + assert fact.measure.concept_evidence_url == ( + "https://www.bea.gov/sites/default/files/methodologies/SPI-Methodology.pdf" + ) + assert fact.filters == {"bea_nipa.series_code": "B4923C"} + assert [ + (item.variable, item.operator, item.value) for item in fact.constraints + ] == [("bea_nipa.series_code", "==", "B4923C")] + assert fact.dimension_labels == {"bea_nipa.series_code": "BEA NIPA series code"} + assert fact.dimension_value_labels == { + "bea_nipa.series_code": {"B4923C": "Private group health insurance"} + } + + assert fact.source.source_name == "bea" + assert fact.source.source_table == ( + "NIPA Table 7.8. Supplements to Wages and Salaries by Type" + ) + assert fact.source.source_file == "Section7All_xls.xlsx" + assert fact.source.url == ( + "https://apps.bea.gov/national/Release/XLS/Survey/Section7All_xls.xlsx" + ) + assert fact.source.source_sha256 == WORKBOOK_SHA256 + assert fact.source.source_size_bytes == 944_273 + assert fact.source.vintage == "bea_nipa_annual_update_2026_09_30" + assert fact.source.raw_r2_uri == ( + "r2://ledger-raw/raw/bea/bea-nipa-private-group-health-insurance/2025/" + f"{WORKBOOK_SHA256}/Section7All_xls.xlsx" + ) + + +def test_the_declared_series_label_is_the_guarded_line_17_stub(): + package = load_source_package(ALIAS) + cells = {cell.address: cell for cell in package.build_source_cells(2024)} + + stub = cells[f"B{LINE_17_ROW}"].raw_value + assert stub == " Private group health insurance\\3\\" + declared = package.build_facts(2024)[0].dimension_value_labels[ + "bea_nipa.series_code" + ]["B4923C"] + assert stub.strip().removesuffix("\\3\\") == declared + assert cells[f"C{LINE_17_ROW}"].raw_value == "B4923C" + assert str(cells[f"A{LINE_17_ROW}"].raw_value) == "17" + assert cells["A2"].raw_value == "[Millions of dollars]" + assert cells["A5"].raw_value == "Data published September 30, 2026" + + +def test_consumer_rows_carry_the_value_the_workbook_digest_and_the_labels(): + package = load_source_package(ALIAS) + facts = [package.build_facts(year)[0] for year in sorted(PUBLISHED_MILLIONS)] + + assert validate_consumer_fact_contract(facts).valid + rows = consumer_fact_rows(facts) + for index, (year, row) in enumerate(zip(sorted(PUBLISHED_MILLIONS), rows), 1): + validate_consumer_fact_row(row, index, PACKAGE_DIR) + assert row["value"] == PUBLISHED_MILLIONS[year] * 1_000_000 + assert row["value_type"] == "integer" + assert row["assertion"] == "observation" + assert row["period"] == {"type": "calendar_year", "value": year} + assert row["lineage"]["source_record_id"] == _record_id(year) + assert row["layout"]["record_set_id"] == ( + f"bea_nipa.cy{year}.private_group_health_insurance" + ) + assert row["dimensions"] == {"bea_nipa.series_code": "B4923C"} + assert row["source"]["source_sha256"] == WORKBOOK_SHA256 + assert row["concept_alignment"]["authority"] == "bea" + assert row["concept_alignment"]["relation"] == "source_label" + # The exported note carries BEA's two coverage quotations and the + # scope inference Chronicle draws from them, marked as its own. + notes = row["concept_alignment"]["evidence_notes"] + assert ( + '"health insurance purchased by employers for their active and ' + 'retired employees"' + ) in notes + assert '"active and retired federal civilian employees"' in notes + assert "Chronicle's reading" in notes + assert "not an active-employee amount" in notes + assert row["concept_alignment"]["evidence_url"].endswith( + "/methodologies/SPI-Methodology.pdf" + ) + # Three years of one series: one record-set spec, three distinct facts. + assert {row["layout"]["record_set_spec_id"] for row in rows} == { + "bea_nipa.private_group_health_insurance.v1" + } + assert len({row["aggregate_fact_key"] for row in rows}) == 3 + assert len({row["semantic_fact_key"] for row in rows}) == 3 + # Deterministic: a second build writes the same rows. + again = consumer_fact_rows( + [package.build_facts(year)[0] for year in sorted(PUBLISHED_MILLIONS)] + ) + assert again == rows + + +@pytest.mark.parametrize("year", [2022, 2026]) +def test_a_year_without_a_declared_column_is_refused_not_restamped(year): + package = load_source_package(ALIAS) + + with pytest.raises(ValueError, match=f"No source artifact for year {year}"): + package.build_facts(year) + + +@pytest.mark.parametrize("year", [2022, 2023, 2024, 2026]) +def test_no_build_year_restamps_the_artifact_year_cell(year): + assert artifact_year_restamp_issues(load_source_package(ALIAS), year) == [] + + +@pytest.mark.parametrize( + ("label", "edit", "message"), + [ + ( + "line 18 (life insurance)", + lambda payload: _row(payload).update(row_number=27), + "expected row header 'B4923C', got 'A2211C'", + ), + ( + "line 16 (publicly administered funds)", + lambda payload: _row(payload).update(row_number=25), + "expected row header 'B4923C', got 'Y390RC'", + ), + ( + "the 2023 column under the 2024 label", + lambda payload: _measure(payload)["column_by_year"].update({2024: "CA"}), + "expected column header 2024, got '2023'", + ), + ], +) +def test_a_selector_that_leaves_line_17_or_its_year_is_refused( + tmp_path, label, edit, message +): + package = load_source_package(_variant(tmp_path, edit)) + + with pytest.raises(ValueError) as raised: + package.build_facts(2024) + assert message in str(raised.value), label + + +@pytest.mark.parametrize( + ("column", "row"), + [ + ("A", 1), + ("A", 2), + ("A", 5), + ("A", 8), + ("A", "start"), + ("B", "start"), + ("C", "start"), + ("C", 21), + ("B", 21), + ("C", 11), + ("A", 43), + ], +) +def test_every_guard_cell_binds(tmp_path, column, row): + """Each guard is checked against the bytes: a wrong expectation refuses.""" + + def edit(payload: dict) -> None: + guards = _row(payload)["guard_cells"] + (guard,) = [g for g in guards if g["column"] == column and g["row"] == row] + guard["expected_value"] = "not what BEA published" + + package = load_source_package(_variant(tmp_path, edit)) + + with pytest.raises(ValueError, match="not what BEA published"): + package.build_facts(2024) + + +def test_the_guards_cover_the_table_identity(): + payload = yaml.safe_load((PACKAGE_DIR / "source_package.yaml").read_text()) + guards = { + (guard["column"], guard["row"]): guard["expected_value"] + for guard in deepcopy(_row(payload)["guard_cells"]) + } + + assert guards[("A", 1)] == "Table 7.8. Supplements to Wages and Salaries by Type" + assert guards[("A", 2)] == "[Millions of dollars]" + assert guards[("A", 5)] == "Data published September 30, 2026" + assert guards[("A", "start")] == 17 + assert guards[("B", "start")] == " Private group health insurance\\3\\" + assert guards[("C", "start")] == "B4923C" + assert _row(payload)["expected_row_header"] == "B4923C" + assert _measure(payload)["value_scale"] == 1_000_000 + assert _measure(payload)["column_by_year"] == YEAR_COLUMN + assert _measure(payload)["expected_column_header_by_year"] == { + year: year for year in YEAR_COLUMN + } + + +def test_the_row_matches_the_earlier_pinned_flat_file_except_at_2024_and_2025(): + """Differential: two BEA publications of series B4923C. + + The Section 7 workbook (data published 30 September 2026) and the annual + flat file the other ``bea_nipa`` packages pin (fetched 11 May 2026) carry + the same series. They agree for every year from 1948 to 2023. For 2024 the + flat file has $1,002.920 billion and the workbook $977.034 billion, and + only the workbook has 2025. + """ + package = load_source_package(ALIAS) + cells = {cell.address: cell for cell in package.build_source_cells(2025)} + workbook = {} + for number in range(4, 82): + column = _column_letters(number) + year = int(cells[f"{column}{HEADER_ROW}"].raw_value) + workbook[year] = cells[f"{column}{LINE_17_ROW}"].raw_value + assert sorted(workbook) == list(range(1948, 2026)) + assert {year: workbook[year] for year in PUBLISHED_MILLIONS} == PUBLISHED_MILLIONS + + flat_file = {} + for line in FLAT_FILE.read_text(encoding="utf-8").splitlines(): + if line.startswith("B4923C,"): + _code, year, value = line.split(",", 2) + flat_file[int(year)] = int(value.strip('"').replace(",", "")) + assert sorted(flat_file) == list(range(1948, 2025)) + + differing = { + year: (flat_file.get(year), workbook[year]) + for year in workbook + if flat_file.get(year) != workbook[year] + } + assert differing == {2024: (1_002_920, 977_034), 2025: (None, 1_027_929)} diff --git a/tests/test_chronicle_bundle.py b/tests/test_chronicle_bundle.py index 8d9b29fb..0cba886d 100644 --- a/tests/test_chronicle_bundle.py +++ b/tests/test_chronicle_bundle.py @@ -232,13 +232,18 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): # with only treasury-estimates-of-appropriations-2026-27 at bundle year # 2023: one package/table, existing Treasury/NZ/government keys and six # existing fiscal-year keys, with no bundle warnings or duplicate keys. + # BEA NIPA Table 7.8 line 17 (B4923C, private group health insurance) adds + # one CY2023 person fact at US country geography from one new package and + # source table; the publisher, period, geography and entity keys already + # exist. Measured with only bea-nipa-private-group-health-insurance at + # bundle year 2023: no warnings and no duplicate keys. assert summary["counts"] == { "aggregate_duplicate_key_count": 0, "entity_count": 12, "error_count": 0, # chronicle#322 adds 4,947 ONS observations and 3,200 DESNZ ones. # Inner/Outer London also add two region-level geography keys. - "fact_count": 1003803, + "fact_count": 1003804, "geography_count": 20547, "period_count": 511, # 467 before chronicle#292 moved the congressional-district and @@ -261,7 +266,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "semantic_duplicate_key_count": 2324, "skipped_source_count": 10, "source_count": 56, - "source_package_count": 322, + "source_package_count": 323, # 1 semantic-duplicate warning, plus the publisher wording Chronicle # keeps as published: values two packages word differently, groupby # rows that drift inside one package (chronicle#265, #266), and the @@ -296,14 +301,14 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): if row_count == 1: assert row["aggregate_fact_key"].startswith("ledger.aggregate_fact.v2:") assert row["semantic_fact_key"].startswith("ledger.semantic_fact.v2:") - assert row_count == 1003803 + assert row_count == 1003804 assert provenance_classes <= { "administrative", "census", "model_output", "survey_aggregate", } - assert source_packages["source_package_count"] == 322 + assert source_packages["source_package_count"] == 323 assert source_packages["skipped_source_count"] == 10 assert sorted(item["source"] for item in source_packages["skipped_sources"]) == [ "census-acs-s0101-congressional-district-age-2024", @@ -317,9 +322,9 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "jct-obbba-revenue-estimates-2025", "jct-tax-expenditures-2024", ] - assert coverage["fact_count"] == 1003803 + assert coverage["fact_count"] == 1003804 assert coverage["counts"]["by_source"] == { - "bea": 445, + "bea": 446, "bfp_economic_outlook": 5, "cbo": 7, "census_acs": 468, @@ -377,7 +382,11 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "welshgov": 9567, } table_counts = coverage["counts"]["by_source_table"] - assert len(table_counts) == 316 + assert len(table_counts) == 317 + assert ( + table_counts["bea:NIPA Table 7.8. Supplements to Wages and Salaries by Type"] + == 1 + ) assert ( table_counts[ "stats_nz:Household net worth statistics: Year ended June 2024, " @@ -1601,6 +1610,8 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): expected_period_counts["fiscal_year:2023"] += 1413 # chronicle#291: IRS SOI Historic Table 2 TY2023 state AGI-band facts. expected_period_counts["tax_year:2023"] += 1020 + # BEA NIPA Table 7.8 line 17 (B4923C): the bundle year's own column. + expected_period_counts["calendar_year:2023"] += 1 assert coverage["counts"]["by_period"] == expected_period_counts # country:NZ: 330 WFF (#209) + 5 national population (#211) + 27 Treasury # (#321) + 3,522 IRD I1 (#318) + 510 IRD I2 (#210) + 103 MSD (#319) @@ -1623,7 +1634,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): assert coverage["counts"]["by_geography"]["nuts1:BE2"] == 3673 assert coverage["counts"]["by_geography"]["nuts1:BE3"] == 3662 assert coverage["counts"]["by_geography"]["commune:11002"] == 1 - assert coverage["counts"]["by_geography"]["country:0100000US"] == 2109 + assert coverage["counts"]["by_geography"]["country:0100000US"] == 2110 assert coverage["counts"]["by_geography"]["state:0400000US06"] == 249 assert ( coverage["counts"]["by_geography"]["congressional_district:5001700US0601"] == 56 @@ -1662,7 +1673,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "household": 154636, "institutional_sector": 4468, "pension_plan": 2, - "person": 496035, + "person": 496036, "return": 14600, "social_protection_scheme": 36, "tax_unit": 42388, @@ -1815,6 +1826,12 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): assert ( output_dir / "sources" / "cms-nhe-table-24" / "consumer_facts.jsonl" ).exists() + assert ( + output_dir + / "sources" + / "bea-nipa-private-group-health-insurance" + / "consumer_facts.jsonl" + ).exists() assert ( output_dir / "sources"