diff --git a/BELGIUM_PUBLIC_FACTS_REPORT.md b/BELGIUM_PUBLIC_FACTS_REPORT.md new file mode 100644 index 00000000..f9856e2a --- /dev/null +++ b/BELGIUM_PUBLIC_FACTS_REPORT.md @@ -0,0 +1,196 @@ +# Belgium public facts wave final report + +## Outcome + +The locally reviewable wave is complete at source/parser implementation head +`fe3fd81670f106c5509847312ba61dd4c4742626`, based on `origin/main` +`10597ae602767b046ca1b294949e5df7bfd3b367` on branch +`be-public-calibration-facts`. The commit containing this report necessarily +follows that reviewed implementation head; its exact hash is reported in the +final handoff response because a Git commit cannot contain its own hash. + +The existing `sfpd-legal-pension-caseload-2025` selector now builds four direct +publisher facts from the official SFPD February 2025 monthly-statistics PDF. +The intermediary handwritten CSV was removed. No regional child-benefit or +HFCS value was transcribed or represented by a placeholder fact: six blocked +official artifacts have deterministic offline-fetch instructions and five +separate follow-up issue bodies. + +Chronicle remains facts-only. This wave adds no aging, period alignment, +cross-source reconciliation, imputation, take-up mechanics, support-aware +activation, target profile, model binding, solver construction, +PolicyEngine-computed value, or Axiom concept. + +## SFPD package and source pin + +Artifact: + +- publisher: Service fédéral des Pensions (SFPD); +- URL: `https://www.sfpd.fgov.be/files/3432/fr_stat_2502.pdf`; +- filename: `fr_stat_2502.pdf`; +- SHA-256: + `0d6173e71a0e9c2cd220cd024a5b5fecbb6ca79f8b791cbd2b3368e7f8412106`; +- size: 2,319,705 bytes; +- vintage: `monthly_social_benefits_2025_02`; +- parser: `pdf_text_numbers`, preserving 13,644 full-document cells. + +| Fact | Exact publisher period | Source cell / number text | Value | +|---|---|---|---:| +| Employee and/or self-employed legal-pension beneficiaries | Snapshot at 2025-02-01 | Table 2.2 `E881` / `D881` `2.435.457` | 2,435,457 | +| Employee and/or self-employed legal-pension monthly expenditure | February 2025 | Table 2.2 `E878` / `D878` `3.759.582.728,06` | EUR 3,759,582,728.06 | +| GRAPA beneficiaries | February 2025 | Table 2.4.1 `E1756` / `D1756` `117.650` | 117,650 | +| GRAPA monthly expenditure | February 2025 | Table 2.4.1 `E1757` / `D1757` `86.398.449,47` | EUR 86,398,449.47 | + +Every fact has `assertion: observation`, `provenance_class: administrative`, +country geography `BE` / `current`, exact URL/hash/size/vintage provenance, +and source-cell keys. Page/line/number/table-title guards bind table 2.2 to +printed page 22 and table 2.4.1 to printed page 36. The GRAPA count uses a +declared `value_scale: 1000` solely to restore SFPD's single-dot grouped integer +from the parser's backward-compatible decimal representation. + +The pension population is deliberately narrowed to the employee and/or +self-employed regimes in table 2.2; it is not labelled as an all-regime total +and does not include a civil-servant column. Its published expenditure total +includes the table's pension-plus-bonus, other pension-bonus, and well-being +bonus components; Chronicle preserves that publisher total without +reclassification. + +The manifest's R2 metadata reuses the content-addressed object path first +recorded for the same PDF on the inspected unmerged GRAPA branch. An +authenticated, read-only download from the `ledger-raw` bucket returned +2,319,705 bytes with SHA-256 +`0d6173e71a0e9c2cd220cd024a5b5fecbb6ca79f8b791cbd2b3368e7f8412106`, +exactly matching the tracked publisher PDF. No object was uploaded, replaced, +or otherwise mutated during verification. + +## Original issue-69 selector audit + +Issue 69 is open. The six original selectors all remain registered and unique +on current main and resolve 587 valid facts in total. PR 207 supplied geography +and offline-fetch authoring prerequisites; it did not supply these fact +packages. + +| Alias | Selector `(source, geography, measure, period)` | Facts | Geography vintage | +|---|---|---:|---| +| `statbel-population-structure-2026` | `statbel_population_structure`, `nuts1`, `people`, 2026 | 18 | `NUTS_2024` | +| `statbel-fiscal-income-2023-nis-2025` | `statbel_fiscal_income`, `commune`, `belgium_pit_taxable_income`, 2023 | 565 | `NIS_2025` | +| `spf-finances-pit-2023` | `spf_finances_pit`, `country`, `belgium_pit_federal_and_local_tax_before_withholding`, 2023 | 1 | `current` | +| `onss-contributions-2024` | `onss_contributions`, `country`, `belgium_worker_article_17_uncapped_component_contribution`, 2024 | 1 | `current` | +| `onem-rva-unemployment-2024` | `onem_rva_unemployment`, `country`, `receives_unemployment_benefit`, 2024 | 1 | `current` | +| `nbb-national-accounts-household-disposable-income-2024` | `nbb_national_accounts`, `country`, `household_disposable_income`, 2024 | 1 | `current` | + +Current artifact pins behind those selectors: + +| Alias | Filename | SHA-256 | Bytes | +|---|---|---|---:| +| `statbel-population-structure-2026` | `statbel_population_structure_nuts1_2026.csv` | `b8456b6a7dfd71caf50184ded4f270206f3b36ae188392ade8c4375aad1ecd52` | 5,210 | +| `statbel-fiscal-income-2023-nis-2025` | `statbel_fiscal_income_commune_2023_nis_2025.csv` | `732bb3945d080c06bba2609bd55d24af0f12859840d6f6b7c3c85de8f93eed75` | 101,316 | +| `spf-finances-pit-2023` | `spf_finances_pit_country_2023.csv` | `7651aaf315f51b78d7fdbe162d6e346cc254793df02833d114c0ca5b1acf3957` | 287 | +| `onss-contributions-2024` | `onss_worker_contributions_2024.csv` | `7fcf5d07c717bcbbd18c2deddc4d7706155f7e4ffcc07161163eae3589ae1958` | 341 | +| `onem-rva-unemployment-2024` | `onem_rva_unemployment_2024.csv` | `b0b42bc0e0e5d449108290be3b235c4736f5e160a817ea5bf32edbc4779c62f7` | 687 | +| `nbb-national-accounts-household-disposable-income-2024` | `nbb_household_disposable_income_2024.csv` | `72067b13a0ae3d85cd10650bfb152aa3228657e12255c32175e11d06eee28bbe` | 292 | + +All six `validate-package` commands returned zero errors and warnings. This is +a resolution audit; it does not claim that every existing curated artifact has +the stronger direct-publisher-byte fidelity added to SFPD in this wave. + +## Unresolved official sources + +`FETCH-MANIFEST-BELGIUM-PUBLIC-FACTS.json` validates under +`ledger.offline_fetch_manifest.v1` with required discovery notes and six +artifacts: + +1. Opgroeien native Groeipakket caseload export; +2. Opgroeien native Groeipakket expenditure export; +3. official AVIQ 2021 annual-report PDF; +4. official Iriscare 2024 annual-report PDF; +5. raw official Ostbelgien Statistik family-allowance HTML/endpoint response; +6. official ECB HFCS Wave 2023 statistical-tables ZIP, version 5.0 / June 2026. + +Each instruction requires unchanged native publisher bytes, exact filters and +period/scope/unit labels, SHA-256, byte size, content-addressed storage, and a +stop when native data are unavailable or ambiguous. No browser card, +screenshot, OCR output, search snippet, accessibility text, or manual +transcription is permitted. + +Ready-to-file issue bodies are committed at: + +- `docs/issue-drafts/belgium-opgroeien-native-exports.md`; +- `docs/issue-drafts/belgium-aviq-family-allowances.md`; +- `docs/issue-drafts/belgium-iriscare-family-allowances.md`; +- `docs/issue-drafts/belgium-ostbelgien-family-allowances.md`; +- `docs/issue-drafts/belgium-ecb-hfcs-wave-2023.md`. + +## Deterministic validation evidence + +- SFPD `validate-package`: PASS, four record sets, four rows, four measures, + four source records, zero errors/warnings. +- SFPD `build-suite`: PASS, 13,644 source cells, four facts, 100% lineage, zero + agent-acceptance errors; source-cell and raw-fact reports valid. +- Package-specific `build-bundle`: PASS, four facts, no duplicate keys or + warnings. +- Facts-only consumer artifact build/load: PASS, four schema-v2 rows, all with + source-cell lineage. +- The European number format is source-scoped to the SFPD package. The default + parser remains byte-for-byte compatible at the token/scalar boundary for the + eight existing `pdf_text_numbers` packages. +- Six original issue-69 package validators: PASS (18 + 565 + 1 + 1 + 1 + 1 + source records). +- Parser/SFPD and all five packages exposed by the first CI run: 27 passed. +- Focused SFPD and Belgium package tests: 3 passed; the current Belgium + selector uniqueness test separately passed. +- `tests/test_chronicle_facts_only.py` plus + `tests/test_chronicle_consumer.py`: 33 passed. +- Full 151-package merged-bundle contract: PASS in 14m11s, with 171,855 + facts, 43 sources, 146 source tables, zero aggregate duplicate keys, and the + one pre-existing semantic-duplicate warning. +- `ruff check .`: PASS. +- `git diff --check origin/main`: PASS. + +The first CI run exposed that the initial implementation had broadened the +shared number grammar and shifted rows in existing Welsh CTR, Medicare, and +LIHEAP packages. The final implementation restores the original default +grammar and selects the European grammar explicitly from the SFPD artifact. +All five directly exposed package regressions pass locally. The first complete +151-package run then exposed only stale bundle goldens caused by the four SFPD +facts moving from calendar-year to February 2025 and the two expenditure facts +being classified as government rather than person. After updating those exact +period and entity histograms, the complete merged-bundle test passed. No fact +value, source selector, or package semantic changed in the golden correction. + +Required judge evidence: + +- `ledger-source-fidelity`: **PASS**, no blocking fidelity finding; +- `ledger-boundary`: **PASS**, no facts-only boundary blocker. + +## Governance and external blockers + +The literal `ledger-source-ingestor.allowed_paths` globs omit paths the user +explicitly required for this task: root `PROGRESS.md`, root offline handoff, +`db/data/**` publisher artifacts/manifests, issue drafts, and the pre-existing +`tests/test_belgium_targets.py`. The work makes no contract, schema, core, +geography transform, model, or consumer-owned change, but strict path +enforcement requires maintainer acceptance or a registry correction before +merge. + +The branch is published as draft PR #213, and the five blocked-source handoffs +are filed as issues #214--#218. Nothing is merged. The final head must remain +draft until final-head CI, the path-registry decision, and a renewed exact-head +Fable review are complete. + +The tracked PR body references open issue 69 without closing it and must be +republished and verified with the final parser-correction head. + +## Actual commit messages through the reviewed implementation head + +1. `f5296028ed62722d199af4a88d07e8f5dd439e3c` — `Start Belgium public facts progress log` +2. `9c2ee9eb0d63309da03f580a7425eba491d32445` — `Track Belgium public facts wave at repository root` +3. `6fd6b91b9d3df9f07873fd917f18eb3979a657d9` — `Parse European-formatted publisher document numbers` +4. `339717e108242e5c157df1164f8a2ba29bc76cd3` — `Back SFPD pension and GRAPA facts with publisher PDF` +5. `a8b5cf46c6465e84e0e57f717049482a289e0667` — `Add offline handoffs for blocked Belgium sources` +6. `9abffbf830779840660e7ebe3b2dc218a7a92dc8` — `Draft follow-up issues for blocked Belgium facts` +7. `6d3694dfa03626625a1af9f802469d40f19d14b1` — `Normalize Belgium issue draft endings` +8. `94a298adf9909edfd15453f387a23b4aeec83125` — `Record Belgium validation and bundle coverage` +9. `206ebfa38bdc65ed4d89f416cf3cbf78eada81dd` — `Prepare Belgium public facts draft PR` +10. `ea809d084ed4913d41f9c629b41fe85cb7e23e46` — `Finalize Belgium public facts report` +11. `fe3fd81670f106c5509847312ba61dd4c4742626` — `Scope European number parsing to SFPD` diff --git a/FETCH-MANIFEST-BELGIUM-PUBLIC-FACTS.json b/FETCH-MANIFEST-BELGIUM-PUBLIC-FACTS.json new file mode 100644 index 00000000..248fe32c --- /dev/null +++ b/FETCH-MANIFEST-BELGIUM-PUBLIC-FACTS.json @@ -0,0 +1,144 @@ +{ + "schema_version": "ledger.offline_fetch_manifest.v1", + "generated_for": "Belgium public pension, regional child-benefit, and HFCS source packages for Chronicle issue 69", + "artifacts": [ + { + "package_id": "opgroeien-groeipakket-native-caseload", + "url": "https://app.powerbi.com/view?r=eyJrIjoiZTFiMzllMTUtNmUxNy00MWVkLWJiOGEtNGE3ZmVlZTY3NjZhIiwidCI6IjEzY2ZlMTgyLTY0MmEtNDgwZS1iYzhhLTVlY2Y2NWRiMGFhMCIsImMiOjh9", + "expected_filename": "opgroeien_groeipakket_caseload_native_export.csv", + "destination_path": "db/data/opgroeien/groeipakket_native_exports/opgroeien_groeipakket_caseload_native_export.csv", + "manifest_path": "db/data/opgroeien/groeipakket_native_exports/manifest.yaml", + "manifest_year": 2024, + "discovery_note": "Official Opgroeien Cijfers op maat landing page https://www.opgroeien.be/kennis/cijfers-en-onderzoek/groeipakket/cijfers-op-maat embeds this public Power BI report. This handoff is for a native data export, never a screenshot, copied accessibility text, or manual transcription.", + "r2": { + "provider": "r2", + "bucket": "ledger-raw", + "key_template": "raw/belgium/opgroeien/opgroeien-groeipakket-native-caseload/2024/{sha256}/opgroeien_groeipakket_caseload_native_export.csv", + "uri_template": "r2://ledger-raw/raw/belgium/opgroeien/opgroeien-groeipakket-native-caseload/2024/{sha256}/opgroeien_groeipakket_caseload_native_export.csv" + }, + "post_download_steps": [ + "Reach the report through the official Opgroeien landing page, reset all report filters, open the caseload data-table visual, choose Export data, select the native summarized/underlying-data CSV option with the current layout, and save the response bytes unchanged under expected_filename. If no tabular CSV export is enabled, or an export omits its period and measure labels, STOP; do not copy values from the rendered dashboard.", + "Record the report URL, report-visible update timestamp, every active filter, exported column headers, period field, administrative scope field, measure label, and unit in manifest.yaml. The artifact may contain several periods; do not relabel any row to 2024 merely because manifest_year is 2024.", + "Compute SHA-256 and size_bytes over the unchanged exported bytes, archive them at the content-addressed R2 template, and record the exact source URL, source table/visual name, hash, size, and R2 pointer in manifest.yaml.", + "Inspect whether the export's geography is an Opgroeien administrative service population, the Flemish Region, the Flemish Community, or another published scope. Preserve that publisher scope exactly; STOP rather than assigning BE2 unless the export itself supports BE2.", + "Only after the artifact is pinned, author exact-period caseload facts selected from native export cells and run validate-package plus source-cell, fact-load, consumer-artifact, and raw-facts-boundary tests." + ] + }, + { + "package_id": "opgroeien-groeipakket-native-expenditure", + "url": "https://app.powerbi.com/view?r=eyJrIjoiMzU3YjY1ZTItNTAxYi00MjdhLTg3OGItMDQ5ZjJiMjU5NjlkIiwidCI6IjEzY2ZlMTgyLTY0MmEtNDgwZS1iYzhhLTVlY2Y2NWRiMGFhMCIsImMiOjh9", + "expected_filename": "opgroeien_groeipakket_expenditure_native_export.csv", + "destination_path": "db/data/opgroeien/groeipakket_native_exports/opgroeien_groeipakket_expenditure_native_export.csv", + "manifest_path": "db/data/opgroeien/groeipakket_native_exports/manifest.yaml", + "manifest_year": 2024, + "discovery_note": "Official Opgroeien Cijfers op maat landing page links this Groeipakket budget Power BI report. Retrieve the dashboard's native expenditure data rather than reading rendered cards or charts.", + "r2": { + "provider": "r2", + "bucket": "ledger-raw", + "key_template": "raw/belgium/opgroeien/opgroeien-groeipakket-native-expenditure/2024/{sha256}/opgroeien_groeipakket_expenditure_native_export.csv", + "uri_template": "r2://ledger-raw/raw/belgium/opgroeien/opgroeien-groeipakket-native-expenditure/2024/{sha256}/opgroeien_groeipakket_expenditure_native_export.csv" + }, + "post_download_steps": [ + "Reach the report through the official Opgroeien landing page, reset all report filters, open the expenditure/budget data-table visual, choose Export data, select the native summarized/underlying-data CSV option with the current layout, and save the response bytes unchanged under expected_filename. If no tabular CSV export is enabled, or period, component, currency, and unit labels are absent, STOP without transcribing chart values.", + "Record the report-visible update timestamp, exact visual name, all filters, exported headers, publisher period labels, whether values are budget or realized expenditure, currency, and any printed scale in manifest.yaml.", + "Compute SHA-256 and size_bytes over the unchanged export, archive it at the content-addressed R2 template, and record the source URL, hash, size, and R2 pointer.", + "Author only publisher-labelled expenditure facts for exact periods and scopes. Do not derive annual totals from monthly values, reconcile the expenditure export to caseload, or compute per-recipient amounts.", + "Run validate-package plus source-cell, fact-load, consumer-artifact, and raw-facts-boundary tests after the native artifact is pinned." + ] + }, + { + "package_id": "aviq-annual-report-2021-family-allowances", + "url": "https://www.aviq.be/sites/default/files/documents_pro/2022-10/rapport_annuel_AVIQ_2021.pdf", + "expected_filename": "rapport_annuel_AVIQ_2021.pdf", + "destination_path": "db/data/aviq/annual_report_2021/rapport_annuel_AVIQ_2021.pdf", + "manifest_path": "db/data/aviq/annual_report_2021/manifest.yaml", + "manifest_year": 2021, + "discovery_note": "Official AVIQ 2021 annual-report PDF; use the publisher PDF tables that label post-2019 Walloon family-allowance caseload and expenditure. Do not use the secondary parliamentary-answer extract as the agency artifact.", + "r2": { + "provider": "r2", + "bucket": "ledger-raw", + "key_template": "raw/belgium/aviq/aviq-annual-report-2021-family-allowances/2021/{sha256}/rapport_annuel_AVIQ_2021.pdf", + "uri_template": "r2://ledger-raw/raw/belgium/aviq/aviq-annual-report-2021-family-allowances/2021/{sha256}/rapport_annuel_AVIQ_2021.pdf" + }, + "post_download_steps": [ + "Download the PDF response bytes directly from the official AVIQ URL; reject HTML challenge/error content and do not substitute a mirror.", + "Compute SHA-256 and size_bytes, archive the unchanged PDF at the content-addressed R2 template, and record the direct URL, hash, size, publication vintage, and exact table/page labels in manifest.yaml.", + "Parse the full PDF through a deterministic source-cell lane. Select only AVIQ-published family-allowance caseload and expenditure cells whose reference periods, units, and Walloon administrative/geographic scope are explicit in the PDF.", + "Do not reconcile AVIQ values to FAMIWAL, derive missing years, infer recipient units, or compute per-recipient amounts. Ambiguous or rounded values remain documented gaps.", + "Run validate-package plus source-cell, fact-load, consumer-artifact, and raw-facts-boundary tests." + ] + }, + { + "package_id": "iriscare-annual-report-2024-family-allowances", + "url": "https://rapport.iriscare.brussels/wp-content/uploads/2025/09/Rapport-annuel-2024.pdf", + "expected_filename": "Rapport-annuel-2024.pdf", + "destination_path": "db/data/iriscare/annual_report_2024/Rapport-annuel-2024.pdf", + "manifest_path": "db/data/iriscare/annual_report_2024/manifest.yaml", + "manifest_year": 2024, + "discovery_note": "Official Iriscare annual-report PDF for Brussels. It is the preferred static primary artifact to inspect before using the separate Iriscare/Famiris Fabric dashboards.", + "r2": { + "provider": "r2", + "bucket": "ledger-raw", + "key_template": "raw/belgium/iriscare/iriscare-annual-report-2024-family-allowances/2024/{sha256}/Rapport-annuel-2024.pdf", + "uri_template": "r2://ledger-raw/raw/belgium/iriscare/iriscare-annual-report-2024-family-allowances/2024/{sha256}/Rapport-annuel-2024.pdf" + }, + "post_download_steps": [ + "Download the PDF response bytes directly from the official Iriscare annual-report host; reject HTML challenge/error content and do not substitute a mirror.", + "Compute SHA-256 and size_bytes, archive the unchanged PDF at the content-addressed R2 template, and record the direct URL, hash, size, publication vintage, and exact page/table labels in manifest.yaml.", + "Parse the full PDF deterministically and inspect every occurrence of family-allowance caseload and expenditure. If two publisher locations give different values or accounting scopes, preserve them as separately labelled source assertions only when both scopes are explicit; otherwise STOP and document the ambiguity.", + "Preserve the publisher's Brussels administrative/geographic scope, reference period, currency, scale, and accounting label. Do not combine Famiris and other fund values or compute missing totals.", + "Run validate-package plus source-cell, fact-load, consumer-artifact, and raw-facts-boundary tests." + ] + }, + { + "package_id": "ostbelgien-family-allowances-2025", + "url": "https://ostbelgienstatistik.be/desktopdefault.aspx/tabid-3748/6766_read-39090/", + "expected_filename": "ostbelgien_family_allowances_2025.html", + "destination_path": "db/data/ostbelgien/family_allowances_2025/ostbelgien_family_allowances_2025.html", + "manifest_path": "db/data/ostbelgien/family_allowances_2025/manifest.yaml", + "manifest_year": 2025, + "discovery_note": "Official Ostbelgien Statistik family-allowance page. Automated retrieval returns HTTP 403 in the current environment; capture the publisher's raw page response in an authorized browser session, never values copied from search snippets or rendered cards.", + "r2": { + "provider": "r2", + "bucket": "ledger-raw", + "key_template": "raw/belgium/ostbelgien/ostbelgien-family-allowances-2025/2025/{sha256}/ostbelgien_family_allowances_2025.html", + "uri_template": "r2://ledger-raw/raw/belgium/ostbelgien/ostbelgien-family-allowances-2025/2025/{sha256}/ostbelgien_family_allowances_2025.html" + }, + "post_download_steps": [ + "Open the exact official URL in an authorized browser session, verify the Ostbelgien Statistik host and family-allowance page title, then save the raw HTML response/source bytes under expected_filename. Do not save a screenshot, print-to-PDF rendering, accessibility tree, search cache, or manually reconstructed table.", + "Verify that the saved bytes contain the publisher table labels, reference periods, units, and the caseload/expenditure values visible on the official page. If the data are injected from a separate official endpoint, STOP and replace this handoff with that native endpoint rather than packaging incomplete shell HTML.", + "Compute SHA-256 and size_bytes, archive the unchanged publisher response at the content-addressed R2 template, and record URL, HTTP retrieval details, page update timestamp, hash, size, and R2 pointer in manifest.yaml.", + "Parse the HTML deterministically and retain the German-speaking Community administrative/geographic scope and exact publisher periods. Do not transcribe search-index values or infer missing units.", + "Run validate-package plus source-cell, fact-load, consumer-artifact, and raw-facts-boundary tests." + ] + }, + { + "package_id": "ecb-hfcs-wave-2023-statistical-tables", + "url": "https://www.ecb.europa.eu/home/pdf/research/hfcn/HFCS_Statistical_Tables_Wave_2023_June_2026.zip", + "expected_filename": "HFCS_Statistical_Tables_Wave_2023_June_2026.zip", + "destination_path": "db/data/ecb_hfcs/wave_2023_statistical_tables/HFCS_Statistical_Tables_Wave_2023_June_2026.zip", + "manifest_path": "db/data/ecb_hfcs/wave_2023_statistical_tables/manifest.yaml", + "manifest_year": 2023, + "discovery_note": "Official ECB Household Finance and Consumption Survey wave-2023 statistical-tables ZIP, version 5.0 published June 2026. Belgium fieldwork was January-December 2023; wealth components refer to interview time while income refers to 2022.", + "r2": { + "provider": "r2", + "bucket": "ledger-raw", + "key_template": "raw/belgium/ecb_hfcs/ecb-hfcs-wave-2023-statistical-tables/2023/{sha256}/HFCS_Statistical_Tables_Wave_2023_June_2026.zip", + "uri_template": "r2://ledger-raw/raw/belgium/ecb_hfcs/ecb-hfcs-wave-2023-statistical-tables/2023/{sha256}/HFCS_Statistical_Tables_Wave_2023_June_2026.zip" + }, + "post_download_steps": [ + "Download the ZIP response bytes directly from the official ECB URL and verify that the response is a readable ZIP, not an HTML error page.", + "Compute the outer ZIP SHA-256 and size_bytes, list every archive member in deterministic lexical order, compute a SHA-256 for each selected workbook member, and record the outer and member pins plus ECB version 5.0 / June 2026 vintage in manifest.yaml.", + "Inspect the native workbook sheets and headers before authoring. For Belgium, preserve fieldwork/reference timing exactly: assets and liabilities are measured at interview time in 2023, while income variables refer to 2022.", + "Select only ECB-published Belgium aggregates such as net-wealth means/medians/quantiles, distribution shares or ratios, and asset/liability component aggregates with exact cells, units, weighting labels, and survey provenance. Do not derive components, interpolate quantiles, reconcile totals, or map facts to model targets.", + "Run validate-package plus source-cell, fact-load, consumer-artifact, and raw-facts-boundary tests." + ] + } + ], + "final_validation": [ + "Validate this handoff with load_offline_fetch_manifest(..., require_discovery_notes=True) before retrieval.", + "For each retrieved artifact, verify the exact SHA-256 and size recorded in its adjacent manifest and archive the unchanged bytes under the content-addressed R2 key.", + "Do not create a source package until native publisher bytes, exact source cells, periods, geography/scope, units, assertion type, URL, checksum, and deterministic selectors are all available.", + "Run validate-package, source-cell preservation, fact-load, consumer-artifact, raw-facts-boundary, ruff, and git diff --check for each implemented follow-up package." + ] +} diff --git a/PROGRESS.md b/PROGRESS.md index 24773009..fac9a137 100644 --- a/PROGRESS.md +++ b/PROGRESS.md @@ -1,51 +1,84 @@ -# Lane C5 progress +# Belgium public facts wave progress ## State -- Branch: `be-2025-vintages` from `origin/main` at `5c15bfd`. -- Worktree inputs are staged under `.lane-raw/` and must remain uncommitted. -- Lane C5 is complete, validated, independently reviewed, and ready for handoff. -- The requested staged C2 report is absent, but root `LANE_C2_REPORT.md` is byte-identical - to the sibling lane's staged copy (SHA-256 `4590e0dc...50f06e7`) and is the pattern used. +- Active role: `ledger-source-ingestor`. +- Branch: `be-public-calibration-facts`, based on clean `origin/main` at + `10597ae`. +- Phase: draft PR #213 published; final-head bundle goldens corrected and the + complete local bundle gate passed; final-head CI and renewed Fable review + remain pending. ## Done -- Read the repository Chronicle boundary rules in `AGENTS.md`. -- Read `.lane-raw/SOURCES.md` and confirmed all five named publisher artifacts are present. -- Confirmed the worktree is otherwise clean apart from `.lane-raw/` and the shared `.venv` link. -- Verified all five staged artifact SHA-256 pins exactly. -- Mapped FPB workbook cells: 990 facts across T01/T06/T07/T11/T17/T24, with - 2022–2025 observations and 2026–2031 `source_projection` facts. -- Confirmed PDF boundary evidence: printed page 19 calls 2026 the first projection year; - annex table units appear on printed pages 45, 48, 49, 53, 58, and 65. -- Chosen Eurostat layout: two vintage-specific source-package aliases share new manifest - entries, preserving the prior package YAMLs, raw bytes, and fact outputs unchanged. -- Reproduced the Statbel curator logic: 18 NUTS1 × sex × age-band cells totaling 11,825,551. -- Added the hash-pinned FPB workbook and publication PDF plus the - `fpb-economic-outlook-2026-2031-june-2026` package alias. -- Built 990 line-specific publisher facts (99 per year): 396 observations for - 2022–2025 and 594 `source_projection` facts for 2026–2031. -- Passed FPB `validate-package` and `build-suite`: 990 facts, full cell lineage, - zero acceptance errors, and pinned 2025 cells 320578 / 77771 / 5602 million euro. -- Re-ran the Statbel 2026 curator logic on the 2025 ZIP and added the hash-pinned - raw capture plus its deterministic 18-row curated CSV. -- Passed Statbel 2025 `validate-package` and `build-suite`: 18 facts totaling - 11,825,551, 66 constraints, full lineage, and zero acceptance errors. -- Added the Eurostat `gov_10a_taxag` 2025 and `spr_exp_func` 2024 manifest - entries plus vintage-specific package aliases, without modifying either - prior artifact or prior package specification. -- Passed both new Eurostat package validations and suite builds: 12 tax facts - and 9 ESSPROS facts, full lineage, and zero acceptance errors. -- Extended Belgium and Eurostat regressions for FPB table counts/cells and - assertion boundary, vintage non-overlap, prior-output digests, Statbel pins, - and the declared 0.25% Statbel/FPB population comparison tolerance. -- Passed 43 focused tests and the full merged-bundle regression: 157,177 facts, - 148 packages, zero aggregate-key duplicates, and expected goldens throughout. -- Recorded pins, counts, boundary evidence, curator commands, validation tails, - and consumer fact families in `LANE_C5_REPORT.md`. -- Passed independent `ledger-source-fidelity` and `ledger-boundary` reviews with - no required corrections. +- Read `AGENTS.md`, `.github/chronicle-agents.yml`, the complete source-package + harness, and the facts-only ADR before changing package content. +- Confirmed Chronicle must retain publisher-period facts only and must not add + Microcosm bindings, target profiles, aging, reconciliation, imputation, + take-up mechanics, PolicyEngine-computed values, or Axiom concepts. +- Confirmed issue 69 remains open and names SFPD pensions, regionalized child + benefits, NBB national accounts, and other original Belgium source families. +- Confirmed the merged offline-fetch validation contract exists on current main. +- Identified current-main SFPD and Opgroeien packages as curated CSV extracts + whose value artifacts are not immutable publisher downloads; they resolve as + packages but do not satisfy this wave's stronger publisher-byte requirement. +- Inspected the unmerged `be-benefit-participation-facts` branch. Its official + SFPD and Walloon Parliament PDFs are potentially reusable, while its manual + dashboard/HTML transcriptions are excluded by this task's no-transcription + rule. +- Confirmed the official SFPD February 2025 monthly-statistics PDF contains + publisher tables for pension beneficiary counts, monthly pension expenditure, + GRAPA beneficiary counts, and monthly GRAPA expenditure. +- Added a source-scoped European document-number format for the SFPD package, + with regression coverage for its number shapes. The original default parser + remains unchanged for every existing package; single-dot European values + remain backward-compatible decimals. +- Audited the six original issue-69 selectors on current main: all six aliases + are registered, build 587 unique valid facts in total, and pass their package + validators. PR 207 added only geography/offline-fetch prerequisites, not + these fact packages. +- Pinned the verbatim official SFPD February 2025 PDF (SHA-256 + `0d6173e71a0e9c2cd220cd024a5b5fecbb6ca79f8b791cbd2b3368e7f8412106`, + 2,319,705 bytes) and replaced the intermediary pension CSV selectors with + direct publisher-cell selectors for pension and GRAPA beneficiary totals and + monthly expenditure. +- Added a schema-validated offline-fetch handoff for the native Opgroeien + caseload and expenditure exports, official AVIQ and Iriscare annual-report + PDFs, the blocked Ostbelgien Statistik HTML response, and the ECB HFCS wave + 2023 statistical-tables ZIP. No unresolved source has a placeholder fact. +- Prepared and filed separate follow-up issues for Opgroeien, AVIQ, Iriscare, + Ostbelgien, and ECB HFCS. They all reference open issue 69 and the + deterministic handoff. +- Validated the SFPD package with zero errors or warnings; its suite preserves + 13,644 full-document cells, resolves four facts with 100% lineage, and builds + and reloads a four-row facts-only consumer artifact. +- Revalidated the parser/SFPD tests and all five existing packages that failed + when the first implementation accidentally broadened the default grammar; + all 27 tests pass after restoring the default and scoping European parsing. +- Focused parser, offline-fetch, SFPD, selector, facts-only, and consumer tests + passed. Ruff and `git diff --check origin/main` passed. +- Authenticated the content-addressed `ledger-raw` object with a read-only + download: its 2,319,705 bytes and SHA-256 exactly match the tracked SFPD PDF. + No R2 object was mutated. +- Obtained `PASS` verdicts from both required judges: + `ledger-source-fidelity` and `ledger-boundary`. +- Completed the 151-package merged-bundle regression. Its first final-head run + passed validity and the exact aggregate/source/table/geography checks, then + exposed stale period and entity goldens for the four SFPD facts. Updated only + those exact histograms: four facts move from calendar year 2025 to February + 2025, while two expenditure facts move from person to government. The full + contract then passed with 171,855 facts in 14m11s. +- Recorded the governance inconsistency that the literal source-ingestor path + globs omit user-required progress, raw-artifact, offline-handoff, issue-draft, + and pre-existing Belgium-test paths. No contract, schema, core, model, or + consumer-owned behavior changed. +- Completed self-review and committed the exact proposed draft-PR body. +- Published draft PR #213 and filed the five deterministic follow-up handoffs + as issues #214--#218. Nothing was merged. +- Wrote the final report to `BELGIUM_PUBLIC_FACTS_REPORT.md`. ## Next -- None; ready for handoff. No push was performed. +- Complete final-head CI after publishing the bundle-golden correction. +- Resolve the source-ingestor allowed-path registry gap. +- Obtain a renewed exact-head Fable review; keep PR #213 draft and do not merge. diff --git a/chronicle/source_package.py b/chronicle/source_package.py index 993b4d9f..d2c5e55c 100644 --- a/chronicle/source_package.py +++ b/chronicle/source_package.py @@ -387,6 +387,7 @@ class SourceArtifactSpec: extracted_at: str extraction_method: str parser: str = "xls_used_range" + number_format: str = "english" sheet_name: str | None = None archive_member: str | None = None artifact_year: int | None = None @@ -543,9 +544,17 @@ def build_source_cells( if self.parser == "ods_used_range": return source_cells_from_ods(content, artifact) if self.parser == "html_tables_and_text": - return source_cells_from_html_tables_and_text(content, artifact) + return source_cells_from_html_tables_and_text( + content, + artifact, + number_format=self.number_format, + ) if self.parser == "pdf_text_numbers": - return source_cells_from_pdf_text_numbers(content, artifact) + return source_cells_from_pdf_text_numbers( + content, + artifact, + number_format=self.number_format, + ) if self.parser == "delimited_text_selected_rows": return source_cells_from_delimited_text( content, @@ -1664,6 +1673,7 @@ def _artifact_from_mapping(payload: dict[str, Any]) -> SourceArtifactSpec: extracted_at=_required(payload, "extracted_at", "artifact"), extraction_method=_required(payload, "extraction_method", "artifact"), parser=payload.get("parser", "xls_used_range"), + number_format=payload.get("number_format", "english"), sheet_name=payload.get("sheet_name"), archive_member=payload.get("archive_member"), artifact_year=( diff --git a/chronicle/sources/cells.py b/chronicle/sources/cells.py index 25aa2255..13f15e3a 100644 --- a/chronicle/sources/cells.py +++ b/chronicle/sources/cells.py @@ -231,6 +231,8 @@ def source_cells_from_ods( def source_cells_from_html_tables_and_text( content: bytes, artifact: SourceArtifactMetadata, + *, + number_format: str = "english", ) -> list[SourceCell]: """Parse HTML tables and numeric document text into source cells.""" parser = _HtmlTableAndTextParser() @@ -246,13 +248,21 @@ def source_cells_from_html_tables_and_text( sheet_name=f"table_{table_index}", ) ) - cells.extend(_html_document_number_cells(parser.blocks, artifact)) + cells.extend( + _html_document_number_cells( + parser.blocks, + artifact, + number_format=number_format, + ) + ) return cells def source_cells_from_pdf_text_numbers( content: bytes, artifact: SourceArtifactMetadata, + *, + number_format: str = "english", ) -> list[SourceCell]: """Parse PDF text lines into numeric document source cells.""" from pypdf import PdfReader @@ -277,6 +287,7 @@ def source_cells_from_pdf_text_numbers( ) for index, value in enumerate(headers) ] + number_pattern = _document_number_pattern(number_format) row_number = 2 for page_number, page in enumerate(reader.pages, start=1): text = page.extract_text() or "" @@ -292,7 +303,7 @@ def source_cells_from_pdf_text_numbers( max(0, line_index - 2) : min(len(lines), line_index + 3) ] ) - for match in _HTML_NUMBER_RE.finditer(normalized_line): + for match in number_pattern.finditer(normalized_line): number_text = match.group(0) for column_number, raw_value in enumerate( ( @@ -300,7 +311,10 @@ def source_cells_from_pdf_text_numbers( line_number, normalized_line, number_text, - _html_number_scalar(number_text), + _html_number_scalar( + number_text, + number_format=number_format, + ), context_text, ), start=1, @@ -643,6 +657,13 @@ def _ods_cell_text(cell: ElementTree.Element) -> str: r"(?:\s*(?:thousand|million|billion)|bn)?(?![\w.])", flags=re.IGNORECASE, ) +_EUROPEAN_NUMBER_RE = re.compile( + r"(? list[SourceCell]: sheet_name = "document_numbers" headers: tuple[Scalar, ...] = ( @@ -811,9 +834,10 @@ def _html_document_number_cells( ) for index, value in enumerate(headers) ] + number_pattern = _document_number_pattern(number_format) row_number = 2 for block in blocks: - for match in _HTML_NUMBER_RE.finditer(block.text): + for match in number_pattern.finditer(block.text): number_text = match.group(0) for column_number, raw_value in enumerate( ( @@ -821,7 +845,10 @@ def _html_document_number_cells( block.tag, block.text, number_text, - _html_number_scalar(number_text), + _html_number_scalar( + number_text, + number_format=number_format, + ), ), start=1, ): @@ -882,8 +909,20 @@ def _html_scalar(text: str) -> Scalar: return text -def _html_number_scalar(text: str) -> int | float: - normalized = text.replace(",", "").lower().lstrip("£$€") +def _document_number_pattern(number_format: str) -> re.Pattern[str]: + if number_format == "english": + return _HTML_NUMBER_RE + if number_format == "european": + return _EUROPEAN_NUMBER_RE + raise ValueError(f"Unsupported document number format: {number_format}") + + +def _html_number_scalar( + text: str, + *, + number_format: str, +) -> int | float: + normalized = text.lower().lstrip("£$€") multiplier = 1 for suffix, value in ( ("thousand", 1_000), @@ -895,6 +934,29 @@ def _html_number_scalar(text: str) -> int | float: multiplier = value normalized = normalized[: -len(suffix)].strip() break + if number_format == "english": + normalized = normalized.replace(",", "") + elif number_format != "european": + raise ValueError(f"Unsupported document number format: {number_format}") + elif "," in normalized and "." in normalized: + if normalized.rfind(",") > normalized.rfind("."): + # European grouped decimal, for example 3.759.582.728,06. + normalized = normalized.replace(".", "").replace(",", ".") + else: + # English grouped decimal, for example 3,759,582,728.06. + normalized = normalized.replace(",", "") + elif normalized.count(".") > 1: + # Multiple dots unambiguously form European thousands groups. + normalized = normalized.replace(".", "") + elif normalized.count(",") > 1: + normalized = normalized.replace(",", "") + elif "," in normalized: + whole, fraction = normalized.rsplit(",", 1) + normalized = ( + whole + fraction + if len(fraction) == 3 and whole.lstrip("-").isdigit() + else whole + "." + fraction + ) value = float(normalized) scaled = value * multiplier return int(scaled) if scaled.is_integer() else scaled diff --git a/db/data/sfpd/legal_pension_caseload_2025/fr_stat_2502.pdf b/db/data/sfpd/legal_pension_caseload_2025/fr_stat_2502.pdf new file mode 100644 index 00000000..3c76ba75 Binary files /dev/null and b/db/data/sfpd/legal_pension_caseload_2025/fr_stat_2502.pdf differ diff --git a/db/data/sfpd/legal_pension_caseload_2025/manifest.yaml b/db/data/sfpd/legal_pension_caseload_2025/manifest.yaml index 6776ad40..9c0de275 100644 --- a/db/data/sfpd/legal_pension_caseload_2025/manifest.yaml +++ b/db/data/sfpd/legal_pension_caseload_2025/manifest.yaml @@ -1,30 +1,31 @@ -source_id: belgium +source_id: sfpd package_id: sfpd-legal-pension-caseload-2025 -source_name: SFPD/SFP legal pension caseload 2025 -publisher: "Service f\xE9d\xE9ral des Pensions (SFP/SFPD), via PensionStat.be" -source_page: https://www.pensionstat.be/fr +dataset: sfpd_monthly_social_benefits_2025_02 +source_name: SFPD monthly social-benefit statistics +publisher: Service fédéral des Pensions (SFPD) +source_page: https://www.sfpd.fgov.be/fr/centre-de-connaissances/statistiques +table: >- + Statistique mensuelle des prestations sociales - Février 2025, + tables 2.2 and 2.4.1 files: 2025: - filename: sfpd_legal_pension_caseload_2025.csv - source_url: https://www.pensionstat.be/fr - source_table: Legal pension beneficiaries by scheme, January 2025 - sha256: 5e2a249035f0871d81c21a0f44f88b9680990394a6f252fd5684d22b235a5a81 - size_bytes: 370 + filename: fr_stat_2502.pdf + source_url: https://www.sfpd.fgov.be/files/3432/fr_stat_2502.pdf + source_table: >- + Statistique mensuelle des prestations sociales - Février 2025, + tables 2.2 and 2.4.1 + sha256: 0d6173e71a0e9c2cd220cd024a5b5fecbb6ca79f8b791cbd2b3368e7f8412106 + size_bytes: 2319705 + fetched_at: '2026-08-30T12:08:46+00:00' storage: r2: provider: r2 bucket: ledger-raw - key: raw/belgium/sfpd-legal-pension-caseload-2025/2025/5e2a249035f0871d81c21a0f44f88b9680990394a6f252fd5684d22b235a5a81/sfpd_legal_pension_caseload_2025.csv - uri: r2://ledger-raw/raw/belgium/sfpd-legal-pension-caseload-2025/2025/5e2a249035f0871d81c21a0f44f88b9680990394a6f252fd5684d22b235a5a81/sfpd_legal_pension_caseload_2025.csv - source_urls: - - https://www.pensionstat.be/fr - - https://www.pensionstat.be/fr/chiffres-cles/pension-legale/pensionnes - notes: Legal pension beneficiary counts published by the Federal Pensions Service - (SFP/SFPD) via PensionStat.be (a Sigedis / SFP / INASTI initiative), January - 2025 reference. The three scheme counts (employee, self-employed, civil servant) - sum to 3,653,050, which exceeds the 2,674,520 all-schemes total because a person - with a mixed career is counted in each scheme they draw from; the scheme rows - are therefore per-scheme recipient counts, not a partition of the total. Total - legal pension expenditure is published only as a rounded "69 billion EUR" without - an exact figure or reference year on the source page, so it is recorded as a - gap rather than a fact. + key: raw/belgium/sfpd/sfpd-grapa-monthly-statistics-2025-02/source_capture/0d6173e71a0e9c2cd220cd024a5b5fecbb6ca79f8b791cbd2b3368e7f8412106/fr_stat_2502.pdf + uri: r2://ledger-raw/raw/belgium/sfpd/sfpd-grapa-monthly-statistics-2025-02/source_capture/0d6173e71a0e9c2cd220cd024a5b5fecbb6ca79f8b791cbd2b3368e7f8412106/fr_stat_2502.pdf + notes: >- + Verbatim official SFPD February 2025 monthly-statistics PDF. Chronicle + parses the publisher bytes directly. Printed page 22 table 2.2 reports + employee and/or self-employed legal-pension beneficiaries and monthly + expenditure; printed page 36 table 2.4.1 reports February GRAPA + beneficiaries and monthly expenditure. diff --git a/db/data/sfpd/legal_pension_caseload_2025/sfpd_legal_pension_caseload_2025.csv b/db/data/sfpd/legal_pension_caseload_2025/sfpd_legal_pension_caseload_2025.csv deleted file mode 100644 index bbae8c33..00000000 --- a/db/data/sfpd/legal_pension_caseload_2025/sfpd_legal_pension_caseload_2025.csv +++ /dev/null @@ -1,5 +0,0 @@ -value_id,period,scheme,reference_basis,recipients,source_url -all_schemes,2025,all,january_2025,2674520,https://www.pensionstat.be/fr -salarie,2025,employee,january_2025,2357954,https://www.pensionstat.be/fr -independant,2025,self_employed,january_2025,690590,https://www.pensionstat.be/fr -fonctionnaire,2025,civil_servant,january_2025,604506,https://www.pensionstat.be/fr diff --git a/docs/issue-drafts/belgium-aviq-family-allowances.md b/docs/issue-drafts/belgium-aviq-family-allowances.md new file mode 100644 index 00000000..a8c89a97 --- /dev/null +++ b/docs/issue-drafts/belgium-aviq-family-allowances.md @@ -0,0 +1,27 @@ +Refs #69. + +## Source gap + +Chronicle has no direct AVIQ or FAMIWAL post-2019 Walloon child-benefit package +that includes both caseload and expenditure. A parliamentary answer on an +unmerged branch is not the requested agency artifact and omits expenditure. + +## Deterministic handoff + +Retrieve the official AVIQ 2021 annual-report PDF through the AVIQ entry in +`FETCH-MANIFEST-BELGIUM-PUBLIC-FACTS.json`: + +`https://www.aviq.be/sites/default/files/documents_pro/2022-10/rapport_annuel_AVIQ_2021.pdf` + +## Acceptance criteria + +- Pin the unchanged official PDF with exact URL, SHA-256, size, publication + vintage, table/page references, and content-addressed raw-storage pointer. +- Parse the publisher PDF deterministically and add only source-cell-backed + family-allowance caseload and expenditure facts with exact reference periods, + units, and Walloon administrative/geographic scope. +- Keep rounded or ambiguous values as documented gaps; do not substitute the + parliamentary answer, reconcile AVIQ to FAMIWAL, derive missing periods, or + compute per-recipient amounts. +- Run `validate-package`, source-cell preservation, fact-load, + consumer-artifact, raw-facts-boundary, ruff, and `git diff --check` checks. diff --git a/docs/issue-drafts/belgium-ecb-hfcs-wave-2023.md b/docs/issue-drafts/belgium-ecb-hfcs-wave-2023.md new file mode 100644 index 00000000..48434b55 --- /dev/null +++ b/docs/issue-drafts/belgium-ecb-hfcs-wave-2023.md @@ -0,0 +1,32 @@ +Refs #69. + +## Source gap + +Chronicle has no official Belgium HFCS net-wealth/distribution/component +aggregate package. No ECB wave-2023 workbook bytes or checksum are present on +current main or in the inspected branches, so values must not be copied from +the rendered statistical tables. + +## Deterministic handoff + +Retrieve the official ECB HFCS Wave 2023 statistical-tables ZIP, version 5.0 +(June 2026), through the ECB entry in +`FETCH-MANIFEST-BELGIUM-PUBLIC-FACTS.json`: + +`https://www.ecb.europa.eu/home/pdf/research/hfcn/HFCS_Statistical_Tables_Wave_2023_June_2026.zip` + +## Acceptance criteria + +- Pin the unchanged outer ZIP with URL, SHA-256, size, ECB version/vintage, and + content-addressed raw-storage pointer. List archive members deterministically + and pin each selected native workbook member separately. +- Inspect workbook sheet names, row/column labels, units, weighting labels, and + cells before authoring. Belgium asset/liability values refer to interview + time during January-December 2023; income variables refer to 2022. +- Add only ECB-published Belgium net-wealth means/medians/quantiles, + distribution shares/ratios, and component aggregates with exact source-cell + lineage and `survey_aggregate` provenance. +- Do not interpolate quantiles, derive components or totals, reconcile to NBB, + age values, or add model/target bindings. +- Run `validate-package`, source-cell preservation, fact-load, + consumer-artifact, raw-facts-boundary, ruff, and `git diff --check` checks. diff --git a/docs/issue-drafts/belgium-iriscare-family-allowances.md b/docs/issue-drafts/belgium-iriscare-family-allowances.md new file mode 100644 index 00000000..8bc938d2 --- /dev/null +++ b/docs/issue-drafts/belgium-iriscare-family-allowances.md @@ -0,0 +1,33 @@ +Refs #69. + +## Source gap + +Chronicle has no direct Iriscare or Famiris post-2019 Brussels child-benefit +package containing both caseload and expenditure. The official interactive +statistics are not safe to transcribe, and preliminary inspection of the +official annual report found more than one expenditure occurrence that must be +distinguished by publisher scope rather than silently reconciled. + +## Deterministic handoff + +Retrieve the official Iriscare 2024 annual-report PDF through the Iriscare entry +in `FETCH-MANIFEST-BELGIUM-PUBLIC-FACTS.json`: + +`https://rapport.iriscare.brussels/wp-content/uploads/2025/09/Rapport-annuel-2024.pdf` + +The official statistics pages may be used only if a separate reviewed handoff +pins their native Fabric export bytes; screenshots and copied dashboard text +are not artifacts. + +## Acceptance criteria + +- Pin the unchanged official PDF with exact URL, SHA-256, size, publication + vintage, table/page references, and content-addressed raw-storage pointer. +- Parse the full PDF deterministically. If multiple publisher cells differ, + emit separately labelled facts only when their accounting populations/scopes + are explicit; otherwise document the ambiguity and omit the values. +- Preserve exact Brussels administrative/geographic scope, reference period, + recipient unit, currency, scale, and assertion. Do not combine funds, infer a + regional total, reconcile values, or compute per-recipient amounts. +- Run `validate-package`, source-cell preservation, fact-load, + consumer-artifact, raw-facts-boundary, ruff, and `git diff --check` checks. diff --git a/docs/issue-drafts/belgium-opgroeien-native-exports.md b/docs/issue-drafts/belgium-opgroeien-native-exports.md new file mode 100644 index 00000000..76b799b3 --- /dev/null +++ b/docs/issue-drafts/belgium-opgroeien-native-exports.md @@ -0,0 +1,40 @@ +Refs #69. + +## Source gap + +The current `opgroeien-groeipakket-caseload-2025` package is a curated extract +of heterogeneous headline figures. It does not pin a native publisher export, +does not include expenditure, and labels the whole administrative scope as +NUTS BE2 without publisher-byte evidence that every row has that geography. + +Opgroeien's official `Cijfers op maat` page embeds separate Power BI reports +for Groeipakket caseload and budget/expenditure. The current execution +environment cannot retrieve their native data exports. Rendered dashboard +cards, screenshots, accessibility text, and manual transcription are not +acceptable Chronicle artifacts. + +## Deterministic handoff + +Use the two Opgroeien entries in +`FETCH-MANIFEST-BELGIUM-PUBLIC-FACTS.json`: + +- official landing page: + `https://www.opgroeien.be/kennis/cijfers-en-onderzoek/groeipakket/cijfers-op-maat` +- caseload report ID and exact native-export procedure recorded in the handoff; +- expenditure report ID and exact native-export procedure recorded in the + handoff. + +## Acceptance criteria + +- Pin unchanged native publisher export bytes with URL, SHA-256, size, report + vintage/update timestamp, visual name, active filters, headers, and + content-addressed raw-storage pointer. +- Preserve each publisher reference period, recipient unit, component, + currency/scale, and administrative/geographic scope exactly. +- Add post-2019 caseload and expenditure facts only for cells that resolve from + the pinned export. Do not force administrative coverage to BE2 without + evidence in the export. +- Do not calculate annual totals, per-recipient amounts, reconciliations, + imputed periods, take-up, or model bindings. +- Run `validate-package`, source-cell preservation, fact-load, + consumer-artifact, raw-facts-boundary, ruff, and `git diff --check` checks. diff --git a/docs/issue-drafts/belgium-ostbelgien-family-allowances.md b/docs/issue-drafts/belgium-ostbelgien-family-allowances.md new file mode 100644 index 00000000..9883a787 --- /dev/null +++ b/docs/issue-drafts/belgium-ostbelgien-family-allowances.md @@ -0,0 +1,34 @@ +Refs #69. + +## Source gap + +Chronicle has no direct post-2019 German-speaking Community child-benefit +caseload and expenditure package. The official Ostbelgien Statistik page +returns HTTP 403 to automated retrieval in the current environment. Search +snippets and the manual HTML transcription on an unmerged branch are not +publisher artifacts and must not become facts. + +## Deterministic handoff + +Use the Ostbelgien entry in +`FETCH-MANIFEST-BELGIUM-PUBLIC-FACTS.json` to capture the raw official page +response/source from: + +`https://ostbelgienstatistik.be/desktopdefault.aspx/tabid-3748/6766_read-39090/` + +If the values are injected from a separate official endpoint, replace the +handoff with that endpoint and pin its native response rather than saving an +incomplete HTML shell. + +## Acceptance criteria + +- Pin unchanged publisher response bytes with URL, retrieval details, + SHA-256, size, page update timestamp, and content-addressed raw-storage + pointer. +- Verify the bytes themselves contain exact table labels, periods, units, + caseload, and expenditure before authoring any facts. +- Preserve the German-speaking Community scope exactly. Do not use search + snippets, OCR, screenshots, manual transcription, imputation, or derived + amounts. +- Run `validate-package`, source-cell preservation, fact-load, + consumer-artifact, raw-facts-boundary, ruff, and `git diff --check` checks. diff --git a/docs/pr-drafts/belgium-public-facts.md b/docs/pr-drafts/belgium-public-facts.md new file mode 100644 index 00000000..8bd8e3ed --- /dev/null +++ b/docs/pr-drafts/belgium-public-facts.md @@ -0,0 +1,74 @@ +## Summary + +- replace the intermediary SFPD pension CSV with the official February 2025 + monthly-statistics PDF and direct source-cell selectors; +- add four exact February 2025 facts: employee/self-employed pension + beneficiaries and monthly expenditure, plus GRAPA beneficiaries and monthly + expenditure; +- add a source-scoped European document-number format for SFPD while preserving + the original default grammar and existing single-dot behavior; +- audit the six original Belgium selectors from #69: all remain registered, + unique, valid, and resolve 587 facts on current main; +- add a validated six-artifact offline-fetch handoff and five ready-to-file + follow-up issue bodies for Opgroeien, AVIQ, Iriscare, Ostbelgien, and ECB + HFCS, without adding placeholder facts. + +## Source pin and fact scope + +Official SFPD artifact: + +- URL: `https://www.sfpd.fgov.be/files/3432/fr_stat_2502.pdf` +- SHA-256: + `0d6173e71a0e9c2cd220cd024a5b5fecbb6ca79f8b791cbd2b3368e7f8412106` +- size: 2,319,705 bytes +- printed page 22, table 2.2: 2,435,457 employee and/or self-employed + pension beneficiaries and EUR 3,759,582,728.06 monthly expenditure; +- printed page 36, table 2.4.1: 117,650 GRAPA beneficiaries and + EUR 86,398,449.47 monthly expenditure. + +All four facts are February 2025 `observation` assertions with +`administrative` provenance, Belgian country scope, exact publisher periods, +and full source-cell lineage. The pension beneficiary snapshot is explicitly +dated 1 February 2025. The pension total is narrowed to the employee and/or +self-employed regimes actually covered by table 2.2. + +No aging, period alignment, reconciliation, imputation, take-up mechanics, +target profile, model binding, solver construction, PolicyEngine-computed +value, or Axiom concept is included. + +## Validation + +- `validate-package sfpd-legal-pension-caseload-2025 --year 2025`: PASS, + 4 record sets/rows/measures and zero errors/warnings; +- source suite: PASS, 13,644 full-document source cells, 4 facts, 100% lineage, + zero agent-acceptance errors; +- package bundle and consumer artifact build/load: PASS, 4 rows; +- parser/SFPD plus the five existing packages exposed by the first CI run: + 27 passed after restoring the original default grammar; +- six original issue-69 package validators: PASS; +- focused parser/offline/SFPD/selector tests: PASS; +- facts-only and consumer tests: 33 passed; +- ruff: PASS; +- `git diff --check origin/main`: PASS; +- `ledger-source-fidelity`: PASS; +- `ledger-boundary`: PASS. + +The first CI run caught row shifts caused by an overly broad shared number +grammar. The final implementation scopes European parsing to SFPD and restores +the prior default for all existing packages. The relevant single-package +bundle and consumer artifact pass; the all-source 151-package bundle remains a +required final-head CI gate. + +## Follow-ups and governance + +The regional child-benefit and HFCS sources require native exports or official +publisher responses. Their deterministic retrieval instructions are in +`FETCH-MANIFEST-BELGIUM-PUBLIC-FACTS.json`; no values were transcribed. + +The literal `ledger-source-ingestor.allowed_paths` list omits several paths +this task explicitly requires (`PROGRESS.md`, `db/data/**`, the root handoff, +issue drafts, and the pre-existing Belgium test). This PR makes no contract, +schema, core, model, or consumer-owned behavior change; maintainer acceptance +or a registry clarification is still needed for the path inconsistency. + +Refs #69 diff --git a/packages/sfpd/legal_pension_caseload_2025/source_package.yaml b/packages/sfpd/legal_pension_caseload_2025/source_package.yaml index c48b3b59..698d9ba6 100644 --- a/packages/sfpd/legal_pension_caseload_2025/source_package.yaml +++ b/packages/sfpd/legal_pension_caseload_2025/source_package.yaml @@ -1,111 +1,299 @@ +# SFPD's February 2025 monthly-statistics PDF is the value artifact. The +# package selects publisher numbers from tables 2.2 and 2.4.1 through the +# pdf_text_numbers parser; no manually transcribed or computed value artifact +# sits between the PDF and these Chronicle facts. The package retains its +# historical alias while narrowing the pension total to the employee and +# self-employed regimes explicitly covered by table 2.2. schema_version: ledger.source_package.v1 package_id: sfpd-legal-pension-caseload-2025 -label: SFPD legal pension caseload 2025 by scheme +label: SFPD February 2025 pension and GRAPA beneficiaries and monthly expenditure artifact: - source_name: sfpd_pensions - source_table: Legal pension beneficiaries by scheme, January 2025 + source_name: sfpd_monthly_social_benefits + source_table: >- + Statistique mensuelle des prestations sociales - Février 2025, + tables 2.2 and 2.4.1 resource_package: db resource_directory: data/sfpd/legal_pension_caseload_2025 manifest: manifest.yaml - vintage: pensionstat_legal_pension_2025 - extracted_at: '2026-07-04' - extraction_method: curated CSV extract from PensionStat.be published legal pension beneficiary headline figures - parser: delimited_text_full_rows + vintage: monthly_social_benefits_2025_02 + extracted_at: '2026-08-30' + extraction_method: pypdf text-line number extraction from the full publisher PDF + parser: pdf_text_numbers + number_format: european artifact_year: 2025 - sheet_name: sfpd_legal_pension_caseload_2025 + record_sets: -- record_set_id: sfpd.legal_pension.cy2025.recipients.by_scheme +- record_set_id: sfpd.month2025_02.pension.employee_self_employed.beneficiaries provenance_class: administrative - record_set_spec_id: sfpd.legal_pension.recipients.by_scheme.v1 - source_record_id_prefix: sfpd.legal_pension.cy2025.recipients - sheet_name: sfpd_legal_pension_caseload_2025 - period_type: calendar_year - period: 2025 + assertion: observation + record_set_spec_id: sfpd.pension.employee_self_employed.beneficiaries.v1 + source_record_id_prefix: sfpd.month2025_02.pension.employee_self_employed.beneficiaries + sheet_name: document_numbers + period_type: month + period: '2025-02' + period_coverage: + basis: calendar + start_date: '2025-02-01' + end_date: '2025-02-01' + source_period_label: au 01.02.2025 + notes: >- + Table 2.2 is a stock at 1 February 2025, not an annual average or an + annual-any recipient count. geography_id: BE geography_level: country geography_name: Belgium geography_vintage: current entity: person - entity_role: benefit_recipient - domain: pensions - groupby_dimension: sfpd.scheme + entity_role: legal_pension_beneficiary + domain: legal_pensions + groupby_dimension: sfpd.published_total rows: - - value_id: all_schemes - label: All legal pension schemes recipients + - value_id: employee_or_self_employed_total + label: Employee and/or self-employed legal pension beneficiaries ordinal: 0 - row_number: 2 - expected_row_header_column: A - expected_row_header: all_schemes + row_number: 881 + expected_row_header_column: C + expected_row_header: 2.435.457 1.810.192 93.346Nombre table_record_kind: total - filters: - sfpd.scheme: all - constraints: - - variable: sfpd.scheme - operator: == - value: all - label: Pension scheme - - value_id: salarie - label: Employee scheme recipients - ordinal: 1 - row_number: 3 - expected_row_header_column: A - expected_row_header: salarie - table_record_kind: total - filters: - sfpd.scheme: employee - constraints: - - variable: sfpd.scheme - operator: == - value: employee - label: Pension scheme - - value_id: independant - label: Self-employed scheme recipients - ordinal: 2 - row_number: 4 - expected_row_header_column: A - expected_row_header: independant + guard_cells: + - column: A + expected_value: 22 + label: PDF page number + - column: B + expected_value: 56 + label: PDF extracted line number + - column: D + expected_value: 2.435.457 + label: Publisher number text + - column: C + row: 844 + expected_value: 2.2 Répartition du nombre de bénéficiaires et dépense mensuelle (en €) + label: Table title line + measures: + - measure_id: beneficiaries + label: Employee and/or self-employed legal pension beneficiaries + ordinal: 0 + column: E + source_column_id: numeric_value + expected_column_header_row: 1 + expected_column_header: numeric_value + concept: sfpd.employee_or_self_employed_legal_pension_beneficiary_count + source_concept: sfpd.monthly_statistics.table_2_2.total_beneficiaries + concept_relation: source_label + concept_authority: sfpd + concept_evidence_url: https://www.sfpd.fgov.be/files/3432/fr_stat_2502.pdf + concept_evidence_notes: >- + SFPD table 2.2, printed page 22, Total - Men and women, Without + distinction row, Nombre column: 2.435.457. The report states that the + table covers people with an employee and/or self-employed retirement or + survivor pension. It does not include a civil-servant regime column. + unit: count + aggregation: sum + expected_cell_type: number + +- record_set_id: sfpd.month2025_02.pension.employee_self_employed.monthly_expenditure + provenance_class: administrative + assertion: observation + record_set_spec_id: sfpd.pension.employee_self_employed.monthly_expenditure.v1 + source_record_id_prefix: sfpd.month2025_02.pension.employee_self_employed.monthly_expenditure + sheet_name: document_numbers + period_type: month + period: '2025-02' + period_coverage: + basis: calendar + start_date: '2025-02-01' + end_date: '2025-02-28' + source_period_label: Février 2025 + notes: Publisher-labelled monthly expenditure for February 2025. + geography_id: BE + geography_level: country + geography_name: Belgium + geography_vintage: current + entity: government + entity_role: legal_pension_payment_program + domain: legal_pensions + groupby_dimension: sfpd.published_total + rows: + - value_id: employee_or_self_employed_total + label: Employee and/or self-employed legal pension monthly expenditure + ordinal: 0 + row_number: 878 + expected_row_header_column: C + expected_row_header: 3.759.582.728,06 2.748.241.379,33 891.285.866,21 table_record_kind: total - filters: - sfpd.scheme: self_employed - constraints: - - variable: sfpd.scheme - operator: == - value: self_employed - label: Pension scheme - - value_id: fonctionnaire - label: Civil servant scheme recipients - ordinal: 3 - row_number: 5 - expected_row_header_column: A - expected_row_header: fonctionnaire + guard_cells: + - column: A + expected_value: 22 + label: PDF page number + - column: B + expected_value: 55 + label: PDF extracted line number + - column: D + expected_value: 3.759.582.728,06 + label: Publisher number text + - column: C + row: 844 + expected_value: 2.2 Répartition du nombre de bénéficiaires et dépense mensuelle (en €) + label: Table title line + measures: + - measure_id: monthly_expenditure + label: Employee and/or self-employed legal pension monthly expenditure + ordinal: 0 + column: E + source_column_id: numeric_value + expected_column_header_row: 1 + expected_column_header: numeric_value + concept: sfpd.employee_or_self_employed_legal_pension_monthly_expenditure + source_concept: sfpd.monthly_statistics.table_2_2.total_monthly_amount + concept_relation: source_label + concept_authority: sfpd + concept_evidence_url: https://www.sfpd.fgov.be/files/3432/fr_stat_2502.pdf + concept_evidence_notes: >- + SFPD table 2.2, printed page 22, Total - Men and women, Without + distinction row, Montant total column: EUR 3.759.582.728,06 for the + employee and/or self-employed pension population. The table itemizes the + total as employee pension plus bonus, self-employed pension plus bonus, + other pension bonuses, and the well-being bonus; Chronicle preserves the + publisher's total and does not reclassify or subtract those components. + unit: eur + aggregation: sum + expected_cell_type: number + +- record_set_id: sfpd.month2025_02.grapa.beneficiaries + provenance_class: administrative + assertion: observation + record_set_spec_id: sfpd.grapa.beneficiaries.v1 + source_record_id_prefix: sfpd.month2025_02.grapa.beneficiaries + sheet_name: document_numbers + period_type: month + period: '2025-02' + period_coverage: + basis: calendar + start_date: '2025-02-01' + end_date: '2025-02-28' + source_period_label: Février 2025 + notes: >- + Publisher-labelled February monthly GRAPA beneficiary population, not an + annual average, annual-any recipient count, or eligibility estimate. + geography_id: BE + geography_level: country + geography_name: Belgium + geography_vintage: current + entity: person + entity_role: grapa_beneficiary + domain: old_age_income_guarantee + groupby_dimension: sfpd.published_total + rows: + - value_id: total + label: GRAPA beneficiaries + ordinal: 0 + row_number: 1756 + expected_row_header_column: C + expected_row_header: 75.523 57.588.778,24Février 2025 117.650 86.398.449,47 42.127 28.809.671,23 table_record_kind: total - filters: - sfpd.scheme: civil_servant - constraints: - - variable: sfpd.scheme - operator: == - value: civil_servant - label: Pension scheme + guard_cells: + - column: A + expected_value: 36 + label: PDF page number + - column: B + expected_value: 19 + label: PDF extracted line number + - column: D + expected_value: '117.650' + label: Publisher number text + - column: F + row: 1752 + expected_value: >- + 36 Service fédéral des Pensions Statistique mensuelle février 2025 - + 2.4.1 2.4.1 Répartition du nombre de bénéficiaires d’une garantie de + revenus aux personnes âgées en combinaison avec la dépense mensuelle + label: Table title context measures: - - measure_id: recipients - label: Legal pension beneficiaries + - measure_id: beneficiaries + label: GRAPA beneficiaries ordinal: 0 column: E - source_column_id: recipients + source_column_id: numeric_value expected_column_header_row: 1 - expected_column_header: recipients - concept: legal_pension_recipients - source_concept: sfpd.legal_pension.beneficiaries - concept_relation: exact - concept_authority: ledger-be - concept_evidence_url: https://www.pensionstat.be/fr + expected_column_header: numeric_value + concept: sfpd.grapa_beneficiary_count + source_concept: sfpd.monthly_statistics.table_2_4_1.grapa_beneficiaries + concept_relation: source_label + concept_authority: sfpd + concept_evidence_url: https://www.sfpd.fgov.be/files/3432/fr_stat_2502.pdf concept_evidence_notes: >- - PensionStat.be (Federal Pensions Service / Sigedis / INASTI) publishes the - count of people receiving income from the legal pension schemes, January - 2025. The all-schemes row is the published total (2,674,520). The three - scheme rows are per-scheme beneficiary counts; their sum (3,653,050) - exceeds the total because mixed-career pensioners are counted in each - scheme they draw from, so the scheme rows are not a partition of the total. + SFPD table 2.4.1, printed page 36, February 2025 TOTAL Nombre column: + 117.650 GRAPA beneficiaries. The PDF parser retains a single-dot number + as 117.650 for backward compatibility; value_scale 1000 restores the + publisher's dot-grouped integer count. unit: count aggregation: sum expected_cell_type: number + value_scale: 1000 + +- record_set_id: sfpd.month2025_02.grapa.monthly_expenditure + provenance_class: administrative + assertion: observation + record_set_spec_id: sfpd.grapa.monthly_expenditure.v1 + source_record_id_prefix: sfpd.month2025_02.grapa.monthly_expenditure + sheet_name: document_numbers + period_type: month + period: '2025-02' + period_coverage: + basis: calendar + start_date: '2025-02-01' + end_date: '2025-02-28' + source_period_label: Février 2025 + notes: Publisher-labelled monthly GRAPA expenditure for February 2025. + geography_id: BE + geography_level: country + geography_name: Belgium + geography_vintage: current + entity: government + entity_role: grapa_payment_program + domain: old_age_income_guarantee + groupby_dimension: sfpd.published_total + rows: + - value_id: total + label: GRAPA monthly expenditure + ordinal: 0 + row_number: 1757 + expected_row_header_column: C + expected_row_header: 75.523 57.588.778,24Février 2025 117.650 86.398.449,47 42.127 28.809.671,23 + table_record_kind: total + guard_cells: + - column: A + expected_value: 36 + label: PDF page number + - column: B + expected_value: 19 + label: PDF extracted line number + - column: D + expected_value: 86.398.449,47 + label: Publisher number text + - column: F + row: 1752 + expected_value: >- + 36 Service fédéral des Pensions Statistique mensuelle février 2025 - + 2.4.1 2.4.1 Répartition du nombre de bénéficiaires d’une garantie de + revenus aux personnes âgées en combinaison avec la dépense mensuelle + label: Table title context + measures: + - measure_id: monthly_expenditure + label: GRAPA monthly expenditure + ordinal: 0 + column: E + source_column_id: numeric_value + expected_column_header_row: 1 + expected_column_header: numeric_value + concept: sfpd.grapa_monthly_expenditure + source_concept: sfpd.monthly_statistics.table_2_4_1.grapa_monthly_amount + concept_relation: source_label + concept_authority: sfpd + concept_evidence_url: https://www.sfpd.fgov.be/files/3432/fr_stat_2502.pdf + concept_evidence_notes: >- + SFPD table 2.4.1, printed page 36, February 2025 TOTAL Montant mensuel + GRAPA column: EUR 86.398.449,47. + unit: eur + aggregation: sum + expected_cell_type: number diff --git a/tests/test_belgium_targets.py b/tests/test_belgium_targets.py index afec1b2f..c2379b89 100644 --- a/tests/test_belgium_targets.py +++ b/tests/test_belgium_targets.py @@ -555,24 +555,24 @@ def test_belgium_supplementary_publisher_aliases_are_registered(): } <= set(SOURCE_PACKAGE_ALIASES) -def test_sfpd_legal_pension_caseload_matches_published_cells(): +def test_sfpd_pension_and_grapa_totals_match_published_cells(): facts = _facts(SFPD_PENSION_ALIAS, 2025) - by_scheme = {fact.filters["sfpd.scheme"]: fact.value for fact in facts} - - # Exact published counts from PensionStat.be (SFP/SFPD), January 2025. - assert by_scheme == { - "all": 2674520, - "employee": 2357954, - "self_employed": 690590, - "civil_servant": 604506, + by_concept = {fact.measure.concept: fact.value for fact in facts} + + # Exact totals from the official SFPD February 2025 PDF, tables 2.2 and 2.4.1. + assert by_concept == { + "sfpd.employee_or_self_employed_legal_pension_beneficiary_count": 2435457, + "sfpd.employee_or_self_employed_legal_pension_monthly_expenditure": ( + 3759582728.06 + ), + "sfpd.grapa_beneficiary_count": 117650, + "sfpd.grapa_monthly_expenditure": 86398449.47, + } + assert {fact.source.source_name for fact in facts} == { + "sfpd_monthly_social_benefits" } - # Scheme counts are per-scheme recipients (mixed careers), not a partition: - # their sum exceeds the all-schemes total. - scheme_sum = sum(v for k, v in by_scheme.items() if k != "all") - assert scheme_sum > by_scheme["all"] - assert {fact.source.source_name for fact in facts} == {"sfpd_pensions"} assert {fact.geography.level for fact in facts} == {"country"} - assert {fact.measure.unit for fact in facts} == {"count"} + assert {fact.measure.unit for fact in facts} == {"count", "eur"} assert validate_facts(facts).valid diff --git a/tests/test_chronicle_belgium_sfpd.py b/tests/test_chronicle_belgium_sfpd.py new file mode 100644 index 00000000..2ea4698c --- /dev/null +++ b/tests/test_chronicle_belgium_sfpd.py @@ -0,0 +1,81 @@ +"""Direct publisher-artifact tests for Belgium SFPD monthly facts.""" + +from __future__ import annotations + +import hashlib +from functools import lru_cache +from pathlib import Path + +from chronicle.core import validate_facts +from chronicle.source_package import load_source_package, validate_source_package + + +REPO_ROOT = Path(__file__).resolve().parents[1] +SFPD_PACKAGE = "sfpd-legal-pension-caseload-2025" + + +@lru_cache +def _sfpd_outputs(): + package = load_source_package(SFPD_PACKAGE) + cells = tuple(package.build_source_cells(2025)) + facts = tuple(package.build_facts(2025, cells=list(cells))) + return cells, facts + + +def test_sfpd_pension_and_grapa_facts_match_publisher_pdf_cells(): + cells, facts = _sfpd_outputs() + + by_concept = {fact.measure.concept: fact.value for fact in facts} + assert by_concept == { + "sfpd.employee_or_self_employed_legal_pension_beneficiary_count": 2435457, + "sfpd.employee_or_self_employed_legal_pension_monthly_expenditure": ( + 3759582728.06 + ), + "sfpd.grapa_beneficiary_count": 117650, + "sfpd.grapa_monthly_expenditure": 86398449.47, + } + + cells_by_address = {cell.address: cell.raw_value for cell in cells} + assert cells_by_address["D881"] == "2.435.457" + assert cells_by_address["E881"] == 2435457 + assert cells_by_address["D878"] == "3.759.582.728,06" + assert cells_by_address["E878"] == 3759582728.06 + assert cells_by_address["D1756"] == "117.650" + assert cells_by_address["E1756"] == 117.65 + assert cells_by_address["D1757"] == "86.398.449,47" + assert cells_by_address["E1757"] == 86398449.47 + + assert {fact.period.type for fact in facts} == {"month"} + assert {fact.period.value for fact in facts} == {"2025-02"} + assert {fact.source.source_name for fact in facts} == { + "sfpd_monthly_social_benefits" + } + assert {fact.source.source_file for fact in facts} == {"fr_stat_2502.pdf"} + assert {fact.source.url for fact in facts} == { + "https://www.sfpd.fgov.be/files/3432/fr_stat_2502.pdf" + } + assert {fact.geography.id for fact in facts} == {"BE"} + assert {fact.geography.vintage for fact in facts} == {"current"} + assert {fact.assertion for fact in facts} == {"observation"} + assert all(fact.source_cell_keys for fact in facts) + assert validate_facts(facts).valid + + +def test_sfpd_monthly_statistics_artifact_is_hash_pinned(): + data_dir = REPO_ROOT / "db" / "data" / "sfpd" / "legal_pension_caseload_2025" + publisher_pdf = data_dir / "fr_stat_2502.pdf" + package_report = validate_source_package(SFPD_PACKAGE, year=2025) + + assert package_report.valid, package_report.to_dict() + assert package_report.counts == { + "record_set_count": 4, + "row_count": 4, + "measure_count": 4, + "source_record_count": 4, + "source_region_count": 4, + } + assert publisher_pdf.stat().st_size == 2319705 + assert hashlib.sha256(publisher_pdf.read_bytes()).hexdigest() == ( + "0d6173e71a0e9c2cd220cd024a5b5fecbb6ca79f8b791cbd2b3368e7f8412106" + ) + assert not (data_dir / "sfpd_legal_pension_caseload_2025.csv").exists() diff --git a/tests/test_chronicle_bundle.py b/tests/test_chronicle_bundle.py index 6a00ef84..03280d67 100644 --- a/tests/test_chronicle_bundle.py +++ b/tests/test_chronicle_bundle.py @@ -149,7 +149,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "onss_contributions": 1, "opgroeien_groeipakket": 11, "scotgov": 2787, - "sfpd_pensions": 4, + "sfpd_monthly_social_benefits": 4, "slc": 199, "spf_finances_pit": 1, "ssa": 426, @@ -533,7 +533,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "calendar_year:2022": 2075, "calendar_year:2023": 6343, "calendar_year:2024": 33936, - "calendar_year:2025": 4571, + "calendar_year:2025": 4567, "calendar_year:2026": 341, "calendar_year:2027": 320, "calendar_year:2028": 320, @@ -622,7 +622,7 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "month:2024-11": 107, "month:2024-12": 377, "month:2025-01": 109, - "month:2025-02": 107, + "month:2025-02": 111, "month:2025-03": 229, "month:2025-04": 112, "month:2025-05": 6221, @@ -695,11 +695,11 @@ def test_build_bundle_writes_merged_consumer_contract(tmp_path): "dwelling": 12733, "family": 107, "firm": 1439, - "government": 1313, + "government": 1315, "household": 40724, "institutional_sector": 133, "pension_plan": 2, - "person": 60466, + "person": 60464, "return": 14600, "social_protection_scheme": 36, "tax_unit": 40069, diff --git a/tests/test_chronicle_offline_fetch.py b/tests/test_chronicle_offline_fetch.py index 6e399915..78e44043 100644 --- a/tests/test_chronicle_offline_fetch.py +++ b/tests/test_chronicle_offline_fetch.py @@ -80,6 +80,27 @@ def test_existing_v1_fetch_manifest_remains_valid(): ) +def test_belgium_public_facts_handoff_covers_every_blocked_primary_source(): + manifest = load_offline_fetch_manifest( + REPO_ROOT / "FETCH-MANIFEST-BELGIUM-PUBLIC-FACTS.json", + require_discovery_notes=True, + ) + + assert len(manifest.artifacts) == 6 + assert {artifact.package_id for artifact in manifest.artifacts} == { + "opgroeien-groeipakket-native-caseload", + "opgroeien-groeipakket-native-expenditure", + "aviq-annual-report-2021-family-allowances", + "iriscare-annual-report-2024-family-allowances", + "ostbelgien-family-allowances-2025", + "ecb-hfcs-wave-2023-statistical-tables", + } + assert all(artifact.r2 is not None for artifact in manifest.artifacts) + assert all(artifact.expected_sha256 is None for artifact in manifest.artifacts) + assert all(artifact.post_download_steps for artifact in manifest.artifacts) + assert manifest.final_validation + + def test_discovery_notes_are_optional_by_default_but_can_be_required(): payload = _manifest() del payload["artifacts"][0]["discovery_note"] diff --git a/tests/test_chronicle_source_cells.py b/tests/test_chronicle_source_cells.py index 9693d025..b4fefa1a 100644 --- a/tests/test_chronicle_source_cells.py +++ b/tests/test_chronicle_source_cells.py @@ -247,6 +247,48 @@ def test_html_tables_and_text_parser_preserves_tables_and_document_numbers(): assert cells_by_sheet_address[("document_numbers", "E3")].raw_value == 180_000 +def test_document_number_parser_preserves_european_grouped_decimals(): + artifact = SourceArtifactMetadata( + source_name="sfpd", + source_table="test French PDF text", + source_file="test.html", + url="https://example.test/test.html", + vintage="test", + sha256="abc123", + size_bytes=10, + extracted_at="2026-08-30", + extraction_method="test", + ) + html = b""" +
+ 2.435.457 pension beneficiaries; EUR 3.759.582.728,06; + 118.262 GRAPA beneficiaries; EUR 85.039.735,46; adjustment 49,58. +
+ """ + + cells = source_cells_from_html_tables_and_text( + html, + artifact, + number_format="european", + ) + rows = { + cell.row_number: {} + for cell in cells + if cell.sheet_name == "document_numbers" and cell.row_number > 1 + } + for cell in cells: + if cell.row_number in rows: + rows[cell.row_number][cell.column_number] = cell.raw_value + + assert [(row[4], row[5]) for row in rows.values()] == [ + ("2.435.457", 2_435_457), + ("3.759.582.728,06", 3_759_582_728.06), + ("118.262", 118.262), + ("85.039.735,46", 85_039_735.46), + ("49,58", 49.58), + ] + + def _two_sheet_workbook() -> bytes: workbook = openpyxl.Workbook() wanted = workbook.active