diff --git a/runs/seam_reconciliation_draft_v0.json b/runs/seam_reconciliation_draft_v0.json new file mode 100644 index 00000000..06dcaed0 --- /dev/null +++ b/runs/seam_reconciliation_draft_v0.json @@ -0,0 +1,215 @@ +{ + "artifact": "seam_reconciliation", + "version": "draft_v0", + "status": "DRAFT - NOT RATIFIED; C3 not locked; no thresholds", + "issue": "192", + "year_label_convention": "Keys under within_wave_monthly_separation and within_wave_means are SIPP FILE years (pu2022/pu2023), not reference years; file year = reference year + 1. PR #212's artifacts use reference years, so pu2022 here pairs with #212's 2021 and pu2023 with #212's 2022.", + "notes": "Naming collision: the SIPP 'j2j' transition measure from #212 (~0.35% person-level direct employer change per month) and the Census J2J data source's ~4.1% monthly-equivalent main-job separation rate benchmark used here are different universes and definitions (person-level direct employer-to-employer moves vs all main-job separations in UI-covered employment); the ~10x gap is not a contradiction.", + "first_reported": "PolicyEngine/populace-dynamics#192 comment 4982442068 (groundwork conditioned on employed-both-months; this artifact adds the E->N leg via the person-month universe)", + "within_wave_monthly_separation": { + "2022": [ + { + "month_pair": "1->2", + "jobs_held": 16932, + "jobs_kept": 16763, + "sep_rate": 0.01 + }, + { + "month_pair": "2->3", + "jobs_held": 17029, + "jobs_kept": 16843, + "sep_rate": 0.0109 + }, + { + "month_pair": "3->4", + "jobs_held": 17125, + "jobs_kept": 16884, + "sep_rate": 0.0141 + }, + { + "month_pair": "4->5", + "jobs_held": 17181, + "jobs_kept": 16923, + "sep_rate": 0.015 + }, + { + "month_pair": "5->6", + "jobs_held": 17317, + "jobs_kept": 16952, + "sep_rate": 0.0211 + }, + { + "month_pair": "6->7", + "jobs_held": 17395, + "jobs_kept": 17055, + "sep_rate": 0.0195 + }, + { + "month_pair": "7->8", + "jobs_held": 17385, + "jobs_kept": 17039, + "sep_rate": 0.0199 + }, + { + "month_pair": "8->9", + "jobs_held": 17475, + "jobs_kept": 17033, + "sep_rate": 0.0253 + }, + { + "month_pair": "9->10", + "jobs_held": 17466, + "jobs_kept": 17156, + "sep_rate": 0.0177 + }, + { + "month_pair": "10->11", + "jobs_held": 17517, + "jobs_kept": 17187, + "sep_rate": 0.0188 + }, + { + "month_pair": "11->12", + "jobs_held": 17541, + "jobs_kept": 17271, + "sep_rate": 0.0154 + } + ], + "2023": [ + { + "month_pair": "1->2", + "jobs_held": 17559, + "jobs_kept": 17323, + "sep_rate": 0.0134 + }, + { + "month_pair": "2->3", + "jobs_held": 17617, + "jobs_kept": 17370, + "sep_rate": 0.014 + }, + { + "month_pair": "3->4", + "jobs_held": 17699, + "jobs_kept": 17395, + "sep_rate": 0.0172 + }, + { + "month_pair": "4->5", + "jobs_held": 17677, + "jobs_kept": 17373, + "sep_rate": 0.0172 + }, + { + "month_pair": "5->6", + "jobs_held": 17720, + "jobs_kept": 17335, + "sep_rate": 0.0217 + }, + { + "month_pair": "6->7", + "jobs_held": 17737, + "jobs_kept": 17358, + "sep_rate": 0.0214 + }, + { + "month_pair": "7->8", + "jobs_held": 17636, + "jobs_kept": 17326, + "sep_rate": 0.0176 + }, + { + "month_pair": "8->9", + "jobs_held": 17758, + "jobs_kept": 17301, + "sep_rate": 0.0257 + }, + { + "month_pair": "9->10", + "jobs_held": 17658, + "jobs_kept": 17355, + "sep_rate": 0.0172 + }, + { + "month_pair": "10->11", + "jobs_held": 17663, + "jobs_kept": 17351, + "sep_rate": 0.0177 + }, + { + "month_pair": "11->12", + "jobs_held": 17660, + "jobs_kept": 17354, + "sep_rate": 0.0173 + } + ] + }, + "within_wave_means": { + "2022": 0.0171, + "2023": 0.0182 + }, + "across_wave_dec_to_jan": { + "persons_linked": 10051, + "jobs_held": 10828, + "jobs_kept": 9805, + "sep_rate": 0.0945 + }, + "j2j_national_benchmark": [ + { + "year": 2021, + "quarter": 1, + "q_sep_rate": 0.1001, + "monthly_equivalent": 0.0346 + }, + { + "year": 2021, + "quarter": 2, + "q_sep_rate": 0.1186, + "monthly_equivalent": 0.0412 + }, + { + "year": 2021, + "quarter": 3, + "q_sep_rate": 0.1315, + "monthly_equivalent": 0.0459 + }, + { + "year": 2021, + "quarter": 4, + "q_sep_rate": 0.1269, + "monthly_equivalent": 0.0442 + }, + { + "year": 2022, + "quarter": 1, + "q_sep_rate": 0.1095, + "monthly_equivalent": 0.0379 + }, + { + "year": 2022, + "quarter": 2, + "q_sep_rate": 0.1201, + "monthly_equivalent": 0.0418 + }, + { + "year": 2022, + "quarter": 3, + "q_sep_rate": 0.1307, + "monthly_equivalent": 0.0456 + }, + { + "year": 2022, + "quarter": 4, + "q_sep_rate": 0.1162, + "monthly_equivalent": 0.0403 + } + ], + "concept_deltas": [ + "J2J counts jobs (person-employer pairs) in UI-covered non-federal employment; SIPP counts all jobs incl. self-employment per job-month held", + "J2J MSep is main-job separations; SIPP here counts every held job", + "the monthly equivalent 1-(1-q)^(1/3) treats a quarter as three independent monthly draws \u2014 an approximation", + "persons leaving the SIPP sample are excluded from the denominator, not counted as separations", + "UNVERIFIED ASSUMPTION (job-ID linkage): the 9.45% Dec->Jan seam rate assumes SIPP EJB job ids are longitudinally consistent across the pu2022->pu2023 file boundary; if ids are reassigned at wave boundaries, spurious separations inflate the seam rate, so part of the seam-vs-within-wave contrast could be a linkage artifact rather than seam recall bias. This must be checked before C3 rules on it." + ], + "proposed_ruling_note": "NOT RATIFIED: as the plan proposed ex ante, J2J is truth for rate LEVELS and SIPP for persistence STRUCTURE; phase-1 hazards estimated seam-aware (wave-frequency with within-wave interpolation). Ratification belongs to the C3 referee round." +} diff --git a/scripts/build_seam_reconciliation.py b/scripts/build_seam_reconciliation.py new file mode 100644 index 00000000..2b5e33bf --- /dev/null +++ b/scripts/build_seam_reconciliation.py @@ -0,0 +1,271 @@ +"""Build the DRAFT SIPP-vs-J2J seam-reconciliation artifact (#192). + +REPORTED ANCHOR, NOT A GATE RUN — and explicitly a DRAFT: no +thresholds, C3 not locked. This is the reconciliation run the #192 +protocol rules require before E2/E4 thresholds lock ("commit a +documented SIPP-vs-J2J rate-reconciliation run and decide ex ante +which source is truth for rates vs persistence structure"). The +groundwork numbers were first posted to #192 (comment 4982442068); +this script commits the reproducible version and adds the E->N leg +the groundwork lacked. + +Three measurements, every concept delta NAMED, none adjusted away: + +(a) **Within-wave monthly job separation** per SIPP file: a job held + in month m and absent in m+1, among persons present in the + panel in both months (presence from the person-month universe, + so exits to nonemployment COUNT as separations — unlike the + groundwork, which conditioned on employed-both-months). +(b) **Across-wave separation** at the wave boundary: December of + one file's reference year to January of the next file's, + linking persons on ``SSUID``-``PNUM`` and the wave-consistent + ``EJB`` job ids. Seam bias concentrates here. +(c) **The J2J benchmark**: national all-industry main-job + separations (``MSep/MainB``) by quarter from the committed + extract, converted to a monthly equivalent + ``1 - (1 - q)**(1/3)``. + +Named concept deltas that remain: J2J counts jobs (person-employer +pairs) in UI-covered non-federal employment and separations of the +*main* job; SIPP here counts all jobs including self-employment +(JBORSE 2/3 excludable in later cuts) per job-month held. J2J +"quarterly separation" is not literally three independent monthly +draws, so the monthly equivalent is an approximation stated as such. + +UNVERIFIED ASSUMPTION (job-ID linkage): the across-wave Dec->Jan +seam rate assumes SIPP ``EJB`` job ids are longitudinally consistent +across the pu2022->pu2023 file boundary. If ids are reassigned at +wave boundaries, spurious separations inflate the seam rate, so part +of the seam-vs-within-wave contrast could be a linkage artifact +rather than seam recall bias. This must be checked before C3 rules +on it. + +Year-label convention: JSON keys under +``within_wave_monthly_separation`` are SIPP FILE years +(pu2022/pu2023), not reference years (file year = reference year ++ 1); PR #212's artifacts use reference years. + +Usage:: + + python scripts/build_seam_reconciliation.py + +writes ``runs/seam_reconciliation_draft_v0.json``. +""" + +from __future__ import annotations + +import json +import sys +from pathlib import Path + +import pandas as pd + +REPO = Path(__file__).resolve().parents[1] +sys.path.insert(0, str(REPO / "src")) + +from populace_dynamics.data import sipp_jobs # noqa: E402 + +FILE_YEARS = (2022, 2023) +J2J_YEARS = (2021, 2022) +J2J_EXTRACT = REPO / "data/external/j2j_us_firmsize_sector_2015on.csv" +ARTIFACT = REPO / "runs/seam_reconciliation_draft_v0.json" + + +def person_month_presence(year: int) -> pd.DataFrame: + """All person-months in the file (employed or not).""" + import os + + data_dir = Path( + os.environ.get( + "POPULACE_DYNAMICS_SIPP_DIR", + str(Path("~/PolicyEngine/sipp-data").expanduser()), + ) + ).expanduser() + for suffix in (".csv", ".csv.gz"): + path = data_dir / f"pu{year}{suffix}" + if path.exists(): + break + else: + raise FileNotFoundError(f"pu{year}.csv[.gz] not staged") + raw = pd.read_csv( + path, + sep="|", + usecols=["SSUID", "PNUM", "MONTHCODE"], + dtype={"SSUID": "string"}, + ) + raw["person_id"] = raw["SSUID"].astype(str) + "-" + raw["PNUM"].astype(str) + return raw[["person_id", "MONTHCODE"]].rename( + columns={"MONTHCODE": "month"} + ) + + +def within_wave(year: int) -> pd.DataFrame: + """Monthly job-separation rates, presence-conditioned only.""" + job_months = sipp_jobs.read_sipp_job_months(year) + presence = person_month_presence(year) + present = set(zip(presence["person_id"], presence["month"], strict=True)) + jobs = ( + job_months.groupby(["person_id", "month"])["job_id"] + .agg(frozenset) + .reset_index() + ) + jobs_next = { + (p, m): j + for p, m, j in zip( + jobs["person_id"], jobs["month"], jobs["job_id"], strict=True + ) + } + rows = [] + for month in range(1, 12): + held = kept = 0 + month_slice = jobs[jobs["month"] == month] + for person, js in zip( + month_slice["person_id"], month_slice["job_id"], strict=True + ): + if (person, month + 1) not in present: + continue # left the sample, not a separation + held += len(js) + kept += len(js & jobs_next.get((person, month + 1), frozenset())) + rows.append( + { + "month_pair": f"{month}->{month + 1}", + "jobs_held": held, + "jobs_kept": kept, + "sep_rate": round(1 - kept / held, 4), + } + ) + return pd.DataFrame(rows) + + +def across_wave() -> dict: + """Dec (first file) -> Jan (second file), sample-present both.""" + first = sipp_jobs.read_sipp_job_months(FILE_YEARS[0]) + second = sipp_jobs.read_sipp_job_months(FILE_YEARS[1]) + presence_second = person_month_presence(FILE_YEARS[1]) + present_jan = set( + presence_second[presence_second["month"] == 1]["person_id"] + ) + dec = ( + first[first["month"] == 12] + .groupby("person_id")["job_id"] + .agg(frozenset) + ) + jan = ( + second[second["month"] == 1] + .groupby("person_id")["job_id"] + .agg(frozenset) + ) + held = kept = persons = 0 + for person, js in dec.items(): + if person not in present_jan: + continue + persons += 1 + held += len(js) + kept += len(js & jan.get(person, frozenset())) + return { + "persons_linked": persons, + "jobs_held": held, + "jobs_kept": kept, + "sep_rate": round(1 - kept / held, 4), + } + + +def j2j_benchmark() -> list[dict]: + j2j = pd.read_csv(J2J_EXTRACT, dtype={"industry": str}) + national = j2j[(j2j.industry == "00") & (j2j.year.isin(J2J_YEARS))] + out = ( + national.groupby(["year", "quarter"]) + .agg(MainB=("MainB", "sum"), MSep=("MSep", "sum")) + .reset_index() + ) + out["q_sep_rate"] = (out.MSep / out.MainB).round(4) + out["monthly_equivalent"] = ( + 1 - (1 - out.MSep / out.MainB) ** (1 / 3) + ).round(4) + return out[ + ["year", "quarter", "q_sep_rate", "monthly_equivalent"] + ].to_dict("records") + + +def build() -> dict: + within = { + str(year): within_wave(year).to_dict("records") for year in FILE_YEARS + } + seam = across_wave() + bench = j2j_benchmark() + within_means = { + year: round(float(pd.DataFrame(rows)["sep_rate"].mean()), 4) + for year, rows in within.items() + } + return { + "artifact": "seam_reconciliation", + "version": "draft_v0", + "status": "DRAFT - NOT RATIFIED; C3 not locked; no thresholds", + "issue": "192", + "year_label_convention": ( + "Keys under within_wave_monthly_separation and " + "within_wave_means are SIPP FILE years (pu2022/pu2023), " + "not reference years; file year = reference year + 1. " + "PR #212's artifacts use reference years, so pu2022 " + "here pairs with #212's 2021 and pu2023 with #212's " + "2022." + ), + "notes": ( + "Naming collision: the SIPP 'j2j' transition measure " + "from #212 (~0.35% person-level direct employer change " + "per month) and the Census J2J data source's ~4.1% " + "monthly-equivalent main-job separation rate benchmark " + "used here are different universes and definitions " + "(person-level direct employer-to-employer moves vs " + "all main-job separations in UI-covered employment); " + "the ~10x gap is not a contradiction." + ), + "first_reported": ( + "PolicyEngine/populace-dynamics#192 comment 4982442068 " + "(groundwork conditioned on employed-both-months; this " + "artifact adds the E->N leg via the person-month " + "universe)" + ), + "within_wave_monthly_separation": within, + "within_wave_means": within_means, + "across_wave_dec_to_jan": seam, + "j2j_national_benchmark": bench, + "concept_deltas": [ + "J2J counts jobs (person-employer pairs) in UI-covered " + "non-federal employment; SIPP counts all jobs incl. " + "self-employment per job-month held", + "J2J MSep is main-job separations; SIPP here counts " + "every held job", + "the monthly equivalent 1-(1-q)^(1/3) treats a quarter " + "as three independent monthly draws — an approximation", + "persons leaving the SIPP sample are excluded from the " + "denominator, not counted as separations", + "UNVERIFIED ASSUMPTION (job-ID linkage): the 9.45% " + "Dec->Jan seam rate assumes SIPP EJB job ids are " + "longitudinally consistent across the pu2022->pu2023 " + "file boundary; if ids are reassigned at wave " + "boundaries, spurious separations inflate the seam " + "rate, so part of the seam-vs-within-wave contrast " + "could be a linkage artifact rather than seam recall " + "bias. This must be checked before C3 rules on it.", + ], + "proposed_ruling_note": ( + "NOT RATIFIED: as the plan proposed ex ante, J2J is " + "truth for rate LEVELS and SIPP for persistence " + "STRUCTURE; phase-1 hazards estimated seam-aware " + "(wave-frequency with within-wave interpolation). " + "Ratification belongs to the C3 referee round." + ), + } + + +def main() -> None: + artifact = build() + ARTIFACT.write_text(json.dumps(artifact, indent=2) + "\n") + print(f"wrote {ARTIFACT}") + print("within-wave means:", artifact["within_wave_means"]) + print("across-wave:", artifact["across_wave_dec_to_jan"]) + + +if __name__ == "__main__": + main()