From d050e7b2dd16ad8240f243de02fd339488f2ca71 Mon Sep 17 00:00:00 2001 From: Matt McKay Date: Tue, 1 Sep 2026 16:21:27 +1000 Subject: [PATCH 1/2] FRED from GitHub runners: no custom User-Agent; canary one leg per builder MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The first canary run of us_business_cycle_monthly.csv timed out on its first FRED request (#115) — twice. A probe from a runner (PR #116, closed) showed why: FRED's edge answers `Python-urllib/3.12` and `curl/8.5.0` in ~50 ms and stalls `qeld-builder` and even `Mozilla/5.0` until the read times out, while every variant succeeds from a workstation, which is how the custom agent survived local testing. The Fred class now sends no User-Agent by default (urllib's own), with the measurement in its docstring; fred_data.py drops the same header. Two workflow fixes found on the same runs: the canary ran a set-writing builder once per dataset (three identical World Bank fetches), so `snapshots.py list --by-builder` gives it one leg per builder; and the `dataset` dispatch input filtered only the refresh matrix, so it now filters the canary too. See #115. Co-Authored-By: Claude Fable 5 --- .github/workflows/refresh-snapshots.yml | 6 ++++-- builders/_fred.py | 16 +++++++++++----- builders/fred_data.py | 4 +++- scripts/snapshots.py | 18 ++++++++++++++++-- 4 files changed, 34 insertions(+), 10 deletions(-) diff --git a/.github/workflows/refresh-snapshots.yml b/.github/workflows/refresh-snapshots.yml index 683afe6..6f95adc 100644 --- a/.github/workflows/refresh-snapshots.yml +++ b/.github/workflows/refresh-snapshots.yml @@ -71,9 +71,11 @@ jobs: FORCE: ${{ inputs.force }} run: | flags="" - [ -n "$DATASET" ] && flags="$flags --dataset $DATASET" + dsflag="" + [ -n "$DATASET" ] && flags="$flags --dataset $DATASET" && dsflag="--dataset $DATASET" [ "$FORCE" = "true" ] && flags="$flags --all" - all=$(python scripts/snapshots.py list | jq -c .) + # canary: one leg per BUILDER (a set-writing builder fetches once) + all=$(python scripts/snapshots.py list --by-builder $dsflag | jq -c .) due=$(python scripts/snapshots.py due $flags | jq -c .) echo "all=$all" >> "$GITHUB_OUTPUT" echo "due=$due" >> "$GITHUB_OUTPUT" diff --git a/builders/_fred.py b/builders/_fred.py index e07e13a..aa277e2 100644 --- a/builders/_fred.py +++ b/builders/_fred.py @@ -21,7 +21,13 @@ DatetimeIndex named `DATE`, so committed files keep the header the lectures expect - FRED's `.` for a missing observation -> NaN -- a User-Agent header, which fred.stlouisfed.org has been seen to require +- the User-Agent: NONE by default, deliberately. Measured 2026-09-01 from a + GitHub-hosted runner (data-lectures#115): FRED's edge answers + `Python-urllib/3.12` (urllib's own default) and `curl/8.5.0` in ~50 ms, + and STALLS `qeld-builder` and even `Mozilla/5.0` until the read times + out. The same requests all succeed from a workstation, which is how the + custom agent survived local testing. Pass `user_agent=` only if you have + measured that it works from where the builder will actually run - one series per request, aligned with an outer join in frame(), so a series that starts later is simply empty before its first observation @@ -39,14 +45,14 @@ class Fred: - def __init__(self, user_agent='qeld-builder', timeout=60): - self.user_agent = user_agent + def __init__(self, user_agent=None, timeout=60): + self.user_agent = user_agent # None -> urllib's default, see above self.timeout = timeout def _get(self, params): query = urllib.parse.urlencode({k: v for k, v in params.items() if v is not None}) - request = urllib.request.Request(f'{FREDGRAPH}?{query}', - headers={'User-Agent': self.user_agent}) + headers = {'User-Agent': self.user_agent} if self.user_agent else {} + request = urllib.request.Request(f'{FREDGRAPH}?{query}', headers=headers) with urllib.request.urlopen(request, timeout=self.timeout) as response: return response.read() diff --git a/builders/fred_data.py b/builders/fred_data.py index 0f91e20..74150a6 100644 --- a/builders/fred_data.py +++ b/builders/fred_data.py @@ -61,7 +61,9 @@ def _fetch_series(code): url = f'{FRED_CSV}?id={code}&cosd={START}&coed={END}' if code in DAILY_AVERAGED: url += '&fq=Monthly&fam=avg' - request = urllib.request.Request(url, headers={'User-Agent': 'qeld-builder'}) + # No custom User-Agent: FRED's edge stalls unfamiliar agents from GitHub + # runners and answers urllib's default at once (data-lectures#115). + request = urllib.request.Request(url) with urllib.request.urlopen(request) as response: payload = response.read() frame = pd.read_csv(io.BytesIO(payload), index_col=0, parse_dates=True, diff --git a/scripts/snapshots.py b/scripts/snapshots.py index 1f42cda..81cda7c 100644 --- a/scripts/snapshots.py +++ b/scripts/snapshots.py @@ -197,7 +197,20 @@ def sha256(path: pathlib.Path) -> str: # --------------------------------------------------------------------------- def cmd_list(args) -> int: - print(json.dumps(snapshots(load_manifests()), indent=1)) + rows = snapshots(load_manifests()) + if args.dataset: + rows = [r for r in rows if r["dataset"] == args.dataset] + if args.by_builder: + # One leg per builder for the canary: a builder that writes a set + # fetches once for all of them, so running it per dataset only + # repeats the same fetch. The leg is named for its first dataset. + seen, unique = set(), [] + for r in rows: + if r["builder"] not in seen: + seen.add(r["builder"]) + unique.append({**r, "datasets": [x["dataset"] for x in rows if x["builder"] == r["builder"]]}) + rows = unique + print(json.dumps(rows, indent=1)) return 0 @@ -331,7 +344,8 @@ def cmd_pr_body(args) -> int: def main() -> int: ap = argparse.ArgumentParser(description=__doc__.split("\n\n")[0]) sub = ap.add_subparsers(dest="cmd", required=True) - sub.add_parser("list").set_defaults(fn=cmd_list) + p = sub.add_parser("list"); p.add_argument("--dataset"); p.add_argument("--by-builder", action="store_true") + p.set_defaults(fn=cmd_list) p = sub.add_parser("due"); p.add_argument("--all", action="store_true"); p.add_argument("--dataset") p.set_defaults(fn=cmd_due) p = sub.add_parser("stamp"); p.add_argument("dataset"); p.add_argument("--summary", required=True) From e2e77309285732100698f2025cf8f6f734e3d0f6 Mon Sep 17 00:00:00 2001 From: Matt McKay Date: Tue, 1 Sep 2026 16:31:24 +1000 Subject: [PATCH 2/2] Copilot review on #117: single-pass grouping, array flags, accurate docstring snapshots.py groups the canary legs in one pass keyed by builder; the plan step builds its CLI flags as bash arrays so the dataset input is one argument however it is spelled; and _fred.py says "no custom User-Agent" rather than "none", since urllib always sends Python-urllib/x.y. Co-Authored-By: Claude Fable 5 --- .github/workflows/refresh-snapshots.yml | 17 +++++++++++------ builders/_fred.py | 18 ++++++++++-------- scripts/snapshots.py | 12 ++++++------ 3 files changed, 27 insertions(+), 20 deletions(-) diff --git a/.github/workflows/refresh-snapshots.yml b/.github/workflows/refresh-snapshots.yml index 6f95adc..94cccfd 100644 --- a/.github/workflows/refresh-snapshots.yml +++ b/.github/workflows/refresh-snapshots.yml @@ -70,13 +70,18 @@ jobs: DATASET: ${{ inputs.dataset }} FORCE: ${{ inputs.force }} run: | - flags="" - dsflag="" - [ -n "$DATASET" ] && flags="$flags --dataset $DATASET" && dsflag="--dataset $DATASET" - [ "$FORCE" = "true" ] && flags="$flags --all" + # Arrays, not string concatenation: the dataset input is a single + # argument however it is spelled. + due_flags=() + list_flags=() + if [ -n "$DATASET" ]; then + due_flags+=(--dataset "$DATASET") + list_flags+=(--dataset "$DATASET") + fi + [ "$FORCE" = "true" ] && due_flags+=(--all) # canary: one leg per BUILDER (a set-writing builder fetches once) - all=$(python scripts/snapshots.py list --by-builder $dsflag | jq -c .) - due=$(python scripts/snapshots.py due $flags | jq -c .) + all=$(python scripts/snapshots.py list --by-builder "${list_flags[@]}" | jq -c .) + due=$(python scripts/snapshots.py due "${due_flags[@]}" | jq -c .) echo "all=$all" >> "$GITHUB_OUTPUT" echo "due=$due" >> "$GITHUB_OUTPUT" echo "canary: $(echo "$all" | jq -r '.[].dataset' | tr '\n' ' ')" diff --git a/builders/_fred.py b/builders/_fred.py index aa277e2..89db7c5 100644 --- a/builders/_fred.py +++ b/builders/_fred.py @@ -21,13 +21,14 @@ DatetimeIndex named `DATE`, so committed files keep the header the lectures expect - FRED's `.` for a missing observation -> NaN -- the User-Agent: NONE by default, deliberately. Measured 2026-09-01 from a - GitHub-hosted runner (data-lectures#115): FRED's edge answers - `Python-urllib/3.12` (urllib's own default) and `curl/8.5.0` in ~50 ms, - and STALLS `qeld-builder` and even `Mozilla/5.0` until the read times - out. The same requests all succeed from a workstation, which is how the - custom agent survived local testing. Pass `user_agent=` only if you have - measured that it works from where the builder will actually run +- NO CUSTOM User-Agent by default, deliberately: the request goes out with + urllib's own `Python-urllib/x.y`. Measured 2026-09-01 from a GitHub-hosted + runner (data-lectures#115): FRED's edge answers `Python-urllib/3.12` and + `curl/8.5.0` in ~50 ms, and STALLS `qeld-builder` and even `Mozilla/5.0` + until the read times out. The same requests all succeed from a + workstation, which is how the custom agent survived local testing. Pass + `user_agent=` only if you have measured that it works from where the + builder will actually run - one series per request, aligned with an outer join in frame(), so a series that starts later is simply empty before its first observation @@ -46,7 +47,8 @@ class Fred: def __init__(self, user_agent=None, timeout=60): - self.user_agent = user_agent # None -> urllib's default, see above + self.user_agent = user_agent # None -> no custom header; urllib + # sends Python-urllib/x.y (see above) self.timeout = timeout def _get(self, params): diff --git a/scripts/snapshots.py b/scripts/snapshots.py index 81cda7c..fd76e0a 100644 --- a/scripts/snapshots.py +++ b/scripts/snapshots.py @@ -203,13 +203,13 @@ def cmd_list(args) -> int: if args.by_builder: # One leg per builder for the canary: a builder that writes a set # fetches once for all of them, so running it per dataset only - # repeats the same fetch. The leg is named for its first dataset. - seen, unique = set(), [] + # repeats the same fetch. One pass, first-seen order; the leg is + # named for the builder's first dataset. + by_builder: dict[str, dict] = {} for r in rows: - if r["builder"] not in seen: - seen.add(r["builder"]) - unique.append({**r, "datasets": [x["dataset"] for x in rows if x["builder"] == r["builder"]]}) - rows = unique + leg = by_builder.setdefault(r["builder"], {**r, "datasets": []}) + leg["datasets"].append(r["dataset"]) + rows = list(by_builder.values()) print(json.dumps(rows, indent=1)) return 0