Repository navigation
Expand file tree
/
Copy pathMakefile
More file actions
235 lines (208 loc) · 10.2 KB
/
Copy pathMakefile
File metadata and controls
235 lines (208 loc) · 10.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
# Export the environment to a yml file
export_env:
@conda env export > environment.yml;
# Build conda environment from the yml file
build_env:
@conda env create -f environment.yml;
# Install the package to pip
install_package:
@pip install -e .;
# Run the unit test suite (no network calls, runs in seconds)
test:
@python -m pytest tests/ -v;
# Lint source code, exploratory scripts, and tests
# Uses the openpois conda env's binaries regardless of whether it is activated
CONDA_PYTHON := $(shell conda run -n openpois which python 2>/dev/null || echo python)
CONDA_BIN := $(dir $(CONDA_PYTHON))
lint:
@$(CONDA_BIN)flake8 src/ scripts/ tests/
@$(CONDA_BIN)pylint src/openpois/
# Build the site for production
site_build:
@cd site && npm run build;
# Serve the site locally with hot reload
# Note: does not build Sphinx docs; use site_preview for a full build
site_dev:
@cd site && npm run dev;
# Generate site/public/taxonomy.html from the conflation data CSVs
# Requires the openpois conda env to be active (for pandas)
build_taxonomy:
@python scripts/build_taxonomy.py;
# Full build + local preview: Sphinx docs, Vite production build, then serve
# Mirrors the GitHub Actions workflow; serves at http://localhost:4173
# Requires the openpois conda env to be active (for sphinx-build)
# Uses Python's HTTP server instead of vite preview so /docs/ is served
# correctly (vite preview uses SPA fallback which swallows directory requests)
site_preview:
@python scripts/build_taxonomy.py
@sphinx-build -b html docs docs/_build/html -q
@cd site && npm run build
@cp -r docs/_build/html site/dist/docs
@python -m http.server 4173 --directory site/dist;
# -----------------------------------------------------------------------------
# Conflation pipeline (canonical entry point for all national runs)
#
# `make conflate` runs the five steps that produce the published
# conflated.parquet:
#
# 1. build_ghosts.py - reconstruct "ghost" POI dataset
# from OSM history (deletions,
# primary-tag removals, lifecycle
# prefixes, substantial renames).
# 2. conflate.py - OSM x Overture matching as before,
# written to conflated_baseline.parquet
# so the pre-CD result is archived.
# 3. apply_change_detection.py - penalize Overture POIs that shadow-
# match a same-entity ghost; writes
# conflated_cd.parquet.
# 4. calibrate - fit the three Bayesian fixed-rate
# mixture models (overture, osm,
# matched), export them as grid
# curves (stops if any segment fails
# acceptance), and apply them; writes
# the canonical conflated.parquet,
# then the calibration figures and
# the HT review.
# 5. apply_manual_overrides.py - hand-curated exclude/include pins
# (Close triage CSV), rewritten in
# place over conflated.parquet. Runs
# LAST so a forced conf_mean is never
# re-scaled by calibration.
#
# Each sub-step tees a per-run log under ~/data/openpois/logs/.
#
# Pass TEST=1 to scope to the Seattle bbox:
# make conflate # full CONUS
# make conflate TEST=1 # Seattle bbox dry run
# Under TEST=1 the Bayesian fit and export are skipped: the dry run applies
# the curves already in conflation/<version>/calibration/.
#
# In a validation month the run stops after change detection, because the
# validator samples from conflated_cd.parquet and calibration needs that
# round's handoff:
# make conflate_to_cd # steps 1-3
# ... validation round, handoff ...
# make calibrate apply_manual_overrides
#
# Sub-targets (build_ghosts / conflate_baseline / apply_cd / fit_calibration
# / export_calibration / apply_calibration / calibrate /
# apply_manual_overrides) are exposed for partial re-runs when one stage
# is being iterated on.
TEST ?=
TEST_FLAG := $(if $(TEST),--test,)
LOG_DIR := $(HOME)/data/openpois/logs
LOG_TS := $(shell date +%Y%m%d_%H%M%S)
.PHONY: download_history check_history rate conflate conflate_to_cd build_ghosts \
conflate_baseline apply_cd fit_calibration export_calibration apply_calibration \
calibrate apply_manual_overrides
# Build versions.osm_data: the full-history download (history_mode: full) or a
# roll-forward of download.osm.incremental_history.base_version with Geofabrik's
# daily diffs (history_mode: incremental). PLAN=1 prints the diff sequences an
# incremental run would fetch and exits.
download_history:
@mkdir -p $(LOG_DIR)
@$(CONDA_PYTHON) -u scripts/osm_data/download_history.py \
$(if $(PLAN),--plan-only,) \
2>&1 | tee $(LOG_DIR)/osm_history_$(LOG_TS).log
# QA: the history's last state of each snapshot node must match the snapshot.
check_history:
@mkdir -p $(LOG_DIR)
@$(CONDA_PYTHON) -u scripts/osm_data/check_history_vs_snapshot.py \
2>&1 | tee $(LOG_DIR)/check_history_$(LOG_TS).log
# Rate the OSM snapshot with the production random_effects model (per-POI cell
# reconstruction). Uses apply_model.model_stub from config; pass MODEL_VERSION=
# to override. NOTE: this is the correct rater for random_effects — the older
# apply_model.py is per-group only and must not be used for it.
rate:
@mkdir -p $(LOG_DIR)
@$(CONDA_PYTHON) -u scripts/osm_snapshot/apply_model_random_effects.py \
$(if $(MODEL_VERSION),--model-version $(MODEL_VERSION),) $(TEST_FLAG) \
2>&1 | tee $(LOG_DIR)/rate_$(LOG_TS).log
build_ghosts:
@mkdir -p $(LOG_DIR)
@$(CONDA_PYTHON) -u scripts/conflation/build_ghosts.py \
2>&1 | tee $(LOG_DIR)/build_ghosts_$(LOG_TS).log
conflate_baseline:
@mkdir -p $(LOG_DIR)
@$(CONDA_PYTHON) -u scripts/conflation/conflate.py \
--output-suffix=baseline $(TEST_FLAG) \
2>&1 | tee $(LOG_DIR)/conflate_baseline_$(LOG_TS).log
apply_cd:
@mkdir -p $(LOG_DIR)
@$(CONDA_PYTHON) -u scripts/conflation/apply_change_detection.py \
--baseline-suffix=baseline --output-suffix=cd $(TEST_FLAG) \
2>&1 | tee $(LOG_DIR)/apply_cd_$(LOG_TS).log
# Fit the per-segment existence-confidence curves from the validation rounds in
# conflation.calibration.pooled_rounds, then map every POI through them.
# Calibration runs AFTER change detection: the CD penalty multiplies conf_mean,
# so calibrating first would leave a calibrated probability scaled by delta.
#
# fit_calibration runs the three Bayesian fixed-rate mixture fits in parallel
# (run_bayes_phase1.sh MODE=mixture; settings in conflation.calibration.bayes);
# export_calibration (--overwrite: a calibrate run replaces this version's curves)
# turns their draws into grid curves in
# conflation/<version>/calibration/ and exits non-zero, writing nothing
# deployable, if any segment fails the acceptance rule. The v4
# fit_calibration.py is retired. pipefail makes a failing step stop make
# despite the tee.
fit_calibration export_calibration apply_calibration calibrate: SHELL := /bin/bash
fit_calibration export_calibration apply_calibration calibrate: .SHELLFLAGS := -o pipefail -c
fit_calibration:
@mkdir -p $(LOG_DIR)
ifeq ($(TEST),)
@MODE=mixture bash scripts/conflation/run_bayes_phase1.sh \
2>&1 | tee $(LOG_DIR)/fit_calibration_$(LOG_TS).log
else
@echo "TEST=1: skipping the Bayesian fit (applying the existing curves)"
endif
export_calibration:
@mkdir -p $(LOG_DIR)
ifeq ($(TEST),)
@$(CONDA_PYTHON) -u scripts/conflation/export_bayes_curves.py --overwrite \
2>&1 | tee $(LOG_DIR)/export_calibration_$(LOG_TS).log
else
@echo "TEST=1: skipping the curve export (applying the existing curves)"
endif
apply_calibration:
@mkdir -p $(LOG_DIR)
@$(CONDA_PYTHON) -u scripts/conflation/apply_calibration.py \
--input-suffix=cd --output-suffix="" $(TEST_FLAG) \
2>&1 | tee $(LOG_DIR)/apply_calibration_$(LOG_TS).log
calibrate: fit_calibration export_calibration apply_calibration
@$(CONDA_PYTHON) -u scripts/conflation/plot_calibration.py \
2>&1 | tee $(LOG_DIR)/plot_calibration_$(LOG_TS).log
@$(CONDA_PYTHON) -u scripts/conflation/ht_review.py \
2>&1 | tee $(LOG_DIR)/ht_review_$(LOG_TS).log
# Manual exclude/include pins from the Close triage CSV. Must run AFTER
# calibrate: it rewrites conflated.parquet in place and a forced conf_mean
# of 0 / 1 must not be re-scaled by the curves. A missing CSV is a no-op.
apply_manual_overrides:
@mkdir -p $(LOG_DIR)
@$(CONDA_PYTHON) -u scripts/conflation/apply_manual_overrides.py \
$(TEST_FLAG) \
2>&1 | tee $(LOG_DIR)/apply_manual_overrides_$(LOG_TS).log
# Steps 1-3 only: stop at conflated_cd.parquet, the frame a validation round
# samples from. Run `make calibrate apply_manual_overrides` after the handoff.
conflate_to_cd: build_ghosts conflate_baseline apply_cd
@echo
@echo "Conflation through change detection complete."
@echo " Output: ~/data/openpois/conflation/<version>/conflated_cd.parquet"
@echo " Next, after the validation handoff: make calibrate apply_manual_overrides"
conflate: build_ghosts conflate_baseline apply_cd calibrate apply_manual_overrides
@echo
@echo "Conflation pipeline complete."
@echo " Canonical output: ~/data/openpois/conflation/<version>/conflated.parquet"
@echo " (calibrated + manual overrides applied in place)"
@echo " (pre-calibration: conflated_cd.parquet)"
@echo " (no-CD archive: conflated_baseline.parquet)"
@echo " Bayesian fits: conflation/<version>/calibration_bayes/"
@echo " Grid curves + fit report + HT review: conflation/<version>/calibration/"
@echo " Logs under: $(LOG_DIR)/{build_ghosts,conflate_baseline,apply_cd,fit_calibration,export_calibration,apply_calibration,plot_calibration,ht_review,apply_manual_overrides}_$(LOG_TS).log"
# Convenience target to print all of the available targets in this file
# From https://stackoverflow.com/questions/4219255
.PHONY: list
list:
@LC_ALL=C $(MAKE) -pRrq -f $(lastword $(MAKEFILE_LIST)) : 2>/dev/null | \
awk -v RS= -F: '/^# File/,/^# Finished Make data base/ \
{if ($$1 !~ "^[#.]") {print $$1}}' | \
sort | egrep -v -e '^[^[:alnum:]]' -e '^$@$$'