"""Create the consolidated data-acquisition layout for all proposal states. Builds per-state target folders and writes a reproducible SEARCH_RECIPE.txt into each SERFF state, a CA-DOI note (CA is not on SERFF Filing Access), and a DOWNLOAD_MANIFEST.md enumerating every data source, its access method, and status. Creates folders only — it never downloads SERFF filings (the NAIC ToS prohibits automated download; retrieval is manual, see RETRIEVAL_RUNBOOK.md). """ from __future__ import annotations from pathlib import Path from . import corpus_queries as cq DATA_DIR = Path(__file__).resolve().parents[2] / "data" RAW = DATA_DIR / "raw" CA_NOTE = """\ California is NOT on SERFF Filing Access. Rate filings are retrieved from the California Department of Insurance rate-filing search (Prior Approval / SERFF filings surfaced via the CDI system), which has its own access workflow and terms. Document the CA access path and ToS here before any retrieval, then log saved filings via retrieval_ledger.py with state=CA. """ DOI_NOTE = """\ State DOI annual-report "state pages" supplement Schedule P for accident years after the public CAS Loss Reserve Database release. One source per state; record the URL and retrieval date alongside each downloaded file. """ def write_serff_recipes() -> list[Path]: """Write a reproducible search recipe into each SERFF state folder.""" written: list[Path] = [] for state in cq.SERFF_STATES: # bounded: 5 states dest = RAW / "serff" / state dest.mkdir(parents=True, exist_ok=True) recipe = dest / "SEARCH_RECIPE.txt" recipe.write_text(cq.render_steps(cq.build_query(state)), encoding="utf-8") written.append(recipe) return written def write_supplementary_dirs() -> None: """Create CA-DOI, DOI annual-report, and Schedule P target folders + notes.""" ca = RAW / "ca_doi" ca.mkdir(parents=True, exist_ok=True) (ca / "ACCESS_NOTE.txt").write_text(CA_NOTE, encoding="utf-8") doi = RAW / "doi_annual_reports" doi.mkdir(parents=True, exist_ok=True) (doi / "ACCESS_NOTE.txt").write_text(DOI_NOTE, encoding="utf-8") (RAW / "schedule_p").mkdir(parents=True, exist_ok=True) def write_manifest() -> Path: """Write DOWNLOAD_MANIFEST.md enumerating every source + access method.""" rows = [ "| Source | States | Access | Auto-downloadable | Target | Status |", "|---|---|---|---|---|---|", "| SERFF rate filings | TX, IL, GA, WA, IN | Manual intended-workflow " "(NAIC ToS bans automation) | No | `raw/serff//` | recipes written |", "| CA rate filings | CA | CA-DOI search (separate system) | No | " "`raw/ca_doi/` | note written |", "| CAS Loss Reserve DB (Schedule P) | all | Public CSV download | Yes | " "`raw/schedule_p/` | `schedule_p.download_all()` |", "| DOI annual-report state pages | all | Public, per-state | Manual | " "`raw/doi_annual_reports//` | note written |", ] manifest = DATA_DIR / "DOWNLOAD_MANIFEST.md" header = "# Data Download Manifest\n\nGenerated by `setup_data_dirs.py`.\n\n" manifest.write_text(header + "\n".join(rows) + "\n", encoding="utf-8") return manifest def main() -> int: """Build the full per-state acquisition layout.""" recipes = write_serff_recipes() write_supplementary_dirs() manifest = write_manifest() print(f"Wrote {len(recipes)} SERFF recipes, supplementary dirs, and {manifest.name}") return 0 if __name__ == "__main__": raise SystemExit(main())