cas-serff-extractor / serff_pipeline /setup_data_dirs.py
pramodmisra's picture
Upload folder using huggingface_hub
1e8f25c verified
Raw History Blame Contribute Delete
3.57 kB
"""Create the consolidated data-acquisition layout for all proposal states.
Builds per-state target folders and writes a reproducible SEARCH_RECIPE.txt into
each SERFF state, a CA-DOI note (CA is not on SERFF Filing Access), and a
DOWNLOAD_MANIFEST.md enumerating every data source, its access method, and
status. Creates folders only — it never downloads SERFF filings (the NAIC ToS
prohibits automated download; retrieval is manual, see RETRIEVAL_RUNBOOK.md).
"""
from __future__ import annotations
from pathlib import Path
from . import corpus_queries as cq
DATA_DIR = Path(__file__).resolve().parents[2] / "data"
RAW = DATA_DIR / "raw"
CA_NOTE = """\
California is NOT on SERFF Filing Access. Rate filings are retrieved from the
California Department of Insurance rate-filing search (Prior Approval / SERFF
filings surfaced via the CDI system), which has its own access workflow and
terms. Document the CA access path and ToS here before any retrieval, then log
saved filings via retrieval_ledger.py with state=CA.
"""
DOI_NOTE = """\
State DOI annual-report "state pages" supplement Schedule P for accident years
after the public CAS Loss Reserve Database release. One source per state; record
the URL and retrieval date alongside each downloaded file.
"""
def write_serff_recipes() -> list[Path]:
"""Write a reproducible search recipe into each SERFF state folder."""
written: list[Path] = []
for state in cq.SERFF_STATES: # bounded: 5 states
dest = RAW / "serff" / state
dest.mkdir(parents=True, exist_ok=True)
recipe = dest / "SEARCH_RECIPE.txt"
recipe.write_text(cq.render_steps(cq.build_query(state)), encoding="utf-8")
written.append(recipe)
return written
def write_supplementary_dirs() -> None:
"""Create CA-DOI, DOI annual-report, and Schedule P target folders + notes."""
ca = RAW / "ca_doi"
ca.mkdir(parents=True, exist_ok=True)
(ca / "ACCESS_NOTE.txt").write_text(CA_NOTE, encoding="utf-8")
doi = RAW / "doi_annual_reports"
doi.mkdir(parents=True, exist_ok=True)
(doi / "ACCESS_NOTE.txt").write_text(DOI_NOTE, encoding="utf-8")
(RAW / "schedule_p").mkdir(parents=True, exist_ok=True)
def write_manifest() -> Path:
"""Write DOWNLOAD_MANIFEST.md enumerating every source + access method."""
rows = [
"| Source | States | Access | Auto-downloadable | Target | Status |",
"|---|---|---|---|---|---|",
"| SERFF rate filings | TX, IL, GA, WA, IN | Manual intended-workflow "
"(NAIC ToS bans automation) | No | `raw/serff/<ST>/` | recipes written |",
"| CA rate filings | CA | CA-DOI search (separate system) | No | "
"`raw/ca_doi/` | note written |",
"| CAS Loss Reserve DB (Schedule P) | all | Public CSV download | Yes | "
"`raw/schedule_p/` | `schedule_p.download_all()` |",
"| DOI annual-report state pages | all | Public, per-state | Manual | "
"`raw/doi_annual_reports/<ST>/` | note written |",
]
manifest = DATA_DIR / "DOWNLOAD_MANIFEST.md"
header = "# Data Download Manifest\n\nGenerated by `setup_data_dirs.py`.\n\n"
manifest.write_text(header + "\n".join(rows) + "\n", encoding="utf-8")
return manifest
def main() -> int:
"""Build the full per-state acquisition layout."""
recipes = write_serff_recipes()
write_supplementary_dirs()
manifest = write_manifest()
print(f"Wrote {len(recipes)} SERFF recipes, supplementary dirs, and {manifest.name}")
return 0
if __name__ == "__main__":
raise SystemExit(main())