Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
170 changes: 170 additions & 0 deletions src/spicy_regs/ingest_sam_extract.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,170 @@
"""Ingest the SAM.gov public monthly *entity extract* into ``sam_entities``.

The SAM Entity Management API is daily-quota-gated (≈1,000 req/day for a
non-federal keyed account) and caps a single query's pagination at ~5,000
records, so the live-API reader (``sources/sam_entities.py``) can only ever
sample the ~885K registered entities. The **public monthly extract** is the
full set in one flat file — this module parses it and publishes the complete
``sam_entities`` table.

Get the file (no API quota involved): SAM.gov → **Data Services** → *Entity
Registration* → *Public* → download ``SAM_PUBLIC_MONTHLY_V2_YYYYMMDD.ZIP`` and
unzip the ``.dat``. (Or the extract API:
``https://api.sam.gov/data-services/v1/extracts?api_key=...&fileType=ENTITY&sensitivity=PUBLIC``.)

Usage::

uv run python -m spicy_regs.ingest_sam_extract SAM_PUBLIC_MONTHLY_V2_20260705.dat
uv run python -m spicy_regs.ingest_sam_extract <file>.dat --output out.parquet --upload

File format: pipe-delimited, **142 fields**, no quoting; a ``BOF PUBLIC V2 ...``
header line and an ``EOF PUBLIC V2 ...`` trailer line wrap the data records.

Field map (1-indexed extract position → published column), confirmed against the
2026-07 extract:

=================================== ======== ================================
column position notes
=================================== ======== ================================
uei 1 12-char Unique Entity ID (PK)
cage_code 4
legal_business_name 12
dba_name 13
entity_structure_desc 28 raw structure CODE (see note)
state 19 physical-address state
city 18
zip_code 20
congressional_district 23
primary_naics 33 primary NAICS (list starts here)
registration_status 6 A→Active, E→Expired
registration_date 8 YYYYMMDD → YYYY-MM-DD
registration_expiration_date 9 YYYYMMDD → YYYY-MM-DD
purpose_of_registration_desc 7 Z1/Z2/Z5 mapped, else raw code
entity_url 27
=================================== ======== ================================

Extract limitations vs the API: the extract does not cleanly expose
``entity_type_desc``, ``profit_structure_desc``, or ``exclusion_status_flag``
(exclusions are a separate SAM extract), so those are published NULL; and
``entity_structure_desc`` carries SAM's raw structure code rather than the API's
prose description.
"""

from __future__ import annotations

import argparse
from pathlib import Path

import pyarrow.parquet as pq
from loguru import logger

OUTPUT = "sam_entities.parquet"

# 1-indexed extract position -> 0-indexed DuckDB column name (read as c0..c141).
POS = {
"uei": "c0",
"cage_code": "c3",
"legal_business_name": "c11",
"dba_name": "c12",
"entity_structure_desc": "c27",
"state": "c18",
"city": "c17",
"zip_code": "c19",
"congressional_district": "c22",
"primary_naics": "c32",
"registration_status_code": "c5",
"purpose_code": "c6",
"registration_date_raw": "c7",
"registration_expiration_date_raw": "c8",
"entity_url": "c26",
}

# SAM code -> description for the small, stable code sets we surface.
STATUS_MAP = {"A": "Active", "E": "Expired"}
PURPOSE_MAP = {"Z1": "Federal Assistance Awards", "Z2": "All Awards", "Z5": "IGT Only"}

_NUM_FIELDS = 142


def _read_csv_expr(src: Path) -> str:
cols = "{" + ",".join(f"'c{i}':'VARCHAR'" for i in range(_NUM_FIELDS)) + "}"
# No quoting (flat file with literal quotes in data); skip the BOF header;
# the EOF trailer is dropped by the UEI-length filter below.
return (
f"read_csv('{src}', delim='|', header=false, auto_detect=false, quote='', "
f"escape='', null_padding=true, ignore_errors=true, parallel=false, skip=1, "
f"max_line_size=20000000, columns={cols})"
)


def _case_map(col: str, mapping: dict[str, str]) -> str:
whens = " ".join(f"WHEN '{k}' THEN '{v}'" for k, v in mapping.items())
return f"CASE trim({col}) {whens} ELSE NULLIF(trim({col}),'') END"


def build_sam_entities_from_extract(src: Path, out: Path) -> Path:
"""Parse a SAM public entity extract ``.dat`` into ``sam_entities`` parquet."""
import duckdb

def nz(c: str) -> str:
return f"NULLIF(trim({c}),'')"

def dt(c: str) -> str:
return f"CASE WHEN length(trim({c}))=8 THEN substr({c},1,4)||'-'||substr({c},5,2)||'-'||substr({c},7,2) END"

logger.info("SAM extract: parsing {}", src)
con = duckdb.connect()
con.execute(
f"""
COPY (
SELECT uei, cage_code, legal_business_name, dba_name, entity_structure_desc,
entity_type_desc, profit_structure_desc, state, city, zip_code,
congressional_district, primary_naics, registration_status,
registration_date, registration_expiration_date, exclusion_status_flag,
purpose_of_registration_desc, entity_url
FROM (
SELECT {nz(POS["uei"])} uei, {nz(POS["cage_code"])} cage_code,
{nz(POS["legal_business_name"])} legal_business_name, {nz(POS["dba_name"])} dba_name,
{nz(POS["entity_structure_desc"])} entity_structure_desc,
CAST(NULL AS VARCHAR) entity_type_desc, CAST(NULL AS VARCHAR) profit_structure_desc,
{nz(POS["state"])} state, {nz(POS["city"])} city, {nz(POS["zip_code"])} zip_code,
{nz(POS["congressional_district"])} congressional_district, {nz(POS["primary_naics"])} primary_naics,
{_case_map(POS["registration_status_code"], STATUS_MAP)} registration_status,
{dt(POS["registration_date_raw"])} registration_date,
{dt(POS["registration_expiration_date_raw"])} registration_expiration_date,
CAST(NULL AS VARCHAR) exclusion_status_flag,
{_case_map(POS["purpose_code"], PURPOSE_MAP)} purpose_of_registration_desc,
{nz(POS["entity_url"])} entity_url,
ROW_NUMBER() OVER (
PARTITION BY {POS["uei"]}
ORDER BY ({POS["registration_status_code"]}='A') DESC, {POS["registration_expiration_date_raw"]} DESC
) rn
FROM {_read_csv_expr(src)}
WHERE length(trim({POS["uei"]}))=12
) WHERE rn=1
) TO '{out}' (FORMAT PARQUET, COMPRESSION ZSTD, ROW_GROUP_SIZE 100000);
"""
)
con.close()
rows = pq.ParquetFile(out).metadata.num_rows
logger.info("SAM extract: {:,} entities -> {}", rows, out)
return out


def main() -> None:
ap = argparse.ArgumentParser(description="Ingest a SAM public entity extract into sam_entities.")
ap.add_argument("dat", type=Path, help="Path to SAM_PUBLIC_MONTHLY_V2_YYYYMMDD.dat")
ap.add_argument("--output", type=Path, default=Path("output") / OUTPUT, help="Output parquet path")
ap.add_argument("--upload", action="store_true", help="Publish the result to R2 (needs R2 creds)")
args = ap.parse_args()
args.output.parent.mkdir(parents=True, exist_ok=True)
out = build_sam_entities_from_extract(args.dat, args.output)
if args.upload:
from spicy_regs.sources import r2

r2.upload_file(out, remote_key=OUTPUT)
logger.info("Uploaded {} to R2", OUTPUT)


if __name__ == "__main__":
main()
107 changes: 107 additions & 0 deletions tests/test_ingest_sam_extract.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,107 @@
"""Hermetic test for the SAM public-extract parser (no network)."""

from __future__ import annotations

import duckdb

from spicy_regs.ingest_sam_extract import build_sam_entities_from_extract


def _record(**pos: str) -> str:
"""Build one 142-field pipe-delimited record with given 0-indexed positions set."""
fields = [""] * 142
for k, v in pos.items():
fields[int(k[1:])] = v # keys like "c0", "c18"
return "|".join(fields)


_FIXTURE = "\n".join(
[
"BOF PUBLIC V2 00000000 20260705 0000002 0000000",
_record(
c0="ABCDEFGHIJ12",
c3="1A2B3",
c5="A",
c6="Z2",
c7="20200115",
c8="20270115",
c11="ACME WIDGETS INC",
c12="ACME",
c17="RANCHO CORDOVA",
c18="CA",
c19="95742",
c22="06",
c26="www.acme.example",
c27="2L",
c32="444110",
),
_record(
c0="ZYXWVUTSRQ98",
c3="",
c5="E",
c6="Z1",
c7="20150301",
c8="20240301",
c11="OLD CO LLC",
c18="TX",
c19="75001",
c32="",
),
"EOF PUBLIC V2 00000000 20260705 0000002 0000000",
]
)


def test_extract_parses_and_maps(tmp_path):
src = tmp_path / "SAM_PUBLIC_MONTHLY_V2_20260705.dat"
src.write_text(_FIXTURE + "\n")
out = tmp_path / "sam_entities.parquet"
build_sam_entities_from_extract(src, out)

rows = duckdb.sql(f"SELECT * FROM read_parquet('{out}') ORDER BY uei").fetchall()
cols = [d[0] for d in duckdb.sql(f"DESCRIBE SELECT * FROM read_parquet('{out}')").fetchall()]
assert cols == [
"uei",
"cage_code",
"legal_business_name",
"dba_name",
"entity_structure_desc",
"entity_type_desc",
"profit_structure_desc",
"state",
"city",
"zip_code",
"congressional_district",
"primary_naics",
"registration_status",
"registration_date",
"registration_expiration_date",
"exclusion_status_flag",
"purpose_of_registration_desc",
"entity_url",
]
assert len(rows) == 2 # BOF/EOF sentinel lines dropped

r = dict(zip(cols, rows[0])) # ABCDEFGHIJ12
assert r["uei"] == "ABCDEFGHIJ12"
assert r["cage_code"] == "1A2B3"
assert r["legal_business_name"] == "ACME WIDGETS INC"
assert r["dba_name"] == "ACME"
assert r["state"] == "CA" and r["city"] == "RANCHO CORDOVA" and r["zip_code"] == "95742"
assert r["congressional_district"] == "06"
assert r["primary_naics"] == "444110"
assert r["entity_structure_desc"] == "2L"
assert r["entity_url"] == "www.acme.example"
assert r["registration_status"] == "Active" # A -> Active
assert r["purpose_of_registration_desc"] == "All Awards" # Z2 -> mapped
assert r["registration_date"] == "2020-01-15" # YYYYMMDD -> YYYY-MM-DD
assert r["registration_expiration_date"] == "2027-01-15"
# Fields the extract doesn't cleanly provide are NULL.
assert r["entity_type_desc"] is None
assert r["profit_structure_desc"] is None
assert r["exclusion_status_flag"] is None

r2 = dict(zip(cols, rows[1])) # ZYXWVUTSRQ98
assert r2["registration_status"] == "Expired" # E -> Expired
assert r2["purpose_of_registration_desc"] == "Federal Assistance Awards" # Z1
assert r2["cage_code"] is None and r2["primary_naics"] is None # empty -> NULL
Loading