diff --git a/scripts/parse_importinfo_shipments.py b/scripts/parse_importinfo_shipments.py new file mode 100755 index 0000000..d441a07 --- /dev/null +++ b/scripts/parse_importinfo_shipments.py @@ -0,0 +1,329 @@ +#!/usr/bin/env python3 +import argparse +import csv +import html +from dataclasses import dataclass +from html.parser import HTMLParser +from pathlib import Path + + +FIELD_MAP = { + "run date": "run_date", + "master bol": "master_bol", + "house bol": "house_bol", + "voyage #": "voyage", + "bill type": "bill_type", + "carrier code": "carrier_code", + "imo #": "imo", + "vessel name": "vessel_name", + "arrival date": "arrival_date", + "us port": "us_port", + "foreign port": "foreign_port", + "quantity": "quantity", + "weight": "weight", + "type of service": "type_of_service", + "shipper": "shipper", + "consignee": "consignee", + "notify party": "notify_party", + "commodity": "commodity", +} + +OUTPUT_FIELDS = [ + "source", + "query", + "run_date", + "master_bol", + "house_bol", + "voyage", + "bill_type", + "carrier_code", + "imo", + "vessel_name", + "arrival_date", + "us_port", + "foreign_port", + "quantity", + "weight", + "type_of_service", + "shipper", + "consignee", + "notify_party", + "commodity", + "source_url", +] + +PARTY_TERMS = ("VALVE", "CEVA", "INGRAM MICRO", "TECH-FRONT", "CHENG UEI") +PRODUCT_TERMS = ( + "GAME CONSOLE", + "VR CONTROLLER", + "CONTROLLER", + "STEAM", + "BASE STATION", + "HEADSET", + "DONGLE", +) + +SOURCE_URLS = { + "ceva-valve": "https://www.importinfo.com/search?s=CEVA%20C%2FO%20VALVE%20CORPORATION", + "ingram-valve": "https://www.importinfo.com/search?s=INGRAM%20MICRO%20C%2FO%20VALVE%20CORPORATION", + "tech-front-game-console": "https://www.importinfo.com/search?s=TECH-FRONT%20GAME%20CONSOLE%20VALVE", + "valve-corporation-game-console": "https://www.importinfo.com/search?s=VALVE%20CORPORATION%20GAME%20CONSOLE", +} + + +@dataclass +class InputSpec: + query: str + path: Path + source_url: str + + +class TableParser(HTMLParser): + def __init__(self): + super().__init__() + self.headers = [] + self.rows = [] + self._in_row = False + self._in_cell = False + self._cell_tag = "" + self._current_cells = [] + self._current_cell_parts = [] + self._row_has_header = False + self._row_has_data = False + + def handle_starttag(self, tag, attrs): + if tag == "tr": + self._in_row = True + self._current_cells = [] + self._row_has_header = False + self._row_has_data = False + elif self._in_row and tag in ("th", "td"): + self._in_cell = True + self._cell_tag = tag + self._current_cell_parts = [] + if tag == "th": + self._row_has_header = True + else: + self._row_has_data = True + + def handle_endtag(self, tag): + if self._in_cell and tag == self._cell_tag: + value = html.unescape("".join(self._current_cell_parts)) + self._current_cells.append(" ".join(value.split())) + self._in_cell = False + self._cell_tag = "" + self._current_cell_parts = [] + elif self._in_row and tag == "tr": + if self._row_has_header and self._current_cells: + self.headers = self._current_cells + elif self._row_has_data and self._current_cells: + self.rows.append(self._current_cells) + self._in_row = False + + def handle_data(self, data): + if self._in_cell: + self._current_cell_parts.append(data) + + +def normalize_header(value): + return FIELD_MAP.get(value.strip().lower(), "") + + +def parse_rows(spec): + parser = TableParser() + parser.feed(spec.path.read_text(encoding="utf-8", errors="ignore")) + fields = [normalize_header(header) for header in parser.headers] + rows = [] + for parsed_row in parser.rows: + record = {field: "" for field in OUTPUT_FIELDS} + record["source"] = "importinfo" + record["query"] = spec.query + record["source_url"] = spec.source_url + for index, field in enumerate(fields): + if field and index < len(parsed_row): + record[field] = parsed_row[index] + if is_relevant(record): + rows.append(record) + return rows + + +def is_relevant(record): + party_text = " ".join( + ( + record["shipper"], + record["consignee"], + record["notify_party"], + ) + ).upper() + commodity_text = record["commodity"].upper() + return any(term in party_text for term in PARTY_TERMS) and any( + term in commodity_text for term in PRODUCT_TERMS + ) + + +def row_identity(row): + if row["house_bol"]: + return row["house_bol"] + if row["master_bol"]: + return row["master_bol"] + return "\t".join( + ( + row["arrival_date"], + row["shipper"], + row["consignee"], + row["quantity"], + row["weight"], + row["commodity"], + ) + ) + + +def dedupe(rows): + seen = set() + output = [] + for row in rows: + identity = row_identity(row) + if identity in seen: + continue + seen.add(identity) + output.append(row) + return output + + +def write_tsv(rows, output): + output.parent.mkdir(parents=True, exist_ok=True) + with output.open("w", encoding="utf-8", newline="") as handle: + writer = csv.DictWriter(handle, fieldnames=OUTPUT_FIELDS, delimiter="\t") + writer.writeheader() + writer.writerows(rows) + + +def key_line(row): + bol = row["house_bol"] or row["master_bol"] + return "\t".join( + ( + row["arrival_date"], + row["consignee"], + row["shipper"], + row["commodity"], + row["quantity"], + row["weight"], + bol, + ) + ) + + +def write_key_lines(rows, output): + output.parent.mkdir(parents=True, exist_ok=True) + content = "\n".join(key_line(row) for row in rows) + if rows: + content += "\n" + output.write_text(content, encoding="utf-8") + + +def write_report(rows, output): + output.parent.mkdir(parents=True, exist_ok=True) + lines = [ + "# Customs Shipments", + "", + f"Relevant shipment count: {len(rows)}", + "", + "## Newest Relevant Shipments", + "", + ] + if rows: + lines.extend( + [ + "| Arrival Date | Consignee | Shipper | Commodity | Quantity | Weight | BOL |", + "| --- | --- | --- | --- | --- | --- | --- |", + ] + ) + for row in rows[:25]: + lines.append( + "| " + + " | ".join( + ( + row["arrival_date"], + row["consignee"], + row["shipper"], + row["commodity"], + row["quantity"], + row["weight"], + row["house_bol"] or row["master_bol"], + ) + ) + + " |" + ) + else: + lines.append("No relevant ImportInfo shipment rows found.") + lines.extend( + [ + "", + "## Limitations", + "", + "Customs data is corroborating evidence, not a standalone confirmation. " + "GAME CONSOLE descriptions are medium confidence without other identifiers.", + "", + ] + ) + output.write_text("\n".join(lines), encoding="utf-8") + + +def parse_input(value): + slug, separator, path = value.partition("=") + if not separator or not slug or not path: + raise argparse.ArgumentTypeError("--input values must use slug=path") + return InputSpec(slug, Path(path), SOURCE_URLS.get(slug, "")) + + +def parse_manifest(path): + specs = [] + with path.open(encoding="utf-8", newline="") as handle: + reader = csv.reader(handle, delimiter="\t") + for row in reader: + if not row or not any(cell.strip() for cell in row): + continue + if row[0].strip().lower() == "slug": + continue + if len(row) < 3: + continue + slug = row[0].strip() + source_url = row[1].strip() or SOURCE_URLS.get(slug, "") + html_path = row[2].strip() + if not slug or not html_path: + continue + specs.append(InputSpec(slug, Path(html_path), source_url)) + return specs + + +def main(): + parser = argparse.ArgumentParser() + parser.add_argument("--input", action="append", default=[], type=parse_input) + parser.add_argument("--manifest", action="append", default=[], type=Path) + parser.add_argument("--report-dir", required=True, type=Path) + args = parser.parse_args() + + specs = list(args.input) + for manifest in args.manifest: + if manifest.exists(): + specs.extend(parse_manifest(manifest)) + + rows = [] + for spec in specs: + if spec.path.exists(): + rows.extend(parse_rows(spec)) + + rows = sorted( + dedupe(rows), + key=lambda row: (row["arrival_date"], row["run_date"]), + reverse=True, + ) + + report_dir = args.report_dir + write_tsv(rows, report_dir / "customs-shipments.tsv") + write_key_lines(rows, report_dir / "customs-shipments-key-lines.txt") + write_report(rows, report_dir / "customs-shipments.md") + + +if __name__ == "__main__": + main() diff --git a/tests/test_customs_shipments.py b/tests/test_customs_shipments.py new file mode 100644 index 0000000..0d405bc --- /dev/null +++ b/tests/test_customs_shipments.py @@ -0,0 +1,161 @@ +import csv +import subprocess +import textwrap +import unittest +from pathlib import Path +from tempfile import TemporaryDirectory + + +ROOT = Path(__file__).resolve().parents[1] + + +IMPORTINFO_HTML = textwrap.dedent( + """ + + +

Search Results

+ + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Run DateMaster BOLHouse BOLVoyage #Bill TypeCarrier CodeIMO #Vessel NameArrival DateUS PortForeign PortQuantityWeightType of ServiceShipperConsigneeNotify PartyCommodity
2026-05-01EGLV142653125618SNHBSHALAX264014084EHouse BillSNHB9604081EVER LOGIC2026-05-01LOS ANGELES, CALIFORNIASHANGHAI CHINA (MAINLAND)42 PKG12,578 KHouse to HouseTECH-FRONT (CHONGQING) COMPUTER COCEVA C/O VALVE CORPORATIONCEVA C/O VALVE CORPORATIONGAME CONSOLE
2026-04-28CMDUCHN3170474EXDO6211281510XRAJHouse BillEXDO9436379CMA CGM SAMSON2026-04-28SAVANNAH, GEORGIASHANGHAI CHINA (MAINLAND)15 PKG4,143 KGPier to PierTECH-FRONT (CHONGQING) COMPUTER COPROMETHEAN INC.CHROMEBOX HTS:
+ + + """ +) + + +class CustomsShipmentParserTests(unittest.TestCase): + def test_extracts_relevant_importinfo_rows(self): + with TemporaryDirectory() as tmp: + tmp_path = Path(tmp) + html = tmp_path / "ceva-valve.html" + html.write_text(IMPORTINFO_HTML, encoding="utf-8") + reports = tmp_path / "reports" + + result = subprocess.run( + [ + "python3", + str(ROOT / "scripts" / "parse_importinfo_shipments.py"), + "--input", + f"ceva-valve={html}", + "--report-dir", + str(reports), + ], + cwd=ROOT, + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + check=False, + ) + + self.assertEqual("", result.stderr) + self.assertEqual(0, result.returncode) + + rows = list( + csv.DictReader( + (reports / "customs-shipments.tsv") + .read_text(encoding="utf-8") + .splitlines(), + delimiter="\t", + ) + ) + self.assertEqual(1, len(rows)) + self.assertEqual("SNHBSHALAX264014", rows[0]["house_bol"]) + self.assertEqual("GAME CONSOLE", rows[0]["commodity"]) + self.assertEqual("CEVA C/O VALVE CORPORATION", rows[0]["consignee"]) + + key_lines = (reports / "customs-shipments-key-lines.txt").read_text( + encoding="utf-8" + ) + self.assertIn("2026-05-01", key_lines) + self.assertIn("SNHBSHALAX264014", key_lines) + self.assertIn("GAME CONSOLE", key_lines) + + report = (reports / "customs-shipments.md").read_text(encoding="utf-8") + self.assertIn("## Newest Relevant Shipments", report) + self.assertIn("CEVA C/O VALVE CORPORATION", report) + self.assertIn("TECH-FRONT (CHONGQING) COMPUTER CO", report) + + def test_preserves_distinct_same_day_bols(self): + with TemporaryDirectory() as tmp: + tmp_path = Path(tmp) + html = tmp_path / "same-day.html" + html.write_text( + IMPORTINFO_HTML.replace( + "", + textwrap.dedent( + """ + + 2026-05-01EGLV142653125669SNHBSHALAX264015084E + House BillSNHB9604081EVER LOGIC + 2026-05-01LOS ANGELES, CALIFORNIASHANGHAI CHINA (MAINLAND) + 42 PKG12,596 KHouse to House + TECH-FRONT (CHONGQING) COMPUTER COCEVA C/O VALVE CORPORATION + CEVA C/O VALVE CORPORATIONGAME CONSOLE + + + """ + ), + ), + encoding="utf-8", + ) + reports = tmp_path / "reports" + + result = subprocess.run( + [ + "python3", + str(ROOT / "scripts" / "parse_importinfo_shipments.py"), + "--input", + f"ceva-valve={html}", + "--report-dir", + str(reports), + ], + cwd=ROOT, + text=True, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + check=False, + ) + + self.assertEqual(0, result.returncode) + rows = list( + csv.DictReader( + (reports / "customs-shipments.tsv") + .read_text(encoding="utf-8") + .splitlines(), + delimiter="\t", + ) + ) + self.assertEqual( + ["SNHBSHALAX264014", "SNHBSHALAX264015"], + [row["house_bol"] for row in rows], + ) + + +if __name__ == "__main__": + unittest.main()