Add ImportInfo customs shipment parser

This commit is contained in:
seanbetts committed 2026-05-05 09:03:32 +01:00
1 parent f294a131b7
commit ac79b88e20
2 files changed
+490

No files matched your search

+329
View File
@@ -0,0 +1,329 @@
#!/usr/bin/env python3
import argparse
import csv
import html
from dataclasses import dataclass
from html.parser import HTMLParser
from pathlib import Path
FIELD_MAP = {
"run date": "run_date",
"master bol": "master_bol",
"house bol": "house_bol",
"voyage #": "voyage",
"bill type": "bill_type",
"carrier code": "carrier_code",
"imo #": "imo",
"vessel name": "vessel_name",
"arrival date": "arrival_date",
"us port": "us_port",
"foreign port": "foreign_port",
"quantity": "quantity",
"weight": "weight",
"type of service": "type_of_service",
"shipper": "shipper",
"consignee": "consignee",
"notify party": "notify_party",
"commodity": "commodity",
}
OUTPUT_FIELDS = [
"source",
"query",
"run_date",
"master_bol",
"house_bol",
"voyage",
"bill_type",
"carrier_code",
"imo",
"vessel_name",
"arrival_date",
"us_port",
"foreign_port",
"quantity",
"weight",
"type_of_service",
"shipper",
"consignee",
"notify_party",
"commodity",
"source_url",
]
PARTY_TERMS = ("VALVE", "CEVA", "INGRAM MICRO", "TECH-FRONT", "CHENG UEI")
PRODUCT_TERMS = (
"GAME CONSOLE",
"VR CONTROLLER",
"CONTROLLER",
"STEAM",
"BASE STATION",
"HEADSET",
"DONGLE",
)
SOURCE_URLS = {
"ceva-valve": "https://www.importinfo.com/search?s=CEVA%20C%2FO%20VALVE%20CORPORATION",
"ingram-valve": "https://www.importinfo.com/search?s=INGRAM%20MICRO%20C%2FO%20VALVE%20CORPORATION",
"tech-front-game-console": "https://www.importinfo.com/search?s=TECH-FRONT%20GAME%20CONSOLE%20VALVE",
"valve-corporation-game-console": "https://www.importinfo.com/search?s=VALVE%20CORPORATION%20GAME%20CONSOLE",
}
@dataclass
class InputSpec:
query: str
path: Path
source_url: str
class TableParser(HTMLParser):
def __init__(self):
super().__init__()
self.headers = []
self.rows = []
self._in_row = False
self._in_cell = False
self._cell_tag = ""
self._current_cells = []
self._current_cell_parts = []
self._row_has_header = False
self._row_has_data = False
def handle_starttag(self, tag, attrs):
if tag == "tr":
self._in_row = True
self._current_cells = []
self._row_has_header = False
self._row_has_data = False
elif self._in_row and tag in ("th", "td"):
self._in_cell = True
self._cell_tag = tag
self._current_cell_parts = []
if tag == "th":
self._row_has_header = True
else:
self._row_has_data = True
def handle_endtag(self, tag):
if self._in_cell and tag == self._cell_tag:
value = html.unescape("".join(self._current_cell_parts))
self._current_cells.append(" ".join(value.split()))
self._in_cell = False
self._cell_tag = ""
self._current_cell_parts = []
elif self._in_row and tag == "tr":
if self._row_has_header and self._current_cells:
self.headers = self._current_cells
elif self._row_has_data and self._current_cells:
self.rows.append(self._current_cells)
self._in_row = False
def handle_data(self, data):
if self._in_cell:
self._current_cell_parts.append(data)
def normalize_header(value):
return FIELD_MAP.get(value.strip().lower(), "")
def parse_rows(spec):
parser = TableParser()
parser.feed(spec.path.read_text(encoding="utf-8", errors="ignore"))
fields = [normalize_header(header) for header in parser.headers]
rows = []
for parsed_row in parser.rows:
record = {field: "" for field in OUTPUT_FIELDS}
record["source"] = "importinfo"
record["query"] = spec.query
record["source_url"] = spec.source_url
for index, field in enumerate(fields):
if field and index < len(parsed_row):
record[field] = parsed_row[index]
if is_relevant(record):
rows.append(record)
return rows
def is_relevant(record):
party_text = " ".join(
(
record["shipper"],
record["consignee"],
record["notify_party"],
)
).upper()
commodity_text = record["commodity"].upper()
return any(term in party_text for term in PARTY_TERMS) and any(
term in commodity_text for term in PRODUCT_TERMS
)
def row_identity(row):
if row["house_bol"]:
return row["house_bol"]
if row["master_bol"]:
return row["master_bol"]
return "\t".join(
(
row["arrival_date"],
row["shipper"],
row["consignee"],
row["quantity"],
row["weight"],
row["commodity"],
)
)
def dedupe(rows):
seen = set()
output = []
for row in rows:
identity = row_identity(row)
if identity in seen:
continue
seen.add(identity)
output.append(row)
return output
def write_tsv(rows, output):
output.parent.mkdir(parents=True, exist_ok=True)
with output.open("w", encoding="utf-8", newline="") as handle:
writer = csv.DictWriter(handle, fieldnames=OUTPUT_FIELDS, delimiter="\t")
writer.writeheader()
writer.writerows(rows)
def key_line(row):
bol = row["house_bol"] or row["master_bol"]
return "\t".join(
(
row["arrival_date"],
row["consignee"],
row["shipper"],
row["commodity"],
row["quantity"],
row["weight"],
bol,
)
)
def write_key_lines(rows, output):
output.parent.mkdir(parents=True, exist_ok=True)
content = "\n".join(key_line(row) for row in rows)
if rows:
content += "\n"
output.write_text(content, encoding="utf-8")
def write_report(rows, output):
output.parent.mkdir(parents=True, exist_ok=True)
lines = [
"# Customs Shipments",
"",
f"Relevant shipment count: {len(rows)}",
"",
"## Newest Relevant Shipments",
"",
]
if rows:
lines.extend(
[
"| Arrival Date | Consignee | Shipper | Commodity | Quantity | Weight | BOL |",
"| --- | --- | --- | --- | --- | --- | --- |",
]
)
for row in rows[:25]:
lines.append(
"| "
+ " | ".join(
(
row["arrival_date"],
row["consignee"],
row["shipper"],
row["commodity"],
row["quantity"],
row["weight"],
row["house_bol"] or row["master_bol"],
)
)
+ " |"
)
else:
lines.append("No relevant ImportInfo shipment rows found.")
lines.extend(
[
"",
"## Limitations",
"",
"Customs data is corroborating evidence, not a standalone confirmation. "
"GAME CONSOLE descriptions are medium confidence without other identifiers.",
"",
]
)
output.write_text("\n".join(lines), encoding="utf-8")
def parse_input(value):
slug, separator, path = value.partition("=")
if not separator or not slug or not path:
raise argparse.ArgumentTypeError("--input values must use slug=path")
return InputSpec(slug, Path(path), SOURCE_URLS.get(slug, ""))
def parse_manifest(path):
specs = []
with path.open(encoding="utf-8", newline="") as handle:
reader = csv.reader(handle, delimiter="\t")
for row in reader:
if not row or not any(cell.strip() for cell in row):
continue
if row[0].strip().lower() == "slug":
continue
if len(row) < 3:
continue
slug = row[0].strip()
source_url = row[1].strip() or SOURCE_URLS.get(slug, "")
html_path = row[2].strip()
if not slug or not html_path:
continue
specs.append(InputSpec(slug, Path(html_path), source_url))
return specs
def main():
parser = argparse.ArgumentParser()
parser.add_argument("--input", action="append", default=[], type=parse_input)
parser.add_argument("--manifest", action="append", default=[], type=Path)
parser.add_argument("--report-dir", required=True, type=Path)
args = parser.parse_args()
specs = list(args.input)
for manifest in args.manifest:
if manifest.exists():
specs.extend(parse_manifest(manifest))
rows = []
for spec in specs:
if spec.path.exists():
rows.extend(parse_rows(spec))
rows = sorted(
dedupe(rows),
key=lambda row: (row["arrival_date"], row["run_date"]),
reverse=True,
)
report_dir = args.report_dir
write_tsv(rows, report_dir / "customs-shipments.tsv")
write_key_lines(rows, report_dir / "customs-shipments-key-lines.txt")
write_report(rows, report_dir / "customs-shipments.md")
if __name__ == "__main__":
main()
+161
View File
@@ -0,0 +1,161 @@
import csv
import subprocess
import textwrap
import unittest
from pathlib import Path
from tempfile import TemporaryDirectory
ROOT = Path(__file__).resolve().parents[1]
IMPORTINFO_HTML = textwrap.dedent(
"""
<html>
<body>
<h2>Search Results</h2>
<table>
<thead>
<tr>
<th>Run Date</th><th>Master BOL</th><th>House BOL</th><th>Voyage #</th>
<th>Bill Type</th><th>Carrier Code</th><th>IMO #</th><th>Vessel Name</th>
<th>Arrival Date</th><th>US Port</th><th>Foreign Port</th><th>Quantity</th>
<th>Weight</th><th>Type of Service</th><th>Shipper</th><th>Consignee</th>
<th>Notify Party</th><th>Commodity</th>
</tr>
</thead>
<tbody>
<tr>
<td>2026-05-01</td><td>EGLV142653125618</td><td>SNHBSHALAX264014</td><td>084E</td>
<td>House Bill</td><td>SNHB</td><td>9604081</td><td>EVER LOGIC</td>
<td>2026-05-01</td><td>LOS ANGELES, CALIFORNIA</td><td>SHANGHAI CHINA (MAINLAND)</td>
<td>42 PKG</td><td>12,578 K</td><td>House to House</td>
<td>TECH-FRONT (CHONGQING) COMPUTER CO</td><td>CEVA C/O VALVE CORPORATION</td>
<td>CEVA C/O VALVE CORPORATION</td><td>GAME CONSOLE</td>
</tr>
<tr>
<td>2026-04-28</td><td>CMDUCHN3170474</td><td>EXDO621128151</td><td>0XRAJ</td>
<td>House Bill</td><td>EXDO</td><td>9436379</td><td>CMA CGM SAMSON</td>
<td>2026-04-28</td><td>SAVANNAH, GEORGIA</td><td>SHANGHAI CHINA (MAINLAND)</td>
<td>15 PKG</td><td>4,143 KG</td><td>Pier to Pier</td>
<td>TECH-FRONT (CHONGQING) COMPUTER CO</td><td>PROMETHEAN INC.</td>
<td></td><td>CHROMEBOX HTS:</td>
</tr>
</tbody>
</table>
</body>
</html>
"""
)
class CustomsShipmentParserTests(unittest.TestCase):
def test_extracts_relevant_importinfo_rows(self):
with TemporaryDirectory() as tmp:
tmp_path = Path(tmp)
html = tmp_path / "ceva-valve.html"
html.write_text(IMPORTINFO_HTML, encoding="utf-8")
reports = tmp_path / "reports"
result = subprocess.run(
[
"python3",
str(ROOT / "scripts" / "parse_importinfo_shipments.py"),
"--input",
f"ceva-valve={html}",
"--report-dir",
str(reports),
],
cwd=ROOT,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
self.assertEqual("", result.stderr)
self.assertEqual(0, result.returncode)
rows = list(
csv.DictReader(
(reports / "customs-shipments.tsv")
.read_text(encoding="utf-8")
.splitlines(),
delimiter="\t",
)
)
self.assertEqual(1, len(rows))
self.assertEqual("SNHBSHALAX264014", rows[0]["house_bol"])
self.assertEqual("GAME CONSOLE", rows[0]["commodity"])
self.assertEqual("CEVA C/O VALVE CORPORATION", rows[0]["consignee"])
key_lines = (reports / "customs-shipments-key-lines.txt").read_text(
encoding="utf-8"
)
self.assertIn("2026-05-01", key_lines)
self.assertIn("SNHBSHALAX264014", key_lines)
self.assertIn("GAME CONSOLE", key_lines)
report = (reports / "customs-shipments.md").read_text(encoding="utf-8")
self.assertIn("## Newest Relevant Shipments", report)
self.assertIn("CEVA C/O VALVE CORPORATION", report)
self.assertIn("TECH-FRONT (CHONGQING) COMPUTER CO", report)
def test_preserves_distinct_same_day_bols(self):
with TemporaryDirectory() as tmp:
tmp_path = Path(tmp)
html = tmp_path / "same-day.html"
html.write_text(
IMPORTINFO_HTML.replace(
"</tbody>",
textwrap.dedent(
"""
<tr>
<td>2026-05-01</td><td>EGLV142653125669</td><td>SNHBSHALAX264015</td><td>084E</td>
<td>House Bill</td><td>SNHB</td><td>9604081</td><td>EVER LOGIC</td>
<td>2026-05-01</td><td>LOS ANGELES, CALIFORNIA</td><td>SHANGHAI CHINA (MAINLAND)</td>
<td>42 PKG</td><td>12,596 K</td><td>House to House</td>
<td>TECH-FRONT (CHONGQING) COMPUTER CO</td><td>CEVA C/O VALVE CORPORATION</td>
<td>CEVA C/O VALVE CORPORATION</td><td>GAME CONSOLE</td>
</tr>
</tbody>
"""
),
),
encoding="utf-8",
)
reports = tmp_path / "reports"
result = subprocess.run(
[
"python3",
str(ROOT / "scripts" / "parse_importinfo_shipments.py"),
"--input",
f"ceva-valve={html}",
"--report-dir",
str(reports),
],
cwd=ROOT,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
self.assertEqual(0, result.returncode)
rows = list(
csv.DictReader(
(reports / "customs-shipments.tsv")
.read_text(encoding="utf-8")
.splitlines(),
delimiter="\t",
)
)
self.assertEqual(
["SNHBSHALAX264014", "SNHBSHALAX264015"],
[row["house_bol"] for row in rows],
)
if __name__ == "__main__":
unittest.main()