Merge branch 'codex/customs-shipment-monitor'

This commit is contained in:
seanbetts committed 2026-05-05 09:43:15 +01:00
commit 92c0eb3606
8 files changed
+948 -7

No files matched your search

+13 -6
View File
@@ -52,6 +52,7 @@ Use the source helpers when available:
- [scripts/check_steamvr_depots.sh](scripts/check_steamvr_depots.sh)
- [scripts/check_steamos_mirror.sh](scripts/check_steamos_mirror.sh)
- [scripts/check_valve_endpoints.sh](scripts/check_valve_endpoints.sh)
- [scripts/check_customs_shipments.sh](scripts/check_customs_shipments.sh)
- [scripts/compare_runs.py](scripts/compare_runs.py)
- [scripts/save_visual_assets.py](scripts/save_visual_assets.py)
- [scripts/draft_status_update.py](scripts/draft_status_update.py)
@@ -207,14 +208,20 @@ Do not treat generic Steam Deck site media as relevant unless it is directly tie
### 7. Customs / Regulatory
Run when:
Use [scripts/check_customs_shipments.sh](scripts/check_customs_shipments.sh) for shipment-level customs checks.
- Komodo gets fresh media or sections
- SteamDB gets fresh app or package movement
- SteamTracking gets materially new rollout strings
- or on a slower periodic cadence
Run customs checks every normal watch run because ImportInfo is quick and high-signal. Use customs data for confirmation, not as the primary source of truth.
Use for confirmation, not as the primary source of truth.
Primary automated customs source:
- ImportInfo search pages for `CEVA C/O VALVE CORPORATION`, `INGRAM MICRO C/O VALVE CORPORATION`, `TECH-FRONT GAME CONSOLE VALVE`, and `VALVE CORPORATION GAME CONSOLE`
Corroborating/manual sources:
- NBD Valve and Ingram/Valve trader pages
- ImportGenius public previews for CEVA/Valve, Ingram/Valve, and Tech-Front
Do not use HMRC UK Trade Info for launch monitoring. It is lagged monthly trader/commodity presence, not shipment-level evidence.
## What Counts As Meaningful
+8 -1
View File
@@ -138,12 +138,19 @@ Check for newly exposed assets or support flows:
## Customs / Regulatory
Use as confirmation when primary signals move:
Run ImportInfo customs checks every normal watch run because they are quick and high-signal. Use customs data as corroborating evidence, not as primary proof.
- customs and import records
- ImportInfo automated search for CEVA/Valve: `https://www.importinfo.com/search?s=CEVA%20C%2FO%20VALVE%20CORPORATION`
- ImportInfo automated search for Ingram/Valve: `https://www.importinfo.com/search?s=INGRAM%20MICRO%20C%2FO%20VALVE%20CORPORATION`
- ImportInfo automated search for Tech-Front game console Valve: `https://www.importinfo.com/search?s=TECH-FRONT%20GAME%20CONSOLE%20VALVE`
- ImportInfo automated search for Valve Corporation game console: `https://www.importinfo.com/search?s=VALVE%20CORPORATION%20GAME%20CONSOLE`
- ImportInfo manual/corroborating Tech-Front supplier page: `https://www.importinfo.com/tech-front-chongqing-computer-co`
- ImportInfo manual/corroborating Valve Corporation page: `https://www.importinfo.com/valve-corporation`
- NBD Valve Corporation public company page: `https://en.nbd.ltd/trader/info/NBDD3Y527621220`
- NBD buyer search for Valve Corporation: `https://en.nbd.ltd/customs-data?t=2&v=Valve%20Corporation`
- ImportGenius Ingram Micro C/O Valve Corporation page: `https://www.importgenius.cn/importers/ingram-micro-c-o-valve-corporation`
- HMRC UK Trade Info: rejected for this workflow because it is lagged monthly aggregate/trader data rather than shipment-level data.
- FCC
- Bluetooth SIG
- Wi-Fi Alliance
+87
View File
@@ -0,0 +1,87 @@
#!/bin/sh
set -eu
if [ "$#" -lt 1 ]; then
echo "usage: $0 RUN_DIR" >&2
exit 1
fi
RUN_DIR=$1
OUT_DIR=$RUN_DIR/api/customs
REPORT_DIR=$RUN_DIR/reports
SCRIPT_DIR=$(CDPATH= cd "$(dirname "$0")" && pwd)
mkdir -p "$OUT_DIR" "$REPORT_DIR"
ERROR_FILE=$REPORT_DIR/customs-shipments-errors.txt
: > "$ERROR_FILE"
IMPORTINFO_USER_AGENT=${IMPORTINFO_USER_AGENT:-"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/124.0 Safari/537.36"}
MANIFEST=$OUT_DIR/importinfo-inputs.tsv
: > "$MANIFEST"
looks_like_importinfo_shipment_table() {
path=$1
for header in "Master BOL" "House BOL" "Arrival Date" "Commodity"; do
if ! grep -qi "$header" "$path"; then
return 1
fi
done
return 0
}
fetch_importinfo() {
slug=$1
query=$2
url=$3
tmp=$OUT_DIR/$slug.html.tmp
out=$OUT_DIR/$slug.html
if curl \
-A "$IMPORTINFO_USER_AGENT" \
-H "Accept: text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8" \
-H "Accept-Language: en-US,en;q=0.9" \
--retry 2 \
--retry-delay 2 \
--max-time 45 \
-fsSL "$url" > "$tmp" 2>/dev/null
then
if ! looks_like_importinfo_shipment_table "$tmp"; then
rm -f "$tmp" "$out"
printf 'unexpected or blocked content from %s\n' "$url" >> "$ERROR_FILE"
return 0
fi
mv "$tmp" "$out"
printf '%s\t%s\t%s\n' "$query" "$url" "$out" >> "$MANIFEST"
return 0
fi
rm -f "$tmp" "$out"
printf 'failed to fetch %s\n' "$url" >> "$ERROR_FILE"
return 1
}
fetch_importinfo \
"importinfo-ceva-valve" \
"ceva-valve" \
"https://www.importinfo.com/search?s=CEVA%20C%2FO%20VALVE%20CORPORATION" || true
fetch_importinfo \
"importinfo-ingram-valve" \
"ingram-valve" \
"https://www.importinfo.com/search?s=INGRAM%20MICRO%20C%2FO%20VALVE%20CORPORATION" || true
fetch_importinfo \
"importinfo-tech-front-game-console" \
"tech-front-game-console" \
"https://www.importinfo.com/search?s=TECH-FRONT%20GAME%20CONSOLE%20VALVE" || true
fetch_importinfo \
"importinfo-valve-corporation" \
"valve-corporation-game-console" \
"https://www.importinfo.com/search?s=VALVE%20CORPORATION%20GAME%20CONSOLE" || true
python3 "$SCRIPT_DIR/parse_importinfo_shipments.py" \
--manifest "$MANIFEST" \
--report-dir "$REPORT_DIR"
echo "Saved customs shipment pages to $OUT_DIR"
echo "Saved customs shipment report to $REPORT_DIR/customs-shipments.md"
+20
View File
@@ -51,6 +51,14 @@ def build_draft(run_dir: Path):
tracking_lines = first_n_nonempty(reports / "steamtracking-pairing-focus.txt", 8)
steamvr_lines = first_n_nonempty(reports / "steamvr-depots-key-lines.txt", 8)
steamos_lines = first_n_nonempty(reports / "steamos-mirror-key-lines.txt", 8)
customs_key_lines_path = reports / "customs-shipments-key-lines.txt"
customs_errors_path = reports / "customs-shipments-errors.txt"
customs_report_path = reports / "customs-shipments.md"
customs_lines = first_n_nonempty(customs_key_lines_path, 8)
customs_outputs_available = any(
path.exists()
for path in (customs_key_lines_path, customs_errors_path, customs_report_path)
)
discovered_count = count_lines(reports / "discovered-visual-assets.tsv")
retrieved_count = count_lines(reports / "retrieved-visual-assets.tsv")
blocked_count = count_lines(reports / "blocked-visual-assets.tsv")
@@ -60,6 +68,7 @@ def build_draft(run_dir: Path):
steamvr_blocked = has_nonempty(reports / "steamvr-depots-errors.txt")
steamos_blocked = has_nonempty(reports / "steamos-mirror-errors.txt")
valve_blocked = has_nonempty(reports / "valve-errors.txt")
customs_blocked = has_nonempty(customs_errors_path)
lines = [
f"# Status Draft: {run_date}",
@@ -133,6 +142,16 @@ def build_draft(run_dir: Path):
else:
lines.append("- No Valve key-line report found.")
lines.extend(["", "#### Customs / Shipments"])
if not customs_outputs_available:
lines.append("- Customs shipment outputs were not generated or are unavailable for this run.")
elif customs_blocked:
lines.append("- Customs shipment fetches failed or were partially blocked in this run. See `customs-shipments-errors.txt`.")
if customs_lines:
lines.extend([f"- `{line}`" for line in customs_lines])
elif customs_outputs_available and not customs_blocked:
lines.append("- No relevant customs shipment rows found.")
lines.extend(
[
"",
@@ -149,6 +168,7 @@ def build_draft(run_dir: Path):
f"- Retrieved visual assets: `{reports / 'retrieved-visual-assets.tsv'}`",
f"- Blocked visual assets: `{reports / 'blocked-visual-assets.tsv'}`",
f"- Manual retry URLs: `{reports / 'manual-asset-urls.txt'}`",
f"- Customs shipments: `{reports / 'customs-shipments.md'}`",
]
)
+352
View File
@@ -0,0 +1,352 @@
#!/usr/bin/env python3
import argparse
import csv
import html
from dataclasses import dataclass
from html.parser import HTMLParser
from pathlib import Path
FIELD_MAP = {
"run date": "run_date",
"master bol": "master_bol",
"house bol": "house_bol",
"voyage #": "voyage",
"bill type": "bill_type",
"carrier code": "carrier_code",
"imo #": "imo",
"vessel name": "vessel_name",
"arrival date": "arrival_date",
"us port": "us_port",
"foreign port": "foreign_port",
"quantity": "quantity",
"weight": "weight",
"type of service": "type_of_service",
"shipper": "shipper",
"consignee": "consignee",
"notify party": "notify_party",
"commodity": "commodity",
}
OUTPUT_FIELDS = [
"source",
"query",
"run_date",
"master_bol",
"house_bol",
"voyage",
"bill_type",
"carrier_code",
"imo",
"vessel_name",
"arrival_date",
"us_port",
"foreign_port",
"quantity",
"weight",
"type_of_service",
"shipper",
"consignee",
"notify_party",
"commodity",
"source_url",
]
PARTY_TERMS = ("VALVE", "CEVA", "INGRAM MICRO", "TECH-FRONT", "CHENG UEI")
PRODUCT_TERMS = (
"GAME CONSOLE",
"VR CONTROLLER",
"CONTROLLER",
"STEAM",
"BASE STATION",
"HEADSET",
"DONGLE",
)
SOURCE_URLS = {
"ceva-valve": "https://www.importinfo.com/search?s=CEVA%20C%2FO%20VALVE%20CORPORATION",
"ingram-valve": "https://www.importinfo.com/search?s=INGRAM%20MICRO%20C%2FO%20VALVE%20CORPORATION",
"tech-front-game-console": "https://www.importinfo.com/search?s=TECH-FRONT%20GAME%20CONSOLE%20VALVE",
"valve-corporation-game-console": "https://www.importinfo.com/search?s=VALVE%20CORPORATION%20GAME%20CONSOLE",
}
@dataclass
class InputSpec:
query: str
path: Path
source_url: str
class TableParser(HTMLParser):
def __init__(self):
super().__init__()
self.rows = []
self._current_headers = []
self._in_row = False
self._in_cell = False
self._cell_tag = ""
self._current_cells = []
self._current_cell_parts = []
self._row_has_header = False
self._row_has_data = False
def handle_starttag(self, tag, attrs):
if tag == "tr":
self._in_row = True
self._current_cells = []
self._row_has_header = False
self._row_has_data = False
elif self._in_row and tag in ("th", "td"):
self._in_cell = True
self._cell_tag = tag
self._current_cell_parts = []
if tag == "th":
self._row_has_header = True
else:
self._row_has_data = True
elif self._in_cell and tag == "br":
self._current_cell_parts.append(" ")
def handle_endtag(self, tag):
if self._in_cell and tag == self._cell_tag:
value = html.unescape(" ".join(self._current_cell_parts))
self._current_cells.append(" ".join(value.split()))
self._in_cell = False
self._cell_tag = ""
self._current_cell_parts = []
elif self._in_row and tag == "tr":
if self._row_has_header and self._current_cells:
self._current_headers = self._current_cells
elif self._row_has_data and self._current_cells:
self.rows.append((list(self._current_headers), self._current_cells))
self._in_row = False
def handle_data(self, data):
if self._in_cell:
self._current_cell_parts.append(data)
def normalize_header(value):
return FIELD_MAP.get(value.strip().lower(), "")
def parse_rows(spec):
parser = TableParser()
parser.feed(spec.path.read_text(encoding="utf-8", errors="ignore"))
rows = []
for headers, parsed_row in parser.rows:
fields = [normalize_header(header) for header in headers]
record = {field: "" for field in OUTPUT_FIELDS}
record["source"] = "importinfo"
record["query"] = spec.query
record["source_url"] = spec.source_url
for index, field in enumerate(fields):
if field and index < len(parsed_row):
record[field] = parsed_row[index]
if is_relevant(record):
rows.append(record)
return rows
def is_relevant(record):
party_text = " ".join(
(
record["shipper"],
record["consignee"],
record["notify_party"],
)
).upper()
record_text = " ".join(
(
record["query"],
record["run_date"],
record["master_bol"],
record["house_bol"],
record["voyage"],
record["bill_type"],
record["carrier_code"],
record["imo"],
record["vessel_name"],
record["arrival_date"],
record["us_port"],
record["foreign_port"],
record["quantity"],
record["weight"],
record["type_of_service"],
record["commodity"],
)
).upper()
commodity_text = record["commodity"].upper()
has_valve_signal = "VALVE" in party_text or "VALVE CORPORATION" in record_text
return has_valve_signal and any(term in party_text for term in PARTY_TERMS) and any(
term in commodity_text for term in PRODUCT_TERMS
)
def row_identity(row):
if row["house_bol"]:
return row["house_bol"]
if row["master_bol"]:
return row["master_bol"]
return "\t".join(
(
row["arrival_date"],
row["shipper"],
row["consignee"],
row["quantity"],
row["weight"],
row["commodity"],
)
)
def dedupe(rows):
seen = set()
output = []
for row in rows:
identity = row_identity(row)
if identity in seen:
continue
seen.add(identity)
output.append(row)
return output
def write_tsv(rows, output):
output.parent.mkdir(parents=True, exist_ok=True)
with output.open("w", encoding="utf-8", newline="") as handle:
writer = csv.DictWriter(handle, fieldnames=OUTPUT_FIELDS, delimiter="\t")
writer.writeheader()
writer.writerows(rows)
def key_line(row):
bol = row["house_bol"] or row["master_bol"]
return "\t".join(
(
row["arrival_date"],
row["consignee"],
row["shipper"],
row["commodity"],
row["quantity"],
row["weight"],
bol,
)
)
def write_key_lines(rows, output):
output.parent.mkdir(parents=True, exist_ok=True)
content = "\n".join(key_line(row) for row in rows)
if rows:
content += "\n"
output.write_text(content, encoding="utf-8")
def write_report(rows, output):
output.parent.mkdir(parents=True, exist_ok=True)
lines = [
"# Customs Shipments",
"",
f"Relevant shipment count: {len(rows)}",
"",
"## Newest Relevant Shipments",
"",
]
if rows:
lines.extend(
[
"| Arrival Date | Consignee | Shipper | Commodity | Quantity | Weight | BOL |",
"| --- | --- | --- | --- | --- | --- | --- |",
]
)
for row in rows[:25]:
lines.append(
"| "
+ " | ".join(
(
row["arrival_date"],
row["consignee"],
row["shipper"],
row["commodity"],
row["quantity"],
row["weight"],
row["house_bol"] or row["master_bol"],
)
)
+ " |"
)
else:
lines.append("No relevant ImportInfo shipment rows found.")
lines.extend(
[
"",
"## Limitations",
"",
"Customs data is corroborating evidence, not a standalone confirmation. "
"GAME CONSOLE descriptions are medium confidence without other identifiers.",
"",
]
)
output.write_text("\n".join(lines), encoding="utf-8")
def parse_input(value):
slug, separator, path = value.partition("=")
if not separator or not slug or not path:
raise argparse.ArgumentTypeError("--input values must use slug=path")
return InputSpec(slug, Path(path), SOURCE_URLS.get(slug, ""))
def parse_manifest(path):
specs = []
with path.open(encoding="utf-8", newline="") as handle:
reader = csv.reader(handle, delimiter="\t")
for row in reader:
if not row or not any(cell.strip() for cell in row):
continue
if row[0].strip().lower() == "slug":
continue
if len(row) < 3:
continue
slug = row[0].strip()
source_url = row[1].strip() or SOURCE_URLS.get(slug, "")
html_path = row[2].strip()
if not slug or not html_path:
continue
specs.append(InputSpec(slug, Path(html_path), source_url))
return specs
def main():
parser = argparse.ArgumentParser()
parser.add_argument("--input", action="append", default=[], type=parse_input)
parser.add_argument("--manifest", action="append", default=[], type=Path)
parser.add_argument("--report-dir", required=True, type=Path)
args = parser.parse_args()
specs = list(args.input)
for manifest in args.manifest:
if manifest.exists():
specs.extend(parse_manifest(manifest))
rows = []
for spec in specs:
if spec.path.exists():
rows.extend(parse_rows(spec))
rows = sorted(
dedupe(rows),
key=lambda row: (row["arrival_date"], row["run_date"]),
reverse=True,
)
report_dir = args.report_dir
write_tsv(rows, report_dir / "customs-shipments.tsv")
write_key_lines(rows, report_dir / "customs-shipments-key-lines.txt")
write_report(rows, report_dir / "customs-shipments.md")
if __name__ == "__main__":
main()
+1
View File
@@ -49,6 +49,7 @@ printf '%s\n' "Run note: $RUN_NOTE"
"$SCRIPT_DIR/check_steamvr_depots.sh" "$RUN_DIR"
"$SCRIPT_DIR/check_steamos_mirror.sh" "$RUN_DIR"
"$SCRIPT_DIR/check_valve_endpoints.sh" "$RUN_DIR"
"$SCRIPT_DIR/check_customs_shipments.sh" "$RUN_DIR"
python3 "$SCRIPT_DIR/save_visual_assets.py" --run-dir "$RUN_DIR" --base-dir "$BASE_DIR"
PREVIOUS_DIR=""
+14
View File
@@ -90,6 +90,17 @@ def count_block_reason(path: Path, reason: str):
return total
def customs_status(reports: Path):
key_lines = reports / "customs-shipments-key-lines.txt"
errors = reports / "customs-shipments-errors.txt"
report = reports / "customs-shipments.md"
if not any(path.exists() for path in (key_lines, errors, report)):
return "unavailable"
if count(errors):
return "blocked"
return "available"
def build(run_dir: Path):
reports = run_dir / "reports"
compare = sorted(reports.glob("compare-vs-*.md"))
@@ -112,6 +123,7 @@ def build(run_dir: Path):
f"- SteamVR depot metadata blocked: `{'yes' if count(reports / 'steamvr-depots-errors.txt') else 'no'}`",
f"- SteamOS mirror metadata blocked: `{'yes' if count(reports / 'steamos-mirror-errors.txt') else 'no'}`",
f"- Valve blocked: `{'yes' if count(reports / 'valve-errors.txt') else 'no'}`",
f"- Customs shipments status: `{customs_status(reports)}`",
f"- Discovered visual assets: `{discovered}`",
f"- Retrieved visual assets: `{retrieved}`",
f"- Blocked visual assets: `{blocked}`",
@@ -124,6 +136,7 @@ def build(run_dir: Path):
f"- Retrieved assets: `{reports / 'retrieved-visual-assets.tsv'}`",
f"- Blocked assets: `{reports / 'blocked-visual-assets.tsv'}`",
f"- Manual retry URLs: `{reports / 'manual-asset-urls.txt'}`",
f"- Customs shipments: `{reports / 'customs-shipments.md'}`",
]
if compare_file:
@@ -137,6 +150,7 @@ def build(run_dir: Path):
notable.extend([f"- SteamVR depots: `{line}`" for line in filtered_first(reports / "steamvr-depots-key-lines.txt", 5)])
notable.extend([f"- SteamOS mirror: `{line}`" for line in filtered_first(reports / "steamos-mirror-key-lines.txt", 5)])
notable.extend([f"- Valve: `{line}`" for line in filtered_first(reports / "valve-key-lines.txt", 5)])
notable.extend([f"- Customs shipments: `{line}`" for line in filtered_first(reports / "customs-shipments-key-lines.txt", 8)])
notable.extend([f"- Asset blocked: `{line}`" for line in filtered_first(reports / "blocked-visual-assets.tsv", 5)])
if notable:
+453
View File
@@ -0,0 +1,453 @@
import csv
import os
import stat
import subprocess
import textwrap
import unittest
from pathlib import Path
from tempfile import TemporaryDirectory
ROOT = Path(__file__).resolve().parents[1]
IMPORTINFO_HTML = textwrap.dedent(
"""
<html>
<body>
<h2>Search Results</h2>
<table>
<thead>
<tr>
<th>Run Date</th><th>Master BOL</th><th>House BOL</th><th>Voyage #</th>
<th>Bill Type</th><th>Carrier Code</th><th>IMO #</th><th>Vessel Name</th>
<th>Arrival Date</th><th>US Port</th><th>Foreign Port</th><th>Quantity</th>
<th>Weight</th><th>Type of Service</th><th>Shipper</th><th>Consignee</th>
<th>Notify Party</th><th>Commodity</th>
</tr>
</thead>
<tbody>
<tr>
<td>2026-05-01</td><td>EGLV142653125618</td><td>SNHBSHALAX264014</td><td>084E</td>
<td>House Bill</td><td>SNHB</td><td>9604081</td><td>EVER LOGIC</td>
<td>2026-05-01</td><td>LOS ANGELES, CALIFORNIA</td><td>SHANGHAI CHINA (MAINLAND)</td>
<td>42 PKG</td><td>12,578 K</td><td>House to House</td>
<td>TECH-FRONT (CHONGQING) COMPUTER CO</td><td>CEVA C/O VALVE CORPORATION</td>
<td>CEVA C/O VALVE CORPORATION</td><td>GAME CONSOLE</td>
</tr>
<tr>
<td>2026-04-28</td><td>CMDUCHN3170474</td><td>EXDO621128151</td><td>0XRAJ</td>
<td>House Bill</td><td>EXDO</td><td>9436379</td><td>CMA CGM SAMSON</td>
<td>2026-04-28</td><td>SAVANNAH, GEORGIA</td><td>SHANGHAI CHINA (MAINLAND)</td>
<td>15 PKG</td><td>4,143 KG</td><td>Pier to Pier</td>
<td>TECH-FRONT (CHONGQING) COMPUTER CO</td><td>PROMETHEAN INC.</td>
<td></td><td>CHROMEBOX HTS:</td>
</tr>
</tbody>
</table>
</body>
</html>
"""
)
def run_parser(args, reports):
return subprocess.run(
[
"python3",
str(ROOT / "scripts" / "parse_importinfo_shipments.py"),
*args,
"--report-dir",
str(reports),
],
cwd=ROOT,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
def read_rows(reports):
return list(
csv.DictReader(
(reports / "customs-shipments.tsv").read_text(encoding="utf-8").splitlines(),
delimiter="\t",
)
)
def write_executable(path, content):
path.write_text(textwrap.dedent(content).lstrip(), encoding="utf-8")
path.chmod(path.stat().st_mode | stat.S_IXUSR)
class CustomsShipmentParserTests(unittest.TestCase):
def test_extracts_relevant_importinfo_rows(self):
with TemporaryDirectory() as tmp:
tmp_path = Path(tmp)
html = tmp_path / "ceva-valve.html"
html.write_text(IMPORTINFO_HTML, encoding="utf-8")
reports = tmp_path / "reports"
result = run_parser(["--input", f"ceva-valve={html}"], reports)
self.assertEqual("", result.stderr)
self.assertEqual(0, result.returncode)
rows = read_rows(reports)
self.assertEqual(1, len(rows))
self.assertEqual("SNHBSHALAX264014", rows[0]["house_bol"])
self.assertEqual("GAME CONSOLE", rows[0]["commodity"])
self.assertEqual("CEVA C/O VALVE CORPORATION", rows[0]["consignee"])
key_lines = (reports / "customs-shipments-key-lines.txt").read_text(
encoding="utf-8"
)
self.assertIn("2026-05-01", key_lines)
self.assertIn("SNHBSHALAX264014", key_lines)
self.assertIn("GAME CONSOLE", key_lines)
report = (reports / "customs-shipments.md").read_text(encoding="utf-8")
self.assertIn("## Newest Relevant Shipments", report)
self.assertIn("CEVA C/O VALVE CORPORATION", report)
self.assertIn("TECH-FRONT (CHONGQING) COMPUTER CO", report)
def test_preserves_distinct_same_day_bols(self):
with TemporaryDirectory() as tmp:
tmp_path = Path(tmp)
html = tmp_path / "same-day.html"
html.write_text(
IMPORTINFO_HTML.replace(
"</tbody>",
textwrap.dedent(
"""
<tr>
<td>2026-05-01</td><td>EGLV142653125669</td><td>SNHBSHALAX264015</td><td>084E</td>
<td>House Bill</td><td>SNHB</td><td>9604081</td><td>EVER LOGIC</td>
<td>2026-05-01</td><td>LOS ANGELES, CALIFORNIA</td><td>SHANGHAI CHINA (MAINLAND)</td>
<td>42 PKG</td><td>12,596 K</td><td>House to House</td>
<td>TECH-FRONT (CHONGQING) COMPUTER CO</td><td>CEVA C/O VALVE CORPORATION</td>
<td>CEVA C/O VALVE CORPORATION</td><td>GAME CONSOLE</td>
</tr>
</tbody>
"""
),
),
encoding="utf-8",
)
reports = tmp_path / "reports"
result = run_parser(["--input", f"ceva-valve={html}"], reports)
self.assertEqual(0, result.returncode)
rows = read_rows(reports)
self.assertEqual(
["SNHBSHALAX264014", "SNHBSHALAX264015"],
[row["house_bol"] for row in rows],
)
def test_ignores_later_unrelated_table_headers(self):
with TemporaryDirectory() as tmp:
tmp_path = Path(tmp)
html = tmp_path / "extra-table.html"
html.write_text(
IMPORTINFO_HTML.replace(
"</body>",
textwrap.dedent(
"""
<table>
<tr><th>Name</th><th>Description</th></tr>
<tr><td>PROMETHEAN INC.</td><td>CHROMEBOX HTS:</td></tr>
</table>
</body>
"""
),
),
encoding="utf-8",
)
reports = tmp_path / "reports"
result = run_parser(["--input", f"ceva-valve={html}"], reports)
self.assertEqual("", result.stderr)
self.assertEqual(0, result.returncode)
rows = read_rows(reports)
self.assertEqual(["SNHBSHALAX264014"], [row["house_bol"] for row in rows])
def test_matches_relevance_terms_split_by_inline_markup(self):
with TemporaryDirectory() as tmp:
tmp_path = Path(tmp)
html = tmp_path / "split-cells.html"
html.write_text(
IMPORTINFO_HTML.replace(
"CEVA C/O VALVE CORPORATION</td><td>GAME CONSOLE",
"CEVA C/O <span>VALVE</span> CORPORATION</td><td>GAME<br>CONSOLE",
),
encoding="utf-8",
)
reports = tmp_path / "reports"
result = run_parser(["--input", f"ceva-valve={html}"], reports)
self.assertEqual("", result.stderr)
self.assertEqual(0, result.returncode)
rows = read_rows(reports)
self.assertEqual(1, len(rows))
self.assertEqual("CEVA C/O VALVE CORPORATION", rows[0]["consignee"])
self.assertEqual("GAME CONSOLE", rows[0]["commodity"])
def test_filters_generic_logistics_product_rows_without_valve_signal(self):
with TemporaryDirectory() as tmp:
tmp_path = Path(tmp)
html = tmp_path / "generic-logistics.html"
html.write_text(
IMPORTINFO_HTML.replace("CEVA C/O VALVE CORPORATION", "CEVA LOGISTICS US INC."),
encoding="utf-8",
)
reports = tmp_path / "reports"
result = run_parser(["--input", f"tech-front-game-console={html}"], reports)
self.assertEqual("", result.stderr)
self.assertEqual(0, result.returncode)
self.assertEqual([], read_rows(reports))
def test_manifest_inputs_dedupe_duplicate_bols(self):
with TemporaryDirectory() as tmp:
tmp_path = Path(tmp)
first = tmp_path / "first.html"
second = tmp_path / "second.html"
manifest = tmp_path / "manifest.tsv"
first.write_text(IMPORTINFO_HTML, encoding="utf-8")
second.write_text(IMPORTINFO_HTML, encoding="utf-8")
manifest.write_text(
"\n".join(
(
"slug\tsource_url\thtml_path",
f"ceva-valve\thttps://example.test/ceva\t{first}",
f"ceva-valve\thttps://example.test/ceva\t{second}",
)
)
+ "\n",
encoding="utf-8",
)
reports = tmp_path / "reports"
result = run_parser(["--manifest", str(manifest)], reports)
self.assertEqual("", result.stderr)
self.assertEqual(0, result.returncode)
rows = read_rows(reports)
self.assertEqual(["SNHBSHALAX264014"], [row["house_bol"] for row in rows])
self.assertEqual("https://example.test/ceva", rows[0]["source_url"])
class CheckCustomsShipmentsShellTests(unittest.TestCase):
def test_fetches_pages_and_runs_parser(self):
with TemporaryDirectory() as tmp:
tmp_path = Path(tmp)
run_dir = tmp_path / "run"
fake_bin = tmp_path / "bin"
fake_bin.mkdir()
write_executable(
fake_bin / "curl",
f"""
#!/bin/sh
cat <<'HTML'
{IMPORTINFO_HTML}
HTML
""",
)
env = os.environ.copy()
env["PATH"] = f"{fake_bin}{os.pathsep}{env['PATH']}"
result = subprocess.run(
[str(ROOT / "scripts" / "check_customs_shipments.sh"), str(run_dir)],
cwd=ROOT,
env=env,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
self.assertEqual("", result.stderr)
self.assertEqual(0, result.returncode)
self.assertTrue(
(run_dir / "api" / "customs" / "importinfo-ceva-valve.html").exists()
)
key_lines = (
run_dir / "reports" / "customs-shipments-key-lines.txt"
).read_text(encoding="utf-8")
self.assertIn("SNHBSHALAX264014", key_lines)
def test_records_blocked_fetches_without_failing_run(self):
with TemporaryDirectory() as tmp:
tmp_path = Path(tmp)
run_dir = tmp_path / "run"
fake_bin = tmp_path / "bin"
fake_bin.mkdir()
write_executable(
fake_bin / "curl",
"""
#!/bin/sh
exit 22
""",
)
env = os.environ.copy()
env["PATH"] = f"{fake_bin}{os.pathsep}{env['PATH']}"
result = subprocess.run(
[str(ROOT / "scripts" / "check_customs_shipments.sh"), str(run_dir)],
cwd=ROOT,
env=env,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
self.assertEqual("", result.stderr)
self.assertEqual(0, result.returncode)
errors = (run_dir / "reports" / "customs-shipments-errors.txt").read_text(
encoding="utf-8"
)
self.assertIn("failed to fetch", errors)
def test_records_200_challenge_page_without_key_lines(self):
with TemporaryDirectory() as tmp:
tmp_path = Path(tmp)
run_dir = tmp_path / "run"
fake_bin = tmp_path / "bin"
fake_bin.mkdir()
write_executable(
fake_bin / "curl",
"""
#!/bin/sh
cat <<'HTML'
<html><body><h1>Checking your browser</h1><p>Please verify you are human.</p></body></html>
HTML
""",
)
env = os.environ.copy()
env["PATH"] = f"{fake_bin}{os.pathsep}{env['PATH']}"
result = subprocess.run(
[str(ROOT / "scripts" / "check_customs_shipments.sh"), str(run_dir)],
cwd=ROOT,
env=env,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
self.assertEqual("", result.stderr)
self.assertEqual(0, result.returncode)
errors = (run_dir / "reports" / "customs-shipments-errors.txt").read_text(
encoding="utf-8"
)
self.assertIn("unexpected or blocked content from", errors)
self.assertFalse(
(run_dir / "api" / "customs" / "importinfo-ceva-valve.html").exists()
)
key_lines = (
run_dir / "reports" / "customs-shipments-key-lines.txt"
).read_text(encoding="utf-8")
self.assertEqual("", key_lines)
class CustomsSummaryIntegrationTests(unittest.TestCase):
def test_run_summary_and_status_draft_include_customs_lines(self):
with TemporaryDirectory() as tmp:
run_dir = Path(tmp) / "2026-05-05"
reports = run_dir / "reports"
reports.mkdir(parents=True)
(reports / "customs-shipments-key-lines.txt").write_text(
"2026-05-01\tCEVA C/O VALVE CORPORATION\tTECH-FRONT (CHONGQING) COMPUTER CO\tGAME CONSOLE\t42 PKG\t12596 Kgs\tSNHBSHALAX264015\n",
encoding="utf-8",
)
(reports / "customs-shipments-errors.txt").write_text("", encoding="utf-8")
summary = subprocess.run(
[
"python3",
str(ROOT / "scripts" / "write_run_summary.py"),
"--run-dir",
str(run_dir),
],
cwd=ROOT,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
draft = subprocess.run(
[
"python3",
str(ROOT / "scripts" / "draft_status_update.py"),
"--run-dir",
str(run_dir),
],
cwd=ROOT,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
self.assertEqual(0, summary.returncode)
self.assertEqual(0, draft.returncode)
run_summary = (reports / "run-summary.md").read_text(encoding="utf-8")
status_draft = (reports / "status-draft.md").read_text(encoding="utf-8")
self.assertIn("Customs shipments", run_summary)
self.assertIn("SNHBSHALAX264015", run_summary)
self.assertIn("Customs / Shipments", status_draft)
self.assertIn("GAME CONSOLE", status_draft)
def test_status_draft_reports_missing_customs_outputs_as_unavailable(self):
with TemporaryDirectory() as tmp:
run_dir = Path(tmp) / "2026-05-05"
reports = run_dir / "reports"
reports.mkdir(parents=True)
draft = subprocess.run(
[
"python3",
str(ROOT / "scripts" / "draft_status_update.py"),
"--run-dir",
str(run_dir),
],
cwd=ROOT,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
summary = subprocess.run(
[
"python3",
str(ROOT / "scripts" / "write_run_summary.py"),
"--run-dir",
str(run_dir),
],
cwd=ROOT,
text=True,
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
check=False,
)
self.assertEqual(0, draft.returncode)
self.assertEqual(0, summary.returncode)
status_draft = (reports / "status-draft.md").read_text(encoding="utf-8")
run_summary = (reports / "run-summary.md").read_text(encoding="utf-8")
self.assertIn("Customs / Shipments", status_draft)
self.assertIn("not generated", status_draft)
self.assertNotIn("No relevant customs shipment rows found.", status_draft)
self.assertIn("Customs shipments status: `unavailable`", run_summary)
if __name__ == "__main__":
unittest.main()