CsvImport: header-driven column mapping
The /contacts LiveTable export produces a 12-column CSV with label-style
headers (Station 1 / Grid 1 / QSO (UTC) / ...) in a different column
order plus extras (id, distance_km, hrrr_status, flagged_invalid,
inserted_at). The old positional import rejected it with "expected 6 or
7 columns, got 12".
Replace the positional parser with header-driven mapping:
- Parse the header row, normalize each cell (downcase + trim), and look
it up in an alias map that covers both the raw field names and the
export labels.
- Required columns: station1, station2, grid1, grid2, band, qso_timestamp
(mode stays optional). If any are missing the header, return
{:error, {:missing_required_columns, [...]}}.
- Unknown columns are silently ignored so the export's extras don't break
round-trip.
- Per-row validation stays the same (changeset errors still reported by
row number).
Existing 6- or 7-column positional CSVs that already have a header still
work — the header parser recognizes `station1`, `station2`, etc.
directly.
This commit is contained in:
parent
c7f9aac3f9
commit
be6f1a1d28
2 changed files with 153 additions and 27 deletions
|
|
@ -30,8 +30,27 @@ defmodule Microwaveprop.Radio.CsvImport do
|
||||||
|
|
||||||
@dedup_window_seconds 3600
|
@dedup_window_seconds 3600
|
||||||
|
|
||||||
@columns_with_mode ~w(station1 station2 grid1 grid2 band mode qso_timestamp)
|
# Header parsing is case-insensitive and whitespace-insensitive. Accepts
|
||||||
@columns_without_mode ~w(station1 station2 grid1 grid2 band qso_timestamp)
|
# both the raw field names (station1, grid1, ...) and the human labels the
|
||||||
|
# /contacts live_table export produces (Station 1, Grid 1, QSO (UTC), ...)
|
||||||
|
# so export → import round-trips. Unknown columns are ignored.
|
||||||
|
@header_aliases %{
|
||||||
|
"station1" => "station1",
|
||||||
|
"station 1" => "station1",
|
||||||
|
"station2" => "station2",
|
||||||
|
"station 2" => "station2",
|
||||||
|
"grid1" => "grid1",
|
||||||
|
"grid 1" => "grid1",
|
||||||
|
"grid2" => "grid2",
|
||||||
|
"grid 2" => "grid2",
|
||||||
|
"band" => "band",
|
||||||
|
"mode" => "mode",
|
||||||
|
"qso_timestamp" => "qso_timestamp",
|
||||||
|
"qso timestamp" => "qso_timestamp",
|
||||||
|
"qso (utc)" => "qso_timestamp"
|
||||||
|
}
|
||||||
|
|
||||||
|
@required_columns ~w(station1 station2 grid1 grid2 band qso_timestamp)
|
||||||
|
|
||||||
@doc """
|
@doc """
|
||||||
Parse-and-insert in one shot. Kept for backward compatibility — does NOT
|
Parse-and-insert in one shot. Kept for backward compatibility — does NOT
|
||||||
|
|
@ -56,7 +75,10 @@ defmodule Microwaveprop.Radio.CsvImport do
|
||||||
}
|
}
|
||||||
|
|
||||||
@spec import(String.t(), String.t()) ::
|
@spec import(String.t(), String.t()) ::
|
||||||
{:ok, import_result()} | {:error, :empty_csv} | {:error, :no_data_rows}
|
{:ok, import_result()}
|
||||||
|
| {:error, :empty_csv}
|
||||||
|
| {:error, :no_data_rows}
|
||||||
|
| {:error, {:missing_required_columns, [String.t()]}}
|
||||||
def import(csv_string, submitter_email) do
|
def import(csv_string, submitter_email) do
|
||||||
with {:ok, rows} <- parse_csv(csv_string, submitter_email) do
|
with {:ok, rows} <- parse_csv(csv_string, submitter_email) do
|
||||||
{valid, invalid} = split_parsed(rows)
|
{valid, invalid} = split_parsed(rows)
|
||||||
|
|
@ -88,7 +110,10 @@ defmodule Microwaveprop.Radio.CsvImport do
|
||||||
(`:earlier_in_upload`).
|
(`:earlier_in_upload`).
|
||||||
"""
|
"""
|
||||||
@spec preview(String.t(), String.t()) ::
|
@spec preview(String.t(), String.t()) ::
|
||||||
{:ok, preview_result()} | {:error, :empty_csv} | {:error, :no_data_rows}
|
{:ok, preview_result()}
|
||||||
|
| {:error, :empty_csv}
|
||||||
|
| {:error, :no_data_rows}
|
||||||
|
| {:error, {:missing_required_columns, [String.t()]}}
|
||||||
def preview(csv_string, submitter_email) do
|
def preview(csv_string, submitter_email) do
|
||||||
with {:ok, rows} <- parse_csv(csv_string, submitter_email) do
|
with {:ok, rows} <- parse_csv(csv_string, submitter_email) do
|
||||||
{valid, invalid} = split_parsed(rows)
|
{valid, invalid} = split_parsed(rows)
|
||||||
|
|
@ -146,39 +171,72 @@ defmodule Microwaveprop.Radio.CsvImport do
|
||||||
[{_header, _}] ->
|
[{_header, _}] ->
|
||||||
{:error, :no_data_rows}
|
{:error, :no_data_rows}
|
||||||
|
|
||||||
[{_header, _} | data_lines] ->
|
[{header_line, _} | data_lines] ->
|
||||||
rows = Enum.map(data_lines, fn {line, row_num} -> parse_row(line, row_num, submitter_email) end)
|
with {:ok, column_map} <- parse_header(header_line) do
|
||||||
{:ok, rows}
|
{:ok, parse_data_rows(data_lines, submitter_email, column_map)}
|
||||||
|
end
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
|
defp parse_data_rows(data_lines, submitter_email, column_map) do
|
||||||
|
Enum.map(data_lines, fn {line, row_num} ->
|
||||||
|
parse_row(line, row_num, submitter_email, column_map)
|
||||||
|
end)
|
||||||
|
end
|
||||||
|
|
||||||
defp blank?(line), do: String.trim(line) == ""
|
defp blank?(line), do: String.trim(line) == ""
|
||||||
|
|
||||||
defp parse_row(line, row_num, submitter_email) do
|
# Builds an index → field-name map from the header line. Unknown headers
|
||||||
fields = parse_csv_fields(line)
|
# are silently skipped (that's how we accept the richer /contacts export).
|
||||||
|
defp parse_header(header_line) do
|
||||||
|
column_map =
|
||||||
|
header_line
|
||||||
|
|> parse_csv_fields()
|
||||||
|
|> Enum.with_index()
|
||||||
|
|> Enum.reduce(%{}, fn {raw, index}, acc ->
|
||||||
|
case Map.fetch(@header_aliases, raw |> String.downcase() |> String.trim()) do
|
||||||
|
{:ok, field} -> Map.put(acc, index, field)
|
||||||
|
:error -> acc
|
||||||
|
end
|
||||||
|
end)
|
||||||
|
|
||||||
case columns_for(length(fields)) do
|
present = column_map |> Map.values() |> MapSet.new()
|
||||||
nil ->
|
missing = Enum.reject(@required_columns, &MapSet.member?(present, &1))
|
||||||
{:invalid,
|
|
||||||
%{
|
|
||||||
row_num: row_num,
|
|
||||||
messages: ["expected 6 columns (no mode) or 7 columns (with mode), got #{length(fields)}"]
|
|
||||||
}}
|
|
||||||
|
|
||||||
columns ->
|
if missing == [] do
|
||||||
attrs =
|
{:ok, column_map}
|
||||||
columns
|
else
|
||||||
|> Enum.zip(fields)
|
{:error, {:missing_required_columns, missing}}
|
||||||
|> Map.new()
|
|
||||||
|> Map.put("submitter_email", submitter_email)
|
|
||||||
|
|
||||||
parse_row_with_timestamp(attrs, row_num)
|
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
defp columns_for(7), do: @columns_with_mode
|
defp parse_row(line, row_num, submitter_email, column_map) do
|
||||||
defp columns_for(6), do: @columns_without_mode
|
fields = parse_csv_fields(line)
|
||||||
defp columns_for(_), do: nil
|
|
||||||
|
attrs =
|
||||||
|
column_map
|
||||||
|
|> Enum.reduce(%{}, fn {index, field}, acc ->
|
||||||
|
case Enum.at(fields, index) do
|
||||||
|
nil -> acc
|
||||||
|
value -> Map.put(acc, field, value)
|
||||||
|
end
|
||||||
|
end)
|
||||||
|
|> Map.put("submitter_email", submitter_email)
|
||||||
|
|
||||||
|
if required_present?(attrs) do
|
||||||
|
parse_row_with_timestamp(attrs, row_num)
|
||||||
|
else
|
||||||
|
missing_on_row = Enum.reject(@required_columns, &Map.has_key?(attrs, &1))
|
||||||
|
|
||||||
|
{:invalid,
|
||||||
|
%{
|
||||||
|
row_num: row_num,
|
||||||
|
messages: ["row is missing required columns: #{Enum.join(missing_on_row, ", ")}"]
|
||||||
|
}}
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
|
defp required_present?(attrs), do: Enum.all?(@required_columns, &Map.has_key?(attrs, &1))
|
||||||
|
|
||||||
defp parse_row_with_timestamp(attrs, row_num) do
|
defp parse_row_with_timestamp(attrs, row_num) do
|
||||||
case normalize_timestamp(attrs["qso_timestamp"]) do
|
case normalize_timestamp(attrs["qso_timestamp"]) do
|
||||||
|
|
|
||||||
|
|
@ -239,6 +239,74 @@ defmodule Microwaveprop.Radio.CsvImportTest do
|
||||||
end
|
end
|
||||||
end
|
end
|
||||||
|
|
||||||
|
describe "import/2 with header-driven column mapping" do
|
||||||
|
# The /contacts export produces a rich CSV with label-style headers and
|
||||||
|
# extra columns (id, distance_km, status fields, inserted_at). The import
|
||||||
|
# must read the header row, match by field name / label, and ignore
|
||||||
|
# anything it doesn't recognize so export → import round-trips.
|
||||||
|
|
||||||
|
test "accepts /contacts export format (labels + extra columns + different order)" do
|
||||||
|
# Mirrors the live_table export: 12 columns, labels like "Station 1",
|
||||||
|
# grids interleaved (station1, grid1, station2, grid2…), and extra
|
||||||
|
# columns the importer should ignore.
|
||||||
|
csv = """
|
||||||
|
ID,Station 1,Grid 1,Station 2,Grid 2,Band,Mode,Distance (km),Enriched,Flag,QSO (UTC),Added (UTC)
|
||||||
|
abc-123,W5XD,EM12,K5TR,EM00,10000,CW,295,Complete,,2026-03-28 18:00,2026-03-28 18:05
|
||||||
|
"""
|
||||||
|
|
||||||
|
assert {:ok, %{imported: [contact], errors: []}} = CsvImport.import(csv, @submitter)
|
||||||
|
assert contact.station1 == "W5XD"
|
||||||
|
assert contact.station2 == "K5TR"
|
||||||
|
assert contact.grid1 == "EM12"
|
||||||
|
assert contact.grid2 == "EM00"
|
||||||
|
assert contact.mode == "CW"
|
||||||
|
end
|
||||||
|
|
||||||
|
test "accepts arbitrary column order" do
|
||||||
|
csv = """
|
||||||
|
qso_timestamp,band,mode,grid2,grid1,station2,station1
|
||||||
|
2026-03-28T18:00:00Z,10000,CW,EM00,EM12,K5TR,W5XD
|
||||||
|
"""
|
||||||
|
|
||||||
|
assert {:ok, %{imported: [contact], errors: []}} = CsvImport.import(csv, @submitter)
|
||||||
|
assert contact.station1 == "W5XD"
|
||||||
|
assert contact.grid1 == "EM12"
|
||||||
|
end
|
||||||
|
|
||||||
|
test "is case-insensitive on header names" do
|
||||||
|
csv = """
|
||||||
|
STATION1,STATION2,GRID1,GRID2,BAND,MODE,QSO_TIMESTAMP
|
||||||
|
W5XD,K5TR,EM12,EM00,10000,CW,2026-03-28T18:00:00Z
|
||||||
|
"""
|
||||||
|
|
||||||
|
assert {:ok, %{imported: [contact], errors: []}} = CsvImport.import(csv, @submitter)
|
||||||
|
assert contact.station1 == "W5XD"
|
||||||
|
end
|
||||||
|
|
||||||
|
test "ignores unknown columns" do
|
||||||
|
csv = """
|
||||||
|
station1,notes,station2,grid1,ignored_col,grid2,band,mode,qso_timestamp
|
||||||
|
W5XD,some note,K5TR,EM12,xyz,EM00,10000,CW,2026-03-28T18:00:00Z
|
||||||
|
"""
|
||||||
|
|
||||||
|
assert {:ok, %{imported: [contact], errors: []}} = CsvImport.import(csv, @submitter)
|
||||||
|
assert contact.station1 == "W5XD"
|
||||||
|
end
|
||||||
|
|
||||||
|
test "rejects when a required column header is missing" do
|
||||||
|
# No qso_timestamp column at all.
|
||||||
|
csv = """
|
||||||
|
station1,station2,grid1,grid2,band,mode
|
||||||
|
W5XD,K5TR,EM12,EM00,10000,CW
|
||||||
|
"""
|
||||||
|
|
||||||
|
assert {:error, {:missing_required_columns, missing}} =
|
||||||
|
CsvImport.import(csv, @submitter)
|
||||||
|
|
||||||
|
assert "qso_timestamp" in missing
|
||||||
|
end
|
||||||
|
end
|
||||||
|
|
||||||
describe "normalize_timestamp/1" do
|
describe "normalize_timestamp/1" do
|
||||||
test "parses ISO 8601 with Z" do
|
test "parses ISO 8601 with Z" do
|
||||||
assert {:ok, "2024-06-15T14:30:00Z"} = CsvImport.normalize_timestamp("2024-06-15T14:30:00Z")
|
assert {:ok, "2024-06-15T14:30:00Z"} = CsvImport.normalize_timestamp("2024-06-15T14:30:00Z")
|
||||||
|
|
|
||||||
Loading…
Add table
Reference in a new issue