using System.Data; using GB5Shared.DTO.Ice; using IceImportDAL.DTO.IceImportPipeline; using Microsoft.Extensions.Logging; namespace IceImportDAL.CustomCode.Parsers; // Shared helper for ExcelSourceFileParser/CsvSourceFileParser. // // Design note (documented simplification): GB5Shared.FileImport.FileToDataTableService reads a // single flat DataTable per file — ExcelReaderFactory's AsDataSet() result is truncated to // ds.Tables[0] (see FileToDataTableService.ReadExcelFromBytesAsync), and the CSV reader has no // concept of multiple sheets/sections either. The Phase 1 plan's "one sheet/filter-pass per // entity" convention (legacy's per-entity-sheet Excel / per-entity-filter CSV split) is therefore // implemented here as a column-subset split of that single table: every DestinationMemberEntityId // referenced by a ValueType=0 ("Source") IceMapDetails row gets its own DataTable containing just // the source columns that entity's fields map from, sharing the same row set as the parent table. // If a future IceMap genuinely needs distinct rows per entity (not just distinct columns), promoting // FileToDataTableService to accept a sheet name/index is the correct fix — not reimplementing Excel // reading here. internal static class FlatTableEntitySplitter { public static ParsedImportDataSetDTO Split( DataTable flat, string sourceFileName, IceMapDTO mapDefinition, ILogger? logger = null) { var result = new ParsedImportDataSetDTO { SourceFileName = sourceFileName }; // Temporary diagnostic — remove once the BOM parent/detail-array column resolution is // confirmed stable. Shows the RAW parsed table's actual column names/row count BEFORE any // per-entity splitting, so a "split table has zero columns" symptom is diagnosable from the // log alone (distinguishes "file didn't parse as expected" from "SourceFieldName mismatch"). logger?.LogInformation( "IceImport FlatTableEntitySplitter: flat table has {RowCount} row(s), Columns=[{Columns}]", flat.Rows.Count, string.Join(",", flat.Columns.Cast().Select(c => c.ColumnName))); // ValueType 4/5 (Code/Name -> Id) rows read a raw source column too, just like ValueType 0 // ("Source") — they just transform it via a lookup afterwards. Excluding them here would // drop their source column from the split table entirely, leaving CodeToIdLookupResolver/ // NameToIdLookupResolver (and EntityLookupService.CollectDistinctSourceValues) with no // column to read from. var groups = mapDefinition.IceMapDetailsArray .Where(d => d.IceMapDetailsValueType is 0 or 4 or 5 && !string.IsNullOrWhiteSpace(d.IceMapDetailsSourceFieldName)) .GroupBy(d => d.DestinationMemberEntityId) .ToList(); if (groups.Count == 0) { // No source-column mappings declared yet — hand back the whole table under the root // entity. Key by the resolved DTO-side entity id (IceMapRootEntityId), not the POCO-side // IceMapEntityId, so ResolveTableForEntity's exact-match lookup (keyed by // DestinationMemberEntityId) can find it. result.Tables[mapDefinition.IceMapRootEntityId.ToString()] = flat; return result; } // Keyed by DestinationMemberEntityId (numeric) — the entity a row of data belongs to/will // be saved as. Not IceMapDetails.EntityId, which (for ValueType 4/5 lookup rows only) // identifies a DIFFERENT entity: the one being looked up, not the one owning this table. foreach (var group in groups) { var tableKey = group.Key.ToString(); var wantedColumns = group .Select(d => d.IceMapDetailsSourceFieldName!.Trim().ToUpperInvariant()) .Where(c => flat.Columns.Contains(c)) .Distinct() .ToList(); var table = new DataTable(tableKey); foreach (var col in wantedColumns) table.Columns.Add(col, typeof(string)); foreach (DataRow srcRow in flat.Rows) { var newRow = table.NewRow(); foreach (var col in wantedColumns) newRow[col] = srcRow[col]; table.Rows.Add(newRow); } result.Tables[tableKey] = table; } return result; } }