diff --git a/msp/importers/assistant.py b/msp/importers/assistant.py index 904c739..0ae39f2 100644 --- a/msp/importers/assistant.py +++ b/msp/importers/assistant.py @@ -41,8 +41,11 @@ def analyze_file(file_url: str) -> dict: if not os.path.exists(path): frappe.throw(f"Datei nicht gefunden: {file_url}") + # ZIPs müssen komplett gelesen werden (Central Directory steht am Ende); + # bei CSVs reichen 8 KiB für den Header. + read_full = path.lower().endswith(".zip") with open(path, "rb") as f: - sample = f.read(8192) + sample = f.read() if read_full else f.read(8192) filename = os.path.basename(path) best_cls, confidence, scores = detect_best_handler(sample, filename) @@ -59,7 +62,8 @@ def analyze_file(file_url: str) -> dict: "error": None, } if best_cls is None: - result["error"] = "Kein passender Handler — Format nicht erkannt." + # Kein Fehler — sondern „Unmatched": Frontend soll eine manuelle + # Handler-Zuordnung über das Dropdown anbieten. error bleibt null. return result result["handler_key"] = best_cls.handler_key diff --git a/msp/importers/handlers/adn_monthly_csv_handler.py b/msp/importers/handlers/adn_monthly_csv_handler.py index 4c11858..04af31b 100644 --- a/msp/importers/handlers/adn_monthly_csv_handler.py +++ b/msp/importers/handlers/adn_monthly_csv_handler.py @@ -44,13 +44,17 @@ class ADNMonthlyCSVHandler(BaseFileHandler): def sniff(cls, sample_bytes: bytes, filename: str) -> float: name = (filename or "").lower() - # ZIP-Dateien können wir nur nach Extension & Namensheuristik erkennen — - # der tatsächliche Header kommt erst nach dem Entpacken. Hohe Confidence - # bei typischen ADN-Dateinamen, moderate sonst. + # ZIP: Inhalt auspacken und die erste CSV-Kopfzeile prüfen. So funktioniert + # die Erkennung auch bei umbenannten ZIPs. if name.endswith(".zip"): - if "rechnungen" in name or "433148" in name: - return 0.85 - return 0.3 # irgendein ZIP — lieber unbestimmt + first_line = cls._peek_first_line_in_zip(sample_bytes) + if first_line is None: + # Kein CSV im ZIP zu sehen (vielleicht verschlüsselt oder zu groß + # im Sample) — greifen zurück auf Namens-Heuristik. + if "rechnungen" in name or "433148" in name: + return 0.65 + return 0.0 + return cls._score_first_line(first_line) if not name.endswith(".csv"): return 0.0 @@ -61,15 +65,42 @@ class ADNMonthlyCSVHandler(BaseFileHandler): except Exception: return 0.0 - first_line = head.split("\n", 1)[0].lstrip("\ufeff").strip() + first_line = head.split("\n", 1)[0] + return cls._score_first_line(first_line) + + @classmethod + def _score_first_line(cls, first_line: str) -> float: + first_line = (first_line or "").lstrip("\ufeff").strip() + if not first_line: + return 0.0 cols = {c.strip().upper() for c in first_line.split(";")} matching = cls._SIGNATURE_COLUMNS & cols if not matching: return 0.0 ratio = len(matching) / len(cls._SIGNATURE_COLUMNS) - # ratio=1.0 ⇒ alle Signatur-Spalten drin, perfekter Match return min(1.0, 0.4 + 0.6 * ratio) + @classmethod + def _peek_first_line_in_zip(cls, sample_bytes: bytes) -> str | None: + """Öffnet das ZIP in-memory und gibt die erste Zeile der enthaltenen CSV + zurück. None, wenn keine CSV gefunden oder der Sample zu klein ist.""" + import io + import zipfile + + if not sample_bytes: + return None + try: + with zipfile.ZipFile(io.BytesIO(sample_bytes)) as z: + csvs = [m for m in z.namelist() + if m.lower().endswith(".csv") and not m.endswith("/")] + if not csvs: + return None + with z.open(csvs[0]) as fh: + head = fh.read(2048).decode("utf-8", errors="replace") + return head.split("\n", 1)[0] + except (zipfile.BadZipFile, EOFError, KeyError): + return None + # ------------------------------------------------------------------ # Preview # ------------------------------------------------------------------ diff --git a/msp/msp/page/import_assistant/import_assistant.js b/msp/msp/page/import_assistant/import_assistant.js index a5e4c62..4719500 100644 --- a/msp/msp/page/import_assistant/import_assistant.js +++ b/msp/msp/page/import_assistant/import_assistant.js @@ -35,6 +35,26 @@ frappe.pages["import-assistant"].on_page_load = function (wrapper) { } inject_style(root); + + // Falls keine Preview mitgereicht wurde (z. B. bei manueller Handler-Zuordnung + // in der Dropzone), hole sie jetzt nach — damit Stats angezeigt und die + // Drift-Checkbox korrekt positioniert werden kann. + if (!ctx.preview) { + root.innerHTML = `
${__("Analysiere Datei …")}
`; + frappe.call({ + method: "msp.importers.assistant.analyze_file", + args: { file_url: ctx.file_url }, + callback: (r) => { + const m = r.message || {}; + if (m.preview) ctx.preview = m.preview; + if (m.display_name) ctx.display_name = ctx.display_name || m.display_name; + if (m.preview && m.preview.format_drift) ctx.format_drift = true; + render_page(root, ctx, page); + }, + }); + return; + } + render_page(root, ctx, page); }; @@ -133,13 +153,17 @@ function render_config_panel(ctx) {
${__("Leer = der Customer.billing_mode entscheidet pro Kunde.")}
-
+
-
${__("Die Analyse hat ein abweichendes CSV-Layout erkannt. Nur aktivieren, wenn du die Quelle kennst.")}
+
+ ${ctx.format_drift + ? __("Die Analyse hat ein abweichendes CSV-Layout erkannt. Nur aktivieren, wenn du die Quelle kennst.") + : __("Aktivieren, falls der Parser die Datei ablehnt, obwohl sie inhaltlich passt.")} +