""" Invoice2Data Service Wrapper around invoice2data library for template-based invoice extraction """ import logging import re from datetime import datetime from pathlib import Path from typing import Dict, List, Optional, Any import yaml logger = logging.getLogger(__name__) class Invoice2DataService: """Service for extracting invoice data using invoice2data templates""" def __init__(self): self.template_dir = Path(__file__).parent.parent.parent / "data" / "invoice_templates" self.templates = self._load_templates() logger.info(f"📋 Loaded {len(self.templates)} invoice2data templates") def _load_templates(self) -> Dict[str, Dict]: """Load all YAML templates from template directory""" templates = {} if not self.template_dir.exists(): logger.warning(f"Template directory not found: {self.template_dir}") return templates for template_file in self.template_dir.glob("*.yml"): try: with open(template_file, 'r', encoding='utf-8') as f: template_data = yaml.safe_load(f) template_name = template_file.stem templates[template_name] = template_data logger.debug(f" ✓ Loaded template: {template_name}") except Exception as e: logger.error(f" ✗ Failed to load template {template_file}: {e}") return templates def match_template(self, text: str) -> Optional[str]: """ Find matching template based on keywords Returns template name or None """ text_lower = text.lower() for template_name, template_data in self.templates.items(): keywords = template_data.get('keywords', []) # Check if all keywords are present matches = sum(1 for keyword in keywords if str(keyword).lower() in text_lower) if matches >= len(keywords) * 0.7: # 70% of keywords must match logger.info(f"✅ Matched template: {template_name} ({matches}/{len(keywords)} keywords)") return template_name logger.warning("⚠️ No template matched") return None def _parse_amount(self, value: Any, decimal_separator: str = ",", thousands_separator: str = ".") -> Optional[float]: if value is None: return None if isinstance(value, (int, float)): return float(value) cleaned = re.sub(r"\s+", "", str(value).strip()) if not cleaned: return None if thousands_separator in cleaned and decimal_separator in cleaned: cleaned = cleaned.replace(thousands_separator, "").replace(decimal_separator, ".") elif thousands_separator in cleaned: cleaned = cleaned.replace(thousands_separator, "") elif decimal_separator == "," and "," in cleaned: cleaned = cleaned.replace(",", ".") try: return float(cleaned) except ValueError: return None def _parse_date_value(self, value: Any, date_formats: Optional[List[str]] = None) -> Optional[str]: if value is None: return None raw = str(value).strip() if not raw: return None normalized = raw replacements = { "januar": "January", "februar": "February", "marts": "March", "april": "April", "maj": "May", "juni": "June", "juli": "July", "august": "August", "september": "September", "oktober": "October", "november": "November", "december": "December", } for da_name, en_name in replacements.items(): normalized = re.sub(rf"\b{da_name}\b", en_name, normalized, flags=re.IGNORECASE) candidates = date_formats or [ "%d.%m.%Y", "%d-%m-%Y", "%d/%m-%Y", "%d. %B %Y", "%d. %B %Y.", "%d. %B %Y", "%d. %B %Y", ] normalized = re.sub(r"\s+", " ", normalized).strip() for candidate in candidates: try: return datetime.strptime(normalized, candidate).strftime("%Y-%m-%d") except ValueError: continue return None def _is_globalconnect_noise_line(self, line: str) -> bool: compact = re.sub(r"\s+", " ", str(line or "")).strip() if not compact: return True if len(compact) > 140: return True noise_patterns = ( r"Skandinaviska Enskilda Banken", r"\bSWIFT-kode\b", r"\bIBAN\b", r"www\.globalconnect\.dk", r"CMsupport@globalconnect\.dk", r"\+45\s*77\s*30\s*30\s*00", r"Faktura\s+BMC Denmark ApS", r"\bBeskrivelse\s+Antal\s+Enhed\s+Enhedspris\s+Beløb\b", r"\bI alt DKK\b", r"\b25%\s+moms\b", r"\bSE/CVR-nr\.?\b", r"\bPBS-nummer\b", r"\bBS Kundenr\.?\b", r"\bDeb\. grp\. nr\.?\b", r"\bBetalingsbetingelser\b", r"\bEfter forfald beregnes rente\b", r"\bTeleydelser uden moms\b", r"\bAdministrations gebyr\b", ) return any(re.search(pattern, compact, re.IGNORECASE) for pattern in noise_patterns) def _extract_globalconnect(self, text: str, template_name: str, template: Dict[str, Any]) -> Dict[str, Any]: options = template.get("options", {}) extracted: Dict[str, Any] = { "template": template_name, "issuer": template.get("issuer"), "country": template.get("country"), "currency": options.get("currency", "DKK"), } invoice_number_match = re.search(r"(?:Fakturanr\.?|Kreditnotanr\.?)\s*(\d+)", text, re.IGNORECASE) if invoice_number_match: extracted["invoice_number"] = int(invoice_number_match.group(1)) if re.search(r"\bKreditnota\b|\bKreditnotanr\.?\b", text, re.IGNORECASE): extracted["document_type"] = "credit_note" customer_reference_match = re.search(r"Kundenr\.?\s*([A-Z0-9-]+)", text, re.IGNORECASE) if customer_reference_match: extracted["customer_reference"] = customer_reference_match.group(1).strip() invoice_date_match = re.search(r"Bilagsdato\s+([^\n]+)", text, re.IGNORECASE) if invoice_date_match: parsed = self._parse_date_value(invoice_date_match.group(1), ["%d. %B %Y"]) if parsed: extracted["invoice_date"] = parsed due_date_match = re.search(r"Forfaldsdato\s+([^\n]+)", text, re.IGNORECASE) if due_date_match: parsed = self._parse_date_value(due_date_match.group(1), ["%d. %B %Y"]) if parsed: extracted["due_date"] = parsed untaxed_match = re.search(r"I\s+alt\s+DKK\s+ekskl\.\s+moms\s+([\d.,]+)", text, re.IGNORECASE) if untaxed_match: extracted["amount_untaxed"] = self._parse_amount(untaxed_match.group(1)) vat_match = re.search(r"25%\s+moms\s+([\d.,]+)", text, re.IGNORECASE) if vat_match: extracted["vat_amount"] = self._parse_amount(vat_match.group(1)) total_match = re.search(r"I\s+alt\s+DKK\s+inkl\.\s+moms\s+([\d.,]+)", text, re.IGNORECASE) if total_match: extracted["amount_total"] = self._parse_amount(total_match.group(1)) cvr_matches = [match.group(1) for match in re.finditer(r"SE/CVR-nr\.\s+(\d{8})", text, re.IGNORECASE)] vendor_cvrs = [cvr for cvr in cvr_matches if cvr != "29522790"] if vendor_cvrs: extracted["vendor_vat"] = vendor_cvrs[0] lines: List[Dict[str, Any]] = [] current_context: Dict[str, Any] = {} pending_street: Optional[str] = None pending_line_for_continuation: Optional[Dict[str, Any]] = None for raw_line in text.splitlines(): line = re.sub(r"\s+", " ", raw_line).strip() if not line: continue contract_match = re.match(r"Kontrakt:\s*(.+)$", line, re.IGNORECASE) if contract_match: current_context["contract_number"] = contract_match.group(1).strip() pending_line_for_continuation = None continue vedr_match = re.match(r"Vedr:\s*(.+)$", line, re.IGNORECASE) if vedr_match: provider_reference = vedr_match.group(1).strip() current_context["provider_reference"] = provider_reference current_context["circuit_id"] = provider_reference pending_line_for_continuation = None continue customer_match = re.match(r"Slutkunde:\s*(.+)$", line, re.IGNORECASE) if customer_match: current_context["end_customer_name"] = customer_match.group(1).strip() pending_line_for_continuation = None continue period_match = re.match(r"Periode:\s*(\d{2}-\d{2}-\d{4})\s*-\s*(\d{2}-\d{2}-\d{4})", line, re.IGNORECASE) if period_match: current_context["period_start"] = self._parse_date_value(period_match.group(1), ["%d-%m-%Y"]) current_context["period_end"] = self._parse_date_value(period_match.group(2), ["%d-%m-%Y"]) pending_line_for_continuation = None continue cidr_match = re.match(r"(\d{1,3}(?:\.\d{1,3}){3}/\d{1,2})(?:\s+\(([^)]+)\))?$", line) if cidr_match: cidr = cidr_match.group(1) reference = cidr_match.group(2).strip() if cidr_match.group(2) else None if pending_line_for_continuation and "ip" in str(pending_line_for_continuation.get("description") or "").lower(): pending_line_for_continuation["ip_address"] = cidr if reference: pending_line_for_continuation["provider_reference"] = reference pending_line_for_continuation["circuit_id"] = reference current_context["ip_address"] = cidr if reference: current_context["provider_reference"] = reference current_context["circuit_id"] = reference continue reference_match = re.match(r"((?:NKA|EB|DSL-)[A-Z0-9-]+)$", line, re.IGNORECASE) if reference_match: reference = reference_match.group(1).strip() current_context["provider_reference"] = reference current_context["circuit_id"] = reference if pending_line_for_continuation and not pending_line_for_continuation.get("provider_reference"): pending_line_for_continuation["provider_reference"] = reference pending_line_for_continuation["circuit_id"] = reference pending_street = None continue street_only_match = re.match(r"(.+?\d+[A-ZÆØÅa-zæøå]?)$", line) if street_only_match and not re.search(r"(?:Fakturanr|Bilagsdato|Forfaldsdato|SE/CVR|Kundenr|Kontrakt|Vedr|Slutkunde|Periode)", line, re.IGNORECASE): postal_hint = re.search(r"\b\d{4}\b", line) if not postal_hint and not re.search(r"\b(?:Gbps|Mbps|Kbps|Måneder|Måned|Stk|pcs)\b", line, re.IGNORECASE): pending_street = street_only_match.group(1).strip() continue city_line_match = re.match(r"(\d{4})\s+([A-ZÆØÅa-zæøå].+)$", line) if city_line_match and pending_street: postal_code = city_line_match.group(1).strip() city = city_line_match.group(2).strip() current_context["location_street"] = pending_street current_context["location_zip"] = postal_code current_context["location_city"] = city current_context["service_address"] = f"{pending_street}, {postal_code} {city}" if pending_line_for_continuation and not pending_line_for_continuation.get("service_address"): pending_line_for_continuation["location_street"] = pending_street pending_line_for_continuation["location_zip"] = postal_code pending_line_for_continuation["location_city"] = city pending_line_for_continuation["service_address"] = f"{pending_street}, {postal_code} {city}" pending_street = None continue address_match = re.match(r"(.+?)\s+(\d{4})\s+([A-ZÆØÅa-zæøå].+)$", line) if address_match and not re.search(r"(?:Fakturanr|Bilagsdato|Forfaldsdato|SE/CVR)", line, re.IGNORECASE): street = address_match.group(1).strip() postal_code = address_match.group(2).strip() city = address_match.group(3).strip() current_context["location_street"] = street current_context["location_zip"] = postal_code current_context["location_city"] = city current_context["service_address"] = f"{street}, {postal_code} {city}" if pending_line_for_continuation and not pending_line_for_continuation.get("service_address"): pending_line_for_continuation["location_street"] = street pending_line_for_continuation["location_zip"] = postal_code pending_line_for_continuation["location_city"] = city pending_line_for_continuation["service_address"] = f"{street}, {postal_code} {city}" pending_street = None continue line_match = re.match( r"(.+?)\s+(\d+(?:[.,]\d+)?)\s+(Måneder|Måned|Stk\.?|Stk|pcs\.?)\s+([\d.]+,\d{2})\s+([\d.]+,\d{2})$", line, re.IGNORECASE, ) if not line_match: if pending_line_for_continuation and not self._is_globalconnect_noise_line(line) and not re.search( r"(?:I alt DKK|25% moms|SE/CVR|Kundenr\.?|Fakturanr\.?|Bilagsdato|Forfaldsdato)", line, re.IGNORECASE, ): existing = str(pending_line_for_continuation.get("description") or "").strip() if line.lower() not in existing.lower(): combined = f"{existing} {line}".strip() pending_line_for_continuation["description"] = combined[:250].strip() continue description = line_match.group(1).strip() quantity = self._parse_amount(line_match.group(2)) unit = line_match.group(3).strip() unit_price = self._parse_amount(line_match.group(4)) line_total = self._parse_amount(line_match.group(5)) line_data: Dict[str, Any] = { "line_number": len(lines) + 1, "description": description, "quantity": quantity, "unit": unit, "unit_price": unit_price, "line_total": line_total, "customer_reference": extracted.get("customer_reference"), } line_data.update(current_context) lines.append(line_data) pending_line_for_continuation = line_data pending_street = None if "ip_address" in current_context: current_context.pop("ip_address", None) if lines: extracted["lines"] = lines self._validate_amounts(extracted) return extracted def extract_with_template(self, text: str, template_name: str) -> Dict[str, Any]: """ Extract invoice data using specific template """ if template_name not in self.templates: raise ValueError(f"Template not found: {template_name}") template = self.templates[template_name] if template_name == "dk.globalconnect": return self._extract_globalconnect(text, template_name, template) fields = template.get('fields', {}) options = template.get('options', {}) extracted = { 'template': template_name, 'issuer': template.get('issuer'), 'country': template.get('country'), 'currency': options.get('currency', 'DKK') } # Extract each field using its regex for field_name, field_config in fields.items(): if field_config.get('parser') != 'regex': continue pattern = field_config.get('regex') field_type = field_config.get('type', 'string') group = field_config.get('group', 1) try: match = re.search(pattern, text, re.IGNORECASE | re.MULTILINE) if match: value = match.group(group).strip() logger.debug(f" 🔍 Extracted raw value for {field_name}: '{value}' (type: {field_type})") # Handle CVR filtering (avoid customer CVR) if field_name == 'vendor_vat': # Find ALL CVR numbers all_cvr_matches = re.finditer(r'SE/CVR-nr\.\s+(\d{8})', text, re.IGNORECASE) cvr_numbers = [m.group(1) for m in all_cvr_matches] # Filter out BMC's CVR (29522790) vendor_cvrs = [cvr for cvr in cvr_numbers if cvr != '29522790'] if vendor_cvrs: value = vendor_cvrs[0] logger.debug(f" ✓ {field_name}: {value} (filtered from {cvr_numbers})") else: logger.warning(f" ⚠️ Only customer CVR found, no vendor CVR") continue # Convert type if field_type == 'float': decimal_sep = options.get('decimal_separator', ',') thousands_sep = options.get('thousands_separator', '.') value = self._parse_amount(value, decimal_sep, thousands_sep) elif field_type == 'int': value = int(value) elif field_type == 'date': date_formats = options.get('date_formats', ['%B %d, %Y', '%d-%m-%Y']) value = self._parse_date_value(value, date_formats) or value extracted[field_name] = value logger.debug(f" ✓ {field_name}: {value}") else: logger.debug(f" ✗ {field_name}: No match") except Exception as e: logger.warning(f" ✗ Failed to extract {field_name}: {e}") # Extract line items if defined in template lines_config = template.get('lines', []) if lines_config: extracted['lines'] = self._extract_lines(text, lines_config, options) # Calculate due_date if field has '+Xd' value if 'due_date' in fields: due_config = fields['due_date'] if due_config.get('parser') == 'static': value = due_config.get('value', '') if value.startswith('+') and value.endswith('d') and extracted.get('invoice_date'): try: from datetime import timedelta days = int(value[1:-1]) inv_date = datetime.strptime(extracted['invoice_date'], '%Y-%m-%d') due_date = inv_date + timedelta(days=days) extracted['due_date'] = due_date.strftime('%Y-%m-%d') logger.info(f"✅ Calculated due_date: {extracted['due_date']} ({days} days from invoice_date)") except Exception as e: logger.warning(f"Failed to calculate due_date: {e}") # Validate amounts (sum of lines vs total, VAT calculation) self._validate_amounts(extracted) return extracted def _extract_lines(self, text: str, lines_configs: List[Dict], options: Dict) -> List[Dict]: """Extract line items from invoice text""" all_lines = [] logger.debug(f"🔍 Extracting lines with {len(lines_configs)} configurations") for lines_config in lines_configs: start_pattern = lines_config.get('start') end_pattern = lines_config.get('end') line_config = lines_config.get('line', {}) if not start_pattern or not line_config: continue try: # Find section between start and end patterns if end_pattern: section_pattern = f"{start_pattern}(.*?){end_pattern}" section_match = re.search(section_pattern, text, re.DOTALL | re.IGNORECASE) else: section_pattern = f"{start_pattern}(.*?)$" section_match = re.search(section_pattern, text, re.DOTALL | re.IGNORECASE) if not section_match: logger.debug(f" ✗ Line section not found (start: {start_pattern[:50]}, end: {end_pattern[:50] if end_pattern else 'None'})") continue section_text = section_match.group(1) logger.debug(f" ✓ Found line section ({len(section_text)} chars)") # Extract individual lines line_pattern = line_config.get('regex') field_names = line_config.get('fields', []) field_types = line_config.get('types', {}) context_config = line_config.get('context_before', {}) if not line_pattern or not field_names: continue # Split section into lines for context processing section_lines = section_text.split('\n') line_matches = [] # Find all matching lines with their indices for line_idx, line_text in enumerate(section_lines): match = re.search(line_pattern, line_text, re.MULTILINE) if match: line_matches.append((line_idx, line_text, match)) logger.debug(f" ✓ Found {len(line_matches)} matching lines") for line_idx, line_text, match in line_matches: line_data = {} # Extract main line fields for idx, field_name in enumerate(field_names, start=1): try: value = match.group(idx).strip() field_type = field_types.get(field_name, 'string') # Convert type if field_type == 'float': thousands_sep = options.get('thousands_separator', ',') decimal_sep = options.get('decimal_separator', '.') value = re.sub(r'\s+', '', value) if thousands_sep in value and decimal_sep in value: value = value.replace(thousands_sep, '').replace(decimal_sep, '.') elif thousands_sep in value: value = value.replace(thousands_sep, '') elif decimal_sep in value and decimal_sep == ',': value = value.replace(',', '.') value = float(value) elif field_type == 'int': value = int(value) line_data[field_name] = value except Exception as e: logger.debug(f" ✗ Failed to extract line field {field_name}: {e}") # Extract context_before if configured if context_config and line_idx > 0: max_lines = context_config.get('max_lines', 5) patterns = context_config.get('patterns', []) # Look at lines BEFORE this line start_idx = max(0, line_idx - max_lines) context_lines = section_lines[start_idx:line_idx] for pattern_config in patterns: pattern_regex = pattern_config.get('regex') pattern_fields = pattern_config.get('fields', []) if not pattern_regex or not pattern_fields: continue # Try to match against context lines (most recent first) for ctx_line in reversed(context_lines): ctx_match = re.search(pattern_regex, ctx_line) if ctx_match: # Extract fields from context for ctx_idx, ctx_field_name in enumerate(pattern_fields, start=1): try: ctx_value = ctx_match.group(ctx_idx).strip() line_data[ctx_field_name] = ctx_value except Exception as e: logger.debug(f" ✗ Failed to extract context field {ctx_field_name}: {e}") break # Stop after first match for this pattern # If header is line-wrapped (e.g. "Husleje" on one line and "(inkl...)" on next), # stitch them together so description becomes "Husleje (inkl...) ...". try: description = line_data.get('description') if isinstance(description, str) and description.lstrip().startswith('('): prefix = None for candidate in reversed(context_lines): candidate_stripped = candidate.strip() if not candidate_stripped: continue if candidate_stripped.startswith('('): continue if re.match(r'^\d', candidate_stripped): continue if candidate_stripped.lower().startswith('periode:'): continue if ' Kr ' in f" {candidate_stripped} ": continue # Avoid picking calculation/detail lines if any(token in candidate_stripped for token in ('*', '=', 'm2', 'm²')): continue # Prefer short header-like prefixes (e.g. "Husleje") if len(candidate_stripped) <= 40: prefix = candidate_stripped break if prefix and not description.strip().lower().startswith(prefix.lower()): line_data['description'] = f"{prefix} {description.strip()}".strip() except Exception as e: logger.debug(f" ✗ Failed to stitch wrapped description: {e}") # Safety: skip subtotal/totals artifacts that may match loosely # e.g. a line like "Kr 3 9.048,75" (no letters) should not become a line item. try: details = line_data.get('_line_details') description = line_data.get('description') if isinstance(details, str) and details.strip() == '': continue text_to_check = None if isinstance(description, str) and description.strip() and description.strip() != '-': text_to_check = description elif isinstance(details, str) and details.strip(): text_to_check = details if isinstance(text_to_check, str): if not re.search(r'[A-Za-zÆØÅæøå]', text_to_check): continue lowered = text_to_check.lower() if lowered.startswith('kr') or ' moms' in lowered or lowered.startswith('total') or lowered.startswith('netto'): continue except Exception: pass if line_data: all_lines.append(line_data) logger.info(f" ✓ Extracted {len(all_lines)} line items") except Exception as e: logger.warning(f" ✗ Failed to extract lines: {e}") return all_lines def _validate_amounts(self, extracted: Dict) -> None: """Validate that line totals sum to subtotal/total, and VAT calculation is correct""" try: lines = extracted.get('lines', []) total_amount = extracted.get('total_amount') vat_amount = extracted.get('vat_amount') if not lines or total_amount is None: return # Calculate sum of line_total values line_sum = 0.0 for line in lines: line_total = line.get('line_total') if line_total is not None: if isinstance(line_total, str): # Parse Danish format: "25.000,00" or "1 .530,00" cleaned = line_total.replace(' ', '').replace('.', '').replace(',', '.') try: line_sum += float(cleaned) except ValueError: pass elif isinstance(line_total, (int, float)): line_sum += float(line_total) # If we have VAT amount, subtract it from total to get subtotal subtotal = total_amount if vat_amount is not None: subtotal = total_amount - vat_amount validation_details = { 'line_sum': round(line_sum, 2), 'subtotal': round(subtotal, 2), 'difference': round(abs(line_sum - subtotal), 2), 'subtotal_matches': abs(line_sum - subtotal) <= 1.0, 'vat_amount': round(float(vat_amount), 2) if vat_amount is not None else None, 'vat_expected': None, 'vat_difference': None, 'vat_matches': None, } # Check if line sum matches subtotal (allow 1 DKK difference for rounding) if abs(line_sum - subtotal) > 1.0: logger.warning(f"⚠️ Amount validation: Line sum {line_sum:.2f} != subtotal {subtotal:.2f} (diff: {abs(line_sum - subtotal):.2f})") extracted['_validation_warning'] = f"Varelinjer sum ({line_sum:.2f}) passer ikke med subtotal ({subtotal:.2f})" else: logger.info(f"✅ Amount validation: Line sum matches subtotal ({line_sum:.2f})") # Check VAT calculation (25%) if vat_amount is not None: expected_vat = subtotal * 0.25 validation_details['vat_expected'] = round(expected_vat, 2) validation_details['vat_difference'] = round(abs(vat_amount - expected_vat), 2) validation_details['vat_matches'] = abs(vat_amount - expected_vat) <= 1.0 if abs(vat_amount - expected_vat) > 1.0: logger.warning(f"⚠️ VAT validation: VAT {vat_amount:.2f} != 25% of {subtotal:.2f} ({expected_vat:.2f})") extracted['_vat_warning'] = f"Moms ({vat_amount:.2f}) passer ikke med 25% af subtotal ({expected_vat:.2f})" else: logger.info(f"✅ VAT validation: 25% VAT calculation correct ({vat_amount:.2f})") extracted['_validation_details'] = validation_details except Exception as e: logger.warning(f"⚠️ Amount validation failed: {e}") def extract(self, text: str, template_name: Optional[str] = None) -> Optional[Dict[str, Any]]: """ Extract invoice data from text If template_name is None, auto-detect template """ try: # Auto-detect template if not specified if template_name is None: template_name = self.match_template(text) if template_name is None: return None # Extract with template result = self.extract_with_template(text, template_name) logger.info(f"✅ Extracted {len(result)} fields using template: {template_name}") return result except Exception as e: logger.error(f"❌ Extraction failed: {e}") return None def get_template_list(self) -> List[Dict[str, str]]: """Get list of available templates""" return [ { 'name': name, 'issuer': template.get('issuer'), 'country': template.get('country') } for name, template in self.templates.items() ] # Singleton instance _invoice2data_service = None def get_invoice2data_service() -> Invoice2DataService: """Get singleton instance of Invoice2Data service""" global _invoice2data_service if _invoice2data_service is None: _invoice2data_service = Invoice2DataService() return _invoice2data_service