from __future__ import annotations import ipaddress import logging import re log = logging.getLogger("xray-lists.normalize") _DOMAIN_RE = re.compile(r"^(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,63}$") def normalize_ip_lines(text: str, source_label: str = "") -> set[str]: out: set[str] = set() for raw_line in text.splitlines(): line = raw_line.strip().rstrip("\r") if not line or line.startswith("#"): continue token = line.split()[0] try: if "/" in token: net = ipaddress.ip_network(token, strict=False) else: net = ipaddress.ip_network(f"{token}/32", strict=False) except ValueError: log.warning("%s: skipping invalid ip/cidr line: %r", source_label, line) continue out.add(str(net)) return out _SKIP_PREFIXES = ("keyword:", "regexp:", "include:") _STRIP_PREFIXES = ("full:", "domain:") def normalize_domain_lines(text: str, source_label: str = "") -> set[str]: out: set[str] = set() for raw_line in text.splitlines(): line = raw_line.strip().rstrip("\r").lower() if not line or line.startswith("#"): continue token = line.split()[0] if token.startswith(_SKIP_PREFIXES): continue for pfx in _STRIP_PREFIXES: if token.startswith(pfx): token = token[len(pfx):] break token = token.lstrip(".") if not _DOMAIN_RE.match(token): log.warning("%s: skipping invalid domain line: %r", source_label, line) continue out.add(token) return out def sort_ips(cidrs: set[str]) -> list[str]: return sorted(cidrs, key=lambda c: ipaddress.ip_network(c)) def sort_domains(domains: set[str]) -> list[str]: return sorted(domains)