Files
2026-08-16 23:21:59 +00:00

63 lines
1.8 KiB
Python

from __future__ import annotations
import ipaddress
import logging
import re
log = logging.getLogger("xray-lists.normalize")
_DOMAIN_RE = re.compile(r"^(?:[a-z0-9](?:[a-z0-9-]{0,61}[a-z0-9])?\.)+[a-z]{2,63}$")
def normalize_ip_lines(text: str, source_label: str = "") -> set[str]:
out: set[str] = set()
for raw_line in text.splitlines():
line = raw_line.strip().rstrip("\r")
if not line or line.startswith("#"):
continue
token = line.split()[0]
try:
if "/" in token:
net = ipaddress.ip_network(token, strict=False)
else:
net = ipaddress.ip_network(f"{token}/32", strict=False)
except ValueError:
log.warning("%s: skipping invalid ip/cidr line: %r", source_label, line)
continue
out.add(str(net))
return out
_SKIP_PREFIXES = ("keyword:", "regexp:", "include:")
_STRIP_PREFIXES = ("full:", "domain:")
def normalize_domain_lines(text: str, source_label: str = "") -> set[str]:
out: set[str] = set()
for raw_line in text.splitlines():
line = raw_line.strip().rstrip("\r").lower()
if not line or line.startswith("#"):
continue
token = line.split()[0]
if token.startswith(_SKIP_PREFIXES):
continue
for pfx in _STRIP_PREFIXES:
if token.startswith(pfx):
token = token[len(pfx):]
break
token = token.lstrip(".")
if not _DOMAIN_RE.match(token):
log.warning("%s: skipping invalid domain line: %r", source_label, line)
continue
out.add(token)
return out
def sort_ips(cidrs: set[str]) -> list[str]:
return sorted(cidrs, key=lambda c: ipaddress.ip_network(c))
def sort_domains(domains: set[str]) -> list[str]:
return sorted(domains)