core.py 32 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761762763764765766767768769770771772773774775776777778779780781782783784785786787788789790791792793794795796797798799800801802803804805806807808809810811812813814815816817818819820821822823824825826827828829830831832833834835836837838839840841842843844845846847848849850851852853854855856857858859860861862863
  1. from __future__ import annotations
  2. import bisect
  3. import re
  4. import unicodedata
  5. import warnings
  6. from typing import Literal
  7. from . import idnadata
  8. from .intranges import intranges_contain
  9. _virama_combining_class = 9
  10. _alabel_prefix = b"xn--"
  11. _max_input_length = 1024
  12. _max_domain_length = 253 # RFC 1035 octets, excluding any trailing dot
  13. _STATUS_VALID, _STATUS_MAPPED, _STATUS_DEVIATION, _STATUS_IGNORED = b"VMDI"
  14. _unicode_dots_re = re.compile("[\u002e\u3002\uff0e\uff61]")
  15. _std3_disallowed_re = re.compile("[\x00-\x2c\x2f\x3a-\x40A-Z\x5b-\x60\x7b-\x7f]")
  16. _bidi_rtl_first = frozenset({"R", "AL"})
  17. _bidi_rtl_categories = frozenset({"R", "AL", "AN"})
  18. _bidi_rtl_allowed = frozenset({"R", "AL", "AN", "EN", "ES", "CS", "ET", "ON", "BN", "NSM"})
  19. _bidi_rtl_valid_ending = frozenset({"R", "AL", "EN", "AN"})
  20. _bidi_rtl_numeric = frozenset({"AN", "EN"})
  21. _bidi_ltr_allowed = frozenset({"L", "EN", "ES", "CS", "ET", "ON", "BN", "NSM"})
  22. _bidi_ltr_valid_ending = frozenset({"L", "EN"})
  23. _bidi_joiner_l_or_d = frozenset({"L", "D"})
  24. _bidi_joiner_r_or_d = frozenset({"R", "D"})
  25. def _joining_type(cp: int) -> str | None:
  26. for jt, ranges in idnadata.joining_types.items():
  27. if intranges_contain(cp, ranges):
  28. return jt
  29. return None
  30. # Machine-readable identifiers for the rule an :class:`IDNAError` reports.
  31. # These strings are stable and documented; exception message wording is not.
  32. _ErrorCode = Literal[
  33. "input_too_long",
  34. "label_too_long",
  35. "domain_too_long",
  36. "empty_label",
  37. "empty_domain",
  38. "not_nfc",
  39. "hyphen_3_4",
  40. "hyphen_start_end",
  41. "leading_combiner",
  42. "disallowed_codepoint",
  43. "contextj",
  44. "contexto",
  45. "unknown_codepoint",
  46. "bidi_rule_1",
  47. "bidi_rule_2",
  48. "bidi_rule_3",
  49. "bidi_rule_4",
  50. "bidi_rule_5",
  51. "bidi_rule_6",
  52. "bidi_unknown_direction",
  53. "invalid_alabel",
  54. "non_canonical_alabel",
  55. "invalid_ascii",
  56. "invalid_utf8",
  57. "uts46_disallowed",
  58. "uts46_std3",
  59. "unsupported_errors",
  60. ]
  61. class IDNAError(UnicodeError):
  62. """Base exception for all IDNA-encoding related problems.
  63. ``str(err)`` is a human-readable description of the failure. The
  64. exception also carries machine-readable attributes so callers do not
  65. need to parse the message:
  66. * ``code`` -- a short, stable identifier for the rule that failed, such
  67. as ``"disallowed_codepoint"`` or ``"bidi_rule_2"``; the full list is
  68. documented in the README. Message wording, by contrast, may change
  69. between releases.
  70. * ``text`` -- the label (or, for UTS #46 processing, the domain) that
  71. was being validated;
  72. * ``codepoint`` -- the offending codepoint, as an ``int``;
  73. * ``position`` -- the 1-based index of the offending character within
  74. ``text``, matching the position quoted in the message.
  75. Each is ``None`` when it does not apply.
  76. """
  77. code: str | None
  78. text: str | None
  79. codepoint: int | None
  80. position: int | None
  81. def __init__(
  82. self,
  83. *args: object,
  84. code: _ErrorCode | None = None,
  85. text: str | None = None,
  86. codepoint: int | None = None,
  87. position: int | None = None,
  88. ) -> None:
  89. super().__init__(*args)
  90. self.code = code
  91. self.text = text
  92. self.codepoint = codepoint
  93. self.position = position
  94. class IDNABidiError(IDNAError):
  95. """Exception when bidirectional requirements are not satisfied"""
  96. class InvalidCodepoint(IDNAError):
  97. """Exception when a disallowed or unallocated codepoint is used"""
  98. class InvalidCodepointContext(IDNAError):
  99. """Exception when the codepoint is not valid in the context it is used"""
  100. def _combining_class(cp: int) -> int:
  101. v = unicodedata.combining(chr(cp))
  102. if v == 0 and not unicodedata.name(chr(cp)):
  103. raise ValueError("Unknown character in unicodedata")
  104. return v
  105. def _is_script(cp: str, script: str) -> bool:
  106. return intranges_contain(ord(cp), idnadata.scripts[script])
  107. def _punycode(s: str) -> bytes:
  108. return s.encode("punycode")
  109. def _unot(s: int) -> str:
  110. return f"U+{s:04X}"
  111. def valid_label_length(label: bytes | str) -> bool:
  112. """Check that a label does not exceed the maximum permitted length.
  113. Per :rfc:`1035` (and :rfc:`5891` §4.2.4) a DNS label must not exceed
  114. 63 octets. The argument may be either a :class:`str` (a U-label, where
  115. length is measured in characters) or :class:`bytes` (an A-label, where
  116. length is measured in octets).
  117. :param label: The label to check.
  118. :returns: ``True`` if the label is within the length limit, otherwise
  119. ``False``.
  120. """
  121. return len(label) <= 63
  122. def valid_string_length(domain: bytes | str, trailing_dot: bool) -> bool:
  123. """Check that a full domain name does not exceed the maximum length.
  124. Per :rfc:`1035`, a domain name is limited to 253 octets when no trailing
  125. dot is present, or 254 octets when one is included.
  126. :param domain: The full (possibly multi-label) domain name.
  127. :param trailing_dot: ``True`` if ``domain`` includes a trailing ``.``.
  128. :returns: ``True`` if the domain is within the length limit, otherwise
  129. ``False``.
  130. """
  131. return len(domain) <= _max_domain_length + trailing_dot
  132. def check_bidi(label: str, check_ltr: bool = False) -> bool:
  133. """Validate the Bidi Rule from :rfc:`5893` for a single label.
  134. The Bidi Rule constrains how bidirectional characters (Hebrew, Arabic,
  135. etc.) may appear within a label. By default the check is only applied
  136. when the label contains at least one right-to-left character (Unicode
  137. bidirectional categories ``R``, ``AL``, or ``AN``); set ``check_ltr``
  138. to ``True`` to apply it to LTR-only labels as well.
  139. :param label: The label to validate, as a Unicode string.
  140. :param check_ltr: If ``True``, apply the rules even when the label
  141. contains no RTL characters.
  142. :returns: ``True`` if the label satisfies the Bidi Rule.
  143. :raises IDNABidiError: If any of Bidi Rule conditions 1-6 are violated,
  144. or if the directional category of a codepoint cannot be determined.
  145. """
  146. if len(label) > _max_input_length:
  147. raise IDNAError("Label too long", code="input_too_long")
  148. # Bidi rules should only be applied if string contains RTL characters
  149. bidi_label = False
  150. for idx, cp in enumerate(label, 1):
  151. direction = unicodedata.bidirectional(cp)
  152. if direction == "":
  153. # String likely comes from a newer version of Unicode
  154. raise IDNABidiError(
  155. f"Unknown directionality in label {label!r} at position {idx}",
  156. code="bidi_unknown_direction",
  157. text=label,
  158. codepoint=ord(cp),
  159. position=idx,
  160. )
  161. if direction in _bidi_rtl_categories:
  162. bidi_label = True
  163. if not bidi_label and not check_ltr:
  164. return True
  165. # Bidi rule 1
  166. direction = unicodedata.bidirectional(label[0])
  167. if direction in _bidi_rtl_first:
  168. rtl = True
  169. elif direction == "L":
  170. rtl = False
  171. else:
  172. raise IDNABidiError(
  173. f"First codepoint in label {label!r} must be directionality L, R or AL",
  174. code="bidi_rule_1",
  175. text=label,
  176. codepoint=ord(label[0]),
  177. position=1,
  178. )
  179. valid_ending = False
  180. ending_idx = 1
  181. number_type: str | None = None
  182. for idx, cp in enumerate(label, 1):
  183. direction = unicodedata.bidirectional(cp)
  184. if rtl:
  185. # Bidi rule 2
  186. if direction not in _bidi_rtl_allowed:
  187. raise IDNABidiError(
  188. f"Invalid direction for codepoint at position {idx} in a right-to-left label",
  189. code="bidi_rule_2",
  190. text=label,
  191. codepoint=ord(cp),
  192. position=idx,
  193. )
  194. # Bidi rule 3
  195. if direction in _bidi_rtl_valid_ending:
  196. valid_ending = True
  197. ending_idx = idx
  198. elif direction != "NSM":
  199. valid_ending = False
  200. ending_idx = idx
  201. # Bidi rule 4
  202. if direction in _bidi_rtl_numeric:
  203. if not number_type:
  204. number_type = direction
  205. elif number_type != direction:
  206. raise IDNABidiError(
  207. "Can not mix numeral types in a right-to-left label",
  208. code="bidi_rule_4",
  209. text=label,
  210. codepoint=ord(cp),
  211. position=idx,
  212. )
  213. else:
  214. # Bidi rule 5
  215. if direction not in _bidi_ltr_allowed:
  216. raise IDNABidiError(
  217. f"Invalid direction for codepoint at position {idx} in a left-to-right label",
  218. code="bidi_rule_5",
  219. text=label,
  220. codepoint=ord(cp),
  221. position=idx,
  222. )
  223. # Bidi rule 6
  224. if direction in _bidi_ltr_valid_ending:
  225. valid_ending = True
  226. ending_idx = idx
  227. elif direction != "NSM":
  228. valid_ending = False
  229. ending_idx = idx
  230. if not valid_ending:
  231. # Rules 3 and 6 concern the last character that is not a
  232. # non-spacing mark, which is what ``ending_idx`` tracks.
  233. raise IDNABidiError(
  234. "Label ends with illegal codepoint directionality",
  235. code="bidi_rule_3" if rtl else "bidi_rule_6",
  236. text=label,
  237. codepoint=ord(label[ending_idx - 1]),
  238. position=ending_idx,
  239. )
  240. return True
  241. def check_initial_combiner(label: str) -> bool:
  242. """Reject labels that begin with a combining mark.
  243. Per :rfc:`5891` §4.2.3.2 a label must not start with a character of
  244. Unicode general category ``M`` (Mark).
  245. :param label: The label to check.
  246. :returns: ``True`` if the first character is not a combining mark.
  247. :raises IDNAError: If the label begins with a combining character.
  248. """
  249. if label and unicodedata.category(label[0])[0] == "M":
  250. raise IDNAError(
  251. "Label begins with an illegal combining character",
  252. code="leading_combiner",
  253. text=label,
  254. codepoint=ord(label[0]),
  255. position=1,
  256. )
  257. return True
  258. def check_hyphen_ok(label: str) -> bool:
  259. """Validate the hyphen restrictions for a label.
  260. Per :rfc:`5891` §4.2.3.1 a label must not start or end with a hyphen
  261. (``U+002D``), and must not have hyphens in both the third and fourth
  262. positions (the prefix reserved for A-labels).
  263. :param label: The label to check.
  264. :returns: ``True`` if the hyphen restrictions are satisfied.
  265. :raises IDNAError: If any of the hyphen restrictions are violated.
  266. """
  267. if label[2:4] == "--":
  268. raise IDNAError("Label has disallowed hyphens in 3rd and 4th position", code="hyphen_3_4")
  269. if label.startswith("-") or label.endswith("-"):
  270. raise IDNAError("Label must not start or end with a hyphen", code="hyphen_start_end")
  271. return True
  272. def check_nfc(label: str) -> None:
  273. """Require that a label is in Unicode Normalization Form C.
  274. :param label: The label to check.
  275. :raises IDNAError: If ``label`` differs from its NFC normalisation.
  276. """
  277. if len(label) > _max_input_length:
  278. raise IDNAError("Label too long", code="input_too_long")
  279. if unicodedata.normalize("NFC", label) != label:
  280. raise IDNAError("Label must be in Normalization Form C", code="not_nfc")
  281. def valid_contextj(label: str, pos: int) -> bool:
  282. """Validate the CONTEXTJ rules from :rfc:`5892` Appendix A.
  283. These rules govern the contextual use of the joiner codepoints
  284. ``U+200C`` (ZERO WIDTH NON-JOINER, Appendix A.1) and ``U+200D``
  285. (ZERO WIDTH JOINER, Appendix A.2) within a label.
  286. :param label: The label containing the codepoint.
  287. :param pos: Index of the joiner codepoint within ``label``.
  288. :returns: ``True`` if the codepoint at ``pos`` satisfies its CONTEXTJ
  289. rule, ``False`` otherwise (including when the codepoint at
  290. ``pos`` is not a recognised joiner).
  291. :raises ValueError: If an adjacent codepoint has no Unicode name when
  292. determining its combining class.
  293. :raises IDNAError: If ``label`` exceeds the defensive input length limit.
  294. """
  295. if len(label) > _max_input_length:
  296. raise IDNAError("Label too long", code="input_too_long")
  297. cp_value = ord(label[pos])
  298. if cp_value == 0x200C:
  299. if pos > 0 and _combining_class(ord(label[pos - 1])) == _virama_combining_class:
  300. return True
  301. ok = False
  302. for i in range(pos - 1, -1, -1):
  303. joining_type = _joining_type(ord(label[i]))
  304. if joining_type == "T":
  305. continue
  306. if joining_type in _bidi_joiner_l_or_d:
  307. ok = True
  308. break
  309. break
  310. if not ok:
  311. return False
  312. ok = False
  313. for i in range(pos + 1, len(label)):
  314. joining_type = _joining_type(ord(label[i]))
  315. if joining_type == "T":
  316. continue
  317. if joining_type in _bidi_joiner_r_or_d:
  318. ok = True
  319. break
  320. break
  321. return ok
  322. if cp_value == 0x200D:
  323. return pos > 0 and _combining_class(ord(label[pos - 1])) == _virama_combining_class
  324. return False
  325. def valid_contexto(label: str, pos: int, exception: bool = False) -> bool:
  326. """Validate the CONTEXTO rules from :rfc:`5892` Appendix A.
  327. Covers the contextual rules for codepoints such as MIDDLE DOT
  328. (``U+00B7``), Greek lower numeral sign, Hebrew punctuation, Katakana
  329. middle dot, and the Arabic-Indic / Extended Arabic-Indic digit ranges.
  330. :param label: The label containing the codepoint.
  331. :param pos: Index of the codepoint within ``label``.
  332. :param exception: Reserved for forward compatibility; currently unused.
  333. :returns: ``True`` if the codepoint at ``pos`` satisfies its CONTEXTO
  334. rule, ``False`` otherwise (including when the codepoint is not a
  335. recognised CONTEXTO codepoint).
  336. :raises IDNAError: If ``label`` exceeds the defensive input length limit.
  337. """
  338. if len(label) > _max_input_length:
  339. raise IDNAError("Label too long", code="input_too_long")
  340. cp_value = ord(label[pos])
  341. if cp_value == 0x00B7:
  342. return 0 < pos < len(label) - 1 and ord(label[pos - 1]) == 0x006C and ord(label[pos + 1]) == 0x006C
  343. if cp_value == 0x0375:
  344. if pos < len(label) - 1 and len(label) > 1:
  345. return _is_script(label[pos + 1], "Greek")
  346. return False
  347. if cp_value in {0x05F3, 0x05F4}:
  348. if pos > 0:
  349. return _is_script(label[pos - 1], "Hebrew")
  350. return False
  351. if cp_value == 0x30FB:
  352. for cp in label:
  353. if cp == "\u30fb":
  354. continue
  355. if _is_script(cp, "Hiragana") or _is_script(cp, "Katakana") or _is_script(cp, "Han"):
  356. return True
  357. return False
  358. if 0x660 <= cp_value <= 0x669:
  359. return not any(0x6F0 <= ord(cp) <= 0x06F9 for cp in label)
  360. if 0x6F0 <= cp_value <= 0x6F9:
  361. return not any(0x660 <= ord(cp) <= 0x0669 for cp in label)
  362. return False
  363. def check_label(label: str | bytes | bytearray) -> None:
  364. """Run the full set of IDNA 2008 validity checks on a single label.
  365. Applies, in order: NFC normalisation (:func:`check_nfc`), hyphen
  366. restrictions (:func:`check_hyphen_ok`), the no-leading-combiner rule
  367. (:func:`check_initial_combiner`), per-codepoint validity (PVALID,
  368. CONTEXTJ, CONTEXTO classes from :rfc:`5892`), and the Bidi Rule
  369. (:func:`check_bidi`).
  370. :param label: The label to validate. ``bytes`` or ``bytearray`` input
  371. is decoded as UTF-8 first.
  372. :raises IDNAError: If the label is empty or fails a structural rule.
  373. :raises InvalidCodepoint: If the label contains a DISALLOWED or
  374. UNASSIGNED codepoint.
  375. :raises InvalidCodepointContext: If a CONTEXTJ or CONTEXTO codepoint
  376. is not valid in its context.
  377. :raises IDNABidiError: If the Bidi Rule is violated.
  378. """
  379. if len(label) > _max_input_length:
  380. raise IDNAError("Label too long", code="input_too_long")
  381. if isinstance(label, (bytes, bytearray)):
  382. try:
  383. label = label.decode("utf-8")
  384. except UnicodeDecodeError as err:
  385. raise IDNAError("Invalid UTF-8 in label", code="invalid_utf8") from err
  386. if len(label) == 0:
  387. raise IDNAError("Empty Label", code="empty_label")
  388. # Check against the domain length rather than the label length to
  389. # support some UTS #46 use cases, while still bounding the work done
  390. # by the label contextual rules below.
  391. if not valid_string_length(label, trailing_dot=True):
  392. raise IDNAError("Label too long", code="label_too_long")
  393. check_nfc(label)
  394. check_hyphen_ok(label)
  395. check_initial_combiner(label)
  396. for pos, cp in enumerate(label):
  397. cp_value = ord(cp)
  398. if intranges_contain(cp_value, idnadata.codepoint_classes["PVALID"]):
  399. continue
  400. if intranges_contain(cp_value, idnadata.codepoint_classes["CONTEXTJ"]):
  401. try:
  402. contextj_ok = valid_contextj(label, pos)
  403. except ValueError as err:
  404. raise IDNAError(
  405. f"Unknown codepoint adjacent to joiner {_unot(cp_value)} at position {pos + 1} in {label!r}",
  406. code="unknown_codepoint",
  407. text=label,
  408. codepoint=cp_value,
  409. position=pos + 1,
  410. ) from err
  411. if not contextj_ok:
  412. raise InvalidCodepointContext(
  413. f"Joiner {_unot(cp_value)} not allowed at position {pos + 1} in {label!r}",
  414. code="contextj",
  415. text=label,
  416. codepoint=cp_value,
  417. position=pos + 1,
  418. )
  419. elif intranges_contain(cp_value, idnadata.codepoint_classes["CONTEXTO"]):
  420. if not valid_contexto(label, pos):
  421. raise InvalidCodepointContext(
  422. f"Codepoint {_unot(cp_value)} not allowed at position {pos + 1} in {label!r}",
  423. code="contexto",
  424. text=label,
  425. codepoint=cp_value,
  426. position=pos + 1,
  427. )
  428. else:
  429. raise InvalidCodepoint(
  430. f"Codepoint {_unot(cp_value)} at position {pos + 1} of {label!r} not allowed",
  431. code="disallowed_codepoint",
  432. text=label,
  433. codepoint=cp_value,
  434. position=pos + 1,
  435. )
  436. check_bidi(label)
  437. def alabel(label: str) -> bytes:
  438. """Convert a single U-label into its A-label form.
  439. The result is the ASCII-Compatible Encoding (ACE) form per :rfc:`5891`
  440. §4: the label is validated, Punycode-encoded, and prefixed with
  441. ``xn--``. Pure ASCII labels that are already valid IDNA labels are
  442. returned unchanged (as :class:`bytes`).
  443. :param label: The label to convert, as a Unicode string.
  444. :returns: The A-label as ASCII-encoded :class:`bytes`.
  445. :raises IDNAError: If the label is invalid or the resulting A-label
  446. exceeds 63 octets.
  447. """
  448. if len(label) > _max_input_length:
  449. raise IDNAError("Label too long", code="input_too_long")
  450. try:
  451. label_bytes = label.encode("ascii")
  452. except UnicodeEncodeError:
  453. pass
  454. else:
  455. ulabel(label_bytes)
  456. if not valid_label_length(label_bytes):
  457. raise IDNAError("Label too long", code="label_too_long")
  458. return label_bytes
  459. check_label(label)
  460. label_bytes = _alabel_prefix + _punycode(label)
  461. if not valid_label_length(label_bytes):
  462. raise IDNAError("Label too long", code="label_too_long")
  463. return label_bytes
  464. def ulabel(label: str | bytes | bytearray) -> str:
  465. """Convert a single A-label into its U-label form.
  466. Performs the inverse of :func:`alabel`: an ``xn--``-prefixed label is
  467. Punycode-decoded and validated, and is rejected unless it is the
  468. canonical A-label for the decoded U-label (:rfc:`5891` §5.3). Labels
  469. that are already Unicode (or plain ASCII without the ACE prefix) are
  470. validated and returned as a Unicode string.
  471. :param label: The label to convert. ``bytes`` or ``bytearray`` input
  472. is treated as ASCII.
  473. :returns: The U-label as a Unicode string.
  474. :raises IDNAError: If the label is malformed or fails validation.
  475. """
  476. if len(label) > _max_input_length:
  477. raise IDNAError("Label too long", code="input_too_long")
  478. if not isinstance(label, (bytes, bytearray)):
  479. try:
  480. label_bytes = label.encode("ascii")
  481. except UnicodeEncodeError:
  482. check_label(label)
  483. return label
  484. else:
  485. label_bytes = bytes(label)
  486. if not label_bytes.isascii():
  487. raise IDNAError("Invalid ASCII in A-label", code="invalid_ascii")
  488. label_bytes = label_bytes.lower()
  489. if label_bytes.startswith(_alabel_prefix):
  490. label_bytes = label_bytes[len(_alabel_prefix) :]
  491. if not label_bytes:
  492. raise IDNAError("Malformed A-label, no Punycode eligible content found", code="invalid_alabel")
  493. if label_bytes.endswith(b"-"):
  494. raise IDNAError("A-label must not end with a hyphen", code="invalid_alabel")
  495. else:
  496. check_label(label_bytes)
  497. return label_bytes.decode("ascii")
  498. try:
  499. label = label_bytes.decode("punycode")
  500. except UnicodeError as err:
  501. raise IDNAError("Invalid A-label", code="invalid_alabel") from err
  502. # RFC 5891 §5.3: the label is rejected unless re-encoding the decoded
  503. # form reproduces the (lowercased) input. This catches "fake A-labels"
  504. # (RFC 5890 §2.3.2.1) such as ``xn---bbk``, a non-canonical Punycode
  505. # spelling of ``xn--bbk`` that would otherwise decode to the same
  506. # U-label and so display identically to a different wire-format name.
  507. if _punycode(label) != label_bytes:
  508. raise IDNAError("A-label is not the canonical Punycode encoding of its U-label", code="non_canonical_alabel")
  509. check_label(label)
  510. return label
  511. def _check_std3(text: str, domain: str, offset: int) -> None:
  512. """Raise if ``text``, a slice of ``domain`` starting at ``offset`` that
  513. UTS #46 mapping left unchanged, contains an ASCII character disallowed
  514. under ``UseSTD3ASCIIRules``."""
  515. match = _std3_disallowed_re.search(text)
  516. if match:
  517. codepoint = ord(match.group())
  518. position = offset + match.start() + 1
  519. raise InvalidCodepoint(
  520. f"Codepoint {_unot(codepoint)} not allowed at position {position} in {domain!r}",
  521. code="uts46_std3",
  522. text=domain,
  523. codepoint=codepoint,
  524. position=position,
  525. )
  526. def _warn_transitional() -> None:
  527. warnings.warn(
  528. "Transitional processing is deprecated in UTS #46 and has no effect. "
  529. "The transitional argument will be removed in a future version.",
  530. DeprecationWarning,
  531. stacklevel=3,
  532. )
  533. def uts46_remap(domain: str, std3_rules: bool = True, transitional: bool = False) -> str:
  534. """Apply the UTS #46 character mapping to a domain string.
  535. Implements the mapping table from `UTS #46 §4
  536. <https://www.unicode.org/reports/tr46/>`_: each character is kept,
  537. replaced, or rejected based on its status (``V``, ``M``, ``D``,
  538. ``I``, ``X``). The result is returned in Normalisation Form C.
  539. :param domain: The full domain name to remap.
  540. :param std3_rules: If ``True``, apply UTS #46's ``UseSTD3ASCIIRules``:
  541. after mapping, any ASCII character other than a lowercase letter,
  542. digit, hyphen or the label separator ``.`` is rejected, whether it
  543. appeared in the input or was produced by a mapping (e.g. U+FF01
  544. FULLWIDTH EXCLAMATION MARK maps to ``!``). If ``False``, such
  545. characters are passed through.
  546. :param transitional: Deprecated and ignored. UTS #46 deprecated
  547. transitional processing in Unicode 15.1 and deviation (status
  548. ``D``) codepoints are now always kept, so this has no effect
  549. beyond emitting a :class:`DeprecationWarning`. It will be removed
  550. in a future version.
  551. :returns: The remapped domain, in Normalisation Form C.
  552. :raises InvalidCodepoint: If the domain contains a disallowed
  553. codepoint under the chosen rules.
  554. :raises IDNAError: If ``domain`` exceeds the defensive input length limit.
  555. """
  556. if transitional:
  557. _warn_transitional()
  558. if len(domain) > _max_input_length:
  559. raise IDNAError("Domain too long", code="input_too_long")
  560. if domain.isascii():
  561. # The only ASCII mapping in UTS #46 is upper- to lowercase, and
  562. # ASCII is invariant under NFC, so lowercasing is the whole job.
  563. result = domain.lower()
  564. if std3_rules:
  565. _check_std3(result, domain, 0)
  566. return result
  567. from .uts46data import uts46_replacements, uts46_starts, uts46_statuses
  568. # ``start`` marks the run of unchanged input not yet copied; a run is
  569. # only sliced out when a character must be replaced or dropped, so the
  570. # common no-change case makes no copy. STD3 is checked per output piece
  571. # to report a violation at its input position.
  572. output: list[str] = []
  573. start = 0
  574. for pos, char in enumerate(domain):
  575. code_point = ord(char)
  576. i = code_point if code_point < 256 else bisect.bisect_right(uts46_starts, code_point) - 1
  577. status = uts46_statuses[i]
  578. # UTS #46 §4: V valid, D deviation (kept), M mapped, I ignored,
  579. # anything else disallowed.
  580. if status == _STATUS_VALID:
  581. continue
  582. if status == _STATUS_MAPPED:
  583. replacement = uts46_replacements[i]
  584. elif status == _STATUS_DEVIATION:
  585. continue
  586. elif status == _STATUS_IGNORED:
  587. replacement = None
  588. else:
  589. raise InvalidCodepoint(
  590. f"Codepoint {_unot(code_point)} not allowed at position {pos + 1} in {domain!r}",
  591. code="uts46_disallowed",
  592. text=domain,
  593. codepoint=code_point,
  594. position=pos + 1,
  595. )
  596. if start < pos:
  597. run = domain[start:pos]
  598. if std3_rules:
  599. _check_std3(run, domain, start)
  600. output.append(run)
  601. if replacement:
  602. if std3_rules and _std3_disallowed_re.search(replacement):
  603. raise InvalidCodepoint(
  604. f"Codepoint {_unot(code_point)} not allowed at position {pos + 1} in {domain!r}",
  605. code="uts46_std3",
  606. text=domain,
  607. codepoint=code_point,
  608. position=pos + 1,
  609. )
  610. output.append(replacement)
  611. start = pos + 1
  612. if start == 0:
  613. if std3_rules:
  614. _check_std3(domain, domain, 0)
  615. return unicodedata.normalize("NFC", domain)
  616. tail = domain[start:]
  617. if std3_rules:
  618. _check_std3(tail, domain, start)
  619. output.append(tail)
  620. return unicodedata.normalize("NFC", "".join(output))
  621. def encode(
  622. s: str | bytes | bytearray,
  623. strict: bool = False,
  624. uts46: bool = False,
  625. std3_rules: bool = False,
  626. transitional: bool = False,
  627. ) -> bytes:
  628. """Encode a Unicode domain name into its ASCII (A-label) form.
  629. Splits the input on label separators (only ``U+002E`` if ``strict`` is
  630. set; otherwise also IDEOGRAPHIC FULL STOP ``U+3002``, FULLWIDTH FULL
  631. STOP ``U+FF0E``, and HALFWIDTH IDEOGRAPHIC FULL STOP ``U+FF61``),
  632. encodes each label with :func:`alabel`, and rejoins them with ``.``.
  633. Optionally pre-processes the input through :func:`uts46_remap`.
  634. :param s: The domain name to encode.
  635. :param strict: If ``True``, only ``U+002E`` is recognised as a label
  636. separator.
  637. :param uts46: If ``True``, apply UTS #46 mapping before encoding.
  638. :param std3_rules: Forwarded to :func:`uts46_remap` when ``uts46`` is
  639. ``True``.
  640. :param transitional: Deprecated and ignored (see :func:`uts46_remap`):
  641. emits a :class:`DeprecationWarning` and will be removed in a
  642. future version.
  643. :returns: The encoded domain as ASCII :class:`bytes`.
  644. :raises IDNAError: If the domain is empty, contains an invalid label,
  645. or exceeds the maximum domain length.
  646. """
  647. if transitional:
  648. _warn_transitional()
  649. if not isinstance(s, str):
  650. try:
  651. s = str(s, "ascii")
  652. except (UnicodeDecodeError, TypeError) as err:
  653. raise IDNAError(
  654. "should pass a unicode string to the function rather than a byte string.", code="invalid_ascii"
  655. ) from err
  656. if len(s) > _max_input_length:
  657. raise IDNAError("Domain too long", code="input_too_long")
  658. if uts46:
  659. s = uts46_remap(s, std3_rules)
  660. if not valid_string_length(s, trailing_dot=True):
  661. raise IDNAError("Domain too long", code="domain_too_long")
  662. trailing_dot = False
  663. result = []
  664. labels = s.split(".") if strict else _unicode_dots_re.split(s)
  665. if not labels or labels == [""]:
  666. raise IDNAError("Empty domain", code="empty_domain")
  667. if labels[-1] == "":
  668. del labels[-1]
  669. trailing_dot = True
  670. for label in labels:
  671. s = alabel(label)
  672. if s:
  673. result.append(s)
  674. else:
  675. raise IDNAError("Empty label", code="empty_label")
  676. if trailing_dot:
  677. result.append(b"")
  678. s = b".".join(result)
  679. if not valid_string_length(s, trailing_dot):
  680. raise IDNAError("Domain too long", code="domain_too_long")
  681. return s
  682. def decode(
  683. s: str | bytes | bytearray,
  684. strict: bool = False,
  685. uts46: bool = False,
  686. std3_rules: bool = False,
  687. display: bool = False,
  688. ) -> str:
  689. """Decode an A-label-encoded domain name back to Unicode.
  690. Splits the input on label separators (see :func:`encode` for the
  691. rules), decodes each label with :func:`ulabel`, and rejoins them
  692. with ``.``. Optionally pre-processes the input through
  693. :func:`uts46_remap`.
  694. :param s: The domain name to decode.
  695. :param strict: If ``True``, only ``U+002E`` is recognised as a label
  696. separator.
  697. :param uts46: If ``True``, apply UTS #46 mapping before decoding.
  698. :param std3_rules: Forwarded to :func:`uts46_remap` when ``uts46`` is
  699. ``True``.
  700. :param display: If ``True``, any ``xn--`` label that fails IDNA
  701. validation is passed through unchanged (lowercased) rather than
  702. aborting the whole call. Intended for "decode for display"
  703. consumers (e.g. URL libraries, HTTP clients) that want to show
  704. the user the label as it appears on the wire when it cannot be
  705. rendered as Unicode. Matches the per-label recovery prescribed
  706. by UTS #46 §4 and the WHATWG URL "domain to Unicode" algorithm.
  707. :returns: The decoded domain as a Unicode string.
  708. :raises IDNAError: If the input is not valid ASCII, contains an
  709. invalid label, or is empty.
  710. """
  711. if not isinstance(s, str):
  712. try:
  713. s = str(s, "ascii")
  714. except (UnicodeDecodeError, TypeError) as err:
  715. raise IDNAError("Invalid ASCII in A-label", code="invalid_ascii") from err
  716. if len(s) > _max_input_length:
  717. raise IDNAError("Domain too long", code="input_too_long")
  718. if uts46:
  719. s = uts46_remap(s, std3_rules, False)
  720. if not valid_string_length(s, trailing_dot=True):
  721. raise IDNAError("Domain too long", code="domain_too_long")
  722. trailing_dot = False
  723. result = []
  724. labels = s.split(".") if strict else _unicode_dots_re.split(s)
  725. if not labels or labels == [""]:
  726. raise IDNAError("Empty domain", code="empty_domain")
  727. if not labels[-1]:
  728. del labels[-1]
  729. trailing_dot = True
  730. for label in labels:
  731. try:
  732. u = ulabel(label)
  733. except IDNAError:
  734. if display and label[:4].lower() == "xn--":
  735. u = label.lower()
  736. else:
  737. raise
  738. if u:
  739. result.append(u)
  740. else:
  741. raise IDNAError("Empty label", code="empty_label")
  742. if trailing_dot:
  743. result.append("")
  744. return ".".join(result)