Merge branch 'master' into avoid-setuppy-run

executablebooks · Dec 1, 2021 · 1850483 · 1850483
2 parents be41341 + bb6cf6e
commit 1850483
Show file tree

Hide file tree

Showing 18 changed files with 231 additions and 200 deletions.
diff --git a/docs/conf.py b/docs/conf.py
@@ -10,6 +10,7 @@
 # add these directories to sys.path here. If the directory is relative to the
 # documentation root, use os.path.abspath to make it absolute, like shown here.
 #
+from glob import glob
 import os
 
 # import sys
@@ -103,6 +104,15 @@ def run_apidoc(app):
     ignore_paths = [
         os.path.normpath(os.path.join(this_folder, p)) for p in ignore_paths
     ]
+    # functions from these modules are all imported in the __init__.py with __all__
+    for rule in ("block", "core", "inline"):
+        for path in glob(
+            os.path.normpath(
+                os.path.join(this_folder, f"../markdown_it/rules_{rule}/*.py")
+            )
+        ):
+            if os.path.basename(path) not in ("__init__.py", f"state_{rule}.py"):
+                ignore_paths.append(path)
 
     if os.path.exists(api_folder):
         shutil.rmtree(api_folder)

diff --git a/markdown_it/__init__.py b/markdown_it/__init__.py
@@ -1,4 +1,4 @@
-from .main import MarkdownIt  # noqa: F401
-
-
+__all__ = ("MarkdownIt",)
 __version__ = "1.1.0"
+
+from .main import MarkdownIt
diff --git a/markdown_it/_punycode.py b/markdown_it/_punycode.py
@@ -0,0 +1,66 @@
+# Copyright 2014 Mathias Bynens <https://mathiasbynens.be/>
+# Copyright 2021 Taneli Hukkinen
+#
+# Permission is hereby granted, free of charge, to any person obtaining
+# a copy of this software and associated documentation files (the
+# "Software"), to deal in the Software without restriction, including
+# without limitation the rights to use, copy, modify, merge, publish,
+# distribute, sublicense, and/or sell copies of the Software, and to
+# permit persons to whom the Software is furnished to do so, subject to
+# the following conditions:
+#
+# The above copyright notice and this permission notice shall be
+# included in all copies or substantial portions of the Software.
+#
+# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND,
+# EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF
+# MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND
+# NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT HOLDERS BE
+# LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER IN AN ACTION
+# OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN CONNECTION
+# WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
+
+import codecs
+import re
+
+REGEX_SEPARATORS = re.compile(r"[\x2E\u3002\uFF0E\uFF61]")
+REGEX_NON_ASCII = re.compile(r"[^\0-\x7E]")
+
+
+def encode(uni: str) -> str:
+    return codecs.encode(uni, encoding="punycode").decode()
+
+
+def decode(ascii: str) -> str:
+    return codecs.decode(ascii, encoding="punycode")  # type: ignore[call-overload]
+
+
+def map_domain(string, fn):
+    parts = string.split("@")
+    result = ""
+    if len(parts) > 1:
+        # In email addresses, only the domain name should be punycoded. Leave
+        # the local part (i.e. everything up to `@`) intact.
+        result = parts[0] + "@"
+        string = parts[1]
+    labels = REGEX_SEPARATORS.split(string)
+    encoded = ".".join(fn(label) for label in labels)
+    return result + encoded
+
+
+def to_unicode(obj: str) -> str:
+    def mapping(obj: str) -> str:
+        if obj.startswith("xn--"):
+            return decode(obj[4:].lower())
+        return obj
+
+    return map_domain(obj, mapping)
+
+
+def to_ascii(obj: str) -> str:
+    def mapping(obj: str) -> str:
+        if REGEX_NON_ASCII.search(obj):
+            return "xn--" + encode(obj)
+        return obj
+
+    return map_domain(obj, mapping)
diff --git a/markdown_it/common/normalize_url.py b/markdown_it/common/normalize_url.py
@@ -1,70 +1,13 @@
-import html
 import re
 from typing import Callable, Optional
 from urllib.parse import urlparse, urlunparse, quote, unquote  # noqa: F401
 
-from .utils import ESCAPABLE
+import mdurl
 
-# TODO below we port the use of the JS packages:
-# var mdurl        = require('mdurl')
-# var punycode     = require('punycode')
-#
-# e.g. mdurl: parsed = mdurl.parse(url, True)
-#
-# but need to check these fixes from https://www.npmjs.com/package/mdurl:
-#
-# Parse url string. Similar to node's url.parse,
-# but without any normalizations and query string parse.
-#    url - input url (string)
-#    slashesDenoteHost - if url starts with //, expect a hostname after it. Optional, false.
-# Difference with node's url:
+from .. import _punycode
 
-# No leading slash in paths, e.g. in url.parse('http://foo?bar') pathname is ``, not /
-# Backslashes are not replaced with slashes, so http:\\example.org\ is treated like a relative path
-# Trailing colon is treated like a part of the path, i.e. in http://example.org:foo pathname is :foo
-# Nothing is URL-encoded in the resulting object,
-# (in joyent/node some chars in auth and paths are encoded)
-# url.parse() does not have parseQueryString argument
-# Removed extraneous result properties: host, path, query, etc.,
-# which can be constructed using other parts of the url.
 
-
-#  ################# Copied from Commonmark.py #################
-
-ENTITY = "&(?:#x[a-f0-9]{1,6}|#[0-9]{1,7}|[a-z][a-z0-9]{1,31});"
-reBackslashOrAmp = re.compile(r"[\\&]")
-reEntityOrEscapedChar = re.compile(
-    "\\\\" + "[" + ESCAPABLE + "]|" + ENTITY, re.IGNORECASE
-)
-
-
-def unescape_char(s: str) -> str:
-    if s[0] == "\\":
-        return s[1]
-    else:
-        return html.unescape(s)
-
-
-def unescape_string(s: str) -> str:
-    """Replace entities and backslash escapes with literal characters."""
-    if re.search(reBackslashOrAmp, s):
-        return re.sub(reEntityOrEscapedChar, lambda m: unescape_char(m.group()), s)
-    else:
-        return s
-
-
-def normalize_uri(uri: str) -> str:
-    return quote(uri, safe="/@:+?=&()%#*,")
-
-
-##################
-
-
-RECODE_HOSTNAME_FOR = ("http", "https", "mailto")
-
-
-def unescape_normalize_uri(x: str) -> str:
-    return normalize_uri(unescape_string(x))
+RECODE_HOSTNAME_FOR = ("http:", "https:", "mailto:")
 
 
 def normalizeLink(url: str) -> str:
@@ -75,91 +18,49 @@ def normalizeLink(url: str) -> str:
         [label]:   destination   'title'
                 ^^^^^^^^^^^
     """
-    (scheme, netloc, path, params, query, fragment) = urlparse(url)
-    if scheme in RECODE_HOSTNAME_FOR:
-        url = urlunparse(
-            (
-                scheme,
-                unescape_normalize_uri(netloc),
-                normalize_uri(path),
-                unescape_normalize_uri(params),
-                normalize_uri(query),
-                unescape_normalize_uri(fragment),
-            )
-        )
-    else:
-        url = unescape_normalize_uri(url)
-
-    return url
-
-    # TODO the selective encoding below should probably be done here,
-    # something like:
-    # url_check = urllib.parse.urlparse(destination)
-    # if url_check.scheme in RECODE_HOSTNAME_FOR: ...
-
-    # parsed = urlparse(url)
-    # if parsed.hostname:
-    #     # Encode hostnames in urls like:
-    #     # `http:#host/`, `https:#host/`, `mailto:user@host`, `#host/`
-    #     #
-    #     # We don't encode unknown schemas, because it's likely that we encode
-    #     # something we shouldn't (e.g. `skype:name` treated as `skype:host`)
-    #     #
-    #     if (not parsed.scheme) or parsed.scheme in RECODE_HOSTNAME_FOR:
-    #         try:
-    #             parsed.hostname = punycode.toASCII(parsed.hostname)
-    #         except Exception:
-    #             pass
-    # return quote(urlunparse(parsed))
-
-
-def unescape_unquote(x: str) -> str:
-    return unquote(unescape_string(x))
-
-
-def normalizeLinkText(link: str) -> str:
+    parsed = mdurl.parse(url, slashes_denote_host=True)
+
+    if parsed.hostname:
+        # Encode hostnames in urls like:
+        # `http://host/`, `https://host/`, `mailto:user@host`, `//host/`
+        #
+        # We don't encode unknown schemas, because it's likely that we encode
+        # something we shouldn't (e.g. `skype:name` treated as `skype:host`)
+        #
+        if not parsed.protocol or parsed.protocol in RECODE_HOSTNAME_FOR:
+            try:
+                parsed = parsed._replace(hostname=_punycode.to_ascii(parsed.hostname))
+            except Exception:
+                pass
+
+    return mdurl.encode(mdurl.format(parsed))
+
+
+def normalizeLinkText(url: str) -> str:
     """Normalize autolink content
 
     ::
 
         <destination>
          ~~~~~~~~~~~
     """
-    (scheme, netloc, path, params, query, fragment) = urlparse(link)
-    if scheme in RECODE_HOSTNAME_FOR:
-        url = urlunparse(
-            (
-                scheme,
-                unescape_unquote(netloc),
-                unquote(path),
-                unescape_unquote(params),
-                unquote(query),
-                unescape_unquote(fragment),
-            )
-        )
-    else:
-        url = unescape_unquote(link)
-    return url
-
-    # TODO the selective encoding below should probably be done here,
-    # something like:
-    # url_check = urllib.parse.urlparse(destination)
-    # if url_check.scheme in RECODE_HOSTNAME_FOR: ...
-
-    # parsed = urlparse(url)
-    # if parsed.hostname:
-    #     # Encode hostnames in urls like:
-    #     # `http:#host/`, `https:#host/`, `mailto:user@host`, `#host/`
-    #     #
-    #     # We don't encode unknown schemas, because it's likely that we encode
-    #     # something we shouldn't (e.g. `skype:name` treated as `skype:host`)
-    #     #
-    #     if (not parsed.protocol) or parsed.protocol in RECODE_HOSTNAME_FOR:
-    #         try:
-    #             parsed.hostname = punycode.toUnicode(parsed.hostname)
-    #         except Exception:
-    #             pass
-    # return unquote(urlunparse(parsed))
+    parsed = mdurl.parse(url, slashes_denote_host=True)
+
+    if parsed.hostname:
+        # Encode hostnames in urls like:
+        # `http://host/`, `https://host/`, `mailto:user@host`, `//host/`
+        #
+        # We don't encode unknown schemas, because it's likely that we encode
+        # something we shouldn't (e.g. `skype:name` treated as `skype:host`)
+        #
+        if not parsed.protocol or parsed.protocol in RECODE_HOSTNAME_FOR:
+            try:
+                parsed = parsed._replace(hostname=_punycode.to_unicode(parsed.hostname))
+            except Exception:
+                pass
+
+    # add '%' to exclude list because of https://github.com/markdown-it/markdown-it/issues/720
+    return mdurl.decode(mdurl.format(parsed), mdurl.DECODE_DEFAULT_CHARS + "%")
 
 
 BAD_PROTO_RE = re.compile(r"^(vbscript|javascript|file|data):")

diff --git a/markdown_it/common/utils.py b/markdown_it/common/utils.py
@@ -6,8 +6,6 @@
 
 from .entities import entities
 
-# from .normalize_url import unescape_string
-
 
 def charCodeAt(src: str, pos: int) -> Any:
     """
@@ -105,7 +103,7 @@ def fromCodePoint(c: int) -> str:
 UNESCAPE_MD_RE = re.compile(r'\\([!"#$%&\'()*+,\-.\/:;<=>?@[\\\]^_`{|}~])')
 # ENTITY_RE_g       = re.compile(r'&([a-z#][a-z0-9]{1,31})', re.IGNORECASE)
 UNESCAPE_ALL_RE = re.compile(
-    r'\\([!"#$%&\'()*+,\-.\/:;<=>?@[\\\]^_`{|}~])' + "|" + r"&([a-z#][a-z0-9]{1,31})",
+    r'\\([!"#$%&\'()*+,\-.\/:;<=>?@[\\\]^_`{|}~])' + "|" + r"&([a-z#][a-z0-9]{1,31});",
     re.IGNORECASE,
 )
 DIGITAL_ENTITY_TEST_RE = re.compile(r"^#((?:x[a-f0-9]{1,8}|[0-9]{1,8}))", re.IGNORECASE)
@@ -146,7 +144,16 @@ def unescapeMd(string: str) -> str:
 
 
 def unescapeAll(string: str) -> str:
-    return html.unescape(string)
+    def replacer_func(match):
+        escaped = match.group(1)
+        if escaped:
+            return escaped
+        entity = match.group(2)
+        return replaceEntityPattern(match.group(), entity)
+
+    if "\\" not in string and "&" not in string:
+        return string
+    return UNESCAPE_ALL_RE.sub(replacer_func, string)
 
 
 ESCAPABLE = r"""\\!"#$%&'()*+,./:;<=>?@\[\]^`{}|_~-"""

diff --git a/markdown_it/helpers/__init__.py b/markdown_it/helpers/__init__.py
@@ -1,5 +1,6 @@
 """Functions for parsing Links
 """
-from .parse_link_label import parseLinkLabel  # noqa: F401
-from .parse_link_destination import parseLinkDestination  # noqa: F401
-from .parse_link_title import parseLinkTitle  # noqa: F401
+__all__ = ("parseLinkLabel", "parseLinkDestination", "parseLinkTitle")
+from .parse_link_label import parseLinkLabel
+from .parse_link_destination import parseLinkDestination
+from .parse_link_title import parseLinkTitle
diff --git a/markdown_it/helpers/parse_link_title.py b/markdown_it/helpers/parse_link_title.py
@@ -1,6 +1,6 @@
 """Parse link title
 """
-from ..common.utils import unescapeAll, charCodeAt, stripEscape
+from ..common.utils import unescapeAll, charCodeAt
 
 
 class _Result:
@@ -40,7 +40,7 @@ def parseLinkTitle(string: str, pos: int, maximum: int) -> _Result:
         code = charCodeAt(string, pos)
         if code == marker:
             title = string[start + 1 : pos]
-            title = unescapeAll(stripEscape(title))
+            title = unescapeAll(title)
             result.pos = pos + 1
             result.lines = lines
             result.str = title

diff --git a/markdown_it/port.yaml b/markdown_it/port.yaml
@@ -1,7 +1,7 @@
 - package: markdown-it/markdown-it
   version: 12.1.0
-  commit: 13cdeb95abccc78a5ce17acf9f6e8cf5b9ce713b
-  date: Jul 1, 2021
+  commit: e5986bb7cca20ac95dc81e4741c08949bf01bb77
+  date: Jul 15, 2021
   notes:
     - Rename variables that use python built-in names, e.g.
       - `max` -> `maximum`

diff --git a/markdown_it/presets/__init__.py b/markdown_it/presets/__init__.py
@@ -1,4 +1,6 @@
-from . import commonmark, default, zero  # noqa: F401
+__all__ = ("commonmark", "default", "zero", "js_default", "gfm_like")
+
+from . import commonmark, default, zero
 
 js_default = default