Merge branch 'develop'

2021-07-11 13:21:20 +02:00 · 2021-07-11 13:14:29 +02:00 · 2021-05-30 19:10:49 +02:00 · 2021-05-30 11:20:32 +02:00 · 2021-05-21 14:18:14 +02:00 · 2021-05-18 10:42:27 +02:00
9 changed files with 25 additions and 146 deletions
--- a/.gitignore
+++ b/.gitignore
@@ -8,4 +8,3 @@
 /MANIFEST
 /venv
 build/
-.vscode/settings.json
--- a/README.rst
+++ b/README.rst
@@ -32,14 +32,14 @@ Convert some HTML to Markdown:
    from markdownify import markdownify as md
    md('<b>Yay</b> <a href="http://github.com">GitHub</a>')  # > '**Yay** [GitHub](http://github.com)'

-Specify tags to exclude:
+Specify tags to exclude (blacklist):

 .. code:: python

    from markdownify import markdownify as md
    md('<b>Yay</b> <a href="http://github.com">GitHub</a>', strip=['a'])  # > '**Yay** GitHub'

-\...or specify the tags you want to include:
+\...or specify the tags you want to include (whitelist):

 .. code:: python

@@ -53,11 +53,11 @@ Options
 Markdownify supports the following options:

 strip
-  A list of tags to strip. This option can't be used with the
+  A list of tags to strip (blacklist). This option can't be used with the
  ``convert`` option.

 convert
-  A list of tags to convert. This option can't be used with the
+  A list of tags to convert (whitelist). This option can't be used with the
  ``strip`` option.

 autolinks
@@ -96,56 +96,10 @@ newline_style
  newline). While the latter convention is non-standard, it is commonly
  preferred and supported by a lot of interpreters.

-code_language
-  Defines the language that should be assumed for all ``<pre>`` sections.
-  Useful, if all code on a page is in the same programming language and
-  should be annotated with `````python`` or similar.
-  Defaults to ``''`` (empty string) and can be any string.
-
-code_language_callback
-  When the HTML code contains ``pre`` tags that in some way provide the code
-  language, for example as class, this callback can be used to extract the
-  language from the tag and prefix it to the converted ``pre`` tag.
-  The callback gets one single argument, an BeautifylSoup object, and returns
-  a string containing the code language, or ``None``.
-  An example to use the class name as code language could be::
-
-    def callback(el):
-        return el['class'][0] if el.has_attr('class') else None
-
-  Defaults to ``None``.
-
-escape_asterisks
-  If set to ``False``, do not escape ``*`` to ``\*`` in text.
-  Defaults to ``True``.
-
-escape_underscores
-  If set to ``False``, do not escape ``_`` to ``\_`` in text.
-  Defaults to ``True``.
-
-keep_inline_images_in
-  Images are converted to their alt-text when the images are located inside
-  headlines or table cells. If some inline images should be converted to
-  markdown images instead, this option can be set to a list of parent tags
-  that should be allowed to contain inline images, for example ``['td']``.
-  Defaults to an empty list.
-
 Options may be specified as kwargs to the ``markdownify`` function, or as a
 nested ``Options`` class in ``MarkdownConverter`` subclasses.


-Converting BeautifulSoup objects
-================================
-
-.. code:: python
-
-    from markdownify import MarkdownConverter
-
-    # Create shorthand method for conversion
-    def md(soup, **options):
-        return MarkdownConverter(**options).convert_soup(soup)
-
-
 Creating Custom Converters
 ==========================

--- a/markdownify/init.py
+++ b/markdownify/init.py
@@ -25,6 +25,12 @@ ASTERISK = '*'
 UNDERSCORE = '_'


+def escape(text):
+    if not text:
+        return ''
+    return text.replace('_', r'\_')
+
+
 def chomp(text):
    """
    If the text in an inline tag like b, a, or em contains a leading or trailing
@@ -62,14 +68,9 @@ class MarkdownConverter(object):
    class DefaultOptions:
        autolinks = True
        bullets = '*+-'  # An iterable of bullet types.
-        code_language = ''
-        code_language_callback = None
        convert = None
        default_title = False
-        escape_asterisks = True
-        escape_underscores = True
        heading_style = UNDERLINED
-        keep_inline_images_in = []
        newline_style = SPACES
        strip = None
        strong_em_symbol = ASTERISK
@@ -91,21 +92,15 @@ class MarkdownConverter(object):

    def convert(self, html):
        soup = BeautifulSoup(html, 'html.parser')
-        return self.convert_soup(soup)
-
-    def convert_soup(self, soup):
        return self.process_tag(soup, convert_as_inline=False, children_only=True)

    def process_tag(self, node, convert_as_inline, children_only=False):
        text = ''
-
-        # markdown headings or cells can't include
-        # block elements (elements w/newlines)
+        # markdown headings can't include block elements (elements w/newlines)
        isHeading = html_heading_re.match(node.name) is not None
-        isCell = node.name in ['td', 'th']
        convert_children_as_inline = convert_as_inline

-        if not children_only and (isHeading or isCell):
+        if not children_only and isHeading:
            convert_children_as_inline = True

        # Remove whitespace-only textnodes in purely nested nodes
@@ -156,7 +151,7 @@ class MarkdownConverter(object):
            text = whitespace_re.sub(' ', text)

        if el.parent.name != 'code':
-            text = self.escape(text)
+            text = escape(text)

        # remove trailing whitespaces if any of the following condition is true:
        # - current text node is the last node in li
@@ -194,15 +189,6 @@ class MarkdownConverter(object):
        else:
            return True

-    def escape(self, text):
-        if not text:
-            return ''
-        if self.options['escape_asterisks']:
-            text = text.replace('*', r'\*')
-        if self.options['escape_underscores']:
-            text = text.replace('_', r'\_')
-        return text
-
    def indent(self, text, level):
        return line_beginning_re.sub('\t' * level, text) if text else ''

@@ -214,6 +200,8 @@ class MarkdownConverter(object):
        prefix, suffix, text = chomp(text)
        if not text:
            return ''
+        if convert_as_inline:
+            return text
        href = el.get('href')
        title = el.get('title')
        # For the replacement see #29: text nodes underscores are escaped
@@ -282,8 +270,7 @@ class MarkdownConverter(object):
        src = el.attrs.get('src', None) or ''
        title = el.attrs.get('title', None) or ''
        title_part = ' "%s"' % title.replace('"', r'\"') if title else ''
-        if (convert_as_inline
-                and el.parent.name not in self.options['keep_inline_images_in']):
+        if convert_as_inline:
            return alt

        return '![%s](%s%s)' % (alt, src, title_part)
@@ -326,7 +313,7 @@ class MarkdownConverter(object):
                el = el.parent
            bullets = self.options['bullets']
            bullet = bullets[depth % len(bullets)]
-        return '%s %s\n' % (bullet, (text or '').strip())
+        return '%s %s\n' % (bullet, text or '')

    def convert_p(self, el, text, convert_as_inline):
        if convert_as_inline:
@@ -336,12 +323,7 @@ class MarkdownConverter(object):
    def convert_pre(self, el, text, convert_as_inline):
        if not text:
            return ''
-        code_language = self.options['code_language']
-
-        if self.options['code_language_callback']:
-            code_language = self.options['code_language_callback'](el) or code_language
-
-        return '\n```%s\n%s\n```\n' % (code_language, text)
+        return '\n```\n%s\n```\n' % text

    convert_s = convert_del

--- a/setup.py
+++ b/setup.py
@@ -10,7 +10,7 @@ read = lambda filepath: codecs.open(filepath, 'r', 'utf-8').read()
 pkgmeta = {
    '__title__': 'markdownify',
    '__author__': 'Matthew Tretter',
-    '__version__': '0.10.3',
+    '__version__': '0.9.2',
 }


@@ -70,7 +70,7 @@ setup(
    zip_safe=False,
    include_package_data=True,
    setup_requires=[
-        'flake8>=3.8,<5',
+        'flake8>=3.8,<4',
    ],
    tests_require=[
        'pytest>=6.2,<7',
--- a/tests/test_conversions.py
+++ b/tests/test_conversions.py
@@ -133,13 +133,12 @@ def test_hn_nested_simple_tag():

 def test_hn_nested_img():
    image_attributes_to_markdown = [
-        ("", "", ""),
-        ("alt='Alt Text'", "Alt Text", ""),
-        ("alt='Alt Text' title='Optional title'", "Alt Text", " \"Optional title\""),
+        ("", ""),
+        ("alt='Alt Text'", "Alt Text"),
+        ("alt='Alt Text' title='Optional title'", "Alt Text"),
    ]
-    for image_attributes, markdown, title in image_attributes_to_markdown:
-        assert md('<h3>A <img src="/path/to/img.jpg" ' + image_attributes + '/> B</h3>') == '### A ' + markdown + ' B\n\n'
-        assert md('<h3>A <img src="/path/to/img.jpg" ' + image_attributes + '/> B</h3>', keep_inline_images_in=['h3']) == '### A ![' + markdown + '](/path/to/img.jpg' + title + ') B\n\n'
+    for image_attributes, markdown in image_attributes_to_markdown:
+        assert md('<h3>A <img src="/path/to/img.jpg " ' + image_attributes + '/> B</h3>') == '### A ' + markdown + ' B\n\n'


 def test_hn_atx_headings():
@@ -211,17 +210,3 @@ def test_sub():
 def test_sup():
    assert md('<sup>foo</sup>') == 'foo'
    assert md('<sup>foo</sup>', sup_symbol='^') == '^foo^'
-
-
-def test_lang():
-    assert md('<pre>test\n    foo\nbar</pre>', code_language='python') == '\n```python\ntest\n    foo\nbar\n```\n'
-    assert md('<pre><code>test\n    foo\nbar</code></pre>', code_language='javascript') == '\n```javascript\ntest\n    foo\nbar\n```\n'
-
-
-def test_lang_callback():
-    def callback(el):
-        return el['class'][0] if el.has_attr('class') else None
-
-    assert md('<pre class="python">test\n    foo\nbar</pre>', code_language_callback=callback) == '\n```python\ntest\n    foo\nbar\n```\n'
-    assert md('<pre class="javascript"><code>test\n    foo\nbar</code></pre>', code_language_callback=callback) == '\n```javascript\ntest\n    foo\nbar\n```\n'
-    assert md('<pre class="javascript"><code class="javascript">test\n    foo\nbar</code></pre>', code_language_callback=callback) == '\n```javascript\ntest\n    foo\nbar\n```\n'
--- a/tests/test_custom_converter.py
+++ b/tests/test_custom_converter.py
@@ -1,5 +1,4 @@
 from markdownify import MarkdownConverter
-from bs4 import BeautifulSoup


 class ImageBlockConverter(MarkdownConverter):
@@ -17,9 +16,3 @@ def test_img():

    assert md('<img src="/path/to/img.jpg" alt="Alt text" title="Optional title" />') == '![Alt text](/path/to/img.jpg "Optional title")\n\n'
    assert md('<img src="/path/to/img.jpg" alt="Alt text" />') == '![Alt text](/path/to/img.jpg)\n\n'
-
-
-def test_soup():
-    html = '<b>test</b>'
-    soup = BeautifulSoup(html, 'html.parser')
-    assert MarkdownConverter().convert_soup(soup) == '**test**'
--- a/tests/test_escaping.py
+++ b/tests/test_escaping.py
@@ -1,14 +1,8 @@
 from markdownify import markdownify as md


-def test_asterisks():
-    assert md('*hey*dude*') == r'\*hey\*dude\*'
-    assert md('*hey*dude*', escape_asterisks=False) == r'*hey*dude*'
-
-
 def test_underscore():
    assert md('_hey_dude_') == r'\_hey\_dude\_'
-    assert md('_hey_dude_', escape_underscores=False) == r'_hey_dude_'


 def test_xml_entities():
--- a/tests/test_lists.py
+++ b/tests/test_lists.py
@@ -51,14 +51,6 @@ def test_nested_ols():

 def test_ul():
    assert md('<ul><li>a</li><li>b</li></ul>') == '* a\n* b\n'
-    assert md("""<ul>
-     <li>
-             a
-     </li>
-     <li> b </li>
-     <li>   c
-     </li>
- </ul>""") == '* a\n* b\n* c\n'


 def test_inline_ul():
--- a/tests/test_tables.py
+++ b/tests/test_tables.py
@@ -39,25 +39,6 @@ table_with_html_content = """<table>
 </table>"""


-table_with_paragraphs = """<table>
-    <tr>
-        <th>Firstname</th>
-        <th><p>Lastname</p></th>
-        <th>Age</th>
-    </tr>
-    <tr>
-        <td><p>Jill</p></td>
-        <td><p>Smith</p></td>
-        <td><p>50</p></td>
-    </tr>
-    <tr>
-        <td>Eve</td>
-        <td>Jackson</td>
-        <td>94</td>
-    </tr>
-</table>"""
-
-
 table_with_header_column = """<table>
    <tr>
        <th>Firstname</th>
@@ -143,7 +124,6 @@ table_missing_head = """<table>
 def test_table():
    assert md(table) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_with_html_content) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| **Jill** | *Smith* | [50](#) |\n| Eve | Jackson | 94 |\n\n'
-    assert md(table_with_paragraphs) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_with_header_column) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_head_body) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_missing_text) == '\n\n|  | Lastname | Age |\n| --- | --- | --- |\n| Jill |  | 50 |\n| Eve | Jackson | 94 |\n\n'
Author	SHA1	Message	Date
AlexVonB	eaeb0603eb	Merge branch 'develop'	2021-07-11 13:21:20 +02:00
AlexVonB	cb73590623	Merge branch 'develop'	2021-07-11 13:14:29 +02:00
AlexVonB	59417ab115	Merge branch 'develop'	2021-05-30 19:10:49 +02:00
AlexVonB	917b01e548	Merge branch 'develop'	2021-05-30 11:20:32 +02:00
AlexVonB	652714859d	Merge branch 'develop'	2021-05-21 14:18:14 +02:00
AlexVonB	ea5b22824b	Merge branch 'develop'	2021-05-18 10:42:27 +02:00
AlexVonB	ec5858e42f	Merge branch 'develop'	2021-05-16 18:41:24 +02:00
AlexVonB	02bb914ef3	Merge branch 'develop'	2021-05-02 13:49:30 +02:00
AlexVonB	21c0d034d0	Merge branch 'develop'	2021-05-02 10:51:00 +02:00
AlexVonB	e3ddc789a2	Merge branch 'develop'	2021-04-22 12:43:27 +02:00
AlexVonB	2d0cd97323	Merge branch 'develop'	2021-04-22 12:13:03 +02:00
AlexVonB	ec185e2e9c	Merge branch 'develop'	2021-02-21 23:09:55 +01:00
AlexVonB	079d1721aa	Merge branch 'develop'	2021-02-21 20:58:34 +01:00
AlexVonB	bf24df3e2e	bump to v0.6.3	2021-01-12 22:43:18 +01:00
AlexVonB	15329588b1	Merge branch 'develop'	2021-01-12 22:42:58 +01:00
AlexVonB	34ad8485fa	bump to v0.6.2	2021-01-12 22:40:03 +01:00
AlexVonB	f0ce934bf8	Merge branch 'develop'	2021-01-12 22:39:47 +01:00
AlexVonB	99cd237f27	Merge branch 'develop'	2021-01-04 10:22:02 +01:00
AlexVonB	2bde8d3e8e	Merge branch 'develop'	2021-01-02 16:49:28 +01:00
AlexVonB	8c9b029756	Merge branch 'develop'	2020-09-01 18:10:07 +02:00
AlexVonB	ae50065872	Merge branch 'develop'	2020-08-18 18:53:10 +02:00