Merge branch 'develop'

bump to v0.10.3
allow BeautifulSoup objects to be converted
2022-01-23 11:01:45 +01:00 · 2022-01-23 11:01:26 +01:00 · 2022-01-23 11:00:19 +01:00 · 2022-01-23 10:59:24 +01:00 · 2022-01-18 08:56:33 +01:00 · 2022-01-18 08:53:33 +01:00
9 changed files with 140 additions and 24 deletions
--- a/.gitignore
+++ b/.gitignore
@@ -8,3 +8,4 @@
 /MANIFEST
 /venv
 build/
 .vscode/settings.json
--- a/README.rst
+++ b/README.rst
@@ -32,14 +32,14 @@ Convert some HTML to Markdown:
    from markdownify import markdownify as md
    md('<b>Yay</b> <a href="http://github.com">GitHub</a>')  # > '**Yay** [GitHub](http://github.com)'
-Specify tags to exclude (blacklist):
+Specify tags to exclude:
 .. code:: python
    from markdownify import markdownify as md
    md('<b>Yay</b> <a href="http://github.com">GitHub</a>', strip=['a'])  # > '**Yay** GitHub'
-\...or specify the tags you want to include (whitelist):
+\...or specify the tags you want to include:
 .. code:: python
@@ -53,11 +53,11 @@ Options
 Markdownify supports the following options:
 strip
-  A list of tags to strip (blacklist). This option can't be used with the
+  A list of tags to strip. This option can't be used with the
  ``convert`` option.
 convert
-  A list of tags to convert (whitelist). This option can't be used with the
+  A list of tags to convert. This option can't be used with the
  ``strip`` option.
 autolinks
@@ -96,10 +96,55 @@ newline_style
  newline). While the latter convention is non-standard, it is commonly
  preferred and supported by a lot of interpreters.
 code_language
  Defines the language that should be assumed for all ``<pre>`` sections.
  Useful, if all code on a page is in the same programming language and
  should be annotated with `````python`` or similar.
  Defaults to ``''`` (empty string) and can be any string.
 escape_underscores
  If set to ``False``, do not escape ``_`` to ``\_`` in text.
  Defaults to ``True``.
 Options may be specified as kwargs to the ``markdownify`` function, or as a
 nested ``Options`` class in ``MarkdownConverter`` subclasses.
 Converting BeautifulSoup objects
 ================================
 .. code:: python
    from markdownify import MarkdownConverter
    # Create shorthand method for conversion
    def md(soup, **options):
        return ImageBlockConverter(**options).convert_soup(soup)
 Creating Custom Converters
 ==========================
 If you have a special usecase that calls for a special conversion, you can
 always inherit from ``MarkdownConverter`` and override the method you want to
 change:
 .. code:: python
    from markdownify import MarkdownConverter
    class ImageBlockConverter(MarkdownConverter):
        """
        Create a custom MarkdownConverter that adds two newlines after an image
        """
        def convert_img(self, el, text, convert_as_inline):
            return super().convert_img(el, text, convert_as_inline) + '\n\n'
    # Create shorthand method for conversion
    def md(html, **options):
        return ImageBlockConverter(**options).convert(html)
 Development
 ===========
--- a/markdownify/init.py
+++ b/markdownify/init.py
@@ -25,10 +25,12 @@ ASTERISK = '*'
 UNDERSCORE = '_'
-def escape(text):
+def escape(text, escape_underscores):
    if not text:
        return ''
-    return text.replace('_', r'\_')
+    if escape_underscores:
        return text.replace('_', r'\_')
    return text
 def chomp(text):
@@ -68,8 +70,10 @@ class MarkdownConverter(object):
    class DefaultOptions:
        autolinks = True
        bullets = '*+-'  # An iterable of bullet types.
        code_language = ''
        convert = None
        default_title = False
        escape_underscores = True
        heading_style = UNDERLINED
        newline_style = SPACES
        strip = None
@@ -92,15 +96,21 @@ class MarkdownConverter(object):
    def convert(self, html):
        soup = BeautifulSoup(html, 'html.parser')
        return self.convert_soup(soup)
    def convert_soup(self, soup):
        return self.process_tag(soup, convert_as_inline=False, children_only=True)
    def process_tag(self, node, convert_as_inline, children_only=False):
        text = ''
-        # markdown headings can't include block elements (elements w/newlines)
+
        # markdown headings or cells can't include
        # block elements (elements w/newlines)
        isHeading = html_heading_re.match(node.name) is not None
        isCell = node.name in ['td', 'th']
        convert_children_as_inline = convert_as_inline
-        if not children_only and isHeading:
+        if not children_only and (isHeading or isCell):
            convert_children_as_inline = True
        # Remove whitespace-only textnodes in purely nested nodes
@@ -142,22 +152,26 @@ class MarkdownConverter(object):
        return text
    def process_text(self, el):
-        text = six.text_type(el)
+        text = six.text_type(el) or ''
        # dont remove any whitespace when handling pre or code in pre
-        if (el.parent.name == 'pre'
+        if not (el.parent.name == 'pre'
-                or (el.parent.name == 'code' and el.parent.parent.name == 'pre')):
+                or (el.parent.name == 'code'
-            return escape(text or '')
+                    and el.parent.parent.name == 'pre')):
            text = whitespace_re.sub(' ', text)
-        cleaned_text = escape(whitespace_re.sub(' ', text or ''))
+        if el.parent.name != 'code':
            text = escape(text, self.options['escape_underscores'])
        # remove trailing whitespaces if any of the following condition is true:
        # - current text node is the last node in li
        # - current text node is followed by an embedded list
-        if el.parent.name == 'li' and (not el.next_sibling or el.next_sibling.name in ['ul', 'ol']):
+        if (el.parent.name == 'li'
-            return cleaned_text.rstrip()
+                and (not el.next_sibling
                     or el.next_sibling.name in ['ul', 'ol'])):
            text = text.rstrip()
-        return cleaned_text
+        return text
    def __getattr__(self, attr):
        # Handle headings
@@ -196,8 +210,6 @@ class MarkdownConverter(object):
        prefix, suffix, text = chomp(text)
        if not text:
            return ''
        if convert_as_inline:
            return text
        href = el.get('href')
        title = el.get('title')
        # For the replacement see #29: text nodes underscores are escaped
@@ -309,7 +321,7 @@ class MarkdownConverter(object):
                el = el.parent
            bullets = self.options['bullets']
            bullet = bullets[depth % len(bullets)]
-        return '%s %s\n' % (bullet, text or '')
+        return '%s %s\n' % (bullet, (text or '').strip())
    def convert_p(self, el, text, convert_as_inline):
        if convert_as_inline:
@@ -319,7 +331,7 @@ class MarkdownConverter(object):
    def convert_pre(self, el, text, convert_as_inline):
        if not text:
            return ''
-        return '\n```\n%s\n```\n' % text
+        return '\n```%s\n%s\n```\n' % (self.options['code_language'], text)
    convert_s = convert_del
--- a/setup.py
+++ b/setup.py
@@ -10,7 +10,7 @@ read = lambda filepath: codecs.open(filepath, 'r', 'utf-8').read()
 pkgmeta = {
    '__title__': 'markdownify',
    '__author__': 'Matthew Tretter',
-    '__version__': '0.9.0',
+    '__version__': '0.10.3',
 }
@@ -70,7 +70,7 @@ setup(
    zip_safe=False,
    include_package_data=True,
    setup_requires=[
-        'flake8>=3.8,<4',
+        'flake8>=3.8,<5',
    ],
    tests_require=[
        'pytest>=6.2,<7',
--- a/tests/test_conversions.py
+++ b/tests/test_conversions.py
@@ -70,6 +70,7 @@ def test_br():
 def test_code():
    inline_tests('code', '`')
    assert md('<code>this_should_not_escape</code>') == '`this_should_not_escape`'
 def test_del():
@@ -131,8 +132,6 @@ def test_hn_nested_simple_tag():
 def test_hn_nested_img():
    assert md('<img src="/path/to/img.jpg" alt="Alt text" title="Optional title" />') == '![Alt text](/path/to/img.jpg "Optional title")'
    assert md('<img src="/path/to/img.jpg" alt="Alt text" />') == '![Alt text](/path/to/img.jpg)'
    image_attributes_to_markdown = [
        ("", ""),
        ("alt='Alt Text'", "Alt Text"),
@@ -211,3 +210,8 @@ def test_sub():
 def test_sup():
    assert md('<sup>foo</sup>') == 'foo'
    assert md('<sup>foo</sup>', sup_symbol='^') == '^foo^'
 def test_lang():
    assert md('<pre>test\n    foo\nbar</pre>', code_language='python') == '\n```python\ntest\n    foo\nbar\n```\n'
    assert md('<pre><code>test\n    foo\nbar</code></pre>', code_language='javascript') == '\n```javascript\ntest\n    foo\nbar\n```\n'
--- a/tests/test_custom_converter.py
+++ b/tests/test_custom_converter.py
@@ -0,0 +1,25 @@
 from markdownify import MarkdownConverter
 from bs4 import BeautifulSoup
 class ImageBlockConverter(MarkdownConverter):
    """
    Create a custom MarkdownConverter that adds two newlines after an image
    """
    def convert_img(self, el, text, convert_as_inline):
        return super().convert_img(el, text, convert_as_inline) + '\n\n'
 def test_img():
    # Create shorthand method for conversion
    def md(html, **options):
        return ImageBlockConverter(**options).convert(html)
    assert md('<img src="/path/to/img.jpg" alt="Alt text" title="Optional title" />') == '![Alt text](/path/to/img.jpg "Optional title")\n\n'
    assert md('<img src="/path/to/img.jpg" alt="Alt text" />') == '![Alt text](/path/to/img.jpg)\n\n'
 def test_soup():
    html = '<b>test</b>'
    soup = BeautifulSoup(html, 'html.parser')
    assert MarkdownConverter().convert_soup(soup) == '**test**'
--- a/tests/test_escaping.py
+++ b/tests/test_escaping.py
@@ -3,6 +3,7 @@ from markdownify import markdownify as md
 def test_underscore():
    assert md('_hey_dude_') == r'\_hey\_dude\_'
    assert md('_hey_dude_', escape_underscores=False) == r'_hey_dude_'
 def test_xml_entities():
--- a/tests/test_lists.py
+++ b/tests/test_lists.py
@@ -51,6 +51,14 @@ def test_nested_ols():
 def test_ul():
    assert md('<ul><li>a</li><li>b</li></ul>') == '* a\n* b\n'
    assert md("""<ul>
     <li>
             a
     </li>
     <li> b </li>
     <li>   c
     </li>
 </ul>""") == '* a\n* b\n* c\n'
 def test_inline_ul():
--- a/tests/test_tables.py
+++ b/tests/test_tables.py
@@ -39,6 +39,25 @@ table_with_html_content = """<table>
 </table>"""
 table_with_paragraphs = """<table>
    <tr>
        <th>Firstname</th>
        <th><p>Lastname</p></th>
        <th>Age</th>
    </tr>
    <tr>
        <td><p>Jill</p></td>
        <td><p>Smith</p></td>
        <td><p>50</p></td>
    </tr>
    <tr>
        <td>Eve</td>
        <td>Jackson</td>
        <td>94</td>
    </tr>
 </table>"""
 table_with_header_column = """<table>
    <tr>
        <th>Firstname</th>
@@ -124,6 +143,7 @@ table_missing_head = """<table>
 def test_table():
    assert md(table) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_with_html_content) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| **Jill** | *Smith* | [50](#) |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_with_paragraphs) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_with_header_column) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_head_body) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_missing_text) == '\n\n|  | Lastname | Age |\n| --- | --- | --- |\n| Jill |  | 50 |\n| Eve | Jackson | 94 |\n\n'
Author	SHA1	Message	Date
AlexVonB	eb0330bfc6	Merge branch 'develop'	2022-01-23 11:01:45 +01:00
AlexVonB	ddda696396	bump to v0.10.3	2022-01-23 11:01:26 +01:00
AlexVonB	0a1343a538	allow BeautifulSoup objects to be converted	2022-01-23 11:00:19 +01:00
AlexVonB	9d0b839b73	wording	2022-01-23 10:59:24 +01:00
AlexVonB	28793ac0b3	Merge branch 'develop'	2022-01-18 08:56:33 +01:00
AlexVonB	d3eff11617	bump to v0.10.2	2022-01-18 08:53:33 +01:00
AlexVonB	bd6b581122	add option to not escape underscores closes #59	2022-01-18 08:51:44 +01:00
AlexVonB	9231704988	Merge branch 'develop'	2021-12-11 14:44:58 +01:00
AlexVonB	c8f7cf63e3	bump to v0.10.1	2021-12-11 14:44:34 +01:00
AlexVonB	12a68a7d14	allow flake8 v4.x closes #57	2021-12-11 14:43:14 +01:00
AlexVonB	1613c302bc	Merge branch 'develop'	2021-11-17 17:11:01 +01:00
AlexVonB	478b1c7e13	bump to v0.10.0	2021-11-17 17:10:15 +01:00
AlexVonB	ffcf6cbcb2	fix readme for code_language	2021-11-17 17:09:47 +01:00
AlexVonB	0ab0452414	add readme for code_language	2021-11-17 17:08:14 +01:00
AlexVonB	b62b067cbd	Merge branch 'Inzaniak-develop' into develop	2021-11-17 17:05:07 +01:00
AlexVonB	cb2646cd93	differentiated between text and code language	2021-11-17 17:03:31 +01:00
AlexVonB	9692b5e714	satisfy linter	2021-11-17 16:55:00 +01:00
Umberto Grando	ac68c53a7d	added language for multiline code	2021-11-01 21:19:35 +01:00
AlexVonB	55c9e84f38	Merge branch 'develop'	2021-09-04 21:50:34 +02:00
AlexVonB	40dd30419c	bump to v0.9.4	2021-09-04 21:50:05 +02:00
AlexVonB	da56f7f56a	Merge pull request #53 from Hozhyi/fix/bullet_list_tags_in_separate_lines Fixed issue #52 - added stripping of text to list	2021-09-04 21:48:16 +02:00
AlexVonB	8400b39dd9	remove trailing whitespace to satisfy the linter	2021-09-04 21:47:27 +02:00
Viktor Hozhyi	5fc1441fe7	Added appropriate test	2021-09-04 20:51:08 +03:00
Viktor Hozhyi	044615eff1	Fixed issue #52 - added stripping of text to list	2021-09-04 12:39:30 +03:00
AlexVonB	99875683ac	Merge branch 'develop'	2021-08-25 08:53:38 +02:00
AlexVonB	dbd9f3f3d2	bump to v0.9.3	2021-08-25 08:53:17 +02:00
AlexVonB	0fdeb1ff6e	convert tags inside table cells as inline in part resolves #49	2021-08-25 08:48:30 +02:00
AlexVonB	eaeb0603eb	Merge branch 'develop'	2021-07-11 13:21:20 +02:00
AlexVonB	6a2f3a4b42	fix rst syntax error	2021-07-11 13:21:02 +02:00
AlexVonB	cb73590623	Merge branch 'develop'	2021-07-11 13:14:29 +02:00
AlexVonB	22180a166d	bump to v0.9.1	2021-07-11 13:13:31 +02:00
AlexVonB	16d8a0e1f7	Revert "add figure/figcaption" This reverts commit `828e116530`.	2021-07-11 13:12:16 +02:00
AlexVonB	4aa6cf2a24	rewrote text processing to not escape _ in code fixes #47	2021-07-11 13:10:59 +02:00
AlexVonB	828e116530	add figure/figcaption for #46	2021-06-30 13:02:42 +02:00
AlexVonB	62e9f0de02	add examples for custom converters closes #46	2021-06-27 15:53:23 +02:00