Merge branch 'develop'

bump to v0.13.0
fix pytest version to 8
2024-07-14 21:20:04 +02:00 · 2024-07-14 21:19:35 +02:00 · 2024-07-14 21:02:49 +02:00 · 2024-06-23 14:30:07 +02:00 · 2024-06-23 14:28:53 +02:00 · 2024-06-23 13:30:08 +02:00
14 changed files with 424 additions and 99 deletions
--- a/.github/workflows/python-app.yml
+++ b/.github/workflows/python-app.yml
@@ -23,11 +23,7 @@ jobs:
    - name: Install dependencies
      run: |
        python -m pip install --upgrade pip
-        pip install flake8==3.8.4 pytest
-        if [ -f requirements.txt ]; then pip install -r requirements.txt; fi
-    - name: Lint with flake8
+        pip install tox
+    - name: Lint and test
      run: |
-        python setup.py lint
-    - name: Test with pytest
-      run: |
-        python setup.py test
+        tox
--- a/.gitignore
+++ b/.gitignore
@@ -9,3 +9,4 @@
 /venv
 build/
 .vscode/settings.json
+.tox/
--- a/MANIFEST.in
+++ b/MANIFEST.in
@@ -1 +1,2 @@
 include README.rst
+prune tests
--- a/README.rst
+++ b/README.rst
@@ -1,8 +1,8 @@
 |build| |version| |license| |downloads|

-.. |build| image:: https://img.shields.io/github/workflow/status/matthewwithanm/python-markdownify/Python%20application/develop
+.. |build| image:: https://img.shields.io/github/actions/workflow/status/matthewwithanm/python-markdownify/python-app.yml?branch=develop
    :alt: GitHub Workflow Status
-    :target: https://github.com/matthewwithanm/python-markdownify/actions?query=workflow%3A%22Python+application%22
+    :target: https://github.com/matthewwithanm/python-markdownify/actions/workflows/python-app.yml?query=workflow%3A%22Python+application%22

 .. |version| image:: https://img.shields.io/pypi/v/markdownify
    :alt: Pypi version
@@ -87,12 +87,16 @@ strong_em_symbol
 sub_symbol, sup_symbol
  Define the chars that surround ``<sub>`` and ``<sup>`` text. Defaults to an
  empty string, because this is non-standard behavior. Could be something like
-  ``~`` and ``^`` to result in ``~sub~`` and ``^sup^``.
+  ``~`` and ``^`` to result in ``~sub~`` and ``^sup^``.  If the value starts
+  with ``<`` and ends with ``>``, it is treated as an HTML tag and a ``/`` is
+  inserted after the ``<`` in the string used after the text; this allows
+  specifying ``<sub>`` to use raw HTML in the output for subscripts, for
+  example.

 newline_style
  Defines the style of marking linebreaks (``<br>``) in markdown. The default
  value ``SPACES`` of this option will adopt the usual two spaces and a newline,
-  while ``BACKSLASH`` will convert a linebreak to ``\\n`` (a backslash an a
+  while ``BACKSLASH`` will convert a linebreak to ``\\n`` (a backslash and a
  newline). While the latter convention is non-standard, it is commonly
  preferred and supported by a lot of interpreters.

@@ -123,6 +127,11 @@ escape_underscores
  If set to ``False``, do not escape ``_`` to ``\_`` in text.
  Defaults to ``True``.

+escape_misc
+  If set to ``False``, do not escape miscellaneous punctuation characters
+  that sometimes have Markdown significance in text.
+  Defaults to ``True``.
+
 keep_inline_images_in
  Images are converted to their alt-text when the images are located inside
  headlines or table cells. If some inline images should be converted to
@@ -130,6 +139,11 @@ keep_inline_images_in
  that should be allowed to contain inline images, for example ``['td']``.
  Defaults to an empty list.

+wrap, wrap_width
+  If ``wrap`` is set to ``True``, all text paragraphs are wrapped at
+  ``wrap_width`` characters. Defaults to ``False`` and ``80``.
+  Use with ``newline_style=BACKSLASH`` to keep line breaks in paragraphs.
+
 Options may be specified as kwargs to the ``markdownify`` function, or as a
 nested ``Options`` class in ``MarkdownConverter`` subclasses.

@@ -151,7 +165,12 @@ Creating Custom Converters

 If you have a special usecase that calls for a special conversion, you can
 always inherit from ``MarkdownConverter`` and override the method you want to
-change:
+change.
+The function that handles a HTML tag named ``abc`` is called
+``convert_abc(self, el, text, convert_as_inline)`` and returns a string
+containing the converted HTML tag.
+The ``MarkdownConverter`` object will handle the conversion based on the
+function names:

 .. code:: python

@@ -168,14 +187,32 @@ change:
    def md(html, **options):
        return ImageBlockConverter(**options).convert(html)

+.. code:: python
+
+    from markdownify import MarkdownConverter
+
+    class IgnoreParagraphsConverter(MarkdownConverter):
+        """
+        Create a custom MarkdownConverter that ignores paragraphs
+        """
+        def convert_p(self, el, text, convert_as_inline):
+            return ''
+
+    # Create shorthand method for conversion
+    def md(html, **options):
+        return IgnoreParagraphsConverter(**options).convert(html)
+
+
+Command Line Interface
+======================
+
+Use ``markdownify example.html > example.md`` or pipe input from stdin
+(``cat example.html | markdownify > example.md``).
+Call ``markdownify -h`` to see all available options.
+They are the same as listed above and take the same arguments.
+

 Development
 ===========

-To run tests:
-
-``python setup.py test``
-
-To lint:
-
-``python setup.py lint``
+To run tests and the linter run ``pip install tox`` once, then ``tox``.
--- a/markdownify/init.py
+++ b/markdownify/init.py
@@ -1,4 +1,5 @@
 from bs4 import BeautifulSoup, NavigableString, Comment, Doctype
+from textwrap import fill
 import re
 import six

@@ -42,15 +43,22 @@ def abstract_inline_conversion(markup_fn):
    """
    This abstracts all simple inline tags like b, em, del, ...
    Returns a function that wraps the chomped text in a pair of the string
-    that is returned by markup_fn. markup_fn is necessary to allow for
+    that is returned by markup_fn, with '/' inserted in the string used after
+    the text if it looks like an HTML tag. markup_fn is necessary to allow for
    references to self.strong_em_symbol etc.
    """
    def implementation(self, el, text, convert_as_inline):
-        markup = markup_fn(self)
+        markup_prefix = markup_fn(self)
+        if markup_prefix.startswith('<') and markup_prefix.endswith('>'):
+            markup_suffix = '</' + markup_prefix[1:]
+        else:
+            markup_suffix = markup_prefix
+        if el.find_parent(['pre', 'code', 'kbd', 'samp']):
+            return text
        prefix, suffix, text = chomp(text)
        if not text:
            return ''
-        return '%s%s%s%s%s' % (prefix, markup, text, markup, suffix)
+        return '%s%s%s%s%s' % (prefix, markup_prefix, text, markup_suffix, suffix)
    return implementation


@@ -68,6 +76,7 @@ class MarkdownConverter(object):
        default_title = False
        escape_asterisks = True
        escape_underscores = True
+        escape_misc = True
        heading_style = UNDERLINED
        keep_inline_images_in = []
        newline_style = SPACES
@@ -75,6 +84,8 @@ class MarkdownConverter(object):
        strong_em_symbol = ASTERISK
        sub_symbol = ''
        sup_symbol = ''
+        wrap = False
+        wrap_width = 80

    class Options(DefaultOptions):
        pass
@@ -149,13 +160,12 @@ class MarkdownConverter(object):
    def process_text(self, el):
        text = six.text_type(el) or ''

-        # dont remove any whitespace when handling pre or code in pre
-        if not (el.parent.name == 'pre'
-                or (el.parent.name == 'code'
-                    and el.parent.parent.name == 'pre')):
+        # normalize whitespace if we're not inside a preformatted element
+        if not el.find_parent('pre'):
            text = whitespace_re.sub(' ', text)

-        if el.parent.name != 'code':
+        # escape special characters if we're not inside a preformatted or code element
+        if not el.find_parent(['pre', 'code', 'kbd', 'samp']):
            text = self.escape(text)

        # remove trailing whitespaces if any of the following condition is true:
@@ -197,6 +207,9 @@ class MarkdownConverter(object):
    def escape(self, text):
        if not text:
            return ''
+        if self.options['escape_misc']:
+            text = re.sub(r'([\\&<`[>~#=+|-])', r'\\\1', text)
+            text = re.sub(r'([0-9])([.)])', r'\1\\\2', text)
        if self.options['escape_asterisks']:
            text = text.replace('*', r'\*')
        if self.options['escape_underscores']:
@@ -235,7 +248,7 @@ class MarkdownConverter(object):
        if convert_as_inline:
            return text

-        return '\n' + (line_beginning_re.sub('> ', text) + '\n\n') if text else ''
+        return '\n' + (line_beginning_re.sub('> ', text.strip()) + '\n\n') if text else ''

    def convert_br(self, el, text, convert_as_inline):
        if convert_as_inline:
@@ -263,7 +276,7 @@ class MarkdownConverter(object):
            return text

        style = self.options['heading_style'].lower()
-        text = text.rstrip()
+        text = text.strip()
        if style == UNDERLINED and n <= 2:
            line = '=' if n == 1 else '-'
            return self.underline(text, line)
@@ -313,7 +326,7 @@ class MarkdownConverter(object):
    def convert_li(self, el, text, convert_as_inline):
        parent = el.parent
        if parent is not None and parent.name == 'ol':
-            if parent.get("start"):
+            if parent.get("start") and str(parent.get("start")).isnumeric():
                start = int(parent.get("start"))
            else:
                start = 1
@@ -331,6 +344,11 @@ class MarkdownConverter(object):
    def convert_p(self, el, text, convert_as_inline):
        if convert_as_inline:
            return text
+        if self.options['wrap']:
+            text = fill(text,
+                        width=self.options['wrap_width'],
+                        break_long_words=False,
+                        break_on_hyphens=False)
        return '%s\n\n' % text if text else ''

    def convert_pre(self, el, text, convert_as_inline):
@@ -343,6 +361,12 @@ class MarkdownConverter(object):

        return '\n```%s\n%s\n```\n' % (code_language, text)

+    def convert_script(self, el, text, convert_as_inline):
+        return ''
+
+    def convert_style(self, el, text, convert_as_inline):
+        return ''
+
    convert_s = convert_del

    convert_strong = convert_b
@@ -356,22 +380,49 @@ class MarkdownConverter(object):
    def convert_table(self, el, text, convert_as_inline):
        return '\n\n' + text + '\n'

+    def convert_caption(self, el, text, convert_as_inline):
+        return text + '\n'
+
+    def convert_figcaption(self, el, text, convert_as_inline):
+        return '\n\n' + text + '\n\n'
+
    def convert_td(self, el, text, convert_as_inline):
-        return ' ' + text + ' |'
+        colspan = 1
+        if 'colspan' in el.attrs and el['colspan'].isdigit():
+            colspan = int(el['colspan'])
+        return ' ' + text.strip().replace("\n", " ") + ' |' * colspan

    def convert_th(self, el, text, convert_as_inline):
-        return ' ' + text + ' |'
+        colspan = 1
+        if 'colspan' in el.attrs and el['colspan'].isdigit():
+            colspan = int(el['colspan'])
+        return ' ' + text.strip().replace("\n", " ") + ' |' * colspan

    def convert_tr(self, el, text, convert_as_inline):
        cells = el.find_all(['td', 'th'])
-        is_headrow = all([cell.name == 'th' for cell in cells])
+        is_headrow = (
+            all([cell.name == 'th' for cell in cells])
+            or (not el.previous_sibling and not el.parent.name == 'tbody')
+            or (not el.previous_sibling and el.parent.name == 'tbody' and len(el.parent.parent.find_all(['thead'])) < 1)
+        )
        overline = ''
        underline = ''
        if is_headrow and not el.previous_sibling:
            # first row and is headline: print headline underline
-            underline += '| ' + ' | '.join(['---'] * len(cells)) + ' |' + '\n'
-        elif not el.previous_sibling and not el.parent.name != 'table':
-            # first row, not headline, and the parent is sth. like tbody:
+            full_colspan = 0
+            for cell in cells:
+                if 'colspan' in cell.attrs and cell['colspan'].isdigit():
+                    full_colspan += int(cell["colspan"])
+                else:
+                    full_colspan += 1
+            underline += '| ' + ' | '.join(['---'] * full_colspan) + ' |' + '\n'
+        elif (not el.previous_sibling
+              and (el.parent.name == 'table'
+                   or (el.parent.name == 'tbody'
+                       and not el.parent.previous_sibling))):
+            # first row, not headline, and:
+            # - the parent is table or
+            # - the parent is tbody at the beginning of a table.
            # print empty headline above this row
            overline += '| ' + ' | '.join([''] * len(cells)) + ' |' + '\n'
            overline += '| ' + ' | '.join(['---'] * len(cells)) + ' |' + '\n'
--- a/markdownify/main.py
+++ b/markdownify/main.py
@@ -0,0 +1,73 @@
+#!/usr/bin/env python
+
+import argparse
+import sys
+
+from markdownify import markdownify, ATX, ATX_CLOSED, UNDERLINED, \
+    SPACES, BACKSLASH, ASTERISK, UNDERSCORE
+
+
+def main(argv=sys.argv[1:]):
+    parser = argparse.ArgumentParser(
+        prog='markdownify',
+        description='Converts html to markdown.',
+    )
+
+    parser.add_argument('html', nargs='?', type=argparse.FileType('r'),
+                        default=sys.stdin,
+                        help="The html file to convert. Defaults to STDIN if not "
+                        "provided.")
+    parser.add_argument('-s', '--strip', nargs='*',
+                        help="A list of tags to strip. This option can't be used with "
+                        "the --convert option.")
+    parser.add_argument('-c', '--convert', nargs='*',
+                        help="A list of tags to convert. This option can't be used with "
+                        "the --strip option.")
+    parser.add_argument('-a', '--autolinks', action='store_true',
+                        help="A boolean indicating whether the 'automatic link' style "
+                        "should be used when a 'a' tag's contents match its href.")
+    parser.add_argument('--default-title', action='store_false',
+                        help="A boolean to enable setting the title of a link to its "
+                        "href, if no title is given.")
+    parser.add_argument('--heading-style', default=UNDERLINED,
+                        choices=(ATX, ATX_CLOSED, UNDERLINED),
+                        help="Defines how headings should be converted.")
+    parser.add_argument('-b', '--bullets', default='*+-',
+                        help="A string of bullet styles to use; the bullet will "
+                        "alternate based on nesting level.")
+    parser.add_argument('--strong-em-symbol', default=ASTERISK,
+                        choices=(ASTERISK, UNDERSCORE),
+                        help="Use * or _ to convert strong and italics text"),
+    parser.add_argument('--sub-symbol', default='',
+                        help="Define the chars that surround '<sub>'.")
+    parser.add_argument('--sup-symbol', default='',
+                        help="Define the chars that surround '<sup>'.")
+    parser.add_argument('--newline-style', default=SPACES,
+                        choices=(SPACES, BACKSLASH),
+                        help="Defines the style of <br> conversions: two spaces "
+                        "or backslash at the and of the line thet should break.")
+    parser.add_argument('--code-language', default='',
+                        help="Defines the language that should be assumed for all "
+                        "'<pre>' sections.")
+    parser.add_argument('--no-escape-asterisks', dest='escape_asterisks',
+                        action='store_false',
+                        help="Do not escape '*' to '\\*' in text.")
+    parser.add_argument('--no-escape-underscores', dest='escape_underscores',
+                        action='store_false',
+                        help="Do not escape '_' to '\\_' in text.")
+    parser.add_argument('-i', '--keep-inline-images-in', nargs='*',
+                        help="Images are converted to their alt-text when the images are "
+                        "located inside headlines or table cells. If some inline images "
+                        "should be converted to markdown images instead, this option can "
+                        "be set to a list of parent tags that should be allowed to "
+                        "contain inline images.")
+    parser.add_argument('-w', '--wrap', action='store_true',
+                        help="Wrap all text paragraphs at --wrap-width characters.")
+    parser.add_argument('--wrap-width', type=int, default=80)
+
+    args = parser.parse_args(argv)
+    print(markdownify(**vars(args)))
+
+
+if __name__ == '__main__':
+    main()
--- a/setup.cfg
+++ b/setup.cfg
@@ -1,2 +0,0 @@
-[flake8]
-ignore = E501 W503
--- a/setup.py
+++ b/setup.py
@@ -2,7 +2,6 @@
 import codecs
 import os
 from setuptools import setup, find_packages
-from setuptools.command.test import test as TestCommand, Command


 read = lambda filepath: codecs.open(filepath, 'r', 'utf-8').read()
@@ -10,52 +9,10 @@ read = lambda filepath: codecs.open(filepath, 'r', 'utf-8').read()
 pkgmeta = {
    '__title__': 'markdownify',
    '__author__': 'Matthew Tretter',
-    '__version__': '0.10.3',
+    '__version__': '0.13.0',
 }

-
-class PyTest(TestCommand):
-    def finalize_options(self):
-        TestCommand.finalize_options(self)
-        self.test_args = ['tests', '-s']
-        self.test_suite = True
-
-    def run_tests(self):
-        import pytest
-        errno = pytest.main(self.test_args)
-        raise SystemExit(errno)
-
-
-class LintCommand(Command):
-    """
-    A copy of flake8's Flake8Command
-
-    """
-    description = "Run flake8 on modules registered in setuptools"
-    user_options = []
-
-    def initialize_options(self):
-        pass
-
-    def finalize_options(self):
-        pass
-
-    def distribution_files(self):
-        if self.distribution.packages:
-            for package in self.distribution.packages:
-                yield package.replace(".", os.path.sep)
-
-        if self.distribution.py_modules:
-            for filename in self.distribution.py_modules:
-                yield "%s.py" % filename
-
-    def run(self):
-        from flake8.api.legacy import get_style_guide
-        flake8_style = get_style_guide(config_file='setup.cfg')
-        paths = self.distribution_files()
-        report = flake8_style.check_files(paths)
-        raise SystemExit(report.total_errors > 0)
-
+read = lambda filepath: codecs.open(filepath, 'r', 'utf-8').read()

 setup(
    name='markdownify',
@@ -69,14 +26,9 @@ setup(
    packages=find_packages(),
    zip_safe=False,
    include_package_data=True,
-    setup_requires=[
-        'flake8>=3.8,<5',
-    ],
-    tests_require=[
-        'pytest>=6.2,<7',
-    ],
    install_requires=[
-        'beautifulsoup4>=4.9,<5', 'six>=1.15,<2'
+        'beautifulsoup4>=4.9,<5',
+        'six>=1.15,<2',
    ],
    classifiers=[
        'Environment :: Web Environment',
@@ -92,8 +44,9 @@ setup(
        'Programming Language :: Python :: 3.8',
        'Topic :: Utilities'
    ],
-    cmdclass={
-        'test': PyTest,
-        'lint': LintCommand,
-    },
+    entry_points={
+        'console_scripts': [
+            'markdownify = markdownify.main:main'
+        ]
+    }
 )
--- a/shell.nix
+++ b/shell.nix
@@ -0,0 +1,10 @@
+{ pkgs ? import <nixpkgs> {} }:
+pkgs.mkShell {
+  name = "python-shell";
+  buildInputs = with pkgs; [
+    python38
+    python38Packages.tox
+    python38Packages.setuptools
+    python38Packages.virtualenv
+  ];
+}
--- a/tests/test_conversions.py
+++ b/tests/test_conversions.py
@@ -52,6 +52,12 @@ def test_b_spaces():

 def test_blockquote():
    assert md('<blockquote>Hello</blockquote>') == '\n> Hello\n\n'
+    assert md('<blockquote>\nHello\n</blockquote>') == '\n> Hello\n\n'
+
+
+def test_blockquote_with_nested_paragraph():
+    assert md('<blockquote><p>Hello</p></blockquote>') == '\n> Hello\n\n'
+    assert md('<blockquote><p>Hello</p><p>Hello again</p></blockquote>') == '\n> Hello\n> \n> Hello again\n\n'


 def test_blockquote_with_paragraph():
@@ -60,7 +66,7 @@ def test_blockquote_with_paragraph():

 def test_blockquote_nested():
    text = md('<blockquote>And she was like <blockquote>Hello</blockquote></blockquote>')
-    assert text == '\n> And she was like \n> > Hello\n> \n> \n\n'
+    assert text == '\n> And she was like \n> > Hello\n\n'


 def test_br():
@@ -68,9 +74,29 @@ def test_br():
    assert md('a<br />b<br />c', newline_style=BACKSLASH) == 'a\\\nb\\\nc'


+def test_caption():
+    assert md('TEXT<figure><figcaption>Caption</figcaption><span>SPAN</span></figure>') == 'TEXT\n\nCaption\n\nSPAN'
+    assert md('<figure><span>SPAN</span><figcaption>Caption</figcaption></figure>TEXT') == 'SPAN\n\nCaption\n\nTEXT'
+
+
 def test_code():
    inline_tests('code', '`')
-    assert md('<code>this_should_not_escape</code>') == '`this_should_not_escape`'
+    assert md('<code>*this_should_not_escape*</code>') == '`*this_should_not_escape*`'
+    assert md('<kbd>*this_should_not_escape*</kbd>') == '`*this_should_not_escape*`'
+    assert md('<samp>*this_should_not_escape*</samp>') == '`*this_should_not_escape*`'
+    assert md('<code><span>*this_should_not_escape*</span></code>') == '`*this_should_not_escape*`'
+    assert md('<code>this  should\t\tnormalize</code>') == '`this should normalize`'
+    assert md('<code><span>this  should\t\tnormalize</span></code>') == '`this should normalize`'
+    assert md('<code>foo<b>bar</b>baz</code>') == '`foobarbaz`'
+    assert md('<kbd>foo<i>bar</i>baz</kbd>') == '`foobarbaz`'
+    assert md('<samp>foo<del> bar </del>baz</samp>') == '`foo bar baz`'
+    assert md('<samp>foo <del>bar</del> baz</samp>') == '`foo bar baz`'
+    assert md('<code>foo<em> bar </em>baz</code>') == '`foo bar baz`'
+    assert md('<code>foo<code> bar </code>baz</code>') == '`foo bar baz`'
+    assert md('<code>foo<strong> bar </strong>baz</code>') == '`foo bar baz`'
+    assert md('<code>foo<s> bar </s>baz</code>') == '`foo bar baz`'
+    assert md('<code>foo<sup>bar</sup>baz</code>', sup_symbol='^') == '`foobarbaz`'
+    assert md('<code>foo<sub>bar</sub>baz</code>', sub_symbol='^') == '`foobarbaz`'


 def test_del():
@@ -85,6 +111,14 @@ def test_em():
    inline_tests('em', '*')


+def test_header_with_space():
+    assert md('<h3>\n\nHello</h3>') == '### Hello\n\n'
+    assert md('<h4>\n\nHello</h4>') == '#### Hello\n\n'
+    assert md('<h5>\n\nHello</h5>') == '##### Hello\n\n'
+    assert md('<h5>\n\nHello\n\n</h5>') == '##### Hello\n\n'
+    assert md('<h5>\n\nHello   \n\n</h5>') == '##### Hello\n\n'
+
+
 def test_h1():
    assert md('<h1>Hello</h1>') == 'Hello\n=====\n\n'

@@ -177,11 +211,39 @@ def test_kbd():

 def test_p():
    assert md('<p>hello</p>') == 'hello\n\n'
+    assert md('<p>123456789 123456789</p>') == '123456789 123456789\n\n'
+    assert md('<p>123456789 123456789</p>', wrap=True, wrap_width=10) == '123456789\n123456789\n\n'
+    assert md('<p><a href="https://example.com">Some long link</a></p>', wrap=True, wrap_width=10) == '[Some long\nlink](https://example.com)\n\n'
+    assert md('<p>12345<br />67890</p>', wrap=True, wrap_width=10, newline_style=BACKSLASH) == '12345\\\n67890\n\n'
+    assert md('<p>12345678901<br />12345</p>', wrap=True, wrap_width=10, newline_style=BACKSLASH) == '12345678901\\\n12345\n\n'


 def test_pre():
    assert md('<pre>test\n    foo\nbar</pre>') == '\n```\ntest\n    foo\nbar\n```\n'
    assert md('<pre><code>test\n    foo\nbar</code></pre>') == '\n```\ntest\n    foo\nbar\n```\n'
+    assert md('<pre>*this_should_not_escape*</pre>') == '\n```\n*this_should_not_escape*\n```\n'
+    assert md('<pre><span>*this_should_not_escape*</span></pre>') == '\n```\n*this_should_not_escape*\n```\n'
+    assert md('<pre>\t\tthis  should\t\tnot  normalize</pre>') == '\n```\n\t\tthis  should\t\tnot  normalize\n```\n'
+    assert md('<pre><span>\t\tthis  should\t\tnot  normalize</span></pre>') == '\n```\n\t\tthis  should\t\tnot  normalize\n```\n'
+    assert md('<pre>foo<b>\nbar\n</b>baz</pre>') == '\n```\nfoo\nbar\nbaz\n```\n'
+    assert md('<pre>foo<i>\nbar\n</i>baz</pre>') == '\n```\nfoo\nbar\nbaz\n```\n'
+    assert md('<pre>foo\n<i>bar</i>\nbaz</pre>') == '\n```\nfoo\nbar\nbaz\n```\n'
+    assert md('<pre>foo<i>\n</i>baz</pre>') == '\n```\nfoo\nbaz\n```\n'
+    assert md('<pre>foo<del>\nbar\n</del>baz</pre>') == '\n```\nfoo\nbar\nbaz\n```\n'
+    assert md('<pre>foo<em>\nbar\n</em>baz</pre>') == '\n```\nfoo\nbar\nbaz\n```\n'
+    assert md('<pre>foo<code>\nbar\n</code>baz</pre>') == '\n```\nfoo\nbar\nbaz\n```\n'
+    assert md('<pre>foo<strong>\nbar\n</strong>baz</pre>') == '\n```\nfoo\nbar\nbaz\n```\n'
+    assert md('<pre>foo<s>\nbar\n</s>baz</pre>') == '\n```\nfoo\nbar\nbaz\n```\n'
+    assert md('<pre>foo<sup>\nbar\n</sup>baz</pre>', sup_symbol='^') == '\n```\nfoo\nbar\nbaz\n```\n'
+    assert md('<pre>foo<sub>\nbar\n</sub>baz</pre>', sub_symbol='^') == '\n```\nfoo\nbar\nbaz\n```\n'
+
+
+def test_script():
+    assert md('foo <script>var foo=42;</script> bar') == 'foo  bar'
+
+
+def test_style():
+    assert md('foo <style>h1 { font-size: larger }</style> bar') == 'foo  bar'


 def test_s():
@@ -206,11 +268,13 @@ def test_strong_em_symbol():
 def test_sub():
    assert md('<sub>foo</sub>') == 'foo'
    assert md('<sub>foo</sub>', sub_symbol='~') == '~foo~'
+    assert md('<sub>foo</sub>', sub_symbol='<sub>') == '<sub>foo</sub>'


 def test_sup():
    assert md('<sup>foo</sup>') == 'foo'
    assert md('<sup>foo</sup>', sup_symbol='^') == '^foo^'
+    assert md('<sup>foo</sup>', sup_symbol='<sup>') == '<sup>foo</sup>'


 def test_lang():
--- a/tests/test_escaping.py
+++ b/tests/test_escaping.py
@@ -12,7 +12,7 @@ def test_underscore():


 def test_xml_entities():
-    assert md('&amp;') == '&'
+    assert md('&amp;') == r'\&'


 def test_named_entities():
@@ -25,4 +25,23 @@ def test_hexadecimal_entities():


 def test_single_escaping_entities():
-    assert md('&amp;amp;') == '&amp;'
+    assert md('&amp;amp;') == r'\&amp;'
+
+
+def text_misc():
+    assert md('\\*') == r'\\\*'
+    assert md('<foo>') == r'\<foo\>'
+    assert md('# foo') == r'\# foo'
+    assert md('> foo') == r'\> foo'
+    assert md('~~foo~~') == r'\~\~foo\~\~'
+    assert md('foo\n===\n') == 'foo\n\\=\\=\\=\n'
+    assert md('---\n') == '\\-\\-\\-\n'
+    assert md('+ x\n+ y\n') == '\\+ x\n\\+ y\n'
+    assert md('`x`') == r'\`x\`'
+    assert md('[text](link)') == r'\[text](link)'
+    assert md('1. x') == r'1\. x'
+    assert md('not a number. x') == r'not a number. x'
+    assert md('1) x') == r'1\) x'
+    assert md('not a number) x') == r'not a number) x'
+    assert md('|not table|') == r'\|not table\|'
+    assert md(r'\ <foo> &amp;amp; | ` `', escape_misc=False) == r'\ <foo> &amp; | ` `'
--- a/tests/test_lists.py
+++ b/tests/test_lists.py
@@ -43,6 +43,9 @@ nested_ols = """
 def test_ol():
    assert md('<ol><li>a</li><li>b</li></ol>') == '1. a\n2. b\n'
    assert md('<ol start="3"><li>a</li><li>b</li></ol>') == '3. a\n4. b\n'
+    assert md('<ol start="-1"><li>a</li><li>b</li></ol>') == '1. a\n2. b\n'
+    assert md('<ol start="foo"><li>a</li><li>b</li></ol>') == '1. a\n2. b\n'
+    assert md('<ol start="1.5"><li>a</li><li>b</li></ol>') == '1. a\n2. b\n'


 def test_nested_ols():
--- a/tests/test_tables.py
+++ b/tests/test_tables.py
@@ -57,6 +57,26 @@ table_with_paragraphs = """<table>
    </tr>
 </table>"""

+table_with_linebreaks = """<table>
+    <tr>
+        <th>Firstname</th>
+        <th>Lastname</th>
+        <th>Age</th>
+    </tr>
+    <tr>
+        <td>Jill</td>
+        <td>Smith
+        Jackson</td>
+        <td>50</td>
+    </tr>
+    <tr>
+        <td>Eve</td>
+        <td>Jackson
+        Smith</td>
+        <td>94</td>
+    </tr>
+</table>"""
+

 table_with_header_column = """<table>
    <tr>
@@ -99,6 +119,28 @@ table_head_body = """<table>
    </tbody>
 </table>"""

+table_head_body_missing_head = """<table>
+    <thead>
+        <tr>
+            <td>Firstname</td>
+            <td>Lastname</td>
+            <td>Age</td>
+        </tr>
+    </thead>
+    <tbody>
+        <tr>
+            <td>Jill</td>
+            <td>Smith</td>
+            <td>50</td>
+        </tr>
+        <tr>
+            <td>Eve</td>
+            <td>Jackson</td>
+            <td>94</td>
+        </tr>
+    </tbody>
+</table>"""
+
 table_missing_text = """<table>
    <thead>
        <tr>
@@ -139,12 +181,74 @@ table_missing_head = """<table>
    </tr>
 </table>"""

+table_body = """<table>
+    <tbody>
+        <tr>
+            <td>Firstname</td>
+            <td>Lastname</td>
+            <td>Age</td>
+        </tr>
+        <tr>
+            <td>Jill</td>
+            <td>Smith</td>
+            <td>50</td>
+        </tr>
+        <tr>
+            <td>Eve</td>
+            <td>Jackson</td>
+            <td>94</td>
+        </tr>
+    </tbody>
+</table>"""
+
+table_with_caption = """TEXT<table><caption>Caption</caption>
+    <tbody><tr><td>Firstname</td>
+            <td>Lastname</td>
+            <td>Age</td>
+        </tr>
+    </tbody>
+</table>"""
+
+table_with_colspan = """<table>
+    <tr>
+        <th colspan="2">Name</th>
+        <th>Age</th>
+    </tr>
+    <tr>
+        <td colspan="1">Jill</td>
+        <td>Smith</td>
+        <td>50</td>
+    </tr>
+    <tr>
+        <td>Eve</td>
+        <td>Jackson</td>
+        <td>94</td>
+    </tr>
+</table>"""
+
+table_with_undefined_colspan = """<table>
+    <tr>
+        <th colspan="undefined">Name</th>
+        <th>Age</th>
+    </tr>
+    <tr>
+        <td colspan="-1">Jill</td>
+        <td>Smith</td>
+    </tr>
+</table>"""
+

 def test_table():
    assert md(table) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_with_html_content) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| **Jill** | *Smith* | [50](#) |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_with_paragraphs) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
+    assert md(table_with_linebreaks) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith  Jackson | 50 |\n| Eve | Jackson  Smith | 94 |\n\n'
    assert md(table_with_header_column) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_head_body) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
+    assert md(table_head_body_missing_head) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_missing_text) == '\n\n|  | Lastname | Age |\n| --- | --- | --- |\n| Jill |  | 50 |\n| Eve | Jackson | 94 |\n\n'
-    assert md(table_missing_head) == '\n\n|  |  |  |\n| --- | --- | --- |\n| Firstname | Lastname | Age |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
+    assert md(table_missing_head) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
+    assert md(table_body) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
+    assert md(table_with_caption) == 'TEXT\n\nCaption\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n\n'
+    assert md(table_with_colspan) == '\n\n| Name | | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
+    assert md(table_with_undefined_colspan) == '\n\n| Name | Age |\n| --- | --- |\n| Jill | Smith |\n\n'
--- a/tox.ini
+++ b/tox.ini
@@ -0,0 +1,15 @@
+[tox]
+envlist = py38
+
+[testenv]
+passenv = PYTHONPATH
+deps =
+	pytest==8
+	flake8
+	restructuredtext_lint
+	Pygments
+commands =
+	pytest
+	flake8 --ignore=E501,W503 markdownify tests
+	restructuredtext-lint README.rst
+
Author	SHA1	Message	Date
AlexVonB	8c810eb8a8	Merge branch 'develop'	2024-07-14 21:20:04 +02:00
AlexVonB	f6c8daf8a5	bump to v0.13.0	2024-07-14 21:19:35 +02:00
AlexVonB	75a678dab9	fix pytest version to 8	2024-07-14 21:02:49 +02:00
AlexVonB	0a5c89aa49	added test for ol start check	2024-06-23 14:30:07 +02:00
microdnd	51390d7389	handle ol start value is not number (#127 ) Co-authored-by: Mico <mico_wu@trendmicro.com>	2024-06-23 14:28:53 +02:00
AlexVonB	50b4640db2	better naming for markup variables	2024-06-23 13:30:08 +02:00
Joseph Myers	7861b330cd	Special-case use of HTML tags for converting `<sub>` / `<sup>` (#119 ) Allow different strings before / after `<sub>` / `<sup>` content In particular, this allows setting `sub_symbol='<sub>'`, `sup_symbol='<sup>'`, to use raw HTML in the output when converting subscripts and superscripts.	2024-06-23 13:28:05 +02:00
AlexVonB	2ec33384de	handle un-parsable colspan values fixes #126	2024-06-23 13:17:20 +02:00
samypr100	c1672aee44	Update MANIFEST.in to exclude tests during packaging (#125 )	2024-06-23 12:59:14 +02:00
AlexVonB	43dbe20aaf	fixed github action badges see https://github.com/badges/shields/issues/8671	2024-04-04 21:50:02 +02:00
Joseph Myers	46af45bb3c	Escape all characters with Markdown significance (#118 ) * Escape all characters with Markdown significance There are many punctuation characters that sometimes have significance in Markdown; more systematically escape them all (based on a new escape_misc configuration option). A limited attempt is made to limit the escaping of '.' and ')' to the context where they might have Markdown significance (after a number, where they can indicate an ordered list item); no such attempt is made for the other characters (and even that limiting of '.' and ')' may not be entirely safe in all cases, as it's possible the HTML could have the number outside the block being escaped in one go, e.g. `<span>1</span>.`. --------- Co-authored-by: AlexVonB <AlexVonB@users.noreply.github.com>	2024-04-04 21:42:58 +02:00
Joseph Myers	2bd0772685	Avoid inline styles inside `<code>` / `<pre>` conversion (#117 ) * Avoid inline styles inside `<code>` / `<pre>` conversion The check used for this is analogous to that used to avoid escaping potential markup characters inside such tags. Fixes #103 --------- Co-authored-by: AlexVonB <AlexVonB@users.noreply.github.com>	2024-04-04 20:55:54 +02:00
AlexVonB	383847ee86	Merge branch 'develop'	2024-03-26 21:56:09 +01:00
AlexVonB	74ddc408cc	bump to v0.12.1	2024-03-26 21:56:00 +01:00
AlexVonB	be3a7f4672	Merge branch 'develop'	2024-03-26 21:52:16 +01:00
Eric Xu	3b4a014f25	Table merge cell horizontally (#110 ) * Fix #109 Table merge cell horizontally * Add test case for colspan --------- Co-authored-by: AlexVonB <AlexVonB@users.noreply.github.com>	2024-03-26 21:50:54 +01:00
AlexVonB	57d4f37923	fixed tests for table caption	2024-03-26 21:43:25 +01:00
Chris Papademetrious	d5fb0fbb85	make sure there are blank lines around table/figure captions (#114 ) Signed-off-by: chrispy <chrispy@synopsys.com> Co-authored-by: AlexVonB <AlexVonB@users.noreply.github.com>	2024-03-26 21:41:56 +01:00
huuya	e4df41225d	Support conversion of header rows in tables without th tag (#83 ) * Fixed support for header row conversion for tables without th tag	2024-03-26 21:32:36 +01:00
AlexVonB	804a3f8f07	added further readme for custom converters	2024-03-26 21:21:45 +01:00
Chris Papademetrious	7d0bf46057	revert workaround example in README.rst for <script> and <style> now that it is properly fixed (#115 ) Signed-off-by: chrispy <chrispy@synopsys.com>	2024-03-26 21:15:22 +01:00
André van Delft	2f9a42d3b8	Strip text before adding blockquote markers (#76 )	2024-03-26 21:07:28 +01:00
AlexVonB	96a25cfbf3	added tests for linebreaks in table cells	2024-03-26 21:05:31 +01:00
Carina de Oliveira Antunes	0477a0c8a0	convert_td: strip text (#91 )	2024-03-26 20:49:50 +01:00
Veronika Butkevich	f33ccd7c1a	Fix newline start in header tags (#89 ) * Fix newline start in header tags	2024-03-26 20:46:30 +01:00
G	a2f82678f7	Add no css example to readme (#111 ) * Add no css example --------- Co-authored-by: G <17325189+Chichilele@users.noreply.github.com>	2024-03-11 21:10:08 +01:00
Thomas L. Kjeldsen	60967c1c95	ignore script and style content (such as css and javascript) (#112 )	2024-03-11 21:07:24 +01:00
Chris Papademetrious	c7718b6d81	Merge pull request #104 from chrispy-snps/fix/97-101-102 improve text normalization/escaping for preformatted/code contexts	2024-01-15 15:46:51 -05:00
chrispy	2b22d239ad	avoid text normalization/escaping in any preformatted/code context Signed-off-by: chrispy <chrispy@synopsys.com>	2024-01-15 10:53:14 -05:00
AlexVonB	8219d2a673	Merge branch 'develop'	2022-09-02 10:11:08 +02:00
AlexVonB	e6e23fd512	bump to v0.11.6	2022-09-02 10:10:27 +02:00
Alex	433fad2dec	added nix shell file	2022-09-02 08:50:45 +02:00
Alex	4fb451ffa6	fixed cli parameters closes #75	2022-09-02 08:44:41 +02:00
AlexVonB	0c8ac578c9	Merge branch 'develop'	2022-08-31 21:45:38 +02:00
AlexVonB	e8d041c251	bump to v0.11.5	2022-08-31 21:45:24 +02:00
AlexVonB	f729c3ba43	first test, then lint	2022-08-31 21:44:53 +02:00
AlexVonB	eddfdae4ca	fix cli options: default heading, em symbols	2022-08-31 21:44:42 +02:00
AlexVonB	8f047753ae	Merge branch 'develop'	2022-08-28 22:03:22 +02:00
AlexVonB	50b3b73a8f	bump to v0.11.4	2022-08-28 22:03:14 +02:00
AlexVonB	0310216877	fixed readme and added linter to detect this earlier	2022-08-28 22:02:49 +02:00
AlexVonB	194c646a20	Merge branch 'develop'	2022-08-28 21:43:12 +02:00
AlexVonB	9914474828	bump to v0.11.3	2022-08-28 21:42:46 +02:00
AlexVonB	6263f0e5f0	Switch to `tox` for tests (#73 )	2022-08-28 21:40:52 +02:00
Adam Bambuch	17d8586843	don't escape text in pre tag (Fenced Code Blocks) (#67 ) don't escape text in pre tag (Fenced Code Blocks)	2022-08-28 20:58:54 +02:00
AlexVonB	59eb069700	added readme for cli	2022-08-28 20:56:23 +02:00
Daniel J. Perry	e79971a7eb	Add console entry point (#72 ) * Add console entry point * Make entry point conform to linter settings.	2022-08-28 20:53:15 +02:00
AlexVonB	2c533339cf	Merge branch 'develop'	2022-04-24 11:01:54 +02:00
AlexVonB	5adda130b8	bump to v0.11.2	2022-04-24 11:01:29 +02:00
AlexVonB	5f1b98e25d	added wrap option closes #66	2022-04-24 11:00:04 +02:00
AlexVonB	16acd2b763	typo in readme	2022-04-24 10:59:22 +02:00
AlexVonB	2b8cf444f1	Merge branch 'develop'	2022-04-14 10:25:35 +02:00
AlexVonB	207d0f4ec6	bump to v0.11.1	2022-04-14 10:25:25 +02:00
Mikko Korpela	ebb9ea713d	Fix detection of "first row, not headline" (#63 ) Improved handling of "first row, not headline". Works for tables with 1) neither thead nor tbody 2) tbody but no thead	2022-04-14 10:24:32 +02:00
AlexVonB	d375116807	Merge branch 'develop'	2022-04-13 20:47:52 +02:00
AlexVonB	87b9f6c88e	bump to v0.11.0	2022-04-13 20:47:30 +02:00
AlexVonB	bda367dad9	Merge branch 'tdgroot-code_language_callback' into develop closes #64	2022-04-13 20:44:18 +02:00
AlexVonB	eb0330bfc6	Merge branch 'develop'	2022-01-23 11:01:45 +01:00
AlexVonB	28793ac0b3	Merge branch 'develop'	2022-01-18 08:56:33 +01:00
AlexVonB	9231704988	Merge branch 'develop'	2021-12-11 14:44:58 +01:00
AlexVonB	1613c302bc	Merge branch 'develop'	2021-11-17 17:11:01 +01:00
AlexVonB	55c9e84f38	Merge branch 'develop'	2021-09-04 21:50:34 +02:00
AlexVonB	99875683ac	Merge branch 'develop'	2021-08-25 08:53:38 +02:00
AlexVonB	eaeb0603eb	Merge branch 'develop'	2021-07-11 13:21:20 +02:00
AlexVonB	cb73590623	Merge branch 'develop'	2021-07-11 13:14:29 +02:00
AlexVonB	59417ab115	Merge branch 'develop'	2021-05-30 19:10:49 +02:00
AlexVonB	917b01e548	Merge branch 'develop'	2021-05-30 11:20:32 +02:00
AlexVonB	652714859d	Merge branch 'develop'	2021-05-21 14:18:14 +02:00
AlexVonB	ea5b22824b	Merge branch 'develop'	2021-05-18 10:42:27 +02:00
AlexVonB	ec5858e42f	Merge branch 'develop'	2021-05-16 18:41:24 +02:00
AlexVonB	02bb914ef3	Merge branch 'develop'	2021-05-02 13:49:30 +02:00
AlexVonB	21c0d034d0	Merge branch 'develop'	2021-05-02 10:51:00 +02:00
AlexVonB	e3ddc789a2	Merge branch 'develop'	2021-04-22 12:43:27 +02:00
AlexVonB	2d0cd97323	Merge branch 'develop'	2021-04-22 12:13:03 +02:00
AlexVonB	ec185e2e9c	Merge branch 'develop'	2021-02-21 23:09:55 +01:00
AlexVonB	079d1721aa	Merge branch 'develop'	2021-02-21 20:58:34 +01:00
AlexVonB	bf24df3e2e	bump to v0.6.3	2021-01-12 22:43:18 +01:00
AlexVonB	15329588b1	Merge branch 'develop'	2021-01-12 22:42:58 +01:00
AlexVonB	34ad8485fa	bump to v0.6.2	2021-01-12 22:40:03 +01:00
AlexVonB	f0ce934bf8	Merge branch 'develop'	2021-01-12 22:39:47 +01:00
AlexVonB	99cd237f27	Merge branch 'develop'	2021-01-04 10:22:02 +01:00
AlexVonB	2bde8d3e8e	Merge branch 'develop'	2021-01-02 16:49:28 +01:00
AlexVonB	8c9b029756	Merge branch 'develop'	2020-09-01 18:10:07 +02:00
AlexVonB	ae50065872	Merge branch 'develop'	2020-08-18 18:53:10 +02:00