Merge branch 'develop'

bump to v0.11.3
Switch to tox for tests (#73 )
2022-08-28 21:43:12 +02:00 · 2022-08-28 21:42:46 +02:00 · 2022-08-28 21:40:52 +02:00 · 2022-08-28 20:58:54 +02:00 · 2022-08-28 20:56:23 +02:00 · 2022-08-28 20:53:15 +02:00
10 changed files with 147 additions and 76 deletions
--- a/.github/workflows/python-app.yml
+++ b/.github/workflows/python-app.yml
@@ -23,11 +23,7 @@ jobs:
    - name: Install dependencies
      run: |
        python -m pip install --upgrade pip
-        pip install flake8==3.8.4 pytest
-        if [ -f requirements.txt ]; then pip install -r requirements.txt; fi
-    - name: Lint with flake8
+        pip install tox
+    - name: Lint and test
      run: |
-        python setup.py lint
-    - name: Test with pytest
-      run: |
-        python setup.py test
+        tox
--- a/.gitignore
+++ b/.gitignore
@@ -9,3 +9,4 @@
 /venv
 build/
 .vscode/settings.json
+.tox/
--- a/README.rst
+++ b/README.rst
@@ -92,7 +92,7 @@ sub_symbol, sup_symbol
 newline_style
  Defines the style of marking linebreaks (``<br>``) in markdown. The default
  value ``SPACES`` of this option will adopt the usual two spaces and a newline,
-  while ``BACKSLASH`` will convert a linebreak to ``\\n`` (a backslash an a
+  while ``BACKSLASH`` will convert a linebreak to ``\\n`` (a backslash and a
  newline). While the latter convention is non-standard, it is commonly
  preferred and supported by a lot of interpreters.

@@ -130,6 +130,11 @@ keep_inline_images_in
  that should be allowed to contain inline images, for example ``['td']``.
  Defaults to an empty list.

+wrap, wrap_width
+  If ``wrap`` is set to ``True``, all text paragraphs are wrapped at
+  ``wrap_width`` characters. Defaults to ``False`` and ``80``.
+  Use with ``newline_style=BACKSLASH`` to keep line breaks in paragraphs.
+
 Options may be specified as kwargs to the ``markdownify`` function, or as a
 nested ``Options`` class in ``MarkdownConverter`` subclasses.

@@ -169,13 +174,16 @@ change:
        return ImageBlockConverter(**options).convert(html)


+Command Line Interface
+=====================
+
+Use ``markdownify example.html > example.md`` or pipe input from stdin
+(``cat example.html | markdownify > example.md``).
+Call ``markdownify -h`` to see all available options.
+They are the same as listed above and take the same arguments.
+
+
 Development
 ===========

-To run tests:
-
-``python setup.py test``
-
-To lint:
-
-``python setup.py lint``
+To run tests and the linter run ``pip install tox`` once, then ``tox``.
--- a/markdownify/init.py
+++ b/markdownify/init.py
@@ -1,4 +1,5 @@
 from bs4 import BeautifulSoup, NavigableString, Comment, Doctype
+from textwrap import fill
 import re
 import six

@@ -75,6 +76,8 @@ class MarkdownConverter(object):
        strong_em_symbol = ASTERISK
        sub_symbol = ''
        sup_symbol = ''
+        wrap = False
+        wrap_width = 80

    class Options(DefaultOptions):
        pass
@@ -155,7 +158,7 @@ class MarkdownConverter(object):
                    and el.parent.parent.name == 'pre')):
            text = whitespace_re.sub(' ', text)

-        if el.parent.name != 'code':
+        if el.parent.name != 'code' and el.parent.name != 'pre':
            text = self.escape(text)

        # remove trailing whitespaces if any of the following condition is true:
@@ -331,6 +334,11 @@ class MarkdownConverter(object):
    def convert_p(self, el, text, convert_as_inline):
        if convert_as_inline:
            return text
+        if self.options['wrap']:
+            text = fill(text,
+                        width=self.options['wrap_width'],
+                        break_long_words=False,
+                        break_on_hyphens=False)
        return '%s\n\n' % text if text else ''

    def convert_pre(self, el, text, convert_as_inline):
@@ -370,8 +378,13 @@ class MarkdownConverter(object):
        if is_headrow and not el.previous_sibling:
            # first row and is headline: print headline underline
            underline += '| ' + ' | '.join(['---'] * len(cells)) + ' |' + '\n'
-        elif not el.previous_sibling and not el.parent.name != 'table':
-            # first row, not headline, and the parent is sth. like tbody:
+        elif (not el.previous_sibling
+              and (el.parent.name == 'table'
+                   or (el.parent.name == 'tbody'
+                       and not el.parent.previous_sibling))):
+            # first row, not headline, and:
+            # - the parent is table or
+            # - the parent is tbody at the beginning of a table.
            # print empty headline above this row
            overline += '| ' + ' | '.join([''] * len(cells)) + ' |' + '\n'
            overline += '| ' + ' | '.join(['---'] * len(cells)) + ' |' + '\n'
--- a/markdownify/main.py
+++ b/markdownify/main.py
@@ -0,0 +1,65 @@
+#!/usr/bin/env python
+
+import argparse
+import sys
+
+from markdownify import markdownify
+
+
+def main(argv=sys.argv[1:]):
+    parser = argparse.ArgumentParser(
+        prog='markdownify',
+        description='Converts html to markdown.',
+    )
+
+    parser.add_argument('html', nargs='?', type=argparse.FileType('r'),
+                        default=sys.stdin,
+                        help="The html file to convert. Defaults to STDIN if not "
+                        "provided.")
+    parser.add_argument('-s', '--strip', nargs='*',
+                        help="A list of tags to strip. This option can't be used with "
+                        "the --convert option.")
+    parser.add_argument('-c', '--convert', nargs='*',
+                        help="A list of tags to convert. This option can't be used with "
+                        "the --strip option.")
+    parser.add_argument('-a', '--autolinks', action='store_true',
+                        help="A boolean indicating whether the 'automatic link' style "
+                        "should be used when a 'a' tag's contents match its href.")
+    parser.add_argument('--default-title', action='store_false',
+                        help="A boolean to enable setting the title of a link to its "
+                        "href, if no title is given.")
+    parser.add_argument('--heading-style',
+                        choices=('ATX', 'ATX_CLOSED', 'SETEXT', 'UNDERLINED'),
+                        help="Defines how headings should be converted.")
+    parser.add_argument('-b', '--bullets', default='*+-',
+                        help="A string of bullet styles to use; the bullet will "
+                        "alternate based on nesting level.")
+    parser.add_argument('--sub-symbol', default='',
+                        help="Define the chars that surround '<sub>'.")
+    parser.add_argument('--sup-symbol', default='',
+                        help="Define the chars that surround '<sup>'.")
+    parser.add_argument('--code-language', default='',
+                        help="Defines the language that should be assumed for all "
+                        "'<pre>' sections.")
+    parser.add_argument('--no-escape-asterisks', dest='escape_asterisks',
+                        action='store_false',
+                        help="Do not escape '*' to '\\*' in text.")
+    parser.add_argument('--no-escape-underscores', dest='escape_underscores',
+                        action='store_false',
+                        help="Do not escape '_' to '\\_' in text.")
+    parser.add_argument('-i', '--keep-inline-images-in', nargs='*',
+                        help="Images are converted to their alt-text when the images are "
+                        "located inside headlines or table cells. If some inline images "
+                        "should be converted to markdown images instead, this option can "
+                        "be set to a list of parent tags that should be allowed to "
+                        "contain inline images.")
+    parser.add_argument('-w', '--wrap', action='store_true',
+                        help="Wrap all text paragraphs at --wrap-width characters.")
+    parser.add_argument('--wrap-width', type=int, default=80)
+
+    args = parser.parse_args(argv)
+    print(markdownify(**vars(args)))
+
+
+if __name__ == '__main__':
+    main()
--- a/setup.cfg
+++ b/setup.cfg
@@ -1,2 +0,0 @@
-[flake8]
-ignore = E501 W503
--- a/setup.py
+++ b/setup.py
@@ -2,7 +2,6 @@
 import codecs
 import os
 from setuptools import setup, find_packages
-from setuptools.command.test import test as TestCommand, Command


 read = lambda filepath: codecs.open(filepath, 'r', 'utf-8').read()
@@ -10,52 +9,10 @@ read = lambda filepath: codecs.open(filepath, 'r', 'utf-8').read()
 pkgmeta = {
    '__title__': 'markdownify',
    '__author__': 'Matthew Tretter',
-    '__version__': '0.10.3',
+    '__version__': '0.11.3',
 }

-
-class PyTest(TestCommand):
-    def finalize_options(self):
-        TestCommand.finalize_options(self)
-        self.test_args = ['tests', '-s']
-        self.test_suite = True
-
-    def run_tests(self):
-        import pytest
-        errno = pytest.main(self.test_args)
-        raise SystemExit(errno)
-
-
-class LintCommand(Command):
-    """
-    A copy of flake8's Flake8Command
-
-    """
-    description = "Run flake8 on modules registered in setuptools"
-    user_options = []
-
-    def initialize_options(self):
-        pass
-
-    def finalize_options(self):
-        pass
-
-    def distribution_files(self):
-        if self.distribution.packages:
-            for package in self.distribution.packages:
-                yield package.replace(".", os.path.sep)
-
-        if self.distribution.py_modules:
-            for filename in self.distribution.py_modules:
-                yield "%s.py" % filename
-
-    def run(self):
-        from flake8.api.legacy import get_style_guide
-        flake8_style = get_style_guide(config_file='setup.cfg')
-        paths = self.distribution_files()
-        report = flake8_style.check_files(paths)
-        raise SystemExit(report.total_errors > 0)
-
+read = lambda filepath: codecs.open(filepath, 'r', 'utf-8').read()

 setup(
    name='markdownify',
@@ -69,14 +26,9 @@ setup(
    packages=find_packages(),
    zip_safe=False,
    include_package_data=True,
-    setup_requires=[
-        'flake8>=3.8,<5',
-    ],
-    tests_require=[
-        'pytest>=6.2,<7',
-    ],
    install_requires=[
-        'beautifulsoup4>=4.9,<5', 'six>=1.15,<2'
+        'beautifulsoup4>=4.9,<5',
+        'six>=1.15,<2',
    ],
    classifiers=[
        'Environment :: Web Environment',
@@ -92,8 +44,9 @@ setup(
        'Programming Language :: Python :: 3.8',
        'Topic :: Utilities'
    ],
-    cmdclass={
-        'test': PyTest,
-        'lint': LintCommand,
-    },
+    entry_points={
+        'console_scripts': [
+            'markdownify = markdownify.main:main'
+        ]
+    }
 )
--- a/tests/test_conversions.py
+++ b/tests/test_conversions.py
@@ -177,11 +177,17 @@ def test_kbd():

 def test_p():
    assert md('<p>hello</p>') == 'hello\n\n'
+    assert md('<p>123456789 123456789</p>') == '123456789 123456789\n\n'
+    assert md('<p>123456789 123456789</p>', wrap=True, wrap_width=10) == '123456789\n123456789\n\n'
+    assert md('<p><a href="https://example.com">Some long link</a></p>', wrap=True, wrap_width=10) == '[Some long\nlink](https://example.com)\n\n'
+    assert md('<p>12345<br />67890</p>', wrap=True, wrap_width=10, newline_style=BACKSLASH) == '12345\\\n67890\n\n'
+    assert md('<p>12345678901<br />12345</p>', wrap=True, wrap_width=10, newline_style=BACKSLASH) == '12345678901\\\n12345\n\n'


 def test_pre():
    assert md('<pre>test\n    foo\nbar</pre>') == '\n```\ntest\n    foo\nbar\n```\n'
    assert md('<pre><code>test\n    foo\nbar</code></pre>') == '\n```\ntest\n    foo\nbar\n```\n'
+    assert md('<pre>this_should_not_escape</pre>') == '\n```\nthis_should_not_escape\n```\n'


 def test_s():
--- a/tests/test_tables.py
+++ b/tests/test_tables.py
@@ -139,6 +139,26 @@ table_missing_head = """<table>
    </tr>
 </table>"""

+table_body = """<table>
+    <tbody>
+        <tr>
+            <td>Firstname</td>
+            <td>Lastname</td>
+            <td>Age</td>
+        </tr>
+        <tr>
+            <td>Jill</td>
+            <td>Smith</td>
+            <td>50</td>
+        </tr>
+        <tr>
+            <td>Eve</td>
+            <td>Jackson</td>
+            <td>94</td>
+        </tr>
+    </tbody>
+</table>"""
+

 def test_table():
    assert md(table) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
@@ -148,3 +168,4 @@ def test_table():
    assert md(table_head_body) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_missing_text) == '\n\n|  | Lastname | Age |\n| --- | --- | --- |\n| Jill |  | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_missing_head) == '\n\n|  |  |  |\n| --- | --- | --- |\n| Firstname | Lastname | Age |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
+    assert md(table_body) == '\n\n|  |  |  |\n| --- | --- | --- |\n| Firstname | Lastname | Age |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
--- a/tox.ini
+++ b/tox.ini
@@ -0,0 +1,10 @@
+[tox]
+envlist = py38
+
+[testenv]
+deps =
+	flake8
+	pytest
+commands =
+	flake8 --ignore=E501,W503 markdownify tests
+	pytest
Author	SHA1	Message	Date
AlexVonB	194c646a20	Merge branch 'develop'	2022-08-28 21:43:12 +02:00
AlexVonB	9914474828	bump to v0.11.3	2022-08-28 21:42:46 +02:00
AlexVonB	6263f0e5f0	Switch to `tox` for tests (#73 )	2022-08-28 21:40:52 +02:00
Adam Bambuch	17d8586843	don't escape text in pre tag (Fenced Code Blocks) (#67 ) don't escape text in pre tag (Fenced Code Blocks)	2022-08-28 20:58:54 +02:00
AlexVonB	59eb069700	added readme for cli	2022-08-28 20:56:23 +02:00
Daniel J. Perry	e79971a7eb	Add console entry point (#72 ) * Add console entry point * Make entry point conform to linter settings.	2022-08-28 20:53:15 +02:00
AlexVonB	2c533339cf	Merge branch 'develop'	2022-04-24 11:01:54 +02:00
AlexVonB	5adda130b8	bump to v0.11.2	2022-04-24 11:01:29 +02:00
AlexVonB	5f1b98e25d	added wrap option closes #66	2022-04-24 11:00:04 +02:00
AlexVonB	16acd2b763	typo in readme	2022-04-24 10:59:22 +02:00
AlexVonB	2b8cf444f1	Merge branch 'develop'	2022-04-14 10:25:35 +02:00
AlexVonB	207d0f4ec6	bump to v0.11.1	2022-04-14 10:25:25 +02:00
Mikko Korpela	ebb9ea713d	Fix detection of "first row, not headline" (#63 ) Improved handling of "first row, not headline". Works for tables with 1) neither thead nor tbody 2) tbody but no thead	2022-04-14 10:24:32 +02:00
AlexVonB	d375116807	Merge branch 'develop'	2022-04-13 20:47:52 +02:00
AlexVonB	87b9f6c88e	bump to v0.11.0	2022-04-13 20:47:30 +02:00
AlexVonB	bda367dad9	Merge branch 'tdgroot-code_language_callback' into develop closes #64	2022-04-13 20:44:18 +02:00
AlexVonB	eb0330bfc6	Merge branch 'develop'	2022-01-23 11:01:45 +01:00
AlexVonB	28793ac0b3	Merge branch 'develop'	2022-01-18 08:56:33 +01:00
AlexVonB	9231704988	Merge branch 'develop'	2021-12-11 14:44:58 +01:00
AlexVonB	1613c302bc	Merge branch 'develop'	2021-11-17 17:11:01 +01:00
AlexVonB	55c9e84f38	Merge branch 'develop'	2021-09-04 21:50:34 +02:00
AlexVonB	99875683ac	Merge branch 'develop'	2021-08-25 08:53:38 +02:00
AlexVonB	eaeb0603eb	Merge branch 'develop'	2021-07-11 13:21:20 +02:00
AlexVonB	cb73590623	Merge branch 'develop'	2021-07-11 13:14:29 +02:00
AlexVonB	59417ab115	Merge branch 'develop'	2021-05-30 19:10:49 +02:00
AlexVonB	917b01e548	Merge branch 'develop'	2021-05-30 11:20:32 +02:00
AlexVonB	652714859d	Merge branch 'develop'	2021-05-21 14:18:14 +02:00
AlexVonB	ea5b22824b	Merge branch 'develop'	2021-05-18 10:42:27 +02:00
AlexVonB	ec5858e42f	Merge branch 'develop'	2021-05-16 18:41:24 +02:00
AlexVonB	02bb914ef3	Merge branch 'develop'	2021-05-02 13:49:30 +02:00
AlexVonB	21c0d034d0	Merge branch 'develop'	2021-05-02 10:51:00 +02:00
AlexVonB	e3ddc789a2	Merge branch 'develop'	2021-04-22 12:43:27 +02:00
AlexVonB	2d0cd97323	Merge branch 'develop'	2021-04-22 12:13:03 +02:00
AlexVonB	ec185e2e9c	Merge branch 'develop'	2021-02-21 23:09:55 +01:00
AlexVonB	079d1721aa	Merge branch 'develop'	2021-02-21 20:58:34 +01:00
AlexVonB	bf24df3e2e	bump to v0.6.3	2021-01-12 22:43:18 +01:00
AlexVonB	15329588b1	Merge branch 'develop'	2021-01-12 22:42:58 +01:00
AlexVonB	34ad8485fa	bump to v0.6.2	2021-01-12 22:40:03 +01:00
AlexVonB	f0ce934bf8	Merge branch 'develop'	2021-01-12 22:39:47 +01:00
AlexVonB	99cd237f27	Merge branch 'develop'	2021-01-04 10:22:02 +01:00
AlexVonB	2bde8d3e8e	Merge branch 'develop'	2021-01-02 16:49:28 +01:00
AlexVonB	8c9b029756	Merge branch 'develop'	2020-09-01 18:10:07 +02:00
AlexVonB	ae50065872	Merge branch 'develop'	2020-08-18 18:53:10 +02:00