Merge branch 'develop'

bump to v0.10.1
allow flake8 v4.x
2021-12-11 14:44:58 +01:00 · 2021-12-11 14:44:34 +01:00 · 2021-12-11 14:43:14 +01:00 · 2021-11-17 17:11:01 +01:00 · 2021-11-17 17:10:15 +01:00 · 2021-11-17 17:09:47 +01:00
7 changed files with 50 additions and 8 deletions
--- a/.gitignore
+++ b/.gitignore
@@ -8,3 +8,4 @@
 /MANIFEST
 /venv
 build/
+.vscode/settings.json
--- a/README.rst
+++ b/README.rst
@@ -96,6 +96,12 @@ newline_style
  newline). While the latter convention is non-standard, it is commonly
  preferred and supported by a lot of interpreters.

+code_language
+  Defines the language that should be assumed for all ``<pre>`` sections.
+  Useful, if all code on a page is in the same programming language and
+  should be annotated with `````python`` or similar.
+  Defaults to ``''`` (empty string) and can be any string.
+
 Options may be specified as kwargs to the ``markdownify`` function, or as a
 nested ``Options`` class in ``MarkdownConverter`` subclasses.

--- a/markdownify/init.py
+++ b/markdownify/init.py
@@ -76,6 +76,7 @@ class MarkdownConverter(object):
        strong_em_symbol = ASTERISK
        sub_symbol = ''
        sup_symbol = ''
+        code_language = ''

    class Options(DefaultOptions):
        pass
@@ -96,11 +97,14 @@ class MarkdownConverter(object):

    def process_tag(self, node, convert_as_inline, children_only=False):
        text = ''
-        # markdown headings can't include block elements (elements w/newlines)
+
+        # markdown headings or cells can't include
+        # block elements (elements w/newlines)
        isHeading = html_heading_re.match(node.name) is not None
+        isCell = node.name in ['td', 'th']
        convert_children_as_inline = convert_as_inline

-        if not children_only and isHeading:
+        if not children_only and (isHeading or isCell):
            convert_children_as_inline = True

        # Remove whitespace-only textnodes in purely nested nodes
@@ -200,8 +204,6 @@ class MarkdownConverter(object):
        prefix, suffix, text = chomp(text)
        if not text:
            return ''
-        if convert_as_inline:
-            return text
        href = el.get('href')
        title = el.get('title')
        # For the replacement see #29: text nodes underscores are escaped
@@ -313,7 +315,7 @@ class MarkdownConverter(object):
                el = el.parent
            bullets = self.options['bullets']
            bullet = bullets[depth % len(bullets)]
-        return '%s %s\n' % (bullet, text or '')
+        return '%s %s\n' % (bullet, (text or '').strip())

    def convert_p(self, el, text, convert_as_inline):
        if convert_as_inline:
@@ -323,7 +325,7 @@ class MarkdownConverter(object):
    def convert_pre(self, el, text, convert_as_inline):
        if not text:
            return ''
-        return '\n```\n%s\n```\n' % text
+        return '\n```%s\n%s\n```\n' % (self.options['code_language'], text)

    convert_s = convert_del

--- a/setup.py
+++ b/setup.py
@@ -10,7 +10,7 @@ read = lambda filepath: codecs.open(filepath, 'r', 'utf-8').read()
 pkgmeta = {
    '__title__': 'markdownify',
    '__author__': 'Matthew Tretter',
-    '__version__': '0.9.2',
+    '__version__': '0.10.1',
 }


@@ -70,7 +70,7 @@ setup(
    zip_safe=False,
    include_package_data=True,
    setup_requires=[
-        'flake8>=3.8,<4',
+        'flake8>=3.8,<5',
    ],
    tests_require=[
        'pytest>=6.2,<7',
--- a/tests/test_conversions.py
+++ b/tests/test_conversions.py
@@ -210,3 +210,8 @@ def test_sub():
 def test_sup():
    assert md('<sup>foo</sup>') == 'foo'
    assert md('<sup>foo</sup>', sup_symbol='^') == '^foo^'
+
+
+def test_lang():
+    assert md('<pre>test\n    foo\nbar</pre>', code_language='python') == '\n```python\ntest\n    foo\nbar\n```\n'
+    assert md('<pre><code>test\n    foo\nbar</code></pre>', code_language='javascript') == '\n```javascript\ntest\n    foo\nbar\n```\n'
--- a/tests/test_lists.py
+++ b/tests/test_lists.py
@@ -51,6 +51,14 @@ def test_nested_ols():

 def test_ul():
    assert md('<ul><li>a</li><li>b</li></ul>') == '* a\n* b\n'
+    assert md("""<ul>
+     <li>
+             a
+     </li>
+     <li> b </li>
+     <li>   c
+     </li>
+ </ul>""") == '* a\n* b\n* c\n'


 def test_inline_ul():
--- a/tests/test_tables.py
+++ b/tests/test_tables.py
@@ -39,6 +39,25 @@ table_with_html_content = """<table>
 </table>"""


+table_with_paragraphs = """<table>
+    <tr>
+        <th>Firstname</th>
+        <th><p>Lastname</p></th>
+        <th>Age</th>
+    </tr>
+    <tr>
+        <td><p>Jill</p></td>
+        <td><p>Smith</p></td>
+        <td><p>50</p></td>
+    </tr>
+    <tr>
+        <td>Eve</td>
+        <td>Jackson</td>
+        <td>94</td>
+    </tr>
+</table>"""
+
+
 table_with_header_column = """<table>
    <tr>
        <th>Firstname</th>
@@ -124,6 +143,7 @@ table_missing_head = """<table>
 def test_table():
    assert md(table) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_with_html_content) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| **Jill** | *Smith* | [50](#) |\n| Eve | Jackson | 94 |\n\n'
+    assert md(table_with_paragraphs) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_with_header_column) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_head_body) == '\n\n| Firstname | Lastname | Age |\n| --- | --- | --- |\n| Jill | Smith | 50 |\n| Eve | Jackson | 94 |\n\n'
    assert md(table_missing_text) == '\n\n|  | Lastname | Age |\n| --- | --- | --- |\n| Jill |  | 50 |\n| Eve | Jackson | 94 |\n\n'
Author	SHA1	Message	Date
AlexVonB	9231704988	Merge branch 'develop'	2021-12-11 14:44:58 +01:00
AlexVonB	c8f7cf63e3	bump to v0.10.1	2021-12-11 14:44:34 +01:00
AlexVonB	12a68a7d14	allow flake8 v4.x closes #57	2021-12-11 14:43:14 +01:00
AlexVonB	1613c302bc	Merge branch 'develop'	2021-11-17 17:11:01 +01:00
AlexVonB	478b1c7e13	bump to v0.10.0	2021-11-17 17:10:15 +01:00
AlexVonB	ffcf6cbcb2	fix readme for code_language	2021-11-17 17:09:47 +01:00
AlexVonB	0ab0452414	add readme for code_language	2021-11-17 17:08:14 +01:00
AlexVonB	b62b067cbd	Merge branch 'Inzaniak-develop' into develop	2021-11-17 17:05:07 +01:00
AlexVonB	cb2646cd93	differentiated between text and code language	2021-11-17 17:03:31 +01:00
AlexVonB	9692b5e714	satisfy linter	2021-11-17 16:55:00 +01:00
Umberto Grando	ac68c53a7d	added language for multiline code	2021-11-01 21:19:35 +01:00
AlexVonB	55c9e84f38	Merge branch 'develop'	2021-09-04 21:50:34 +02:00
AlexVonB	40dd30419c	bump to v0.9.4	2021-09-04 21:50:05 +02:00
AlexVonB	da56f7f56a	Merge pull request #53 from Hozhyi/fix/bullet_list_tags_in_separate_lines Fixed issue #52 - added stripping of text to list	2021-09-04 21:48:16 +02:00
AlexVonB	8400b39dd9	remove trailing whitespace to satisfy the linter	2021-09-04 21:47:27 +02:00
Viktor Hozhyi	5fc1441fe7	Added appropriate test	2021-09-04 20:51:08 +03:00
Viktor Hozhyi	044615eff1	Fixed issue #52 - added stripping of text to list	2021-09-04 12:39:30 +03:00
AlexVonB	99875683ac	Merge branch 'develop'	2021-08-25 08:53:38 +02:00
AlexVonB	dbd9f3f3d2	bump to v0.9.3	2021-08-25 08:53:17 +02:00
AlexVonB	0fdeb1ff6e	convert tags inside table cells as inline in part resolves #49	2021-08-25 08:48:30 +02:00