SF.net SVN: docutils:[9909 ] trunk/docutils

milde--- via Docutils-checkins <[email protected]>
Newsgroups gmane.text.docutils.cvs
Message-ID <[email protected]>
Revision: 9909
          http://sourceforge.net/p/docutils/code/9909
Author:   milde
Date:     2024-08-15 16:35:34 +0000 (Thu, 15 Aug 2024)
Log Message:
-----------
Refactor "Include" directive class.

Disentangle actions and simplify interface of auxiliary methods.

`Include.run()` sets up instance attributes and calls helper functions
to read the file content and process it according to the options.
(Helper methods depend on setup in `run()`.)

Input sanity checks (too long lines and circular inclusions) are only
required when when inserting the file content into the "input_lines" of
the calling parser (which did these checks on the other "input_lines"
before starting to parsing it).

Adapt formatting.

Adapt tests.

Use shorter URL to the "include" directive documentation.

Modified Paths:
--------------
    trunk/docutils/docutils/parsers/rst/directives/misc.py
    trunk/docutils/test/test_parsers/test_rst/test_directives/test_include.py
    trunk/docutils/test/test_parsers/test_rst/test_line_length_limit.py

Modified: trunk/docutils/docutils/parsers/rst/directives/misc.py
===================================================================
--- trunk/docutils/docutils/parsers/rst/directives/misc.py	2024-08-15 14:14:24 UTC (rev 9908)
+++ trunk/docutils/docutils/parsers/rst/directives/misc.py	2024-08-15 16:35:34 UTC (rev 9909)
@@ -9,6 +9,7 @@
 __docformat__ = 'reStructuredText'
 
 import re
+import sys
 import time
 from pathlib import Path
 from typing import TYPE_CHECKING
@@ -23,11 +24,21 @@
 
 if TYPE_CHECKING:
     import os
+    if sys.version_info[:2] >= (3, 12):
+        from typing import TypeAlias
+    else:
+        from typing_extensions import TypeAlias
 
     from docutils.nodes import Node
 
+    PathString: TypeAlias = str | os.PathLike[str]
+    """File system path."""
 
-def adapt_path(path, source='', root_prefix='/'):
+    SourceString: TypeAlias = str | os.PathLike[str]
+    """Path to or informal description of a Docutils input source."""
+
+
+def adapt_path(path: str, source='', root_prefix='/') -> str:
     # Adapt path to files to include or embed.
     # `root_prefix` is prepended to absolute paths (cf. root_prefix setting),
     # `source` is the `current_source` of the including directive (which may
@@ -52,7 +63,7 @@
     start and end line or text to match before and/or after the text
     to be used.
 
-    https://docutils.sourceforge.io/docs/ref/rst/directives.html#including-an-external-document-fragment
+    https://docutils.sourceforge.io/docs/ref/rst/directives.html#include
     """
 
     required_arguments = 1
@@ -68,7 +79,7 @@
                    'start-after': directives.unchanged_required,
                    'end-before': directives.unchanged_required,
                    # ignored except for 'literal' or 'code':
-                   'number-lines': directives.unchanged,  # integer or None
+                   'number-lines': directives.value_or((None,), int),
                    'class': directives.class_option,
                    'name': directives.unchanged}
 
@@ -77,14 +88,17 @@
     def run(self) -> list[Node]:
         """Include a file as part of the content of this reST file.
 
-        Depending on the options, the file (or a clipping) is
+        Depending on the options, the file content (or a clipping) is
         converted to nodes and returned or inserted into the input stream.
         """
-        settings = self.state.document.settings
+        self.settings = settings = self.state.document.settings
         if not settings.file_insertion_enabled:
             raise self.warning('"%s" directive disabled.' % self.name)
-        tab_width = self.options.get('tab-width', settings.tab_width)
-        current_source = self.state.document.current_source
+        self.tab_width = self.options.get('tab-width', settings.tab_width)
+        self.clip_options = (self.options.get('start-line', None),
+                             self.options.get('end-line', None),
+                             self.options.get('start-after', ''),
+                             self.options.get('end-before', ''))
         path = directives.path(self.arguments[0])
         if path.startswith('<') and path.endswith('>'):
             path = '/' + path[1:-1]
@@ -91,152 +105,92 @@
             root_prefix = self.standard_include_path
         else:
             root_prefix = settings.root_prefix
-        path = adapt_path(path, current_source, root_prefix)
-        rawtext, clip_options = self._read_source(path)
-        include_lines = statemachine.string2lines(rawtext, tab_width,
-                                                  convert_whitespace=True)
-        for i, line in enumerate(include_lines):
-            if len(line) > settings.line_length_limit:
-                raise self.warning('"%s": line %d exceeds the'
-                                   ' line-length-limit.' % (path, i+1))
+        path = adapt_path(path,
+                          self.state.document.current_source,
+                          root_prefix)
+        self.options['source'] = path
 
+        inputstring = self.read_file(path)
+
         if 'literal' in self.options:
-            return self.include_literal_block(
-                source=path,
-                rawtext=rawtext,
-                tab_width=tab_width,
-                include_lines=include_lines,
-            )
-
+            return self.as_literal_block(inputstring)
         if 'code' in self.options:
-            return self.include_code_block(
-                source=path,
-                rawtext=rawtext,
-                tab_width=tab_width,
-                include_lines=include_lines,
-            )
-
-        # Prevent circular inclusion:
-        include_log = self.state.document.include_log
-        # log entries are tuples (<source>, <clip-options>)
-        if not include_log:  # new document, initialize with document source
-            include_log.append((utils.relative_path(None, current_source),
-                                (None, None, None, None)))
-        if (path, clip_options) in include_log:
-            master_paths = (pth for (pth, opt) in reversed(include_log))
-            inclusion_chain = '\n> '.join((path, *master_paths))
-            raise self.warning('circular inclusion in "%s" directive:\n%s'
-                               % (self.name, inclusion_chain))
-
+            return self.as_code_block(inputstring)
         if 'parser' in self.options:
-            return self.include_parsed(
-                source=path,
-                include_lines=include_lines,
-                clip_options=clip_options,
-            )
-
-        # Include as rST source:
-        self.include_text(
-            rawtext,
-            source=path,
-            tab_width=tab_width,
-            include_lines=include_lines,
-        )
-        # update include-log
-        include_log.append((path, clip_options))
+            return self.custom_parse(inputstring)
+        self.insert_into_input_lines(inputstring)
         return []
 
-    def _read_source(
-        self, source: str | os.PathLike[str], /,
-    ) -> tuple[str, tuple[int | None, int | None, str | None, str | None]]:
-        """Read and clip the content of *source*."""
-        settings = self.state.document.settings
-        encoding = self.options.get('encoding', settings.input_encoding)
-        error_handler = settings.input_encoding_error_handler
+    def read_file(self, path: PathString) -> str:
+        """Read text file at `path`. Clip and return content.
+
+        Provisional.
+        """
+        encoding = self.options.get('encoding', self.settings.input_encoding)
+        error_handler = self.settings.input_encoding_error_handler
         try:
-            include_file = io.FileInput(
-                source_path=source,
-                encoding=encoding,
-                error_handler=error_handler,
-            )
+            include_file = io.FileInput(source_path=path,
+                                        encoding=encoding,
+                                        error_handler=error_handler)
         except UnicodeEncodeError:
             raise self.severe(f'Problems with "{self.name}" directive path:\n'
-                              f'Cannot encode input file path "{source}" '
+                              f'Cannot encode input file path "{path}" '
                               '(wrong locale?).')
         except OSError as error:
-            raise self.severe(f'Problems with "{self.name}" directive '
-                              f'path:\n{io.error_string(error)}.')
+            raise self.severe(f'Problems with "{self.name}" directive path:\n'
+                              f'{io.error_string(error)}.')
         else:
-            settings.record_dependencies.add(source)
-
-        # Get to-be-included content
-        startline = self.options.get('start-line', None)
-        endline = self.options.get('end-line', None)
+            self.settings.record_dependencies.add(path)
         try:
-            if startline or (endline is not None):
-                lines = include_file.readlines()
-                rawtext = ''.join(lines[startline:endline])
-            else:
-                rawtext = include_file.read()
+            text = include_file.read()
         except UnicodeError as error:
             raise self.severe(f'Problem with "{self.name}" directive:\n'
                               + io.error_string(error))
+        # Clip to-be-included content
+        startline, endline, starttext, endtext = self.clip_options
+        if startline or (endline is not None):
+            lines = text.splitlines()
+            text = '\n'.join(lines[startline:endline])
         # start-after/end-before: no restrictions on newlines in match-text,
         # and no restrictions on matching inside lines vs. line boundaries
-        after_text = self.options.get('start-after', None)
-        if after_text:
-            # skip content in rawtext before *and incl.* a matching text
-            after_index = rawtext.find(after_text)
+        if starttext:
+            # skip content in text before *and incl.* a matching text
+            after_index = text.find(starttext)
             if after_index < 0:
-                raise self.severe('Problem with "start-after" option of "%s" '
-                                  'directive:\nText not found.' % self.name)
-            rawtext = rawtext[after_index + len(after_text):]
-        before_text = self.options.get('end-before', None)
-        if before_text:
-            # skip content in rawtext after *and incl.* a matching text
-            before_index = rawtext.find(before_text)
+                raise self.severe('Problem with "start-after" option of '
+                                  f'"{self.name}" directive:\nText not found.')
+            text = text[after_index + len(starttext):]
+        if endtext:
+            # skip content in text after *and incl.* a matching text
+            before_index = text.find(endtext)
             if before_index < 0:
-                raise self.severe('Problem with "end-before" option of "%s" '
-                                  'directive:\nText not found.' % self.name)
-            rawtext = rawtext[:before_index]
+                raise self.severe('Problem with "end-before" option of '
+                                  f'"{self.name}" directive:\nText not found.')
+            text = text[:before_index]
+        return text
 
-        clip_options = (startline, endline, before_text, after_text)
-        return rawtext, clip_options
+    def as_literal_block(self, text: str) -> list[nodes.literal_block]:
+        """Return list with literal_block containing `text`.
 
-    def include_literal_block(
-        self,
-        *,
-        source: str | os.PathLike[str],
-        rawtext: str,
-        include_lines: list[str],
-        tab_width: int,
-    ) -> list[nodes.literal_block]:
-        """Return the content from *source* as a literal block."""
-        # Don't convert tabs to spaces, if `tab_width` is negative.
-        if tab_width >= 0:
-            text = rawtext.expandtabs(tab_width)
-        else:
-            text = rawtext
+        Provisional
+        """
+        source = self.options['source']
+        # Convert tabs to spaces unless `tab_width` is negative.
+        if self.tab_width >= 0:
+            text = text.expandtabs(self.tab_width)
         literal_block = nodes.literal_block(
-            rawtext, source=source,
-            classes=self.options.get('class', []))
-        literal_block.line = 1
+            '', source=source, classes=self.options.get('class', []))
+        literal_block.source = source
+        literal_block.line = self.options.get('start-line', 0) + 1
         self.add_name(literal_block)
         if 'number-lines' in self.options:
-            try:
-                startline = int(self.options['number-lines'] or 1)
-            except ValueError:
-                msg = ':number-lines: with non-integer start value'
-                raise self.error(msg)
-            endline = startline + len(include_lines)
-            if text.endswith('\n'):
-                text = text[:-1]
-            tokens = NumberLines([([], text)], startline, endline)
+            firstline = self.options['number-lines'] or 1
+            text = text.removesuffix('\n')
+            lastline = firstline + len(text.splitlines())
+            tokens = NumberLines([([], text)], firstline, lastline)
             for classes, value in tokens:
                 if classes:
-                    literal_block += nodes.inline(
-                        value, value, classes=classes,
-                    )
+                    literal_block += nodes.inline('', value, classes=classes)
                 else:
                     literal_block += nodes.Text(value)
         else:
@@ -243,50 +197,41 @@
             literal_block += nodes.Text(text)
         return [literal_block]
 
-    def include_code_block(
-        self,
-        *,
-        source: str | os.PathLike[str],
-        rawtext: str,
-        include_lines: list[str],
-        tab_width: int,
-    ) -> list[nodes.literal_block]:
-        """Return the content from *source* as a code block."""
-        self.options['source'] = source
-        # Don't convert tabs to spaces, if `tab_width` is negative:
-        if tab_width < 0:
-            include_lines = rawtext.splitlines()
-        codeblock = CodeBlock(
-            self.name,
-            [self.options.pop('code')],  # arguments
-            self.options,
-            include_lines,  # content
-            self.lineno,
-            self.content_offset,
-            self.block_text,
-            self.state,
-            self.state_machine,
-        )
+    def as_code_block(self, text: str) -> list[nodes.literal_block]:
+        """Pass `text` to the `CodeBlock` directive class.
+
+        Provisional.
+        """
+        # convert tabs to spaces unless `tab_width` is negative:
+        if self.tab_width >= 0:
+            text = text.expandtabs(self.tab_width)
+        codeblock = CodeBlock(self.name,
+                              [self.options.pop('code')],  # pass as argument
+                              self.options,
+                              [text.removesuffix('\n')],   # content
+                              self.lineno,
+                              self.content_offset,
+                              self.block_text,
+                              self.state,
+                              self.state_machine,
+                              )
         return codeblock.run()
 
-    def include_parsed(
-        self,
-        *,
-        source: str | os.PathLike[str],
-        include_lines: list[str],
-        clip_options: tuple[int | str, int | str, str, str],
-    ) -> list[Node]:
-        """Return content from *source* after parsing with the given parser."""
-        settings = self.state.document.settings
-        include_log = self.state.document.include_log
+    def custom_parse(self, text: str) -> list[Node]:
+        """Parse with custom parser.
 
-        # parse into a dummy document and return created nodes
-        _settings = settings.copy()
-        _settings._source = source
-        document = utils.new_document(source, _settings)
-        document.include_log = include_log + [(source, clip_options)]
+        Parse with ``self.options['parser']`` into a new (dummy) document,
+        apply the parser's default transforms,
+        return child elements.
+
+        Provisional.
+        """
+        settings = self.settings.copy()
+        settings._source = self.options['source']
+        document = utils.new_document(settings._source, settings)
+        document.include_log = self.state.document.include_log
         parser = self.options['parser']()
-        parser.parse('\n'.join(include_lines), document)
+        parser.parse(text, document)
         self.state.document.parse_messages.extend(document.parse_messages)
         # clean up doctree and complete parsing
         document.transformer.populate_from_components((parser,))
@@ -295,26 +240,42 @@
             document.transform_messages)
         return document.children
 
-    def include_text(
-        self,
-        text: str,
-        *,
-        source: str | os.PathLike[str],
-        tab_width: int,
-        include_lines: list[str] | None = None,
-    ) -> None:
-        """Insert *text* into the input stream."""
-        # *text* is passed for use in subclasses, but if we have already
-        # split the lines and nothing is changing, reuse those lines.
-        if include_lines is not None:
-            include_lines = statemachine.string2lines(
-                text, tab_width, convert_whitespace=True,
-            )
-        # Mark end (cf. parsers.rst.states.Body.comment())
-        include_lines += ['', '.. end of inclusion from "%s"' % source]
-        self.state_machine.insert_input(include_lines, source)
+    def insert_into_input_lines(self, text: str) -> None:
+        """Insert file content into the rST input of the calling parser.
 
+        Returns an empty list to comply with the API of `Directive.run()`.
 
+        Provisional.
+        """
+        source = self.options['source']
+        textlines = statemachine.string2lines(text, self.tab_width,
+                                              convert_whitespace=True)
+        # Sanity checks:
+        # excessively long lines
+        for i, line in enumerate(textlines):
+            if len(line) > self.settings.line_length_limit:
+                line_no = i + 1 + self.options.get('start-line', 0)
+                raise self.warning(f'"{source}": line {line_no} exceeds the'
+                                   ' line-length-limit.')
+        # circular inclusion
+        include_log = self.state.document.include_log
+        if not include_log:  # new document, initialize with document source
+            current_source = utils.relative_path(
+                                None, self.state.document.current_source)
+            include_log.append((current_source, (None, None, '', '')))
+        if (source, self.clip_options) in include_log:
+            source_chain = (pth for (pth, opt) in reversed(include_log))
+            inclusion_chain = '\n> '.join((source, *source_chain))
+            raise self.warning(f'circular inclusion in "{self.name}"'
+                               f' directive:\n{inclusion_chain}')
+        include_log.append((source, self.clip_options))
+        # marker for removing log entry (cf. parsers.rst.states.Body.comment())
+        textlines += ['', f'.. end of inclusion from "{source}"']
+
+        self.state_machine.insert_input(textlines, source)
+        # TODO: if startline != 0, line numbers are wrong.
+
+
 class Raw(Directive):
 
     """

Modified: trunk/docutils/test/test_parsers/test_rst/test_directives/test_include.py
===================================================================
--- trunk/docutils/test/test_parsers/test_rst/test_directives/test_include.py	2024-08-15 14:14:24 UTC (rev 9908)
+++ trunk/docutils/test/test_parsers/test_rst/test_directives/test_include.py	2024-08-15 16:35:34 UTC (rev 9909)
@@ -1457,7 +1457,6 @@
             {include15}
             > {include16}
             > {include15}
-            > test data
         <literal_block xml:space="preserve">
             .. include:: include15.rst
     <paragraph>

Modified: trunk/docutils/test/test_parsers/test_rst/test_line_length_limit.py
===================================================================
--- trunk/docutils/test/test_parsers/test_rst/test_line_length_limit.py	2024-08-15 14:14:24 UTC (rev 9908)
+++ trunk/docutils/test/test_parsers/test_rst/test_line_length_limit.py	2024-08-15 16:35:34 UTC (rev 9909)
@@ -35,6 +35,8 @@
 
 
 class ParserTestCase(unittest.TestCase):
+    maxDiff = None
+
     def test_parser(self):
         parser = Parser()
         settings = get_default_settings(Parser)
@@ -77,7 +79,6 @@
 ============
 
 .. include:: {docutils_conf}
-   :literal:
 
 A paragraph.
 """,
@@ -91,10 +92,30 @@
                 "{docutils_conf}": line 5 exceeds the line-length-limit.
             <literal_block xml:space="preserve">
                 .. include:: {docutils_conf}
-                   :literal:
         <paragraph>
             A paragraph.
 """],
+[f"""\
+Include Test 2
+
+.. include:: {docutils_conf}
+   :start-line: 3
+
+A paragraph.
+""",
+f"""\
+<document source="test data">
+    <paragraph>
+        Include Test 2
+    <system_message level="2" line="3" source="test data" type="WARNING">
+        <paragraph>
+            "{docutils_conf}": line 5 exceeds the line-length-limit.
+        <literal_block xml:space="preserve">
+            .. include:: {docutils_conf}
+               :start-line: 3
+    <paragraph>
+        A paragraph.
+"""],
 ]
 
 

This was sent by the SourceForge.net collaborative development platform, the world's largest Open Source development site.
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.