stouputils.check.python module#

Rules on Python sources: indentation, comments, docstrings and module constants.

CLAUSE_ENDINGS: tuple[str, ...] = ('.', '!', '?', ':', ';', ',')[source]#

Line endings that close a sentence or a clause, where a line break reads naturally.

CLAUSE_BOUNDARY: Pattern[str] = re.compile('[.!?:;,](?:\\s|$)')[source]#

Punctuation followed by a space, which splits a line into its clauses.

NEW_ITEM: Pattern[str] = re.compile('\\*{0,2}\\w+(?:\\s+\\([^)]*\\))?:(?:\\s|$)|\\w+(?:\\[.*?\\])?(?: \\| \\w+(?:\\[.*?\\])?)+:\\s|\\w[.)]\\s|(?:noqa|type:|pyright:|ruff:|fmt:|pragma|stp:|stouputils:)')[source]#

an Args: entry, a union type, a list item, or a tool directive.

Type:

Line starts that open an entry of their own

EXAMPLES_HEADER: Pattern[str] = re.compile('\\s*Examples?:\\s*$')[source]#

Header that doctests do not need, since >>> already marks them.

TYPED_ARGUMENT: Pattern[str] = re.compile('\\s*\\*{0,2}(?!Traceback\\b)\\w+:?\\s+\\(.*\\):')[source]#

Args: entry that repeats the type the signature already carries, as in name (int):.

SPAN_DELIMITERS: tuple[str, ...] = ('``', '`', '"')[source]#

Delimiters of inline code and quotes, the double backtick counted and removed before the single one.

STRANDED_FRAGMENT: str = 'a line break leaves a few words of a clause alone, break at a clause or sentence boundary'[source]#

Message shared by comments and docstrings.

SUPPRESSION: Pattern[str] = re.compile('#\\s*(stp|stouputils):\\s*ignore\\b(?:\\[([^\\]]*)\\])?')[source]#

stp silences its own line, or the string it closes, and stouputils the whole file.

Type:

A suppression comment

LAYOUT_TOKENS: frozenset[int] = frozenset({0, 4, 5, 6, 65})[source]#

Tokens carrying no code, which a suppression comment never attaches to.

python_errors(
source: str,
config: CheckConfig,
) → Iterator[Violation][source]#

Violations of the Python rules, in no particular order.

>>> rules = lambda source: sorted((v.line, v.rule) for v in python_errors(source, CheckConfig()))
>>> rules("def f():\n\tx  = 1\n\ty\t= 2\n")
[(3, 'space-alignment')]
>>> rules("x = [\n    1,\n]\n"), rules('x = f"""\n    {1}\n"""\n')
([(2, 'tab-indentation')], [])
>>> [(v.line, v.end_line) for v in python_errors("if x:\n    a = 1\n\n    b = 2\n", CheckConfig())]
[(2, 4)]
>>> rules("# Roll once per tick. The\n# caller resets the counter.\n")
[(2, 'stranded-fragment')]
>>> rules("LIMIT: int = 3  # Retries\n")
[(1, 'constant-comment')]
suppressions(
source: str,
) → list[Suppression][source]#

Suppression comments of a Python source, none when it does not tokenize.

A stp comment after a multi-line string covers the whole string, the only way to reach the lines of a docstring.

>>> [(s.line, s.lines) for s in suppressions("def f():\n\t'''\n\tlong\n\t'''  # stp: ignore[long-docstring]\n")]
[(4, range(2, 5))]
>>> suppressions("# stouputils: ignore[tab-indentation, long-comment]\n")
[Suppression(line=1, rules=('tab-indentation', 'long-comment'), lines=None)]
>>> suppressions('x = "# stp: ignore[long-comment]"  # stp: ignore\n')
[Suppression(line=1, rules=(), lines=range(1, 2))]
suppression_from_comment(
match: Match[str],
line: int,
statement_start: int,
) → Suppression[source]#

The suppression a comment matching SUPPRESSION declares.

Parameters:
  • line – Line of the comment.

  • statement_start – First line of the statement the comment ends, which a stp comment covers down to its own line.

>>> suppression_from_comment(SUPPRESSION.search("# stp: ignore[long-comment, x]"), 5, statement_start=3)
Suppression(line=5, rules=('long-comment', 'x'), lines=range(3, 6))
statements(
tree: Module,
) → Iterator[stmt][source]#

Every statement of a module, nested ones included, without descending into expressions.

indentation_errors(
source: str,
string_lines: set[int],
tab_aligned: set[int],
) → Iterator[Violation][source]#

Lines of code indented with a space, or aligned with a tab between two tokens, one violation per run of them.

Lines continuing a multi-line string are data, so they are left alone, and neither they nor blank lines end a run.

tab_aligned_lines(
tokens: list[TokenInfo],
) → set[int][source]#

Lines where a tab separates two tokens, which leaves the tabs inside strings and comments alone.

string_content_lines(
tokens: list[TokenInfo],
) → set[int][source]#

Lines whose leading whitespace belongs to a multi-line string rather than to the code.

comment_errors(
tokens: list[TokenInfo],
config: CheckConfig,
) → Iterator[Violation][source]#

Blocks of comment lines over the limit, and fragments stranded across comment lines.

docstring_layout_errors(
tree: Module,
config: CheckConfig,
) → Iterator[Violation][source]#

A module docstring below line 1, and function or class docstrings over the size limit.

constant_errors(
tree: Module,
tokens: list[TokenInfo],
) → Iterator[Violation][source]#

Module constants documented by a trailing comment rather than a docstring below them.

is_docstring(node: stmt) → bool[source]#

Whether a statement is a bare string literal.

docstring_errors(
docstring: str,
first_line: int,
config: CheckConfig,
) → Iterator[Violation][source]#

An Examples: header, types repeated in Args:, and fragments stranded across lines.

Code blocks and doctests are skipped, and a deeper indent continues an entry rather than a sentence.

>>> [v.rule for v in docstring_errors("Args:\n\tlimit (int): Retries\nExamples:\n\t>>> f()", 1, CheckConfig())]
['typed-argument', 'examples-header']
split_spans(
lines: Iterable[tuple[int, str]],
) → Iterator[Violation][source]#

Inline code or a quote opened on one line and closed on a later one, reported at the line opening it.

>>> [v.line for v in split_spans([(1, "use ``f(a,"), (2, "b)`` here"), (3, 'and `g` or "h"'), (4, 'from "A'), (5, 'B" on')])]
[1, 4]
code_block_indent(content: str, indent: int) → int | None[source]#

Indent that lines must exceed to belong to the code a docstring line opens, None when it opens none.

A doctest owns the lines at its own indent until a blank line, an RST block only the deeper ones.

strands_fragment(
previous: str,
current: str,
max_words: int,
) → bool[source]#

Whether a line break cuts a clause and leaves at most max_words of its words alone on one side.

A break after a comma or a semicolon falls between clauses, which reads fine.

>>> strands_fragment("The wrapper is generic, so they come", "from :py:mod:`cli`. Only the rest is ours.", 3)
True
>>> strands_fragment("The wrapper is generic,", "so they come from :py:mod:`cli`.", 3)
False
>>> strands_fragment("Roll once per tick. The", "caller resets the counter.", 3)
True
>>> strands_fragment("Args:", "limit: Retries", 3)
False