PY-9795,PY-4717 Parse Numpy-style docstrings sections and fields

This commit is contained in:
Mikhail Golubev
2015-09-02 14:34:20 +03:00
parent 524510d1a6
commit 72a3cf804d
7 changed files with 155 additions and 81 deletions
@@ -20,8 +20,6 @@ import com.intellij.util.containers.ContainerUtil;
import com.jetbrains.python.toolbox.Substring;
import org.jetbrains.annotations.NotNull;
import java.util.Arrays;
import java.util.Collections;
import java.util.List;
import java.util.regex.Matcher;
import java.util.regex.Pattern;
@@ -32,7 +30,6 @@ import java.util.regex.Pattern;
public class GoogleCodeStyleDocString extends SectionBasedDocString {
public static final Pattern SECTION_HEADER_RE = Pattern.compile("\\s*(\\w+):\\s*", Pattern.MULTILINE);
private static final Pattern FIELD_NAME_AND_TYPE_RE = Pattern.compile("\\s*(.+?)\\s*\\(\\s*(.+?)\\s*\\)\\s*");
private static final Pattern SPHINX_REFERENCE_RE = Pattern.compile("(:\\w+:\\S+:`.+?`|:\\S+:`.+?`|`.+?`)");
public GoogleCodeStyleDocString(@NotNull String text) {
super(text);
@@ -73,12 +70,13 @@ public class GoogleCodeStyleDocString extends SectionBasedDocString {
* </ol>
*/
@Override
protected Pair<SectionField, Integer> parseField(int lineNum,
int sectionIndent,
boolean mayHaveType,
boolean preferType) {
protected Pair<SectionField, Integer> parseSectionField(int lineNum,
int sectionIndent,
boolean mayHaveType,
boolean preferType) {
final Substring line = getLine(lineNum);
Substring name, type = null, description;
final List<Substring> colonSeparatedParts = splitFieldStartLineByColon(getLine(lineNum));
final List<Substring> colonSeparatedParts = splitByFirstColon(line);
assert colonSeparatedParts.size() <= 2;
if (colonSeparatedParts.size() < 2) {
return Pair.create(null, lineNum);
@@ -98,41 +96,17 @@ public class GoogleCodeStyleDocString extends SectionBasedDocString {
name = null;
}
description = colonSeparatedParts.get(1);
final Pair<List<Substring>, Integer> pair = parseIndentedBlock(lineNum + 1, getIndent(getLine(lineNum)), sectionIndent);
// parse line with indentation at least one space greater than indentation of the field
final Pair<List<Substring>, Integer> pair = parseIndentedBlock(lineNum + 1, getIndent(line), sectionIndent);
final List<Substring> nestedBlock = pair.getFirst();
if (!nestedBlock.isEmpty()) {
//noinspection ConstantConditions
description = mergeSubstrings(description, ContainerUtil.getLastItem(nestedBlock));
}
assert description != null;
description = description.trim();
return Pair.create(new SectionField(name, type, description), pair.getSecond());
}
/**
* Partitions line by colon if it contains type references, e.g.
* <p/>
* <pre><code>
* runtime (:class:`Runtime`): Use it to access the environment.
* </code></pre>
*/
@NotNull
private static List<Substring> splitFieldStartLineByColon(@NotNull Substring line) {
final List<Substring> parts = line.split(SPHINX_REFERENCE_RE);
if (parts.size() > 1) {
for (Substring part : parts) {
final int i = part.indexOf(":");
if (i >= 0) {
final Substring beforeColon = new Substring(line.getSuperString(), line.getStartOffset(), part.getStartOffset() + i);
final Substring afterColon = new Substring(line.getSuperString(), part.getStartOffset() + i + 1, line.getEndOffset());
return Arrays.asList(beforeColon, afterColon);
}
}
return Collections.singletonList(line);
}
return line.split(":", 1);
}
@NotNull
@Override
@@ -19,16 +19,17 @@ import com.intellij.openapi.util.Pair;
import com.jetbrains.python.toolbox.Substring;
import org.jetbrains.annotations.NotNull;
import java.util.List;
import java.util.regex.Pattern;
/**
* @author Mikhail Golubev
*/
public class NumpyDocString extends SectionBasedDocString {
private static final Pattern SIGNATURE = Pattern.compile("^([\\w., ]+=)?\\s*[\\w\\.]+\\(.*\\)$");
private static final Pattern SECTION_HEADER = Pattern.compile("^[-=]+");
private static final Pattern SIGNATURE = Pattern.compile("^\\s*([\\w., ]+=)?\\s*[\\w\\.]+\\(.*\\)\\s*$");
private static final Pattern SECTION_HEADER = Pattern.compile("^\\s*[-=]{2,}\\s*$");
private Substring mySignature = null;
private Substring mySignature;
public NumpyDocString(@NotNull String text) {
super(text);
}
@@ -39,8 +40,9 @@ public class NumpyDocString extends SectionBasedDocString {
final Substring line = getLineOrNull(nextNonEmptyLineNum);
if (line != null && SIGNATURE.matcher(line).matches()) {
mySignature = line.trim();
return nextNonEmptyLineNum + 1;
}
return super.parseHeader(startLine);
return nextNonEmptyLineNum;
}
@NotNull
@@ -54,11 +56,32 @@ public class NumpyDocString extends SectionBasedDocString {
}
@Override
protected Pair<SectionField, Integer> parseField(int lineNum,
int sectionIndent,
boolean mayHaveType,
boolean preferType) {
return null;
protected Pair<SectionField, Integer> parseSectionField(int lineNum,
int sectionIndent,
boolean mayHaveType,
boolean preferType) {
final Substring line = getLine(lineNum);
Substring name, type = null, description = null;
if (mayHaveType) {
final List<Substring> colonSeparatedParts = splitByFirstColon(line);
name = colonSeparatedParts.get(0).trim();
if (colonSeparatedParts.size() == 2) {
type = colonSeparatedParts.get(1).trim();
}
}
else {
name = line.trim();
}
if (preferType && type == null) {
type = name;
name = null;
}
final Pair<List<Substring>, Integer> parsedDescription = parseIndentedBlock(lineNum + 1, getIndent(line), sectionIndent);
final List<Substring> descriptionLines = parsedDescription.getFirst();
if (!descriptionLines.isEmpty()) {
description = mergeSubstrings(descriptionLines.get(0), descriptionLines.get(descriptionLines.size() - 1));
}
return Pair.create(new SectionField(name, type, description != null ? description.trim() : null), parsedDescription.getSecond());
}
@NotNull
@@ -30,6 +30,7 @@ import org.jetbrains.annotations.NotNull;
import org.jetbrains.annotations.Nullable;
import java.util.*;
import java.util.regex.Pattern;
/**
* Common base class for docstring styles supported by Napoleon Sphinx extension.
@@ -38,7 +39,7 @@ import java.util.*;
* @see <a href="http://sphinxcontrib-napoleon.readthedocs.org/en/latest/index.html">Napoleon</a>
*/
public abstract class SectionBasedDocString implements StructuredDocString {
/**
* Frequently used section types
*/
@@ -50,7 +51,7 @@ public abstract class SectionBasedDocString implements StructuredDocString {
@NonNls protected static final String METHODS_SECTION = "methods";
@NonNls protected static final String OTHER_PARAMETERS_SECTION = "other parameters";
@NonNls protected static final String YIELDS_SECTION = "yields";
protected static final Map<String, String> SECTION_ALIASES =
ImmutableMap.<String, String>builder()
.put("arguments", PARAMETERS_SECTION)
@@ -76,11 +77,12 @@ public abstract class SectionBasedDocString implements StructuredDocString {
.put("warns", "warnings")
.put("warnings", "warnings")
.build();
private static final Pattern SPHINX_REFERENCE_RE = Pattern.compile("(:\\w+:\\S+:`.+?`|:\\S+:`.+?`|`.+?`)");
public static Set<String> SECTION_NAMES = SECTION_ALIASES.keySet();
private static final ImmutableSet<String> SECTIONS_WITH_NAME_AND_OPTIONAL_TYPE = ImmutableSet.of(ATTRIBUTES_SECTION,
PARAMETERS_SECTION,
KEYWORD_ARGUMENTS_SECTION,
private static final ImmutableSet<String> SECTIONS_WITH_NAME_AND_OPTIONAL_TYPE = ImmutableSet.of(ATTRIBUTES_SECTION,
PARAMETERS_SECTION,
KEYWORD_ARGUMENTS_SECTION,
OTHER_PARAMETERS_SECTION);
private static final ImmutableSet<String> SECTIONS_WITH_TYPE_AND_OPTIONAL_NAME = ImmutableSet.of(RETURNS_SECTION, YIELDS_SECTION);
private static final ImmutableSet<String> SECTIONS_WITH_TYPE = ImmutableSet.of(RAISES_SECTION);
@@ -95,8 +97,8 @@ public abstract class SectionBasedDocString implements StructuredDocString {
protected SectionBasedDocString(@NotNull String text) {
myLines = new Substring(text).splitLines();
List<Substring> summary = Collections.emptyList();
final int startLine = parseHeader(0);
int lineNum = skipEmptyLines(startLine);
int startLine = skipEmptyLines(parseHeader(0));
int lineNum = startLine;
while (lineNum < myLines.size()) {
final Pair<Section, Integer> parsedSection = parseSection(lineNum);
if (parsedSection.getFirst() != null) {
@@ -148,7 +150,7 @@ public abstract class SectionBasedDocString implements StructuredDocString {
final List<SectionField> fields = new ArrayList<SectionField>();
final int sectionIndent = getIndent(getLine(sectionStartLine));
while (!isSectionBreak(lineNum, sectionIndent)) {
final Pair<SectionField, Integer> result = parseField(lineNum, title, sectionIndent);
final Pair<SectionField, Integer> result = parseSectionField(lineNum, title, sectionIndent);
if (result.getFirst() == null) {
break;
}
@@ -159,26 +161,26 @@ public abstract class SectionBasedDocString implements StructuredDocString {
}
@NotNull
protected Pair<SectionField, Integer> parseField(int lineNum, @NotNull String normalizedSectionTitle, int sectionIndent) {
protected Pair<SectionField, Integer> parseSectionField(int lineNum, @NotNull String normalizedSectionTitle, int sectionIndent) {
if (SECTIONS_WITH_NAME_AND_OPTIONAL_TYPE.contains(normalizedSectionTitle)) {
return parseField(lineNum, sectionIndent, true, false);
return parseSectionField(lineNum, sectionIndent, true, false);
}
if (SECTIONS_WITH_TYPE_AND_OPTIONAL_NAME.contains(normalizedSectionTitle)) {
return parseField(lineNum, sectionIndent, true, true);
return parseSectionField(lineNum, sectionIndent, true, true);
}
if (SECTIONS_WITH_NAME.contains(normalizedSectionTitle)) {
return parseField(lineNum, sectionIndent, false, false);
return parseSectionField(lineNum, sectionIndent, false, false);
}
if (SECTIONS_WITH_TYPE.contains(normalizedSectionTitle)) {
return parseField(lineNum, sectionIndent, false, true);
return parseSectionField(lineNum, sectionIndent, false, true);
}
return parseGenericField(lineNum, sectionIndent);
}
protected abstract Pair<SectionField, Integer> parseField(int lineNum,
int sectionIndent,
boolean mayHaveType,
boolean preferType);
protected abstract Pair<SectionField, Integer> parseSectionField(int lineNum,
int sectionIndent,
boolean mayHaveType,
boolean preferType);
@NotNull
protected Pair<SectionField, Integer> parseGenericField(int lineNum, int sectionIndent) {
@@ -186,7 +188,6 @@ public abstract class SectionBasedDocString implements StructuredDocString {
final Substring firstLine = ContainerUtil.getFirstItem(pair.getFirst());
final Substring lastLine = ContainerUtil.getLastItem(pair.getFirst());
if (firstLine != null && lastLine != null) {
//noinspection ConstantConditions
return Pair.create(new SectionField(null, null, mergeSubstrings(firstLine, lastLine).trim()), pair.getSecond());
}
return Pair.create(null, pair.getSecond());
@@ -229,6 +230,8 @@ public abstract class SectionBasedDocString implements StructuredDocString {
/**
* Consumes all lines that are indented more than {@code blockIndent} and don't contain start of a new section.
* Trailing empty lines (e.g. due to indentation of closing triple quotes) are omitted in result.
*
* @param blockIndent indentation threshold, block ends with a line that has greater indentation
*/
@NotNull
protected Pair<List<Substring>, Integer> parseIndentedBlock(int lineNum, int blockIndent, int sectionIndent) {
@@ -238,7 +241,7 @@ public abstract class SectionBasedDocString implements StructuredDocString {
if (getIndent(getLine(lineNum)) > blockIndent) {
// copy all lines after the last non empty including the current one
for (int i = lastNonEmpty + 1; i <= lineNum; i++) {
result.add(getLine(lineNum));
result.add(getLine(i));
}
lastNonEmpty = lineNum;
}
@@ -250,6 +253,31 @@ public abstract class SectionBasedDocString implements StructuredDocString {
return Pair.create(result, lineNum);
}
/**
* Properly partitions line by first colon taking into account possible Sphinx references inside
* <p/>
* <h3>Example</h3>
* <pre><code>
* runtime (:class:`Runtime`): Use it to access the environment.
* </code></pre>
*/
@NotNull
protected static List<Substring> splitByFirstColon(@NotNull Substring line) {
final List<Substring> parts = line.split(SPHINX_REFERENCE_RE);
if (parts.size() > 1) {
for (Substring part : parts) {
final int i = part.indexOf(":");
if (i >= 0) {
final Substring beforeColon = new Substring(line.getSuperString(), line.getStartOffset(), part.getStartOffset() + i);
final Substring afterColon = new Substring(line.getSuperString(), part.getStartOffset() + i + 1, line.getEndOffset());
return Arrays.asList(beforeColon, afterColon);
}
}
return Collections.singletonList(line);
}
return line.split(":", 1);
}
/**
* If both substrings share the same origin, returns new substring that includes both of them. Otherwise return {@code null}.
*
@@ -257,14 +285,14 @@ public abstract class SectionBasedDocString implements StructuredDocString {
* @param s2 substring to concat with
* @return new substring as described
*/
@Nullable
@NotNull
public static Substring mergeSubstrings(@NotNull Substring s1, @NotNull Substring s2) {
if (s1.getSuperString().equals(s2.getSuperString())) {
return new Substring(s1.getSuperString(),
Math.min(s1.getStartOffset(), s2.getStartOffset()),
Math.max(s1.getEndOffset(), s2.getEndOffset()));
if (!s1.getSuperString().equals(s2.getSuperString())) {
throw new IllegalArgumentException(String.format("Substrings '%s' and '%s' must belong to the same origin", s1, s2));
}
return null;
return new Substring(s1.getSuperString(),
Math.min(s1.getStartOffset(), s2.getStartOffset()),
Math.max(s1.getEndOffset(), s2.getEndOffset()));
}
// like Python's textwrap.dedent()
@@ -464,7 +492,7 @@ public abstract class SectionBasedDocString implements StructuredDocString {
final SectionField field = getFirstReturnField();
return field != null ? field.getType() : null;
}
@Nullable
@Override
public Substring getReturnTypeSubstring() {
@@ -498,7 +526,7 @@ public abstract class SectionBasedDocString implements StructuredDocString {
}
});
}
@Nullable
@Override
public String getRaisedExceptionDescription(@Nullable String exceptionName) {
@@ -519,7 +547,7 @@ public abstract class SectionBasedDocString implements StructuredDocString {
result.addAll(section.getFields());
}
}
return result;
return result;
}
@Nullable
@@ -531,7 +559,7 @@ public abstract class SectionBasedDocString implements StructuredDocString {
}
});
}
@Nullable
@Override
public String getAttributeDescription() {
@@ -0,0 +1,14 @@
class ndarray(object):
def diagonal(self, offset=0, axis1=0, axis2=1):
"""
a.diagonal(offset=0, axis1=0, axis2=1)
Return specified diagonals.
Refer to `numpy.diagonal` for full documentation.
See Also
--------
numpy.diagonal : equivalent function
"""
pass
@@ -11,4 +11,4 @@ def func(x, y, *args, **kwargs):
Returns:
None: always
"""
pass
pass
@@ -0,0 +1,21 @@
def func(x, y, *args, **kwargs):
"""Summary
Parameters
----------
x : int
first parameter
y
second parameter
with longer description
Raises
======
Exception
if anything bad happens
Returns
=======
None
always
"""
pass
@@ -20,6 +20,7 @@ import com.intellij.psi.search.PsiElementProcessor;
import com.intellij.psi.util.PsiTreeUtil;
import com.jetbrains.python.documentation.GoogleCodeStyleDocString;
import com.jetbrains.python.documentation.NumpyDocString;
import com.jetbrains.python.documentation.SectionBasedDocString;
import com.jetbrains.python.documentation.SectionBasedDocString.Section;
import com.jetbrains.python.documentation.SectionBasedDocString.SectionField;
import com.jetbrains.python.fixtures.PyTestCase;
@@ -33,13 +34,20 @@ import java.util.List;
* @author Mikhail Golubev
*/
public class PySectionBasedDocStringTest extends PyTestCase {
public void testSimpleFunctionDocString() {
final GoogleCodeStyleDocString docString = findAndParseGoogleStyleDocString();
public void testSimpleGoogleDocString() {
checkSimpleDocstringStructure(findAndParseGoogleStyleDocString());
}
public void testSimpleNumpyDocstring() {
checkSimpleDocstringStructure(findAndParseNumpyStyleDocString());
}
private static void checkSimpleDocstringStructure(@NotNull SectionBasedDocString docString) {
assertEquals("Summary", docString.getSummary());
final List<Section> sections = docString.getSections();
assertSize(3, sections);
assertEquals("parameters", sections.get(0).getTitle());
final List<SectionField> paramFields = sections.get(0).getFields();
assertSize(2, paramFields);
@@ -113,9 +121,9 @@ public class PySectionBasedDocStringTest extends PyTestCase {
public void testSectionStartAfterQuotes() {
final GoogleCodeStyleDocString docString = findAndParseGoogleStyleDocString();
assertEmpty(docString.getSummary());
assertSize(2, docString.getSections());
final Section examplesSection = docString.getSections().get(0);
assertEquals("examples", examplesSection.getTitle());
assertSize(1, examplesSection.getFields());
@@ -124,7 +132,7 @@ public class PySectionBasedDocStringTest extends PyTestCase {
assertEmpty(example1.getType());
assertEquals("Useless call\n" +
"func() == func()", example1.getDescription());
final Section notesSection = docString.getSections().get(1);
assertEquals("notes", notesSection.getTitle());
assertSize(1, notesSection.getFields());
@@ -192,7 +200,7 @@ public class PySectionBasedDocStringTest extends PyTestCase {
assertEquals("status_code", return1.getName());
assertEquals("int", return1.getType());
assertEquals("HTTP status code", return1.getDescription());
final SectionField return2 = returnSection.getFields().get(1);
assertEquals("template", return2.getName());
assertEquals("str", return2.getType());
@@ -207,6 +215,12 @@ public class PySectionBasedDocStringTest extends PyTestCase {
"of the task", yield1.getDescription());
}
public void testNumpySignature() {
final NumpyDocString docString = findAndParseNumpyStyleDocString();
assertEquals("a.diagonal(offset=0, axis1=0, axis2=1)", docString.getSignature());
assertEquals("Return specified diagonals.", docString.getSummary());
}
@Override
protected String getTestDataPath() {
return super.getTestDataPath() + "/docstrings";