PY-4717 Add "pockets" library in python/helpers under VCS

This commit is contained in:
Mikhail Golubev
2015-09-02 14:34:27 +03:00
parent 31e755b87a
commit 57b04f9863
6 changed files with 844 additions and 0 deletions
+37
View File
@@ -0,0 +1,37 @@
# -*- coding: utf-8 -*-
# Copyright (c) 2015, by the Pockets team, see AUTHORS.
# Licensed under the BSD License, see LICENSE for details.
"""*Let me check my pockets...*
Functions available in the `pockets.*` submodules are also imported to the base
package for easy access, so::
from pockets import camel, peek_iter, resolve
works just as well as::
from pockets.inspect import resolve
from pockets.iterators import peek_iter
from pockets.string import camel
"""
from __future__ import absolute_import
from pockets._version import __version__
from pockets.collections import is_listy, listify, mappify
from pockets.inspect import resolve
from pockets.iterators import peek_iter, modify_iter
from pockets.string import camel, uncamel, splitcaps
__all__ = ["__version__",
"camel",
"uncamel",
"splitcaps",
"resolve",
"is_listy",
"listify",
"mappify",
"peek_iter",
"modify_iter"]
+8
View File
@@ -0,0 +1,8 @@
# Package versioning solution originally found here:
# http://stackoverflow.com/q/458550
# Store the version here so:
# 1) we don't load dependencies by storing it in __init__.py
# 2) we can import it in setup.py for the same reason
# 3) we can import it into your module
__version__ = '0.2.4'
+164
View File
@@ -0,0 +1,164 @@
# -*- coding: utf-8 -*-
# Copyright (c) 2015, by the Pockets team, see AUTHORS.
# Licensed under the BSD License, see LICENSE for details.
"""A pocket full of useful collection functions!"""
from __future__ import absolute_import
from collections import Sized, Iterable, Mapping
from inspect import isclass
import six
__all__ = ["is_listy", "listify", "mappify"]
def is_listy(x):
"""Return True if `x` is "listy", i.e. a list-like object.
"Listy" is defined as a sized iterable which is neither a map nor a string:
>>> is_listy(["a", "b"])
True
>>> is_listy(set())
True
>>> is_listy(iter(["a", "b"]))
False
>>> is_listy({"a": "b"})
False
>>> is_listy("a regular string")
False
Note:
Iterables and generators fail the "listy" test because they
are not sized.
Args:
x (any value): The object to test.
Returns:
bool: True if `x` is "listy", False otherwise.
"""
return (isinstance(x, Sized) and
isinstance(x, Iterable) and
not isinstance(x, Mapping) and
not isinstance(x, six.string_types))
def listify(x, minlen=0, default=None, cls=None):
"""Return a listified version of `x`.
If `x` is a non-string iterable, it is wrapped in a list; otherwise
a list is returned with `x` as its only element.
>>> listify("a regular string")
['a regular string']
>>> listify(tuple(["a", "b", "c"]))
['a', 'b', 'c']
>>> listify({'a': 'A'})
[{'a': 'A'}]
Note:
Not guaranteed to return a copy of `x`. If `x` is already a list and
`cls` is not specified, then `x` itself is returned.
Args:
x (any value): Value to listify.
minlen (int): Minimum length of the returned list. If the returned
list would be shorter than `minlen` it is padded with values from
`default`. Defaults to 0.
>>> listify([], minlen=0)
[]
>>> listify([], minlen=1)
[None]
>>> listify("item", minlen=3)
['item', None, None]
default (any value): Value that should be used to pad the list if it
would be shorter than `minlen`:
>>> listify([], minlen=1, default="PADDING")
['PADDING']
>>> listify("item", minlen=3, default="PADDING")
['item', 'PADDING', 'PADDING']
cls (class or callable): Instead of wrapping `x` in a list, wrap it
in an instance of `cls`. `cls` should accept an iterable object
as its single parameter when called:
>>> from collections import deque
>>> listify(["a", "b", "c"], cls=deque)
deque(['a', 'b', 'c'])
Returns:
list or `cls`: A listified version of `x`.
"""
if x is None:
x = []
elif not isinstance(x, list):
x = list(x) if is_listy(x) else [x]
if minlen and len(x) < minlen:
x.extend([default for i in range(minlen - len(x))])
if cls and not (isclass(cls) and issubclass(type(x), cls)):
x = cls(x)
return x
def mappify(x, default=True, cls=None):
"""Return a mappified version of `x`.
If `x` is a string, it becomes the only key of the returned dict. If `x`
is a non-string iterable, the elements of `x` become keys in the returned
dict. The values of the returned dict are set to `default`.
If `x` is a map, it is returned directly.
>>> mappify("a regular string")
{'a regular string': True}
>>> mappify(["a"])
{'a': True}
>>> mappify({'a': "A"})
{'a': 'A'}
Note:
Not guaranteed to return a copy of `x`. If `x` is already a map and
`cls` is not specified, then `x` itself is returned.
Args:
x (str, map, or iterable): Value to mappify.
default (any value): Value used to fill out missing values of the
returned dict.
cls (class or callable): Instead of wrapping `x` in a dict, wrap it
in an instance of `cls`. `cls` should accept a map object as
its single parameter when called:
>>> from collections import defaultdict
>>> mappify("a", cls=lambda x: defaultdict(None, x))
defaultdict(None, {'a': True})
Returns:
dict or `cls`: A mappified version of `x`.
Raises:
TypeError: If `x` is not a map, iterable, or string.
"""
if not isinstance(x, Mapping):
if isinstance(x, six.string_types):
x = {x: default}
elif isinstance(x, Iterable):
x = dict([(v, default) for v in x])
else:
raise TypeError("Unable to mappify {0}".format(type(x)), x)
if cls and not (isclass(cls) and issubclass(type(x), cls)):
x = cls(x)
return x
+105
View File
@@ -0,0 +1,105 @@
# -*- coding: utf-8 -*-
# Copyright (c) 2015, by the Pockets team, see AUTHORS.
# Licensed under the BSD License, see LICENSE for details.
"""A pocket full of useful reflection functions!"""
from __future__ import absolute_import
import inspect
import functools
from pockets.collections import listify
from six import string_types
__all__ = ["resolve"]
def resolve(name, modules=None):
"""Resolve a dotted name to an object (usually class, module, or function).
If `name` is a string, attempt to resolve it according to Python
dot notation, e.g. "path.to.MyClass". If `name` is anything other than a
string, return it immediately:
>>> resolve("calendar.TextCalendar")
<class 'calendar.TextCalendar'>
>>> resolve(object()) #doctest: +ELLIPSIS
<object object at 0x...>
If `modules` is specified, then resolution of `name` is restricted
to the given modules. Leading dots are allowed in `name`, but they are
ignored. Resolution **will not** traverse up the module path if `modules`
is specified.
If `modules` is not specified and `name` has leading dots, then resolution
is first attempted relative to the calling function's module, and then
absolutely. Resolution **will** traverse up the module path. If `name` has
no leading dots, resolution is first attempted absolutely and then
relative to the calling module.
Warning:
Do not resolve strings supplied by an end user without specifying
`modules`. Instantiating an arbitrary object specified by an end user
can introduce a potential security risk.
To avoid this, restrict the search path by explicitly specifying
`modules`.
Restricting `name` resolution to a set of `modules`:
>>> resolve("pockets.camel") #doctest: +ELLIPSIS
<function camel at 0x...>
>>> resolve("pockets.camel", modules=["re", "six"]) #doctest: +ELLIPSIS
Traceback (most recent call last):
...
ValueError: Unable to resolve 'pockets.camel' in modules: ['re', 'six']
Args:
name (str or object): A dotted name.
modules (str or list, optional): A module or list of modules, under
which to search for `name`.
Returns:
object: The object specified by `name`.
Raises:
ValueError: If `name` can't be resolved.
"""
if not isinstance(name, string_types):
return name
obj_path = name.split('.')
search_paths = []
if modules:
while not obj_path[0]:
obj_path.pop(0)
for module_path in listify(modules):
search_paths.append(module_path.split('.') + obj_path)
else:
caller = inspect.getouterframes(inspect.currentframe())[1][0].f_globals
module_path = caller['__name__'].split('.')
if not obj_path[0]:
obj_path.pop(0)
while not obj_path[0]:
obj_path.pop(0)
if module_path:
module_path.pop()
search_paths.append(module_path + obj_path)
search_paths.append(obj_path)
else:
search_paths.append(obj_path)
search_paths.append(module_path + obj_path)
for path in search_paths:
try:
obj = functools.reduce(getattr, path[1:], __import__(path[0]))
except (AttributeError, ImportError):
pass
else:
return obj
raise ValueError("Unable to resolve '{0}' "
"in modules: {1}".format(name, modules))
+253
View File
@@ -0,0 +1,253 @@
# -*- coding: utf-8 -*-
# Copyright (c) 2015, by the Pockets team, see AUTHORS.
# Licensed under the BSD License, see LICENSE for details.
"""A pocket full of useful iterators!"""
from __future__ import absolute_import
import collections
import six
__all__ = ["peek_iter", "modify_iter"]
class peek_iter(object):
"""An iterator object that supports peeking ahead.
>>> p = peek_iter(["a", "b", "c", "d", "e"])
>>> p.peek()
'a'
>>> p.next()
'a'
>>> p.peek(3)
['b', 'c', 'd']
Args:
o (iterable or callable): `o` is interpreted very differently
depending on the presence of `sentinel`.
If `sentinel` is not given, then `o` must be a collection object
which supports either the iteration protocol or the sequence
protocol.
If `sentinel` is given, then `o` must be a callable object.
sentinel (any value, optional): If given, the iterator will call `o`
with no arguments for each call to its `next` method; if the value
returned is equal to `sentinel`, :exc:`StopIteration` will be
raised, otherwise the value will be returned.
See Also:
`peek_iter` can operate as a drop in replacement for the built-in
`iter <http://docs.python.org/2/library/functions.html#iter>`_
function.
Attributes:
sentinel (any value): The value used to indicate the iterator is
exhausted. If `sentinel` was not given when the `peek_iter` was
instantiated, then it will be set to a new object
instance: ``object()``.
"""
def __init__(self, *args):
"""__init__(o, sentinel=None)"""
self._iterable = iter(*args)
self._cache = collections.deque()
self.sentinel = args[1] if len(args) > 1 else object()
def __iter__(self):
return self
def __next__(self, n=None):
# NOTE: Prevent 2to3 from transforming self.next() in next(self),
# which causes an infinite loop!
return getattr(self, 'next')(n)
def _fillcache(self, n):
"""Cache `n` items. If `n` is 0 or None, then 1 item is cached."""
if not n:
n = 1
try:
while len(self._cache) < n:
self._cache.append(next(self._iterable))
except StopIteration:
while len(self._cache) < n:
self._cache.append(self.sentinel)
def has_next(self):
"""Determine if iterator is exhausted.
Returns:
bool: True if iterator has more items, False otherwise.
Note:
Will never raise :exc:`StopIteration`.
"""
return self.peek() != self.sentinel
def next(self, n=None):
"""Get the next item or `n` items of the iterator.
Args:
n (int, optional): The number of items to retrieve. Defaults to
None.
Returns:
item or list of items: The next item or `n` items of the iterator.
If `n` is None, the item itself is returned. If `n` is an int,
the items will be returned in a list. If `n` is 0, an empty
list is returned:
>>> p = peek_iter(["a", "b", "c", "d", "e"])
>>> p.next()
'a'
>>> p.next(0)
[]
>>> p.next(1)
['b']
>>> p.next(2)
['c', 'd']
Raises:
StopIteration: Raised if the iterator is exhausted, even if
`n` is 0.
"""
self._fillcache(n)
if not n:
if self._cache[0] == self.sentinel:
raise StopIteration
if n is None:
result = self._cache.popleft()
else:
result = []
else:
if self._cache[n - 1] == self.sentinel:
raise StopIteration
result = [self._cache.popleft() for i in range(n)]
return result
def peek(self, n=None):
"""Preview the next item or `n` items of the iterator.
The iterator is not advanced when peek is called.
Args:
n (int, optional): The number of items to retrieve. Defaults to
None.
Returns:
item or list of items: The next item or `n` items of the iterator.
If `n` is None, the item itself is returned. If `n` is an int,
the items will be returned in a list. If `n` is 0, an empty
list is returned.
If the iterator is exhausted, `peek_iter.sentinel` is returned,
or placed as the last item in the returned list:
>>> p = peek_iter(["a", "b", "c"])
>>> p.sentinel = "END"
>>> p.peek()
'a'
>>> p.peek(0)
[]
>>> p.peek(1)
['a']
>>> p.peek(2)
['a', 'b']
>>> p.peek(4)
['a', 'b', 'c', 'END']
Note:
Will never raise :exc:`StopIteration`.
"""
self._fillcache(n)
if n is None:
result = self._cache[0]
else:
result = [self._cache[i] for i in range(n)]
return result
class modify_iter(peek_iter):
"""An iterator object that supports modifying items as they are returned.
>>> a = [" A list ",
... " of strings ",
... " with ",
... " extra ",
... " whitespace. "]
>>> modifier = lambda s: s.strip().replace('with', 'without')
>>> for s in modify_iter(a, modifier=modifier):
... print('"%s"' % s)
"A list"
"of strings"
"without"
"extra"
"whitespace."
Args:
o (iterable or callable): `o` is interpreted very differently
depending on the presence of `sentinel`.
If `sentinel` is not given, then `o` must be a collection object
which supports either the iteration protocol or the sequence
protocol.
If `sentinel` is given, then `o` must be a callable object.
sentinel (any value, optional): If given, the iterator will call `o`
with no arguments for each call to its `next` method; if the value
returned is equal to `sentinel`, :exc:`StopIteration` will be
raised, otherwise the value will be returned.
modifier (callable, optional): The function that will be used to
modify each item returned by the iterator. `modifier` should take
a single argument and return a single value. Defaults
to ``lambda x: x``.
If `sentinel` is not given, `modifier` must be passed as a keyword
argument.
Attributes:
modifier (callable): `modifier` is called with each item in `o` as it
is iterated. The return value of `modifier` is returned in lieu of
the item.
Values returned by `peek` as well as `next` are affected by
`modifier`. However, `modify_iter.sentinel` is never passed through
`modifier`; it will always be returned from `peek` unmodified.
"""
def __init__(self, *args, **kwargs):
"""__init__(o, sentinel=None, modifier=lambda x: x)"""
if 'modifier' in kwargs:
self.modifier = kwargs['modifier']
elif len(args) > 2:
self.modifier = args[2]
args = args[:2]
else:
self.modifier = lambda x: x
if not six.callable(self.modifier):
raise TypeError('modify_iter(o, modifier): '
'modifier must be callable')
super(modify_iter, self).__init__(*args)
def _fillcache(self, n):
"""Cache `n` modified items. If `n` is 0 or None, 1 item is cached.
Each item returned by the iterator is passed through the
`modify_iter.modified` function before being cached.
"""
if not n:
n = 1
try:
while len(self._cache) < n:
self._cache.append(self.modifier(next(self._iterable)))
except StopIteration:
while len(self._cache) < n:
self._cache.append(self.sentinel)
+277
View File
@@ -0,0 +1,277 @@
# -*- coding: utf-8 -*-
# Copyright (c) 2015, by the Pockets team, see AUTHORS.
# Licensed under the BSD License, see LICENSE for details.
"""A pocket full of useful string manipulation functions!"""
from __future__ import absolute_import
import re
from pockets.collections import listify
__all__ = ["camel", "uncamel", "splitcaps"]
# Default regular expression flags
_re_flags = re.L | re.M | re.U
_whitespace_group_re = re.compile("(\s+)", _re_flags)
_uncamel_re = re.compile(
"(" # The whole expression is in a single group
# Clause 1
"(?<=[^\sA-Z])" # Preceded by neither a space nor a capital letter
"[A-Z]+[^a-z\s]*" # All non-lowercase beginning with a capital letter
"(?=[A-Z][^A-Z\s]*?[a-z]|\s|$)" # Followed by a capitalized word
"|"
# Clause 2
"(?<=[^\s])" # Preceded by a character that is not a space
"[A-Z][^A-Z\s]*?[a-z]+[^A-Z\s]*" # Capitalized word
")", _re_flags)
_splitcaps_re = re.compile(
# Clause 1
"[A-Z]+[^a-z]*" # All non-lowercase beginning with a capital letter
"(?=[A-Z][^A-Z]*?[a-z]|$)" # Followed by a capitalized word
"|"
# Clause 2
"[A-Z][^A-Z]*?[a-z]+[^A-Z]*" # Capitalized word
"|"
# Clause 3
"[^A-Z]+", # All non-uppercase
_re_flags)
def camel(s, sep="_", lower_initial=False, upper_segments=None,
preserve_upper=False):
"""Convert underscore_separated string (aka snake_case) to CamelCase.
Works on full sentences as well as individual words:
>>> camel("hello_world!")
'HelloWorld!'
>>> camel("Totally works as_expected, even_with_whitespace!")
'Totally Works AsExpected, EvenWithWhitespace!'
Args:
sep (string, optional): Delineates segments of `s` that will be
CamelCased. Defaults to an underscore "_".
For example, if you want to CamelCase a dash separated word:
>>> camel("xml-http-request", sep="-")
'XmlHttpRequest'
lower_initial (bool, int, or list, optional): If True, the initial
character of each camelCased word will be lowercase. If False, the
initial character of each CamelCased word will be uppercase.
Defaults to False:
>>> camel("http_request http_response")
'HttpRequest HttpResponse'
>>> camel("http_request http_response", lower_initial=True)
'httpRequest httpResponse'
Optionally, `lower_initial` can be an int or a list of ints,
indicating which individual segments of each CamelCased word
should start with a lowercase. Supports negative numbers to index
segments from the right:
>>> camel("xml_http_request", lower_initial=0)
'xmlHttpRequest'
>>> camel("xml_http_request", lower_initial=-1)
'XmlHttprequest'
>>> camel("xml_http_request", lower_initial=[0, 1])
'xmlhttpRequest'
upper_segments (int or list, optional): Indicates which segments of
CamelCased words should be fully uppercased, instead of just
capitalizing the first letter.
Can be an int, indicating a single segment, or a list of ints,
indicating multiple segments. Supports negative numbers to index
segments from the right.
`upper_segments` is helpful when dealing with acronyms:
>>> camel("tcp_socket_id", upper_segments=0)
'TCPSocketId'
>>> camel("tcp_socket_id", upper_segments=[0, -1])
'TCPSocketID'
>>> camel("tcp_socket_id", upper_segments=[0, -1], lower_initial=1)
'TCPsocketID'
preserve_upper (bool): If True, existing uppercase characters will
not be automatically lowercased. Defaults to False.
>>> camel("xml_HTTP_reQuest")
'XmlHttpRequest'
>>> camel("xml_HTTP_reQuest", preserve_upper=True)
'XmlHTTPReQuest'
Returns:
str: CamelCased version of `s`.
"""
if isinstance(lower_initial, bool):
lower_initial = [0] if lower_initial else []
else:
lower_initial = listify(lower_initial)
upper_segments = listify(upper_segments)
result = []
for word in _whitespace_group_re.split(s):
segments = [segment for segment in word.split(sep) if segment]
count = len(segments)
for i, segment in enumerate(segments):
upper = i in upper_segments or (i - count) in upper_segments
lower = i in lower_initial or (i - count) in lower_initial
if upper and lower:
if preserve_upper:
segment = segment[0] + segment[1:].upper()
else:
segment = segment[0].lower() + segment[1:].upper()
elif upper:
segment = segment.upper()
elif lower:
if not preserve_upper:
segment = segment.lower()
elif preserve_upper:
segment = segment[0].upper() + segment[1:]
else:
segment = segment[0].upper() + segment[1:].lower()
result.append(segment)
return "".join(result)
def uncamel(s, sep="_"):
"""Convert CamelCase string to underscore_separated (aka snake_case).
A CamelCase word is considered to be any uppercase letter followed by zero
or more lowercase letters. Contiguous groups of uppercase letters – like
you would find in an acronym – are also considered part of a single word:
>>> uncamel("Request")
'request'
>>> uncamel("HTTP")
'http'
>>> uncamel("HTTPRequest")
'http_request'
>>> uncamel("xmlHTTPRequest")
'xml_http_request'
Works on full sentences as well as individual words:
>>> uncamel("HelloWorld!")
'hello_world!'
>>> uncamel("Totally works AsExpected, EvenWithWhitespace!")
'totally works as_expected, even_with_whitespace!'
Args:
sep (str, optional): String used to separate CamelCase words. Defaults
to an underscore "_".
For example, if you want dash separated words:
>>> uncamel("XmlHttpRequest", sep="-")
'xml-http-request'
Returns:
str: uncamel_cased version of `s`.
"""
return _uncamel_re.sub(r'{0}\1'.format(sep), s).lower()
def splitcaps(s, pattern=None, maxsplit=None, flags=0):
"""Intelligently split a string on capitalized words.
A capitalized word is considered to be any uppercase letter followed by
zero or more lowercase letters. Contiguous groups of uppercase letters –
like you would find in an acronym – are also considered part of a single
word:
>>> splitcaps("Request")
['Request']
>>> splitcaps("HTTP")
['HTTP']
>>> splitcaps("HTTPRequest")
['HTTP', 'Request']
>>> splitcaps("HTTP/1.1Request")
['HTTP/1.1', 'Request']
>>> splitcaps("xmlHTTPRequest")
['xml', 'HTTP', 'Request']
If no capitalized words are found in `s`, the whole string is
returned in a single element list:
>>> splitcaps("")
['']
>>> splitcaps("lower case words")
['lower case words']
Does not split on whitespace by default. To also split
on whitespace, pass "\\\s+" for `pattern`:
>>> splitcaps("Without whiteSpace pattern")
['Without white', 'Space pattern']
>>> splitcaps("With whiteSpace pattern", pattern="\s+")
['With', 'white', 'Space', 'pattern']
>>> splitcaps("With whiteSpace group", pattern="(\s+)")
['With', ' ', 'white', 'Space', ' ', 'group']
Args:
s (str): The string to split.
pattern (str, optional): In addition to splitting on capital letters,
also split by the occurrences of `pattern`. If capturing
parentheses are used in `pattern`, then the text of all groups in
`pattern` are also returned as part of the resulting list.
Defaults to None.
maxsplit (int, optional): If maxsplit is not specified or -1, then
there is no limit on the number of splits (all possible splits are
made). If maxsplit is >= 0, at most maxsplit splits occur, and the
remainder of the string is returned as the final element of the
list.
flags (int, optional): Flags to pass to the regular expression created
using `pattern`. Ignored if `pattern` is not specified. Defaults
to (re.LOCALE | re.MULTILINE | re.UNICODE).
Returns:
list: List of capitalized substrings in `s`.
"""
if not maxsplit:
if maxsplit == 0:
return [s]
else:
maxsplit = -1
if pattern:
pattern_re = re.compile(pattern, flags or _re_flags)
else:
pattern_re = None
result = []
post_maxsplit = []
for m in _splitcaps_re.finditer(s):
if pattern_re:
for segment in pattern_re.split(m.group()):
if segment:
if maxsplit > 0 and len(result) >= maxsplit:
post_maxsplit.append(segment)
else:
result.append(segment)
else:
result.append(m.group())
if maxsplit > 0 and len(result) >= maxsplit:
if m.end() < len(s):
post_maxsplit.append(s[m.end():])
post_maxsplit = ''.join(post_maxsplit)
if post_maxsplit:
result.append(post_maxsplit)
break
return result if len(result) > 0 else [s]