diff --git a/.gitignore b/.gitignore index c61a4a98..c6117a14 100644 --- a/.gitignore +++ b/.gitignore @@ -124,3 +124,4 @@ misc/data/ # Claude Code local working files. Handoffs summarize in-flight work and may # name local paths / private packages; they are deliberately never committed. .claude/handoffs/ +docsrc/ diff --git a/README.md b/README.md index 538692fe..d8145cca 100644 --- a/README.md +++ b/README.md @@ -15,6 +15,26 @@ To install: ```pip install dol``` [Documentation here](https://i2mint.github.io/dol/) + +## For AI agents + +`dol` ships tooling for coding agents. If you are one, start here. + +**Skills** ([Agent Skills](https://agentskills.io) format), for any agent host. + +| Skill | Use it to | +|---|---| +| `dol-dev-portability` | keep dol working on Windows as well as Linux/macOS | +| `dol-dev-wrap-kvs` | understand and safely modify dol's core wrapping machinery — wrap_kvs, store_decorator, Store.wrap, and how transforms are applied | +| `dol-store-building` | build a dol store: wrap any storage backend | + +**Instruction files**: `CLAUDE.md` (Claude Code). + +**The documentation, machine-readable**: [`llms.txt`](https://i2mint.github.io/dol/llms.txt) indexes every page; [`dol.md`](https://i2mint.github.io/dol/dol.md) is the whole documentation in one file; every page has a `.md` twin; [`objects.inv`](https://i2mint.github.io/dol/objects.inv) maps symbols to URLs. The full list, with install lines, is on the site's [For AI agents](https://i2mint.github.io/dol/ai-agents.html) page. + +If you like writing your own code, the rest of this README is written for you, starting at [Example use](#example-use). + + ## Example use Say you have a source backend that has pickles of some lists-of-lists-of-strings, @@ -217,13 +237,13 @@ how the content is stored should be specified, but StoreInterface offers a dict- __delitem__ calls: _id_of_key __iter__ calls: _key_of_id -```pydocstring +```python >>> from dol import Store ``` A Store can be instantiated with no arguments. By default it will make a dict and wrap that. -```pydocstring +```python >>> # Default store: no key or value conversion ################################################ >>> s = Store() >>> s['foo'] = 33 @@ -236,7 +256,7 @@ Now let's make stores that have a key and value conversion layer input keys will be upper cased, and output keys lower cased input values (assumed int) will be converted to ascii string, and visa versa -```pydocstring +```python >>> >>> def test_store(s): ... s['foo'] = 33 # write 33 to 'foo' @@ -262,7 +282,7 @@ We can introduce this conversion layer in several ways. Here are few... ## by subclassing -```pydocstring +```python >>> # by subclassing ############################################################################### >>> class MyStore(Store): ... def _id_of_key(self, k): @@ -280,7 +300,7 @@ Here are few... ## by assigning functions to converters -```pydocstring +```python >>> # by assigning functions to converters ########################################################## >>> class MyStore(Store): ... def __init__(self, store, _id_of_key, _key_of_id, _data_of_obj, _obj_of_data): @@ -301,7 +321,7 @@ Here are few... ## using a Mixin class -```pydocstring +```python >>> # using a Mixin class ############################################################################# >>> class Mixin: ... def _id_of_key(self, k): @@ -322,7 +342,7 @@ Here are few... ## adding wrapper methods to an already made Store instance -```pydocstring +```python >>> # adding wrapper methods to an already made Store instance ######################################### >>> s = Store(dict()) >>> s._id_of_key=lambda k: k.upper() diff --git a/dol/__init__.py b/dol/__init__.py index 6ea59f74..9e1dce9e 100644 --- a/dol/__init__.py +++ b/dol/__init__.py @@ -1,4 +1,18 @@ -"""Core tools to build simple interfaces to complex data sources and bend the interface to your will (and need)""" +"""Core tools to build simple interfaces to complex data sources and bend the interface to your will (and need). + +``dol`` wraps any storage backend (files, S3, databases, dicts) behind a dict-like +interface, and transforms that interface with composable layers. Start with +``wrap_kvs`` (key/value transforms), the file stores (``Files``, ``TextFiles``, +``JsonFiles``, ``PickleFiles``), the ready-made codecs (``ValueCodecs``, ``KeyCodecs``), +``filt_iter`` (key filtering) and ``cache_this`` (caching). + + >>> from dol import wrap_kvs + >>> import json + >>> s = wrap_kvs({}, obj_of_data=json.loads, data_of_obj=json.dumps) + >>> s['a'] = {'x': 1} + >>> s['a'], s.store + ({'x': 1}, {'a': '{"x": 1}'}) +""" import os diff --git a/dol/_interface_wrap.py b/dol/_interface_wrap.py index 9a5bce70..96a0ebbd 100644 --- a/dol/_interface_wrap.py +++ b/dol/_interface_wrap.py @@ -490,8 +490,8 @@ def _validate_stack_seams(stack): def _compile_method_plan(name, sites, leaf_method, sig, encoders, decoders): """Compile one boundary method: encode role args, call leaf, decode result. - ``sites``: {param_name_or_'return': ((role, path, ann), ...)}. - Returns a callable(*args, **kwargs) with the leaf method baked in. + ``sites``: ``{param_name_or_'return': ((role, path, ann), ...)}``. + Returns a ``callable(*args, **kwargs)`` with the leaf method baked in. """ # Build per-parameter transformers (outer -> inner). Integer site keys # mean positional index (dict-form specs), resolved against the outer diff --git a/dol/appendable.py b/dol/appendable.py index 57b130f6..cf896c13 100644 --- a/dol/appendable.py +++ b/dol/appendable.py @@ -1,10 +1,17 @@ -""" -Tools to add append-functionality to key-val stores. The main function is - `appendable_store_cls = add_append_functionality_to_store_cls(store_cls, item2kv, ...)` -You give it the `store_cls` you want to sub class, and a item -> (key, val) function, and you get a store (subclass) that -has a `store.append(item)` method. Also includes an extend method (that just called appends in a loop. - -See add_append_functionality_to_store_cls docs for examples. +"""Tools to add append-functionality to key-val stores. + +The main function is ``appendable(store_cls, item2kv=...)``: you give it the store +class you want to subclass and an item -> (key, val) function, and you get a store +(subclass) that has a ``store.append(item)`` method (and an ``extend``, which appends in +a loop). ``mk_item2kv_for`` holds ready-made item2kv factories (timestamps, uuids, +fields of the item, ...). + + >>> from dol.appendable import appendable + >>> S = appendable(dict, item2kv=lambda item: (item['id'], item)) + >>> s = S() + >>> s.append({'id': 1}) + >>> s + {1: {'id': 1}} """ import time @@ -24,7 +31,8 @@ def define_extend_as_seq_of_appends(obj): Args: obj: Class (type) or instance of an object that has an "append" method. - Returns: The obj, but with that extend method. + Returns: + The obj, but with that extend method. >>> class A: ... def __init__(self): @@ -48,7 +56,6 @@ def define_extend_as_seq_of_appends(obj): >>> a.extend([10, 20]) >>> a.t [1, 2, 3, 10, 20] - """ assert hasattr(obj, "append"), ( f"Your object needs to have an append method! Object was: {obj}" @@ -113,7 +120,9 @@ def attr(attr_name): Args: attr_name: The attribute name to use as the key - Returns: an item -> (key, val) function + + Returns: + an item -> (key, val) function >>> ref_getter =mk_item2kv_for.attr("ref") >>> from collections import namedtuple @@ -121,7 +130,6 @@ def attr(attr_name): >>> a = A(ref='some_ref') >>> ref_getter(a) ('some_ref', A(ref='some_ref')) - """ def item2kv(item): @@ -147,7 +155,8 @@ def item_to_key(item2key): Args: item2key: an item -> key function - Returns: an item -> (key, val) function + Returns: + an item -> (key, val) function >>> item2key = lambda item: item['G'] # use value of 'L' as the key >>> item2key({'L': 'let', 'I': 'it', 'G': 'go'}) @@ -166,8 +175,9 @@ def item2kv(item): def field(field, keep_field_in_value=True, dflt_if_missing=NotSpecified): """item2kv that uses a specific key of a (mapping) item as the key - Note: If keep_field_in_value=False, the field will be popped OUT of the item. - If that's not the desired effect, one should feed copies of the items (e.g. map(dict.copy, items)) + Note: + If keep_field_in_value=False, the field will be popped OUT of the item. + If that's not the desired effect, one should feed copies of the items (e.g. map(dict.copy, items)) :param field: The field (value) to use as the returned key :param keep_field_in_value: Whether to leave the field in the item. If False, will pop it out @@ -183,7 +193,6 @@ def field(field, keep_field_in_value=True, dflt_if_missing=NotSpecified): >>> item2kv = mk_item2kv_for.field('G', dflt_if_missing=None) >>> item2kv({'L': 'let', 'I': 'it', 'DIE': 'go'}) (None, {'L': 'let', 'I': 'it', 'DIE': 'go'}) - """ if dflt_if_missing is NotSpecified: if keep_field_in_value: @@ -216,9 +225,11 @@ def utc_key(offset_s=0, factor=1, *, time_postproc: Callable | None = None): or to get a more accurate timestamp of an event. Use case for offset_s: + * Align to another system's clock * Get more accurate timestamping of an event. For example, in situations where the item is a chunk of live - streaming data and we want the key (timestamp) to represent the timestamp of the beginning of the chunk. + streaming data and we want the key (timestamp) to represent the timestamp of the beginning of the chunk. + Without an offset_s, the timestamp would be the timestamp after the last byte of the chunk was produced, plus the time it took to reach the present function. If we know the data production rate (e.g. sample rate) and the average lag to get to the present function, we can get a more accurate timestamp for the beginning @@ -227,14 +238,14 @@ def utc_key(offset_s=0, factor=1, *, time_postproc: Callable | None = None): Args: offset_s: An offset (in seconds, possibly negative) to add to the current time. - Returns: an item -> (current_utc_s, item) function + Returns: + an item -> (current_utc_s, item) function >>> import time >>> item2key = mk_item2kv_for.utc_key() >>> k, v = item2key('some data') >>> assert abs(time.time() - k) < 0.01 # which asserts that k is indeed a (current) utc timestamp >>> assert v == 'some data' # just the item itself - """ if time_postproc is None: @@ -260,7 +271,8 @@ def uuid_key(hex=True): One advantage though, is that the uuid is time-based, so it can be used to sort the keys in the order they were IDed. - Returns: an item -> (uuid, item) function + Returns: + an item -> (uuid, item) function >>> import uuid >>> item2key = mk_item2kv_for.uuid_key() @@ -275,7 +287,6 @@ def uuid_key(hex=True): >>> k, v = item2key('some data') >>> isinstance(k, uuid.UUID) True - """ import uuid @@ -298,12 +309,11 @@ def item_to_key_params_and_val(item_to_key_params_and_val, key_str_format): Args: item_to_key_params_and_val: an item -> (key_params, val) function - key_str_format: A string format such that - key_str_format.format(*key_params) or - key_str_format.format(**key_params) - will produce the desired key string + key_str_format: A string format such that ``key_str_format.format(*key_params)`` + or ``key_str_format.format(**key_params)`` will produce the desired key string - Returns: an item -> (key, val) function + Returns: + an item -> (key, val) function >>> # Using tuple key params with unnamed string format fields >>> item_to_kv = mk_item2kv_for.item_to_key_params_and_val(lambda x: ((x['L'], x['I']), x['G']), '{}/{}') @@ -330,14 +340,16 @@ def item2kv(item): def fields(fields, keep_field_in_value=False, key_as_tuple=False): """Make item2kv from specific fields of a Mapping (i.e. dict-like object) item. - Note: item2kv will not mutate item (even if keep_field_in_value=False). + Note: + item2kv will not mutate item (even if keep_field_in_value=False). Args: fields: The sequence (list, tuple, etc.) of item fields that should be used to create the key. keep_field_in_value: Set to True to return the item as is, as the value key_as_tuple: Set to True if you want keys to be tuples (note that the fields order is important here!) - Returns: an item -> (item[fields], item[not in fields]) function + Returns: + an item -> (item[fields], item[not in fields]) function >>> item_to_kv = mk_item2kv_for.fields('L') >>> item_to_kv({'L': 'let', 'I': 'it', 'G': 'go'}) @@ -351,7 +363,6 @@ def fields(fields, keep_field_in_value=False, key_as_tuple=False): >>> item_to_kv = mk_item2kv_for.fields(('G', 'L'), key_as_tuple=True) # but ('G', 'L') order is respected here >>> item_to_kv({'L': 'let', 'I': 'it', 'G': 'go'}) (('go', 'let'), {'I': 'it'}) - """ if isinstance(fields, str): fields_set = {fields} @@ -392,7 +403,8 @@ def appendable(store_cls=None, *, item2kv, return_keys=False): item2kv: The function that produces a (key, val) pair from an item new_store_name: The name to give the new class (default will be 'Appendable' + store_cls.__name__) - Returns: A subclass of store_cls with two additional methods: append, and extend. + Returns: + A subclass of store_cls with two additional methods: append, and extend. >>> item_to_kv = lambda item: (item['L'], item) # use value of 'L' as the key, and value is the item itself @@ -516,7 +528,6 @@ class Extender: >>> b_extender += ' split' >>> store {'a': 'pplesauce', 'b': 'anana split'} - """ def __init__( diff --git a/dol/base.py b/dol/base.py index b6b030d3..4683c348 100644 --- a/dol/base.py +++ b/dol/base.py @@ -22,6 +22,19 @@ These key converters object serialization methods default to the identity (i.e. they return the input as is). This means that you don't have to implement these as all, and can choose to implement these concerns within the storage methods themselves. + +Main entry points: + +- ``KvReader``: base class for read-only stores (a ``Mapping`` with a ``head``) +- ``KvPersister``: base class for read-write stores (a ``MutableMapping``, ``clear`` disabled) +- ``Store``: a persister with the key/value transform hooks, wrapping a backend +- ``kv_walk``: walk a nested mapping, yielding (path, key, value) triples by default + + >>> from dol.base import Store + >>> s = Store({}) + >>> s['a'] = 1 + >>> s['a'], list(s) + (1, ['a']) """ from functools import partial, update_wrapper @@ -58,6 +71,7 @@ class AttrNames: + """Name sets of the methods that make up each mapping interface (``Collection``, ``Mapping``, ``KvReader``, ``KvPersister``, ...).""" CollectionABC = {"__len__", "__iter__", "__contains__"} Mapping = CollectionABC | { "keys", @@ -86,14 +100,18 @@ class AttrNames: # point to. class Collection(CollectionABC): """The same as collections.abc.Collection, with some modifications: + - Addition of a ``head`` """ def __contains__(self, x) -> bool: """ Check if collection of keys contains k. - Note: This method loops through all contents of collection to see if query element exists. - Therefore it may not be efficient, and in most cases, a method specific to the case should be used. + + Note: + This method loops through all contents of collection to see if query element exists. + Therefore it may not be efficient, and in most cases, a method specific to the case should be used. + :return: True if k is in the collection, and False if not """ for existing_x in iter(self): @@ -104,8 +122,11 @@ def __contains__(self, x) -> bool: def __len__(self) -> int: """ Number of elements in collection of keys. - Note: This method iterates over all elements of the collection and counts them. - Therefore it is not efficient, and in most cases should be overridden with a more efficient version. + + Note: + This method iterates over all elements of the collection and counts them. + Therefore it is not efficient, and in most cases should be overridden with a more efficient version. + :return: The number (int) of elements in the collection of keys. """ # Note: Found that sum(1 for _ in self.__iter__()) was slower for small, slightly faster for big inputs. @@ -140,6 +161,10 @@ def head(self): class MappingViewMixin: + """Make ``keys()``, ``values()`` and ``items()`` build their views from the + ``KeysView``, ``ValuesView`` and ``ItemsView`` class attributes, so a subclass can + swap in its own view classes.""" + KeysView: type = BaseKeysView ValuesView: type = BaseValuesView ItemsView: type = BaseItemsView @@ -184,7 +209,6 @@ def __reversed__(self): .. code-block:: python reversed = sorted(self)[::-1] - """ raise NotImplementedError(__doc__) @@ -210,16 +234,16 @@ class KvPersister(KvReader, MutableMapping): If `s` is a dict, this would have the effect of adding a ('b', 3) item under 'a'. But in the general case, this might + - fail, because the `s['a']` doesn't support sub-scripting (doesn't have a `__getitem__`) - or, worse, will pass silently but not actually persist the write as expected (e.g. LocalFileStore) Another example: `s.popitem()` will pop a `(k, v)` pair off of the `s` store. That is, retrieve the `v` for `k`, delete the entry for `k`, and return a `(k, v)`. Note that unlike modern dicts which will return the last item that was stored - -- that is, LIFO (last-in, first-out) order -- for KvPersisters, - there's no assurance as to what item will be, since it will depend on the backend storage system - and/or how the persister was implemented. - + (that is, LIFO (last-in, first-out) order), for KvPersisters + there's no assurance as to what item will be, since it will depend on the backend storage system + and/or how the persister was implemented. """ clear = _disabled_clear_method @@ -241,6 +265,7 @@ class KvPersister(KvReader, MutableMapping): class NoSuchItem: + """Sentinel type; ``no_such_item`` is its instance.""" pass @@ -250,6 +275,7 @@ class NoSuchItem: class DelegatedAttribute: + """Descriptor forwarding ``attr_name`` lookups to the object held in the instance's ``delegate_name`` attribute.""" def __init__(self, delegate_name, attr_name): self.attr_name = attr_name self.delegate_name = delegate_name @@ -421,6 +447,7 @@ def delegate_to( ignore=frozenset(), ) -> Decorator: # turn include and ignore into sets, if they aren't already + """Class decorator factory: the decorated wrapper class constructs a ``wrapped`` instance and delegates to it, through ``delegation_attr``, the attributes of ``wrapped`` (``dir(wrapped)`` minus ``ignore``, plus ``include``) not already defined on the wrapper.""" if not isinstance(include, Set): include = set(include) if not isinstance(ignore, Set): @@ -564,7 +591,6 @@ def delegator_wrap( >>> WrappedA = Delegator.wrap(A) >>> hasattr(WrappedA, 'foo') True - """ if isinstance(obj, type): if isinstance(delegator, type): @@ -595,7 +621,9 @@ class Store(KvPersister): """ By store we mean key-value store. This could be files in a filesystem, objects in s3, or a database. Where and how the content is stored should be specified, but StoreInterface offers a dict-like interface to this. + :: + __getitem__ calls: _id_of_key _obj_of_data __setitem__ calls: _id_of_key _data_of_obj __delitem__ calls: _id_of_key @@ -699,7 +727,6 @@ class Store(KvPersister): override the `KeysView`, `ValuesView` or `ItemsView` classes that they use. For more, see: https://github.com/i2mint/dol/wiki/Mapping-Views - """ _state_attrs = ["store", "_class_wrapper"] @@ -883,14 +910,17 @@ def __setstate__(self, state: dict): def val_is_mapping(p: PT, k: KT, v: VT) -> bool: + """Whether the walked value ``v`` is a ``Mapping`` (a ``kv_walk`` ``walk_filt``).""" return isinstance(v, Mapping) def asis(p: PT, k: KT, v: VT) -> Any: + """Return ``(p, k, v)`` as is (the default ``kv_walk`` ``leaf_yield``).""" return p, k, v def tuple_keypath_and_val(p: PT, k: KT, v: VT) -> tuple[PT, VT]: + """Extend the path ``p`` with the key ``k`` and return ``(new_path, v)`` (the default ``kv_walk`` ``pkv_to_pv``).""" if p == (): # we're just begining (the root), p = (k,) # so begin the path with the first key. else: @@ -978,10 +1008,11 @@ def kv_walk( ... ] ... ) - Tip: If you want to use ``kv_filt`` to search and extract stuff from a nested - mapping, you can have your ``leaf_yield`` return a sentinel (say, ``None``) to - indicate that the value should be skipped, and then filter out the ``None``s from - your results. + Tip: + If you want to use ``kv_filt`` to search and extract stuff from a nested + mapping, you can have your ``leaf_yield`` return a sentinel (say, ``None``) to + indicate that the value should be skipped, and then filter out the ``None`` values from + your results. >>> mm = { ... 'a': {'b': {'c': 42}}, @@ -1027,9 +1058,9 @@ def kv_walk( [('apple',), ('big', 'apple')] So now, you can get the first apple path by doing: + >>> next(filter(None, walker3(d))) ('apple',) - """ if not breadth_first: # print(f"1: entered with: v={v}, p={p}") @@ -1077,8 +1108,8 @@ def has_kv_store_interface(o): Args: o: object (class or instance) - Returns: True if kv has the four key (in/out) and value (in/out) transformation methods - + Returns: + True if kv has the four key (in/out) and value (in/out) transformation methods """ return ( hasattr(o, "_id_of_key") @@ -1139,6 +1170,7 @@ def __subclasshook__(cls, C): class stream_util: + """Small callbacks for ``Stream``: an always-true filter, a no-op, and rewind (``skip_lines`` currently only rewinds).""" def always_true(*args, **kwargs): return True diff --git a/dol/caching.py b/dol/caching.py index b1847216..2798bb2b 100644 --- a/dol/caching.py +++ b/dol/caching.py @@ -5,6 +5,7 @@ offering flexible and powerful caching solutions for both data stores and method calls. Main Use Cases: + - Property caching: Cache expensive computations that only need to be run once - Method caching: Cache method results based on arguments, with smart key generation - Store caching: Add caching layers to data stores for improved performance @@ -12,28 +13,19 @@ Key Tools: -cache_this: - The main decorator for caching properties and methods. Automatically detects - whether to use property or method caching based on function signature. - Supports custom cache storage, key functions, parameter ignoring, and - serialization hooks. - -CachedProperty: - A descriptor for caching property values with flexible cache storage and - key generation strategies. - -CachedMethod: - A descriptor for caching method results based on arguments, with support - for parameter filtering and custom key functions. - -KeyStrategy Protocol: - Extensible system for defining how cache keys are generated, including - strategies for explicit keys, instance properties, method arguments, and - composite keys. - -Store Decorators: - Tools like cache_vals, mk_sourced_store, and store_cached for adding - caching layers to data stores. +- ``cache_this``: The main decorator for caching properties and methods. + Automatically detects whether to use property or method caching based on + function signature. Supports custom cache storage, key functions, parameter + ignoring, and serialization hooks. +- ``CachedProperty``: A descriptor for caching property values with flexible + cache storage and key generation strategies. +- ``CachedMethod``: A descriptor for caching method results based on arguments, + with support for parameter filtering and custom key functions. +- ``KeyStrategy`` protocol: Extensible system for defining how cache keys are + generated, including strategies for explicit keys, instance properties, + method arguments, and composite keys. +- Store decorators: Tools like ``cache_vals``, ``mk_sourced_store``, and + ``store_cached`` for adding caching layers to data stores. Examples: @@ -59,7 +51,6 @@ ... @cache_this(cache='cache', ignore={'verbose'}) ... def process(self, data, mode='fast', verbose=False): ... return len(data) if mode == 'fast' else sum(data) - """ # ------------------------------------------------------------------------------------- @@ -237,7 +228,7 @@ class FromMethodArgs: """ Apply a function to method arguments to generate the key. - The function receives (self, *args, **kwargs) and should return a cache key. + The function receives ``(self, *args, **kwargs)`` and should return a cache key. """ def __init__(self, func: Callable): @@ -245,8 +236,8 @@ def __init__(self, func: Callable): Initialize with a function to apply to method arguments. Args: - func: A function that takes (self, *args, **kwargs) and returns a key. - Example: lambda self, x, y: f'{x}_{y}' + func: A function that takes ``(self, *args, **kwargs)`` and returns a key, + for example ``lambda self, x, y: f'{x}_{y}'``. """ self.func = func @@ -468,24 +459,26 @@ def __init__( Args: func: The function whose result needs to be cached. cache: The cache storage. Can be: + - A MutableMapping instance (shared across all instances) - A string naming an instance attribute that is a MutableMapping - A callable that takes the instance and returns a MutableMapping - (allows per-instance cache customization) - Example: cache=lambda self: Files(f'/tmp/cache_{self.user_id}/') + (allows per-instance cache customization), + e.g. ``cache=lambda self: Files(f'/tmp/cache_{self.user_id}/')`` + key: The key to store the cache value. Can be: + - A string (treated as an explicit key) - A function (interpreted based on its signature) - A KeyStrategy instance + allow_none_keys: Whether to allow None as a valid key. lock_factory: Factory function to create a lock. pre_cache: If True or a MutableMapping, adds in-memory caching. serialize: Optional function to serialize values before storing in cache. - Signature: (value) -> serialized_value - Example: serialize=pickle.dumps + Signature: ``(value) -> serialized_value``, e.g. ``pickle.dumps``. deserialize: Optional function to deserialize values retrieved from cache. - Signature: (serialized_value) -> value - Example: deserialize=pickle.loads + Signature: ``(serialized_value) -> value``, e.g. ``pickle.loads``. """ self.func = func self.attrname = None @@ -569,6 +562,7 @@ def __get_cache(self, instance): Get the cache for the instance. This method handles the three main cache specification patterns: + 1. Cache factories (functions that create cache instances) 2. Attribute names (strings referring to instance attributes) 3. Direct cache objects (MutableMapping instances) @@ -731,6 +725,7 @@ def _default_method_key(func, self, *args, ignore=None, **kwargs): A string like "x=1;y=2;mode=fast" representing all arguments Examples: + >>> def sample_method(self, x, y, mode='fast'): pass >>> _default_method_key(sample_method, None, 1, 2) 'x=1;y=2;mode=fast' @@ -812,26 +807,26 @@ def __init__( Args: func: The function whose results need to be cached. cache: The cache storage. Can be: + - A MutableMapping instance (shared across instances) - A string naming an instance attribute containing a MutableMapping - A callable taking (instance) and returning a MutableMapping This enables instance-specific caching, e.g.: cache=lambda self: Files(f'/cache/{self.user_id}/') - key: Callable that takes (self, *args, **kwargs) and returns a cache key. + key: Callable that takes ``(self, *args, **kwargs)`` and returns a cache key. Defaults to a function that converts args/kwargs to a string. ignore: Parameter name(s) to exclude from cache key computation. Can be a string (single parameter) or list of strings (multiple parameters). Commonly used to ignore 'self' or parameters like 'verbose' that don't affect the result. + allow_none_keys: Whether to allow None as a valid key. lock_factory: Factory function to create a lock. pre_cache: If True or a MutableMapping, adds in-memory caching. serialize: Optional function to serialize values before storing in cache. - Signature: (value) -> serialized_value - Example: serialize=pickle.dumps + Signature: ``(value) -> serialized_value``, e.g. ``pickle.dumps``. deserialize: Optional function to deserialize values retrieved from cache. - Signature: (serialized_value) -> value - Example: deserialize=pickle.loads + Signature: ``(serialized_value) -> value``, e.g. ``pickle.loads``. """ self.func = func self.attrname = None @@ -1062,14 +1057,16 @@ def cache_this( :param func: The function to be decorated (usually left empty). :param cache: The cache storage. Can be: + - A MutableMapping instance (shared across instances) - A string naming an instance attribute containing a MutableMapping - A callable taking (instance) and returning a MutableMapping This enables instance-specific caching, e.g.: cache=lambda self: Files(f'/cache/{self.user_id}/') + :param key: For properties: the key to store the cache value, can be a callable that will be applied to the method name to make a key, or an explicit string. - For methods: a callable that takes (self, *args, **kwargs) and returns a cache key. + For methods: a callable that takes ``(self, *args, **kwargs)`` and returns a cache key. :param pre_cache: Default is False. If True, adds an in-memory cache to the method to (also) cache the results in memory. If a MutableMapping is given, it will be used as the pre-cache. @@ -1081,13 +1078,13 @@ def cache_this( Can be a string (single parameter) or list of strings (multiple parameters). Commonly used to ignore 'self' or parameters like 'verbose' that don't affect the result. - :param serialize: Optional function to serialize values before caching. - Example: serialize=pickle.dumps for binary file storage - :param deserialize: Optional function to deserialize cached values. - Example: deserialize=pickle.loads + :param serialize: Optional function to serialize values before caching + (e.g. ``pickle.dumps`` for binary file storage). + :param deserialize: Optional function to deserialize cached values + (e.g. ``pickle.loads``). :return: The decorated function. - ## Comprehensive Example + .. rubric:: Comprehensive Example Here's a complete example showcasing all major features of cache_this: @@ -1367,8 +1364,6 @@ def cache_this( In CacheA: getting value of foo.pkl b'\x80\x04K*.' >>> # == b'\x80\x04K*.' - - """ import inspect @@ -1492,9 +1487,10 @@ def add_extension(ext=None, name=None): >>> add_txt_ext('file') 'file.txt' - Note: If you want to add an extension to a name that already has an extension, - you can do that, but it will add the extension to the end of the name, - not replace the existing extension. + Note: + If you want to add an extension to a name that already has an extension, + you can do that, but it will add the extension to the end of the name, + not replace the existing extension. >>> add_txt_ext('file.txt') 'file.txt.txt' @@ -1504,7 +1500,6 @@ def add_extension(ext=None, name=None): >>> add_extension('.txt', 'file') == add_extension('txt', 'file') == 'file.txt' True - """ if ext.startswith(extsep): ext = ext[1:] @@ -1527,17 +1522,18 @@ def cached_method(func=None, *, maxsize=128, typed=False): to the method, excluding the first argument (typically `self`). This allows methods of a class to be cached while ignoring the instance (`self`) in the cache key. - Parameters: - - func (callable, optional): The method to be decorated. If not provided, a partially applied decorator - will be returned for later application. - - maxsize (int, optional): The maximum size of the cache. Defaults to 128. - - typed (bool, optional): If True, cache entries will be different based on argument types, such as - distinguishing between `1` and `1.0`. Defaults to False. + Args: + func: The method to be decorated. If not provided, a partially applied + decorator will be returned for later application. + maxsize: The maximum size of the cache. + typed: If True, cache entries will be different based on argument types, + such as distinguishing between `1` and `1.0`. Returns: - - callable: A wrapped function with LRU caching applied, ignoring the first argument (`self`). + A wrapped function with LRU caching applied, ignoring the first argument (`self`). + + .. rubric:: Example - Example: >>> class MyClass: ... @cached_method(maxsize=2, typed=True) ... def add(self, x, y): @@ -1583,17 +1579,17 @@ def lru_cache_method(func=None, *, maxsize=128, typed=False): to the method, excluding the first argument (typically `self`). This allows methods of a class to be cached while ignoring the instance (`self`) in the cache key. - Parameters: - - func (callable, optional): The method to be decorated. If not provided, a partially applied decorator - will be returned for later application. - - maxsize (int, optional): The maximum size of the cache. Defaults to 128. - - typed (bool, optional): If True, cache entries will be different based on argument types, such as - distinguishing between `1` and `1.0`. Defaults to False. + Args: + func: The method to be decorated. If not provided, a partially applied + decorator will be returned for later application. + maxsize: The maximum size of the cache. + typed: If True, cache entries will be different based on argument types, + such as distinguishing between `1` and `1.0`. Returns: - - callable: A wrapped function with LRU caching applied, ignoring the first argument (`self`). + A wrapped function with LRU caching applied, ignoring the first argument (`self`). - Example: + .. rubric:: Example >>> class MyClass: ... @lru_cache_method @@ -1659,7 +1655,7 @@ def cache_property_method( `cache_this`. One frequent use case would be to use `functools.partial` to fix the cache and key parameters of `cache_this` and inject that. - Example: + .. rubric:: Example >>> @cache_property_method(['normal_method', 'property_method']) ... class TestClass: @@ -1716,8 +1712,6 @@ def cache_property_method( 2 >>> c.property_method 2 - - """ if method_name is None: assert cls is not None, ( @@ -1803,8 +1797,9 @@ def mk_memoizer(cache): """ Make a memoizer that caches the output of a getter function in a cache. - Note: This is a specialized memoizer for getter functions/methods, i.e. - functions/methods that have the signature (instance, key) and return a value. + Note: + This is a specialized memoizer for getter functions/methods, i.e. + functions/methods that have the signature (instance, key) and return a value. :param cache: The cache to use. Must have __getitem__ and __setitem__ methods. :return: A memoizer that caches the output of the function in the cache. @@ -1820,7 +1815,6 @@ def mk_memoizer(cache): 20 >>> getter(None, 2) 20 - """ def memoize(method): @@ -1854,7 +1848,6 @@ def _mk_cache_instance(cache=None, assert_attrs=()): Traceback (most recent call last): ... AssertionError: cache should have the __setitem__ method, but does not: () - """ if isinstance(assert_attrs, str): assert_attrs = (assert_attrs,) @@ -1888,7 +1881,8 @@ def cache_vals(store=None, *, cache=dict): cache: The store you want to use to cache. Anything with a __setitem__(k, v) and a __getitem__(k). By default, it will use a dict - Returns: A subclass of the input store, but with caching (to the cache store) + Returns: + A subclass of the input store, but with caching (to the cache store) >>> from dol.caching import cache_vals >>> import time @@ -1985,8 +1979,11 @@ def mk_sourced_store(store=None, *, source=None, return_source_data=True): store: The class of the store you want to cache cache: The store you want to use to cache. Anything with a __setitem__(k, v) and a __getitem__(k). By default, it will use a dict + return_source_data: - Returns: A subclass of the input store, but with caching (to the cache store) + + Returns: + A subclass of the input store, but with caching (to the cache store) :param store: The class of the store you're talking to. This store acts as the cache @@ -2023,7 +2020,7 @@ def mk_sourced_store(store=None, *, source=None, return_source_data=True): >>> list(s) # the local store has one key ['some'] - # but if we ask for a key that is in the remote store, it provides it + But if we ask for a key that is in the remote store, it provides it: >>> assert s['foo'] == 'bar' looking for foo in Local @@ -2115,7 +2112,9 @@ def __contains__(self, k): def _slow_but_somewhat_general_hash(*args, **kwargs): """ Attempts to create a hash of the inputs, recursively resolving the most common hurdles (dicts, sets, lists) - Returns: A hash value for the input + + Returns: + A hash value for the input >>> _slow_but_somewhat_general_hash(1, [1, 2], a_set={1,2}, a_dict={'a': 1, 'b': [1,2]}) ((1, (1, 2)), (('a_set', (1, 2)), ('a_dict', (('a', 1), ('b', (1, 2)))))) @@ -2145,6 +2144,7 @@ def store_cached(store, key_func: Callable): memory and a key_func to compute the key under which to store the output. The key can be + - a single value under which the output should be stored, regardless of the input. - a key function that is called on the inputs to create a hash under which the function's output should be stored. @@ -2152,11 +2152,8 @@ def store_cached(store, key_func: Callable): store: The key-value store to use for caching. Must support __getitem__ and __setitem__. key_func: The key function that is called on the input of the function to create the key value. - Note: Union[Callable, Any] is equivalent to just Any, but reveals the two cases of a key more clearly. - Note: No, Union[Callable, Hashable] is not better. For one, general store keys are not restricted to hashable keys. - Note: No, they shouldn't. - - See Also: store_cached_with_single_key (for a version where the cache store key doesn't depend on function's args) + See Also: + store_cached_with_single_key (for a version where the cache store key doesn't depend on function's args) >>> # Note: Our doc test will use dict as the store, but to make the functionality useful beyond existing >>> # RAM-memorizer, you should use actual "persisting" stores that store in local files, or DBs, etc. @@ -2211,20 +2208,19 @@ def store_cached_with_single_key(store, key): The key should be a single value under which the output should be stored, regardless of the input. - Note: The wrapped function comes with a empty_cache attribute, which when called, empties the cache (i.e. removes - the key from the store) + Note: + The wrapped function comes with a empty_cache attribute, which when called, empties the cache (i.e. removes + the key from the store) - Note: The wrapped function has a hidden `_cache` attribute pointing to the store in case you need to peep into it. + Note: + The wrapped function has a hidden `_cache` attribute pointing to the store in case you need to peep into it. Args: store: The cache. The key-value store to use for caching. Must support __getitem__ and __setitem__. key: The store key under which to store the output of the function. - Note: Union[Callable, Any] is equivalent to just Any, but reveals the two cases of a key more clearly. - Note: No, Union[Callable, Hashable] is not better. For one, general store keys are not restricted to hashable keys. - Note: No, they shouldn't. - - See Also: store_cached (for a version whose keys are computed from the wrapped function's input. + See Also: + store_cached (for a version whose keys are computed from the wrapped function's input. >>> # Note: Our doc test will use dict as the store, but to make the functionality useful beyond existing >>> # RAM-memorizer, you should use actual "persisting" stores that store in local files, or DBs, etc. @@ -2334,6 +2330,9 @@ def _clear_method(self): # TODO: Normalize using store_decorator and add control over flush_cache method name def flush_on_exit(cls): + """Class decorator: a subclass whose ``__exit__`` calls ``flush_cache()`` (adding a + trivial ``__enter__`` if the class has none), so a write-cached store can be used as + a context manager that flushes on exit. Used by ``mk_write_cached_store``.""" new_cls = type(cls.__name__, (cls,), {}) if not hasattr(new_cls, "__enter__"): @@ -2551,13 +2550,13 @@ class WriteBackChainMap(ChainMap): Example use cases: - You're working with a local and a remote source of data. You'd like to list the - keys available in both, and use the local item if it's available, and if it's not, - you want it to be sourced from remote, but written in local for quicker access - next time. + keys available in both, and use the local item if it's available, and if it's not, + you want it to be sourced from remote, but written in local for quicker access + next time. - You have several sources to look for configuration values: a sequence of - configuration files/folders to look through (like a unix search path for command - resolution) and environment variables. + configuration files/folders to look through (like a unix search path for command + resolution) and environment variables. """ max_key_search_depth = 1 @@ -2612,6 +2611,7 @@ def _mk_cache_method_local_path_key( class HashableMixin: + """Mixin making instances hashable by identity (``id(self)``).""" def __hash__(self): return id(self) @@ -2622,6 +2622,8 @@ class HashableDict(HashableMixin, dict): # NOTE: cache uses (func, args, kwargs). Don't want to make more complex with a bind cast to (func, kwargs) only def cache_func_outputs(cache=HashableDict): + """Decorator factory intended to cache a function's outputs in ``cache``, keyed by ``(func, args, kwargs)``; + only positional-argument calls with an explicitly given ``cache`` actually hit the cache.""" cache = get_cache(cache) def cache_method_decorator(func): diff --git a/dol/content.py b/dol/content.py index 5e8a98a7..ad3a74d6 100644 --- a/dol/content.py +++ b/dol/content.py @@ -209,9 +209,10 @@ def _statically_defines_url_for(obj) -> bool: called per object. Anything this misses falls through to the plain-``getattr`` fallback in :func:`content_url`, so a miss costs correctness nothing. - NOTE: ``DelegatedAttribute`` is defined twice in dol -- ``dol.base`` (the one the wrap - machinery constructs) and an unused duplicate in ``dol.util``. If a wrap path ever - switches to the other copy this check silently stops working, so they must not diverge. + NOTE: + ``DelegatedAttribute`` is defined twice in dol -- ``dol.base`` (the one the wrap + machinery constructs) and an unused duplicate in ``dol.util``. If a wrap path ever + switches to the other copy this check silently stops working, so they must not diverge. """ d = getattr(obj, "__dict__", None) if d is not None and "url_for" in d: diff --git a/dol/dig.py b/dol/dig.py index e43448b1..2cd1ff7f 100644 --- a/dol/dig.py +++ b/dol/dig.py @@ -1,4 +1,10 @@ -"""Layers introspection""" +"""Layers introspection: walk the layers of a wrapped store and trace a key through them. + +Main entry points: + +- ``layers``: the list of nested layers (found through the ``store`` attribute), outermost first +- ``trace_getitem``, ``print_trace_info``: the (layer, method, value) steps of a ``__getitem__`` +""" from functools import partial diff --git a/dol/errors.py b/dol/errors.py index 80328084..55c633a6 100644 --- a/dol/errors.py +++ b/dol/errors.py @@ -1,4 +1,13 @@ -"""Error objects and utils""" +"""Error objects and utils. + +The exception classes dol raises (``NotAllowed``, ``OverWritesNotAllowedError``, +``KeyValidationError``, ...) and ``items_with_caught_exceptions``, an ``items()`` +that skips (or reports) the keys whose value cannot be fetched. + + >>> from dol.errors import items_with_caught_exceptions + >>> list(items_with_caught_exceptions({'a': 1})) + [('a', 1)] +""" from collections.abc import Mapping from inspect import signature @@ -39,9 +48,11 @@ def items_with_caught_exceptions( :param d: Any Mapping :param catch_exceptions: A tuple of exceptions that should be caught :param callback: A function that will be called every time an exception is caught. - The signature of the callback function is required to be: - k (key), e (error obj), d (mapping), i (index) - but + It may take any subset of the arguments ``k`` (key), ``e`` (error obj), + ``d`` (mapping) and ``i`` (index), by name (see the examples below); if its + signature cannot be inspected it is called with all four, positionally. + :param yield_callback_output: If True, also yield the callback's output for the + keys whose value raised. :return: An (key, val) generator with exceptions caught >>> from collections.abc import Mapping diff --git a/dol/explicit.py b/dol/explicit.py index be055f24..fd4ce1ab 100644 --- a/dol/explicit.py +++ b/dol/explicit.py @@ -1,5 +1,16 @@ -""" -utils to make stores based on a the input data itself +"""Stores whose keys are given explicitly, with values fetched lazily from a source. + +Main entry points: + +- ``KeysReader``: a collection of keys plus a ``getter(src, key)`` +- ``ExplicitKeysSource``: explicit keys plus a function reading the value for a key +- ``ExplicitKeysStore``: wrap a store so that its keys come from an explicit iterable +- ``ExplicitKeyMap``: a key mapper given as explicit dicts + + >>> from dol.explicit import KeysReader + >>> r = KeysReader({'apple': 'pie', 'banana': 'split'}, ['banana'], lambda src, k: src[k]) + >>> list(r), r['banana'] + (['banana'], 'split') """ from collections.abc import Mapping @@ -36,7 +47,7 @@ class KeysReader(Mapping): key_error_msg: A function that takes a source and a key, and returns an error message. - Example:: + .. rubric:: Example >>> src = {'apple': 'pie', 'banana': 'split', 'carrot': 'cake'} >>> key_collection = ['carrot', 'apple'] @@ -76,7 +87,6 @@ class KeysReader(Mapping): Traceback (most recent call last): ... KeyError: "Key banana was not found" - """ def __init__( @@ -176,7 +186,6 @@ class ExplicitKeysSource(ExplicitKeys, ObjReader, KvReader): [1, 2, 3] >>> list(s.values()) ['1', '2', '3'] - """ def __init__(self, key_collection: CollectionType, _obj_of_key: Callable): @@ -216,11 +225,16 @@ def __init__(self, store, key_collection): # TODO: Put on the path of deprecation, since KeyCodecs.mapped_keys is a better way to do this. class ExplicitKeyMap: + """A key mapper given as explicit ``key_of_id``/``id_of_key`` dicts (one is enough; + the other is derived, and both are checked to be inverse of each other). + Provides the ``_key_of_id``/``_id_of_key`` methods that ``kv_wrap`` looks for. + """ + def __init__(self, *, key_of_id: Mapping = None, id_of_key: Mapping = None): """ - :param key_of_id: - :param id_of_key: + :param key_of_id: The {inner_key: outer_key, ...} mapping + :param id_of_key: The {outer_key: inner_key, ...} mapping >>> km = ExplicitKeyMap(key_of_id={'a': 1, 'b': 2}) >>> km.id_of_key = {1: 'a', 2: 'b'} diff --git a/dol/filesys.py b/dol/filesys.py index 3341f173..20d00c0a 100644 --- a/dol/filesys.py +++ b/dol/filesys.py @@ -1,4 +1,26 @@ -"""File system access""" +"""File system access: dict-like stores over folders and files. + +``Files`` gives a folder a ``MutableMapping`` interface: keys are paths relative to the +root folder, values are the files' bytes. ``TextFiles``, ``JsonFiles`` and ``PickleFiles`` +add the corresponding value codecs. Writing under a sub-folder that does not exist raises +``KeyError``; wrap the store with ``mk_dirs_if_missing`` to create folders on write. + +Main entry points: + +- ``Files``: bytes of the files under a root folder +- ``TextFiles``: same, with text values +- ``JsonFiles``: same, with JSON-decoded values +- ``PickleFiles``: same, with pickled values +- ``mk_dirs_if_missing``: make a file store create missing directories on write + + >>> import tempfile + >>> s = Files(tempfile.mkdtemp()) + >>> s['hello.txt'] = b'world' + >>> s['hello.txt'] + b'world' + >>> list(s) + ['hello.txt'] +""" import os from os import stat as os_stat @@ -31,6 +53,7 @@ def ensure_slash_suffix(path: str): def paths_in_dir(rootdir, include_hidden=False): + """Yield the paths of the entries of ``rootdir`` (directories with a trailing separator), skipping hidden ones unless ``include_hidden``.""" try: for name in os.listdir(rootdir): if include_hidden or not name.startswith( @@ -85,17 +108,20 @@ def create_directories(dirpath, max_dirs_to_make: int | None = None): """ Create directories up to a specified limit. - Parameters: - dirpath (str): The directory path to create. - max_dirs_to_make (int, optional): The maximum number of directories to create. If None, there's no limit. + Args: + dirpath: The directory path to create. + max_dirs_to_make: The maximum number of directories to create. If None, + there's no limit. Returns: - bool: True if the directory was created successfully, False otherwise. + True if the directory exists (already, or after creation); False if creating + it would need more than ``max_dirs_to_make`` new directories (none are made). Raises: - ValueError: If max_dirs_to_make is negative. + ValueError: If max_dirs_to_make is negative. + + .. rubric:: Examples - Examples: >>> import tempfile, shutil >>> temp_dir = tempfile.mkdtemp() >>> target_dir = os.path.join(temp_dir, 'a', 'b', 'c') @@ -183,7 +209,6 @@ def process_path( ... ) >>> p == os.path.join('root_dir', 'a', 'b', 'c') + os.sep True - """ path = os.path.join(*path) if ensure_endswith_slash and ensure_does_not_end_with_slash: @@ -232,14 +257,13 @@ def ensure_dir( - a ``bool``' a standard message will be printed - a ``callable``; will be called on dirpath before directory is created -- you - can use this to ask the user for confirmation for example + can use this to ask the user for confirmation for example - a ''string``; this string will be printed Usage note: If you want to string or the (argument-less) callable to be dependent on ``dirpath``, you need make them so when calling ensure_dir. - """ if not os.path.exists(dirpath): if verbose: @@ -260,25 +284,17 @@ def temp_dir(dirname="", make_it_if_necessary=True, verbose=False): Create and return a path to a temporary directory that's guaranteed to be accessible to the user. - Parameters: - ---------- - dirname : str - Optional subdirectory name to append to the temporary directory path - make_it_if_necessary : bool - Whether to create the directory if it doesn't exist - verbose : bool, str, or callable - Controls verbosity when creating directories + Args: + dirname: Optional subdirectory name to append to the temporary directory path + make_it_if_necessary: Whether to create the directory if it doesn't exist + verbose: Controls verbosity when creating directories Returns: - ------- - str Path to a temporary directory that the user has access to - Notes: - ----- - This function creates a user-specific temporary directory to avoid permission - issues with system-wide temporary directories. The directory is guaranteed to - be accessible to the current user. + Note: + This function creates a user-specific temporary directory to avoid permission + issues with system-wide temporary directories. """ from tempfile import mkdtemp, gettempdir import uuid @@ -316,6 +332,7 @@ def temp_dir(dirname="", make_it_if_necessary=True, verbose=False): def mk_absolute_path(path_format): + """Expand a leading ``~``, or make a leading ``.`` path absolute; other paths are returned as is.""" if path_format.startswith("~"): path_format = os.path.expanduser(path_format) elif path_format.startswith("."): @@ -332,11 +349,13 @@ def mk_absolute_path(path_format): class KeyValidationError(KeyError): + """A ``KeyError`` for keys that fail a file-system store's validation.""" pass # TODO: The validate and try/except is a frequent pattern. Make it a decorator. def validate_key_and_raise_key_error_on_exception(func): + """Method decorator: validate the key first, and re-raise any exception of the method as a ``KeyError``.""" @wraps(func) def wrapped_method(self, k, *args, **kwargs): self.validate_key(k) @@ -392,6 +411,7 @@ def _for_repr(obj, quote="'"): class FileSysCollection(Collection): # rootdir = None # mentioning here so that the attribute is seen as an attribute before instantiation. + """Base collection of file-system paths under ``rootdir``, optionally restricted by ``subpath``, ``max_levels`` and hidden-file inclusion.""" def __init__( self, rootdir, @@ -443,6 +463,7 @@ def with_relative_paths(self): class DirCollection(FileSysCollection): + """Collection of the directory paths under ``rootdir``.""" def __iter__(self): yield from filter( self.is_valid_key, @@ -458,6 +479,7 @@ def __contains__(self, k): class FileCollection(FileSysCollection): + """Collection of the file paths under ``rootdir``.""" def __iter__(self): """ Iterator of valid filepaths. @@ -498,12 +520,14 @@ def __contains__(self, k): class FileInfoReader(FileCollection, KvReader): + """Reader mapping file paths to their ``os.stat`` result.""" def __getitem__(self, k): self.validate_key(k) return os_stat(k) class FileBytesReader(FileCollection, KvReader): + """Reader mapping file paths under ``rootdir`` to the files' bytes.""" _read_open_kwargs = dict( mode="rb", buffering=-1, @@ -560,6 +584,7 @@ class LocalFileDeleteMixin: to os.remove (with warning). See dol.trash module for available deletion strategies: + - default_delete_func: Safe trash with warning on fallback - permanent_delete: Direct os.remove (no warnings) - trash_only: Error if trash unavailable @@ -607,9 +632,11 @@ def __init__(self, *args, delete_func=None, **kwargs): delete_func: Optional custom deletion function. If None, uses class default (safe trash with fallback). Common options from dol.trash: + - default_delete_func (safe trash, warning on fallback) - permanent_delete (os.remove, no warnings) - trash_only (error if trash unavailable) + **kwargs: Passed to parent classes """ super().__init__(*args, **kwargs) @@ -649,10 +676,12 @@ class Files(FileBytesPersister): class FileStringReader(FileBytesReader): + """Reader mapping file paths to the files' text (files opened in text mode).""" _read_open_kwargs = dict(FileBytesReader._read_open_kwargs, mode="rt") class FileStringPersister(FileBytesPersister): + """Persister mapping file paths to the files' text (files opened in text mode).""" _read_open_kwargs = dict(FileBytesReader._read_open_kwargs, mode="rt") _write_open_kwargs = dict(FileBytesPersister._write_open_kwargs, mode="wt") @@ -686,7 +715,7 @@ class TextFiles(FileStringPersister): def mk_pickle_bytes_wrap( *, loads_kwargs: dict | None = None, dumps_kwargs: dict | None = None ) -> Callable: - """""" + """Make a ``wrap_kvs`` value-codec wrapper for pickle, with kwargs for ``pickle.loads``/``pickle.dumps``.""" return wrap_kvs( value_decoder=partial(pickle.loads, **(loads_kwargs or {})), value_encoder=partial(pickle.dumps, **(dumps_kwargs or {})), @@ -696,6 +725,7 @@ def mk_pickle_bytes_wrap( def mk_json_bytes_wrap( *, loads_kwargs: dict | None = None, dumps_kwargs: dict | None = None ) -> Callable: + """Make a ``wrap_kvs`` value-codec wrapper for JSON, with kwargs for ``json.loads``/``json.dumps``.""" return wrap_kvs( value_decoder=partial(json.loads, **(loads_kwargs or {})), value_encoder=partial(json.dumps, **(dumps_kwargs or {})), @@ -703,6 +733,7 @@ def mk_json_bytes_wrap( class ReprMixin: + """A ``__repr__`` showing the ``_init_kwargs`` the instance was created with.""" def __repr__(self): input_str = ", ".join( f"{k}={_for_repr(v)}" for k, v in getattr(self, "_init_kwargs", {}).items() @@ -736,6 +767,7 @@ class Jsons(ReprMixin, JsonFiles): # @wrap_kvs(key_of_id=lambda x: x[:-1], id_of_key=lambda x: x + path_sep) @mk_relative_path_store(prefix_attr="rootdir") class PickleStores(DirCollection): + """Reader mapping each sub-directory of ``rootdir`` to a ``PickleFiles`` store of it.""" def __getitem__(self, k): return PickleFiles(k) @@ -744,6 +776,7 @@ def __repr__(self): class DirReader(DirCollection, KvReader): + """Reader mapping each sub-directory of ``rootdir`` to a ``DirReader`` of it.""" def __getitem__(self, k): return DirReader(k) diff --git a/dol/kv_codecs.py b/dol/kv_codecs.py index 8444fbda..62f9a4ea 100644 --- a/dol/kv_codecs.py +++ b/dol/kv_codecs.py @@ -1,5 +1,24 @@ -""" -Tools to make Key-Value Codecs (encoder-decoder pairs) from standard library tools. +"""Tools to make Key-Value Codecs (encoder-decoder pairs) from standard library tools. + +A codec is a store wrapper: ``ValueCodecs.json()`` encodes values on write and decodes +them on read, ``KeyCodecs.suffixed('.json')`` adds the suffix on the way in and strips +it on the way out. Codecs compose with ``+``. + +Main entry points: + +- ``ValueCodecs``: ready-made value codecs (json, pickle, gzip, csv, str_to_bytes, ...) +- ``KeyCodecs``: ready-made key codecs (suffixed, prefixed, ...) +- ``key_based_value_trans``: a value codec chosen from the key + + >>> from dol.kv_codecs import ValueCodecs, KeyCodecs + >>> s = ValueCodecs.json()({}) + >>> s['a'] = {'x': 1} + >>> s.store, s['a'] + ({'a': '{"x": 1}'}, {'x': 1}) + >>> k = KeyCodecs.suffixed('.json')({}) + >>> k['a'] = 1 + >>> k.store, list(k) + ({'a.json': 1}, ['a']) """ # ------------------------------------ Codecs ------------------------------------------ @@ -60,6 +79,7 @@ def _csv_dict_extra_sig( # Note: @(_string + _csv_rw_sig) made (ax)black choke @__csv_rw_sig def csv_encode(string, *args, **kwargs): + """Encode rows (an iterable of iterables) into a CSV string (``csv.writer`` arguments accepted).""" with io.StringIO() as buffer: writer = csv.writer(buffer, *args, **kwargs) writer.writerows(string) @@ -68,6 +88,7 @@ def csv_encode(string, *args, **kwargs): @__csv_rw_sig def csv_decode(string, *args, **kwargs): + """Decode a CSV string into a list of rows (``csv.reader`` arguments accepted).""" with io.StringIO(string) as buffer: reader = csv.reader(buffer, *args, **kwargs) return list(reader) @@ -81,7 +102,6 @@ def csv_dict_encode(string, *args, **kwargs): >>> encoded = csv_dict_encode(data, fieldnames=['a', 'b']) >>> encoded 'a,b\r\n1,2\r\n3,4\r\n' - """ _ = kwargs.pop("fieldcasts", None) # this one is for decoder only with io.StringIO() as buffer: @@ -122,7 +142,6 @@ def csv_dict_decode(string, *args, **kwargs): [{'a': 1, 'b': 2}, {'a': 3, 'b': 4}] >>> csv_dict_decode(encoded, fieldnames=['a', 'b'], fieldcasts={'b': float}) [{'a': '1', 'b': 2.0}, {'a': '3', 'b': 4.0}] - """ fieldcasts = kwargs.pop("fieldcasts", lambda row: row) if isinstance(fieldcasts, Iterable): @@ -179,6 +198,7 @@ def _xml_tree_decode( def extract_arguments(func, args, kwargs): + """Map ``args``/``kwargs`` to ``func``'s parameter names, leniently (partial and excess allowed, kinds ignored).""" return Sig(func).map_arguments( args, kwargs, allow_partial=True, allow_excess=True, ignore_kind=True ) @@ -206,6 +226,7 @@ def _codec_wrap(cls, encoder: Callable, decoder: Callable, **kwargs): def codec_wrap(cls, encoder: Callable, decoder: Callable, *, exclude=()): + """Make a ``cls`` codec factory from an ``encoder`` and a ``decoder``, with the merged signature of both.""" _cls_codec_wrap = partial(_codec_wrap, cls) factory = partial(_cls_codec_wrap, encoder, decoder) # TODO: Review this signature here. Should be keyword-only to match what @@ -300,8 +321,6 @@ class ValueCodecs(CodecCollection): { "b": 2 } - - """ # TODO: Clean up module import polution? @@ -547,7 +566,7 @@ def key_based_value_trans( """A factory that creates a value codec that uses the key to determine the codec to use. - # a key_func that gets the extension of a file path + Below, ``key_func`` gets the extension of a file path: >>> import json >>> from functools import partial @@ -557,8 +576,6 @@ def key_based_value_trans( >>> trans = key_based_value_trans( ... key_func, value_trans_mapping, default_factory=lambda: identity_func ... ) - - """ if k is NotGiven: return partial( diff --git a/dol/misc.py b/dol/misc.py index d061af69..ed6d0021 100644 --- a/dol/misc.py +++ b/dol/misc.py @@ -1,5 +1,16 @@ -""" -Functions to read from and write to misc sources +"""Functions to read from and write to misc sources, choosing the codec from the key. + +``get_obj``/``set_obj`` read and write a file with the codec picked from its extension +(``.json``, ``.csv``, ``.pkl``, ...); ``MiscReaderMixin``/``MiscStoreMixin`` add the same +key-conditioned (de)serialization to any store. + + >>> from dol.misc import MiscStoreMixin + >>> class M(MiscStoreMixin, dict): + ... pass + >>> m = M() + >>> m['a.json'] = {'x': 1} + >>> dict.__getitem__(m, 'a.json'), m['a.json'] + (b'{"x": 1}', {'x': 1}) """ # TODO: Completely redo this, using preset and postget and making it into a plugin @@ -307,9 +318,12 @@ def __iter__(self): class MiscStoreMixin(MiscReaderMixin): r"""Mixin to transform incoming and outgoing vals according to the key their under. - Warning: If used as a subclass, this mixin should (in general) be placed before the store - See also: preset and postget args from wrap_kvs decorator from dol.trans. + Warning: + If used as a subclass, this mixin should (in general) be placed before the store + + See also: + preset and postget args from wrap_kvs decorator from dol.trans. >>> # Make a class to wrap a dict with a layer that transforms written and read values >>> class MiscStore(MiscStoreMixin, dict): @@ -355,7 +369,6 @@ class MiscStoreMixin(MiscReaderMixin): a.csv: b'event,year\r\n Magna Carta,1215\r\n Guido,1956\r\n' a.txt: b'this is not a text' a.json: b'{"str": "field", "int": 42, "float": 3.14, "array": [1, 2], "nested": {"a": 1, "b": 2}}' - """ _dflt_outgoing_val_trans_for_key = staticmethod(identity_method) @@ -393,8 +406,8 @@ def set_obj( outgoing_val_trans_for_key=imdict(dflt_outgoing_val_trans_for_key), func_key=lambda k: os.path.splitext(k)[1], ): - """A quick way to get an object, with default... - # everything (but the key, you know, a clue of what you want)""" + """A quick way to set an object, with defaults for everything + (but the key and value, you know, a clue of what you want to store).""" if isinstance(store, Files) and store._prefix in {"", "/"}: k = os.path.abspath(os.path.expanduser(k)) @@ -438,7 +451,6 @@ class MiscGetterAndSetter(MiscGetter): >>> # using bin ... misc_objs[pjoin('tmp.bin')] = b'let us pretend these are bytes of an audio waveform' >>> assert misc_objs[pjoin('tmp.bin')] == b'let us pretend these are bytes of an audio waveform' - """ def __init__( diff --git a/dol/mixins.py b/dol/mixins.py index 76d0a7c3..143c8184 100644 --- a/dol/mixins.py +++ b/dol/mixins.py @@ -1,4 +1,22 @@ -"""Mixins""" +"""Mixins that add or restrict store behaviours. + +Main entry points: + +- ``ReadOnlyMixin``: forbid writes and deletes +- ``OverWritesNotAllowedMixin``: forbid writing to an existing key +- ``SimpleJsonMixin``: JSON-encoded values +- ``IterBasedSizedContainerMixin``: ``__len__`` and ``__contains__`` from ``__iter__`` + + >>> from dol.mixins import OverWritesNotAllowedMixin + >>> class P(OverWritesNotAllowedMixin, dict): + ... pass + >>> p = P() + >>> p['a'] = 1 + >>> p['a'] = 2 # doctest: +ELLIPSIS + Traceback (most recent call last): + ... + dol.errors.OverWritesNotAllowedError: key a already exists and cannot be overwritten... +""" import json from dol.errors import ( @@ -32,6 +50,7 @@ def _id_of_key(self, k): """ Maps an interface identifier (key) to an internal identifier (_id) that is actually used to perform operations. Can also perform validation and permission checks. + :param k: interface identifier of some data :return: internal identifier _id """ @@ -40,6 +59,7 @@ def _id_of_key(self, k): def _key_of_id(self, _id): """ The inverse of _id_of_key. Maps an internal identifier (_id) to an interface identifier (key) + :param _id: :return: """ @@ -55,6 +75,7 @@ class IdentityValsWrapMixin: def _data_of_obj(self, v): """ Serialization of a python object. + :param v: A python object. :return: The serialization of this object, in a format that can be stored by __getitem__ """ @@ -63,6 +84,7 @@ def _data_of_obj(self, v): def _obj_of_data(self, data): """ Deserialization. The inverse of _data_of_obj. + :param data: Serialized data. :return: The python object corresponding to this data. """ @@ -96,8 +118,11 @@ def __iter__(self): def __contains__(self, k) -> bool: """ Check if collection of keys contains k. - Note: This method iterates over all elements of the collection to check if k is present. - Therefore it is not efficient, and in most cases should be overridden with a more efficient version. + + Note: + This method iterates over all elements of the collection to check if k is present. + Therefore it is not efficient, and in most cases should be overridden with a more efficient version. + :return: True if k is in the collection, and False if not """ return self._key_filt(k) and super().__contains__(k) @@ -132,7 +157,9 @@ def pop(self, k): class OverWritesNotAllowedMixin: """Mixin for only allowing a write to a key if they key doesn't already exist. - Note: Should be before the persister in the MRO. + + Note: + Should be before the persister in the MRO. >>> class TestPersister(OverWritesNotAllowedMixin, dict): ... pass @@ -179,8 +206,11 @@ class GetBasedContainerMixin: def __contains__(self, k) -> bool: """ Check if collection of keys contains k. - Note: This method actually fetches the contents for k, returning False if there's a key error trying to do so - Therefore it may not be efficient, and in most cases, a method specific to the case should be used. + + Note: + This method actually fetches the contents for k, returning False if there's a key error trying to do so + Therefore it may not be efficient, and in most cases, a method specific to the case should be used. + :return: True if k is in the collection, and False if not """ try: @@ -194,8 +224,11 @@ class IterBasedContainerMixin: def __contains__(self, k) -> bool: """ Check if collection of keys contains k. - Note: This method iterates over all elements of the collection to check if k is present. - Therefore it is not efficient, and in most cases should be overridden with a more efficient version. + + Note: + This method iterates over all elements of the collection to check if k is present. + Therefore it is not efficient, and in most cases should be overridden with a more efficient version. + :return: True if k is in the collection, and False if not """ for collection_key in self.__iter__(): @@ -208,8 +241,11 @@ class IterBasedSizedMixin: def __len__(self) -> int: """ Number of elements in collection of keys. - Note: This method iterates over all elements of the collection and counts them. - Therefore it is not efficient, and in most cases should be overridden with a more efficient version. + + Note: + This method iterates over all elements of the collection and counts them. + Therefore it is not efficient, and in most cases should be overridden with a more efficient version. + :return: The number (int) of elements in the collection of keys. """ # TODO: some other means to more quickly count files? @@ -222,13 +258,15 @@ def __len__(self) -> int: class IterBasedSizedContainerMixin(IterBasedSizedMixin, IterBasedContainerMixin): """ - An ABC that defines - (a) how to iterate over a collection of elements (keys) (__iter__) - (b) check that a key is contained in the collection (__contains__), and - (c) how to get the number of elements in the collection + An ABC that defines: + + (a) how to iterate over a collection of elements (keys) (``__iter__``) + (b) check that a key is contained in the collection (``__contains__``), and + (c) how to get the number of elements in the collection (``__len__``) + This is exactly what the collections.abc.Collection (from which Keys inherits) does. The difference here, besides the "Keys" purpose-explicit name, is that Keys offers default - __len__ and __contains__ definitions based on what ever __iter__ the concrete class defines. + ``__len__`` and ``__contains__`` definitions based on what ever ``__iter__`` the concrete class defines. Keys is a collection (i.e. a Sized (has __len__), Iterable (has __iter__), Container (has __contains__). It's purpose is to serve as a collection of object identifiers in a key->obj mapping. diff --git a/dol/naming.py b/dol/naming.py index 93ef67a0..6721b164 100644 --- a/dol/naming.py +++ b/dol/naming.py @@ -1,5 +1,14 @@ -""" -This module is about generating, validating, and operating on (parametrized) fields (i.e. stings, e.g. paths). +"""This module is about generating, validating, and operating on (parametrized) fields (i.e. strings, e.g. paths). + +Main entry points: + +- ``StrTupleDict``: convert a templated name between string, tuple and dict forms +- ``mk_pattern_from_template_and_format_dict``: a compiled regex from a template +- ``get_fields_from_template``: the field names of a template + + >>> from dol.naming import get_fields_from_template + >>> get_fields_from_template('this{is}an{example}') + ['is', 'example'] """ import re @@ -50,6 +59,7 @@ def validate_kwargs( Utility to validate a dict. It's main use is to validate function arguments (expressing the validation checks in validation_dict) by doing validate_kwargs(locals()), usually in the beginning of the function (to avoid having more accumulated variables than we need in locals()) + :param kwargs_to_validate: as the name implies... :param validation_dict: A dict specifying what to validate. Keys are usually name of variables (when feeding locals()) and values are dicts, themselves specifying check:check_val pairs where check is a string that @@ -214,8 +224,9 @@ def update_fields_of_namedtuple( def get_fields_from_template(template): """ Get list from {item} items of template string + :param template: a "template" string (a string with {item} items - -- the kind that is used to mark token for str.format) + -- the kind that is used to mark token for str.format) :return: a list of the token items of the string, in the order they appear >>> get_fields_from_template('this{is}an{example}of{a}template') @@ -318,10 +329,13 @@ def mk_extract_pattern( # TODO: Is dependent on path sep -- separate concern def mk_pattern_from_template_and_format_dict(template, format_dict=None, sep=path_sep): r"""Make a compiled regex to match template + Args: template: A format string format_dict: A dict whose keys are template fields and values are regex strings to capture them - Returns: a compiled regex + + Returns: + a compiled regex Assert on *behavior* (matching) rather than the exact pattern string, so the examples hold on every OS (the field separator -- and therefore the default @@ -408,6 +422,7 @@ def _mk(self, *args, **kwargs): function. The required fields are in self.fields. Does NOT check for validity of the vals. + :param kwargs: The name=val arguments needed to construct a valid name :return: an name """ @@ -430,6 +445,12 @@ def _mk(self, *args, **kwargs): # # # @add_wrapper_method class StrTupleDict: + """Convert a parametrized name between its string, tuple and dict forms. + + Built from a string template with ``{field}`` placeholders (and optional regexes + for the fields). See ``__init__`` for the parameters and an example. + """ + def __init__( self, template: str | tuple | list, @@ -453,6 +474,7 @@ def __init__( process_info_dict: A sort of converse of format_dict. This is a {field_name: field_conversion_func, ...} dict that is used to convert info_dict values before returning them. + name_separator: Used >>> ln = StrTupleDict('/home/{user}/fav/{num}.txt', @@ -538,6 +560,7 @@ def _mk(self, *args, **kwargs): function. The required fields are in self.fields. Does NOT check for validity of the vals. + :param kwargs: The name=val arguments needed to construct a valid name :return: an name """ @@ -562,6 +585,7 @@ def _mk(self, *args, **kwargs): def is_valid(self, s: str): """Check if the name has the "upload format" (i.e. the kind of fields that are _ids of fv_mgc, and what name means in most of the iatis system. + :param s: the string to check :return: True iff name has the upload format """ @@ -570,6 +594,7 @@ def is_valid(self, s: str): def str_to_dict(self, s: str): """ Get a dict with the arguments of an name (for example group, user, subuser, etc.) + :param s: :return: a dict holding the argument fields and values """ @@ -634,6 +659,7 @@ def namedtuple_to_str(self, nt): def extract(self, field, s): """Extract a single item from an name + :param field: field of the item to extract :param s: the string from which to extract it :return: the value for name @@ -645,6 +671,7 @@ def extract(self, field, s): def replace_name_elements(self, s: str, **elements_kwargs): """Replace specific name argument values with others + :param s: the string to replace :param elements_kwargs: the arguments to replace (and their values) :return: a new name @@ -697,6 +724,7 @@ class StrTupleDictWithPrefix(StrTupleDict): process_info_dict: A sort of converse of format_dict. This is a {field_name: field_conversion_func, ...} dict that is used to convert info_dict values before returning them. + name_separator: Used >>> ln = StrTupleDictWithPrefix('/home/{user}/fav/{num}.txt', @@ -747,6 +775,7 @@ def __init__(self, *args, **kwargs): def _mk_prefix(self, *args, **kwargs): """ Make a prefix for an uploads name that has has the path up to the first None argument. + :return: A string that is the prefix of a valid name """ assert len(args) + len(kwargs) <= self.n_fields, ( @@ -785,6 +814,7 @@ def _mk_prefix(self, *args, **kwargs): def is_valid_prefix(self, s): """Check if name is a valid prefix. + :param s: a string (that might or might not be a valid prefix) :return: True iff name is a valid prefix """ @@ -940,10 +970,13 @@ def is_valid_name(self, name): class BigDocTest: + """Naming-scheme example holder whose (large) doctest is currently disabled. + + The former doctest is kept as comments in the class body. """ # TODO: Fix this test (maybe test assertions aren't correct) - # This happened when we changed some re.compile to safe_compile + # This happened when we changed some re.compile to safe_compile # >>> # >>> e_name = BigDocTest.mk_e_naming() # >>> u_name = BigDocTest.mk_u_naming() @@ -994,7 +1027,6 @@ class BigDocTest: # >>> u_name.extract('user', u_name_2) # 'ANOTHER_USER' # >>> - # # >>> ####### mk_prefix(self, *args, **kwargs): ###### # >>> e_name.mk_prefix() @@ -1039,7 +1071,6 @@ class BigDocTest: # >>> name = 's3://bucket-redrum/example/files/oopsy@domain.com/ozeip/2008-11-04/1225779243969_1225779246969' # >>> e_name.replace_name_elements(name, user='NEW_USER', group='NEW_GROUP') # 's3://bucket-NEW_GROUP/example/files/NEW_USER/ozeip/2008-11-04/1225779243969_1225779246969' - """ @staticmethod def process_info_dict_for_example(**info_dict): @@ -1159,23 +1190,25 @@ def mk_store_from_path_format_store_cls( store: The instance or class to wrap subpath: The subpath (defining the subset of the data pointed at by the URI store_cls_kwargs: # if store is a class, the kwargs that you would have given the store_cls to make itself - key_type: The key type you want to interface with: - dict, tuple, namedtuple, str or 'dict', 'tuple', 'namedtuple', 'str' + key_type: The key type you want to interface with: ``dict``, ``tuple``, + ``namedtuple``, ``str``, or one of those names as a string keymap: # the keymap instance or class you want to use to map keys keymap_kwargs: # if keymap is a cls, the kwargs to give it (besides the subpath) name: The name to give the class the function will make here - Returns: An instance of a wrapped class + Returns: + An instance of a wrapped class + + + .. rubric:: Example + .. code-block:: python - Example: - ``` - # Get a (session, bt) indexed LocalJsonStore - s = mk_store_from_path_format_store_cls(LocalJsonStore, - os.path.join(root_dir, 'd'), - subpath='{session}/d/{bt}', - keymap_kwargs=dict(process_info_dict={'session': int, 'bt': int})) - ``` + # Get a (session, bt) indexed LocalJsonStore + s = mk_store_from_path_format_store_cls(LocalJsonStore, + os.path.join(root_dir, 'd'), + subpath='{session}/d/{bt}', + keymap_kwargs=dict(process_info_dict={'session': int, 'bt': int})) """ if isinstance(keymap, type): keymap = keymap(subpath, **(keymap_kwargs or {})) # make the keymap instance @@ -1218,11 +1251,14 @@ class PartialFormatter(Formatter): >>> partial_formatter.format(str_template, bar="BAR", b=34) 'foo:{foo} bar=BAR a={a} b=34.00 c={c}' - Note: If you only need a formatting function (not the transformed formatting string), a simpler solution may be: - ``` - import functools - format_str = functools.partial(str_template.format, bar="BAR", b=34) - ``` + Note: + If you only need a formatting function (not the transformed formatting string), a simpler solution may be: + + .. code-block:: python + + import functools + format_str = functools.partial(str_template.format, bar="BAR", b=34) + See https://stackoverflow.com/questions/11283961/partial-string-formatting for more options and discussions. """ diff --git a/dol/paths.py b/dol/paths.py index 4c034c66..520ccedd 100644 --- a/dol/paths.py +++ b/dol/paths.py @@ -1,7 +1,6 @@ """Module for path (and path-like) object manipulation - -Examples:: +Examples: >>> d = {'a': {'b': {'c': 1, 'd': 2}, 'e': 3}} >>> list(path_filter(lambda p, k, v: v == 2, d)) @@ -14,7 +13,6 @@ >>> path_set(d, ('a', 'b', 'new_ab_key'), 42) >>> d {'a': {'b': {'c': 1, 'd': 4, 'new_ab_key': 42}, 'e': 3}} - """ from functools import wraps, partial @@ -122,6 +120,7 @@ def flatten_dict( d: The dictionary to flatten sep: The separator to use for joining keys, or a function that takes a path and a key and returns a new path. + parent_path: The path to the parent of the current dict visit_nested: A function that returns True if a value should be visited egress: A function that takes a generator of key-value pairs and returns a mapping @@ -131,7 +130,6 @@ def flatten_dict( {'a.b': 2, 'c': 3} >>> flatten_dict(d, sep='/') {'a/b': 2, 'c': 3} - """ return egress( flattened_dict_items( @@ -164,10 +162,12 @@ def leaf_paths( d: The nested dictionary to get the leaf paths from sep: The separator to use for joining keys, or a function that takes a path and a key and returns a new path. + parent_path: The path to the parent of the current dict egress: A function that takes a generator of key-value pairs and returns a mapping - Example: + .. rubric:: Example + >>> leaf_paths({'a': {'b': 2}, 'c': 3}) {'a': {'b': 'a.b'}, 'c': 'c'} @@ -205,14 +205,17 @@ def _leaf_paths_recursive( def raise_on_error(d: dict): + """``on_error`` policy for ``path_get``: re-raise the caught error.""" raise def return_none_on_error(d: dict): + """``on_error`` policy for ``path_get``: return ``None``.""" return None def return_empty_tuple_on_error(d: dict): + """``on_error`` policy for ``path_get``: return ``()``.""" return () @@ -251,7 +254,6 @@ def _path_get( # ... 'k': 'c', # ... 'error': KeyError('c') # ... } - """ if path_to_keys is not None: @@ -287,16 +289,19 @@ def _path_get( def split_if_str(obj, sep="."): + """Split ``obj`` on ``sep`` if it is a string; return it unchanged otherwise.""" if isinstance(obj, str): return obj.split(sep) return obj def separate_keys_with_separator(obj, sep="."): + """Split a string path on ``sep`` and cast numeric parts to ``int``; a non-string iterable is only cast element-wise.""" return map(cast_to_int_if_numeric_str, split_if_str(obj, sep)) def getitem(obj, k): + """Return ``obj[k]``.""" return obj[k] @@ -305,11 +310,13 @@ def get_attr_or_item(obj, k): If ``k`` is a string, tries to get ``k`` as an attribute of ``obj`` first, and if that fails, gets it as ``obj[k]`` - WARNING: The hardcoded priority choices of this function regarding when to try - k as an item, index, or attribute, don't apply to every case, so you may want to - use an explicit value getter to be more robust! + WARNING: + The hardcoded priority choices of this function regarding when to try + k as an item, index, or attribute, don't apply to every case, so you may want to + use an explicit value getter to be more robust! # >>> d = {'a': [1, {'items': 2, '3': 33, 3: 42}]} + >>> get_attr_or_item({'items': 2}, 'items') 2 @@ -346,7 +353,6 @@ def get_attr_or_item(obj, k): Traceback (most recent call last): ... KeyError: 2 - """ if isinstance(k, str): if str.isnumeric(k) and not isinstance(obj, Mapping): @@ -382,7 +388,7 @@ def keys_and_indices_path(str_path, *, sep=".", index_pattern=r"\[(\d+)\]"): Returns: tuple: A tuple representation of the path, e.g., ("a21-59c", "message", 2, "user"). - Example: + .. rubric:: Example >>> keys_and_indices_path("a21-59c.message[2].user") ('a21-59c', 'message', 2, 'user') @@ -457,12 +463,14 @@ def path_get( >>> path_get([1, [4, 5, {'a': A}], 3], '1.2.a.an_attribute') 42 - Note: The underlying function is ``_path_get``, but `path_get` has defaults and - flexible input processing for more convenience. + Note: + The underlying function is ``_path_get``, but `path_get` has defaults and + flexible input processing for more convenience. - Note: ``path_get`` contains some ready-made ``OnErrorType`` functions in its - attributes. For example, see how we can make ``path_get`` have the same behavior - as ``dict.get`` by passing ``path_get.return_none_on_error`` as ``on_error``: + Note: + ``path_get`` contains some ready-made ``OnErrorType`` functions in its + attributes. For example, see how we can make ``path_get`` have the same behavior + as ``dict.get`` by passing ``path_get.return_none_on_error`` as ``on_error``: >>> dd = path_get({}, 'no.keys', on_error=path_get.return_none_on_error) >>> dd is None @@ -470,7 +478,6 @@ def path_get( For example, ``path_get.raise_on_error``, ``path_get.return_none_on_error``, and ``path_get.return_empty_tuple_on_error``. - """ if sep is None: if isinstance(path, str): @@ -528,8 +535,9 @@ def paths_getter( get multiple paths, returning the (path, value) pairs in a dict (by default), or via any pairs aggregator (``egress``) function. - Note: For reasons who's clarity is burried in historical legacy, the order of - obj and path are the opposite of path_get. + Note: + For reasons who's clarity is burried in historical legacy, the order of + obj and path are the opposite of path_get. :param paths: The paths to get :param obj: The object to get the paths from @@ -555,7 +563,6 @@ def paths_getter( >>> path_extractor_2 = paths_getter({'california': 'a.c', 'dreaming': 'd'}) >>> path_extractor_2(obj) {'california': 2, 'dreaming': 3} - """ kwargs = dict( on_error=on_error, @@ -599,6 +606,7 @@ def chain_of_getters( @add_as_attribute_of(path_get) def cast_to_int_if_numeric_str(k): + """Cast ``k`` to ``int`` if it is a numeric string; return it unchanged otherwise.""" if isinstance(k, str) and str.isnumeric(k): return int(k) return k @@ -649,7 +657,7 @@ class PathMappedData(KeysReader): data: The mapping to extract data from paths: The paths to extract data from the mapping - Example:: + .. rubric:: Example >>> data = { ... 'a': { @@ -679,7 +687,6 @@ class PathMappedData(KeysReader): Traceback (most recent call last): ... KeyError: "Key a.b.1.c was not found....key_collection attribute)" - """ def __init__( @@ -745,7 +752,6 @@ def path_edit(d: Mapping, edits: Edits = ()) -> Mapping: >>> path_edit(d, {'a': 4, 'd.e.f': 5}) {'a': 4, 'b': {'c': 2}, 'd': {'e': {'f': 5}}} - """ if isinstance(edits, Mapping): @@ -781,7 +787,7 @@ def path_filter( (instead of the default depth-first traversal). :return: An iterator of paths to values that pass the ``pkv_filt`` - Example:: + .. rubric:: Example >>> d = {'a': {'b': {'c': 1, 'd': 2}, 'e': 3}} >>> list(path_filter(lambda p, k, v: v == 2, d)) @@ -815,8 +821,9 @@ def path_filter( >>> vals [42, 'meaning of life'] - Note: pkv_filt is first to match the order of the arguments of the - builtin filter function. + Note: + pkv_filt is first to match the order of the arguments of the + builtin filter function. """ _leaf_yield = partial(_path_matcher_leaf_yield, pkv_filt, None) kwargs = dict(leaf_yield=_leaf_yield, breadth_first=breadth_first) @@ -891,6 +898,7 @@ class KeyPath: Args: path_sep: The path separator (used to make string paths from iterable paths and visa versa + _path_type: The type of the outcoming (inner) path. But really, any function to convert from a list to the outer path type we want. @@ -936,19 +944,19 @@ class KeyPath: >>> s {'a': {'b': {}}} - Note: By default ``KeyPath`` reads with paths only when all the keys of the path - are valid (i.e. have a value), and, just like a ``dict``, will *not* create - intermediate nested values for you on write. Pass ``create_missing=True`` to opt - into write-through autovivification: missing intermediates are created on write - (like ``collections.defaultdict``, but with an optional contextual per-level - ``mk_missing(ctx)`` factory), and the change persists correctly even through - persistent / copy-semantics stores. See ``misc/docs/dol_issue16_design.md``. + Note: + By default ``KeyPath`` reads with paths only when all the keys of the path + are valid (i.e. have a value), and, just like a ``dict``, will *not* create + intermediate nested values for you on write. Pass ``create_missing=True`` to opt + into write-through autovivification: missing intermediates are created on write + (like ``collections.defaultdict``, but with an optional contextual per-level + ``mk_missing(ctx)`` factory), and the change persists correctly even through + persistent / copy-semantics stores. See ``misc/docs/dol_issue16_design.md``. >>> s = KeyPath('.', create_missing=True)({}) >>> s['a.b.c'] = 42 >>> s['a.b.c'] 42 - """ path_sep: str = path_sep @@ -1003,9 +1011,8 @@ class PrefixRelativizationMixin: (assumed to exist). The cannonical use case is when keys are absolute file paths, but we want to identify data through relative paths. Instead of referencing files through an absolute path such as - /A/VERY/LONG/ROOT/FOLDER/the/file/we.want - we can instead reference the file as - the/file/we.want + ``/A/VERY/LONG/ROOT/FOLDER/the/file/we.want`` we can instead reference the file + as ``the/file/we.want``. Note though, that PrefixRelativizationMixin can be used, not only for local paths, but when ever a string reference is involved. @@ -1113,7 +1120,8 @@ def mk_relative_path_store( name: The name of the new store (by default 'RelPath' + store_cls.__name__) with_key_validation: Whether keys should be validated upon access (store_cls must have an is_valid_key method - Returns: A new class that uses relative paths (i.e. where _prefix is automatically added to incoming keys, + Returns: + A new class that uses relative paths (i.e. where _prefix is automatically added to incoming keys, and the len(_prefix) first characters are removed from outgoing keys. >>> # The dynamic way (if you try this at home, be aware of the pitfalls of the dynamic way @@ -1147,7 +1155,6 @@ def mk_relative_path_store( >>> # but under the hood, the dict we wrapped actually contains the '/ROOT/' prefix >>> dict(s.store) {'/ROOT/foo': 'bar'} - """ # name = name or ("RelPath" + store_cls.__name__) # __module__ = __module__ or getattr(store_cls, "__module__", None) @@ -1242,6 +1249,7 @@ def _key_mapped(self, k, *args, __name=_method_name, **kwargs): # TODO: Intended to replace the init-less PrefixRelativizationMixin # (but should change name if so, since Mixins shouldn't have inits) class RelativePathKeyMapper: + """Key mapper adding ``prefix`` on the way in and removing it on the way out.""" def __init__(self, prefix): self._prefix = prefix self._prefix_length = len(self._prefix) @@ -1255,6 +1263,7 @@ def _key_of_id(self, _id): @store_decorator def prefixless_view(store=None, *, prefix=None): + """Wrap ``store`` so that keys are seen without ``prefix`` (added back on access).""" key_mapper = RelativePathKeyMapper(prefix) return wrap_kvs( store, id_of_key=key_mapper._id_of_key, key_of_id=key_mapper._key_of_id @@ -1321,7 +1330,8 @@ def _prefix_filter_with_relativization(store, prefix: str): @store_decorator def add_prefix_filtering(store=None, *, relativize_prefix: bool = False): - """Add prefix filtering to a store. + """Make a missing key that is a prefix of existing keys return the sub-mapping of + those keys (so ``s['a/']`` lists everything "under" ``a/``). >>> d = {'a/b': 1, 'a/c': 2, 'd/e': 3, 'f': 4} >>> s = add_prefix_filtering(d) @@ -1333,7 +1343,6 @@ def add_prefix_filtering(store=None, *, relativize_prefix: bool = False): >>> D = add_prefix_filtering(UserDict) >>> s = D(d) >>> assert s['a/'] == {'a/b': 1, 'a/c': 2} - """ __prefix_filter = _prefix_filter if relativize_prefix: @@ -1362,6 +1371,7 @@ def handle_prefixes( store: The store to wrap prefix: The prefix to use. If None and the store is an instance (not type), will take the longest common prefix as the prefix. + filter_prefix: Whether to filter out keys that don't start with the prefix relativize_prefix: Whether to relativize the prefix default_prefix: The default prefix to use if no prefix is given and the store @@ -1374,7 +1384,6 @@ def handle_prefixes( {'every/thing': 42, 'this/too': 0, 'foo': 'bar'} >>> dict(dd.store) # but see where the underlying store actually wrote 'bar': {'/ROOT/of/every/thing': 42, '/ROOT/of/this/too': 0, '/ROOT/of/foo': 'bar'} - """ if prefix is None: if isinstance(store, type): @@ -1397,6 +1406,7 @@ def handle_prefixes( class PathKeyTypes(Enum): + """Enum of the path key forms: ``str``, ``dict``, ``tuple``, ``namedtuple``.""" str = "str" dict = "dict" tuple = "tuple" @@ -1505,7 +1515,6 @@ def rel_path_wrap(o, _prefix): >>> class MyStore(mk_relative_path_store(dict)): # Indeed, mk_relative_path_store(dict) is a class you can subclass ... def __init__(self, _prefix, *args, **kwargs): ... self._prefix = _prefix - """ from dol import kv_wrap @@ -1554,10 +1563,11 @@ def _return_none_if_none_input(func): >>> assert foo.bar(None) is None >>> assert foo.bar(x=None) is None - Note: On the other hand, this will not return `None`, but should: - ``foo.bar(y=3, x=None)``. To achieve this, we'd need to look into the signature, - which seems like overkill and I might not want that systematic overhead in my - methods. + Note: + On the other hand, this will not return `None`, but should: + ``foo.bar(y=3, x=None)``. To achieve this, we'd need to look into the signature, + which seems like overkill and I might not want that systematic overhead in my + methods. """ @wraps(func) @@ -1619,6 +1629,7 @@ def _field_names(string_template): def identity(x): + """Return ``x``.""" return x @@ -1659,7 +1670,7 @@ class KeyTemplate: from_str_funcs: A dictionary of field names and their functions to convert them from strings. - Examples: + .. rubric:: Examples >>> st = KeyTemplate( ... 'root/{name}/v_{version}.json', @@ -1720,10 +1731,11 @@ class KeyTemplate: >>> store['root/i2/v_4.json'] '{"downloads": 274, "type": "utility"}' - Note: If your store contains keys that don't fit the format, key_codec will - raise a ``ValueError``. To remedy this, you can use the ``st.filt_iter`` to - filter out keys that don't fit the format, before you wrap the store with - ``st.key_codec``. + Note: + If your store contains keys that don't fit the format, key_codec will + raise a ``ValueError``. To remedy this, you can use the ``st.filt_iter`` to + filter out keys that don't fit the format, before you wrap the store with + ``st.key_codec``. >>> store = { ... 'root/meshed/v_151.json': '{"downloads": 41, "type": "productivity"}', @@ -1746,7 +1758,6 @@ class KeyTemplate: {'name': 'i2', 'version': 96} >>> key_codec.decoder({'name': 'fantastic', 'version': 4}) ('fantastic', 4) - """ _formatter = string_formatter @@ -1843,11 +1854,11 @@ def key_codec( >>> store['root/i2/v_4.json'] '{"downloads": 274, "type": "utility"}' - Note: If your store contains keys that don't fit the format, key_codec will - raise a ``ValueError``. To remedy this, you can use the ``st.filt_iter`` to - filter out keys that don't fit the format, before you wrap the store with - ``st.key_codec``. - + Note: + If your store contains keys that don't fit the format, key_codec will + raise a ``ValueError``. To remedy this, you can use the ``st.filt_iter`` to + filter out keys that don't fit the format, before you wrap the store with + ``st.key_codec``. """ self._assert_field_type(decoded, "decoded") self._assert_field_type(encoded, "encoded") @@ -1869,7 +1880,6 @@ def filt_iter(self, field_type: FieldTypeNames = "str"): >>> filtered_store = filt.filt_iter('str')(store) >>> list(filtered_store) ['root/meshed/v_151.json', 'root/dol/v_9.json'] - """ if isinstance(field_type, Mapping): # The user wants to filter a store with the default @@ -1888,7 +1898,6 @@ def str_to_dict(self, s: str) -> dict: ... ) >>> st.str_to_dict('root/life/v_30.json') {'i01_': 'life', 'ver': 30} - """ if s is None: return None @@ -1907,7 +1916,6 @@ def dict_to_str(self, params: dict) -> str: ... ) >>> st.dict_to_str({'i01_': 'life', 'ver': 42}) 'root/life/v_042.json' - """ if params is None: return None @@ -1923,7 +1931,6 @@ def dict_to_tuple(self, params: dict) -> tuple: ... ) >>> st.str_to_tuple('root/life/v_42.json') ('life', 42) - """ if params is None: return None @@ -2140,17 +2147,17 @@ def _extract_template_info(self, template): r"""Extracts information from the template. Namely: - normalized_template: A template where each placeholder has a field name - (if not given, dflt_field_name will be used, which by default is - 'i{:02.0f}_'.format) + (if not given, dflt_field_name will be used, which by default is + 'i{:02.0f}_'.format) - field_names: The tuple of field names in the order they appear in template - to_str_funcs: A dict of field names and their corresponding to_str functions, - which will be used to convert the field values to strings when generating a - string. + which will be used to convert the field values to strings when generating a + string. - - field_patterns_: A dict of field names and their corresponding regex patterns, - which will be used to extract the field values from a string. + - ``field_patterns_``: A dict of field names and their corresponding regex patterns, + which will be used to extract the field values from a string. These four values are used in the init to compute the parameters of the instance. @@ -2169,7 +2176,6 @@ def _extract_template_info(self, template): '003' >>> to_str_funcs['name']('life') 'life' - """ field_names = [] diff --git a/dol/scrap/store_factories.py b/dol/scrap/store_factories.py index 203c6ce2..d63b6a49 100644 --- a/dol/scrap/store_factories.py +++ b/dol/scrap/store_factories.py @@ -24,9 +24,11 @@ def count_by_iteration(collection: Collection) -> int: """ Number of elements in collection of keys. - Note: This method iterates over all elements of the collection and counts them. - Therefore it is not efficient, and in most cases should be overridden with a more - efficient method. + + Note: + This method iterates over all elements of the collection and counts them. + Therefore it is not efficient, and in most cases should be overridden with a more + efficient method. """ count = 0 for _ in iter(collection): @@ -39,8 +41,11 @@ def count_by_iteration(collection: Collection) -> int: def check_by_iteration(collection: Collection[KT], x: KT) -> bool: """ Check if collection of keys contains k. - Note: Method loops through contents of collection to see if query element exists. - Therefore it may not be efficient, and in most cases, a method specific to the case should be used. + + Note: + Method loops through contents of collection to see if query element exists. + Therefore it may not be efficient, and in most cases, a method specific to the case should be used. + :return: True if k is in the collection, and False if not """ for existing_x in iter(collection): @@ -52,9 +57,12 @@ def check_by_iteration(collection: Collection[KT], x: KT) -> bool: def check_by_trying_to_get(mapping: Mapping, x: KT, false_on_error=(KeyError,)) -> bool: """ Check if mapping contains x. - Note: This method tries to get x from the mapping, returning ``False`` if it fails. - Therefore it may not be efficient, and in most cases, - a method specific to the case should be used. + + Note: + This method tries to get x from the mapping, returning ``False`` if it fails. + Therefore it may not be efficient, and in most cases, + a method specific to the case should be used. + :return: True if x is in the mapping, and False if not """ try: @@ -100,8 +108,6 @@ class KvReaderShell(KvReader): True >>> s['__add__']((4, 5)) (1, 2, 3, 4, 5) - - """ src: Any diff --git a/dol/signatures.py b/dol/signatures.py index 887b1801..3c2286d4 100644 --- a/dol/signatures.py +++ b/dol/signatures.py @@ -9,14 +9,15 @@ - merge two or more signatures - + - give a function a specific signature (with a choice of validations) - get an equivalent function with a different order of arguments - get an equivalent function with a subset of arguments (like partial) - - get an equivalent function but with variadic *args and/or **kwargs replaced with - non-variadic args (tuple) and kwargs (dict) + - get an equivalent function but with variadic ``*args`` and/or ``**kwargs`` replaced with + non-variadic args (tuple) and kwargs (dict) - make an f(a) function in to a f(a, b=None) function with b ignored @@ -79,7 +80,6 @@ - PO = Parameter.POSITIONAL_ONLY - KO = Parameter.KEYWORD_ONLY - """ from inspect import Signature, Parameter, signature, unwrap @@ -214,7 +214,6 @@ def validate_signature(func: Callable) -> Callable: ... i2.signatures.InvalidSignature: Invalid signature for function : non-default argument follows default a rgument - """ try: Sig(func) # to get errors if the signature is not valid @@ -311,7 +310,6 @@ def name_of_obj( >>> alt = partial(name_of_obj, base_name_of_obj=attrgetter('__qualname__')) >>> alt(Signature.replace) 'Signature.replace' - """ try: return base_name_of_obj(o) @@ -436,6 +434,7 @@ def _add_optional_keywords(sig, kwarg_and_defaults, kwarg_annotations=None): '(x, y, *, z=3, verbose: bool = False)' Note: + - Annotations for the additional keywords are optional. - All additional keywords are added as keyword-only arguments. """ @@ -498,7 +497,6 @@ def ensure_params(obj: ParamsAble = None): >>> ensure_params() # equivalent to ensure_params(None) [] - """ # obj = inspect.unwrap(obj, stop=(lambda f: hasattr(f, "__signature__"))) @@ -578,8 +576,9 @@ def extract_arguments( dirty details. Returns an (param_args, param_kwargs, remaining_kwargs) tuple where + - param_args are the values of kwargs that are PO (POSITION_ONLY) as defined by - params, + params, - param_kwargs are those names that are both in params and not in param_args, and - remaining_kwargs are the remaining. @@ -588,16 +587,19 @@ def extract_arguments( but you have a kwargs dict of arguments in your hand. You can't just to `func( **kwargs)`. But you can (now) do - ``` - args, kwargs, remaining = extract_arguments(kwargs, func) # extract from kwargs - what you need for func - # ... check if remaing is empty (or not, depending on your paranoia), and then - call the func: - func(*args, **kwargs) - ``` + + .. code-block:: python + + # extract from kwargs what you need for func + args, kwargs, remaining = extract_arguments(kwargs, func) + # ... check if remaining is empty (or not, depending on your paranoia), + # and then call the func: + func(*args, **kwargs) + (And if you doing that a lot: Do put it in a decorator!) - See Also: extract_arguments.without_remainding + See Also: + extract_arguments.without_remainding The most frequent case you'll encounter is when there's no POSITION_ONLY args, your param_args will be empty @@ -642,9 +644,11 @@ def extract_arguments( This is because we don't want to assume that all the kwargs can actually be included in a call to the function behind the params. Instead, the user can chose whether to include the remainder by doing a: - ``` - param_kwargs.update(remaining_kwargs) - ``` + + .. code-block:: python + + param_kwargs.update(remaining_kwargs) + et voilà. That said, we do understand that it may be a common pattern, so we'll do that @@ -668,10 +672,12 @@ def extract_arguments( If you're expecting no remainder you might want to just get the args and kwargs ( not this third expected-to-be-empty remainder). You have two ways to do that, specifying: - `what_to_do_with_remainding='ignore'`, which will just return the (args, - kwargs) pair - `what_to_do_with_remainding='assert_empty'`, which will do the same, but first - assert the remainder is empty + + - `what_to_do_with_remainding='ignore'`, which will just return the (args, + kwargs) pair + - `what_to_do_with_remainding='assert_empty'`, which will do the same, but first + assert the remainder is empty + We suggest to use `functools.partial` to configure the `argument_argument` you need. >>> from functools import partial @@ -731,8 +737,11 @@ def extract_arguments( 'ignore': function will return `param_args`, `param_kwargs` 'assert_empty': function will assert that `remaining_kwargs` is empty and then return `param_args`, `param_kwargs` - :param include_all_when_var_keywords_in_params=False, - :param assert_no_missing_position_only_args=False, + :param include_all_when_var_keywords_in_params: If True and `params` has a + VAR_KEYWORD parameter, the remaining kwargs are merged into `param_kwargs` + (leaving an empty remainder). + :param assert_no_missing_position_only_args: If True, assert that no + position-only argument is missing from `kwargs`. :param kwargs: The kwargs to extract the args from :return: A (param_args, param_kwargs, remaining_kwargs) tuple. """ @@ -935,14 +944,15 @@ def flatten_if_var_kw(kvs, var_kw_name): class Sig(Signature, Mapping): """A subclass of inspect.Signature that has a lot of extra api sugar, such as - - making a signature for a variety of input types (callable, - iterable of callables, parameter lists, strings, etc.) - - has a dict-like interface - - signature merging (with operator interfaces) - - quick access to signature data - - positional/keyword argument mapping. - # Positional/Keyword argument mapping + - making a signature for a variety of input types (callable, + iterable of callables, parameter lists, strings, etc.) + - has a dict-like interface + - signature merging (with operator interfaces) + - quick access to signature data + - positional/keyword argument mapping. + + .. rubric:: Positional/Keyword argument mapping In python, arguments can be positional (args) or keyword (kwargs). ... sometimes both, sometimes a single one is imposed. @@ -958,12 +968,13 @@ class Sig(Signature, Mapping): Two of the base methods for dealing with positional (args) and keyword (kwargs) inputs are: + - `map_arguments`: Map some args/kwargs input to a keyword-only - expression of the inputs. This is useful if you need to do some processing - based on the argument names. + expression of the inputs. This is useful if you need to do some processing + based on the argument names. - `mk_args_and_kwargs`: Translate a fully keyword expression of some - inputs into an (args, kwargs) pair that can be used to call the function. - (Remember, your function can have constraints, so you may need to do this. + inputs into an (args, kwargs) pair that can be used to call the function. + (Remember, your function can have constraints, so you may need to do this. The usual pattern of use of these methods is to use `map_arguments` to map all the inputs to their corresponding name, do what needs to be done with @@ -974,7 +985,7 @@ class Sig(Signature, Mapping): `call_forgivingly`, `tuple_the_args`, `map_arguments_from_variadics`, `extract_args_and_kwargs`, `source_arguments`, and `source_args_and_kwargs`. - # Making a signature + .. rubric:: Making a signature You can construct a `Sig` object from a callable, @@ -1063,7 +1074,6 @@ class Sig(Signature, Mapping): ... ... >>> inspect.signature(some_func) - """ # Adding parameter kinds as class attributes for usage convenience @@ -1082,16 +1092,19 @@ def __init__( __validate_parameters__=True, ): """Initialize a Sig instance. - See Also: `ensure_params` to see what kind of objects you can make `Sig`s with. + + See Also: + `ensure_params` to see what kind of objects you can make `Sig`s with. :param obj: A ParamsAble object, which could be: + - a callable, - and iterable of Parameter instances - an iterable of strings (representing annotation-less, default-less) - argument names, + argument names, - tuples: (argname, default) or (argname, default, annotation), - dicts: ``{'name': REQUIRED,...}`` with optional `kind`, `default` and - `annotation` fields + `annotation` fields - None (which will produce an argument-less Signature) >>> Sig(["a", "b", "c"]) @@ -1185,17 +1198,19 @@ def wrap( signature (not a callable). Also, where as both write to the input func's `__signature__` attribute, here we also write to + - `__defaults__` and `__kwdefaults__`, extracting these from `__signature__` - (functools.wraps doesn't do that at the time of writing this - (see https://github.com/python/cpython/pull/21379)). + (functools.wraps doesn't do that at the time of writing this + (see https://github.com/python/cpython/pull/21379)). - `__annotations__` (also extracted from `__signature__`) - does not write to `__module__`, `__name__`, `__qualname__`, `__doc__` - (because again, we're basinig the injecton on a signature, not a function, - so we have no name, doc, etc...) + (because again, we're basinig the injecton on a signature, not a function, + so we have no name, doc, etc...) - WARNING: The fact that you've modified the signature of your function doesn't - mean that the decorated function will work as expected (or even work at all). - See below for examples. + WARNING: + The fact that you've modified the signature of your function doesn't + mean that the decorated function will work as expected (or even work at all). + See below for examples. >>> def f(w, /, x: float = 1, y=2, z: int = 3): ... return w + x * y ** z @@ -1206,6 +1221,7 @@ def wrap( >>> assert 8 == f(0) == f(0, 1) == f(0, 1, 2) == f(0, 1, 2, 3) Now let's create a very similar function to f, but where: + - w is not position-only - x annot is int instead of float, and doesn't have a default - z's default changes to 10 @@ -1254,9 +1270,8 @@ def wrap( Traceback (most recent call last): ... TypeError: f() takes from 0 to 3 positional arguments but 4 were given - - TODO: Give more explanations why this is. """ + # TODO: Give more explanations why this is. # TODO: Should we make copy_function=False the default, # so as to not override decorated function itself by default? @@ -1325,7 +1340,6 @@ def sig_or_default(cls, obj, default_signature=DFLT_SIGNATURE): >>> str(Sig.sig_or_default(time.time, Sig(lambda: ...))) '()' - """ try: # (try to) return cls(obj) if obj is callable: @@ -1365,7 +1379,6 @@ def sig_or_none(cls, obj): >>> robust_has_signature(print) True - """ return cls.sig_or_default(obj, default_signature=None) @@ -1425,16 +1438,17 @@ def to_signature_kwargs(self): 'return_annotation': } Note that this does NOT return: - ``` - {'parameters': self.parameters, - 'return_annotation': self.return_annotation} - ``` + + .. code-block:: python + + {'parameters': self.parameters, + 'return_annotation': self.return_annotation} + which would not actually work as keyword arguments of ``Signature``. Yeah, I know. Don't ask me, ask the authors of `Signature`! Instead, `parammeters` will be ``list(self.parameters.values())``, which does work. - """ return { "parameters": list(self.parameters.values()), @@ -1449,7 +1463,6 @@ def to_simple_signature(self): ... ... >>> Sig(f).to_simple_signature() - """ return Signature(**self.to_signature_kwargs()) @@ -1609,7 +1622,6 @@ def get_names(self, spec, *, conserve_sig_order=True, allow_excess=False): >>> sig.get_names(['a', 'c', 'e', 'g', 'h'], allow_excess=True) ('a', 'c', 'e') - """ if isinstance(spec, str): spec = spec.split() @@ -1688,7 +1700,8 @@ def names_for_kind(self, kind): @property def has_var_kinds(self): - """ + """Whether the signature has a VAR_POSITIONAL or a VAR_KEYWORD parameter. + >>> Sig(lambda x, *, y: None).has_var_kinds False >>> Sig(lambda x, *y: None).has_var_kinds @@ -1737,7 +1750,6 @@ def index_of_var_keyword(self): And if there's none... >>> assert Sig(lambda a, *args, b=1: 0).index_of_var_keyword is None - """ last_arg_idx = len(self) - 1 if last_arg_idx != -1: @@ -1785,11 +1797,13 @@ def required_names(self): def n_required(self): """The number of required arguments. A required argument is one that doesn't have a default, nor is VAR_POSITIONAL - (*args) or VAR_KEYWORD (**kwargs). - Note: Sometimes a minimum number of arguments in VAR_POSITIONAL and - VAR_KEYWORD are in fact required, - but we can't see this from the signature, so we can't tell you about that! You - do the math. + (``*args``) or VAR_KEYWORD (``**kwargs``). + + Note: + Sometimes a minimum number of arguments in VAR_POSITIONAL and + VAR_KEYWORD are in fact required, + but we can't see this from the signature, so we can't tell you about that! You + do the math. >>> f = lambda a00, /, a11, a12, *a23, a34, a35=1, a36='two', **a47: None >>> Sig(f).n_required @@ -1819,8 +1833,9 @@ def _transform_params(self, changes_for_name: dict): def modified(self, /, _allow_reordering=False, **changes_for_name): """Returns a modified (new) signature object. - Note: This function doesn't modify the signature, but creates a modified copy - of the signature. + Note: + This function doesn't modify the signature, but creates a modified copy + of the signature. IMPORTANT WARNING: This is an advanced feature. Avoid wrapping a function with a modified signature, as this may not have the intended effect. @@ -1832,6 +1847,7 @@ def modified(self, /, _allow_reordering=False, **changes_for_name): >>> assert sig.kinds['pka'] == PK Let's make a signature that is the same as sig, except that + - `poa` is given a PO (POSITIONAL_ONLY) kind insteadk of PK - `koa` is given a default of None - the signature is given a return_annotation of str @@ -1859,7 +1875,6 @@ def modified(self, /, _allow_reordering=False, **changes_for_name): On the other hand, if you decorate a function with a sig that adds or modifies defaults, these defaults will actually be used (unlike with `functools.wraps`). - """ new_return_annotation = changes_for_name.pop( "return_annotation", self.return_annotation @@ -1911,17 +1926,10 @@ def ch_param_attrs( >>> special_foo(5) # should be 5 + 2 * 3 == 11 11 - + """ # TODO: Would like to make this work (reordering) # Now, if you want to set a default for a but not b and c for example, you'll - # get complaints: - # - # ``` - # ValueError: non-default argument follows default argument - # ``` - # - # will tell you. - # + # get complaints (ValueError: non-default argument follows default argument). # It's true. But if you're fine with rearranging the argument order, # `ch_param_attrs` can take care of that for you. # You'll have to tell it explicitly that you wish for this though, because @@ -1940,8 +1948,6 @@ def ch_param_attrs( # >>> another_foo(2, 3) # should be 10 + (2 * 3) = # 16 - """ - if not param_attr in param_attributes: raise ValueError( f"param_attr needs to be one of: {param_attributes}.", @@ -2010,7 +2016,6 @@ def add_optional_keywords( >>> str(Sig(foo)) '(a, *, c: int = 2, d=3, b=1, **kwargs)' - """ # Resolve arguments ( to be able to use this method as a decorator) @@ -2049,7 +2054,7 @@ def merge_with_sig( :param sig: The signature to merge with. :param ch_to_all_pk: Whether to change all kinds of both signatures to PK ( - POSITIONAL_OR_KEYWORD) + POSITIONAL_OR_KEYWORD) :return: >>> def func(a=None, *, b=1, c=2): @@ -2081,7 +2086,6 @@ def merge_with_sig( >>> s.merge_with_sig([("d", 3), ("e", 4)], ch_to_all_pk=True) - """ if ch_to_all_pk: _self = Sig(all_pk_signature(self)) @@ -2187,15 +2191,13 @@ def __add__(self, sig: ParamsAble): before hand) Important Notes: + - The resulting Sig will loose it's return_annotation if it had one. - This is to avoid making too many assumptions about how the sig sum will be - used. - If a return_annotation is needed (say, for composition, the last - return_annotation - summed), one can subclass Sig and overwrite __add__ + This is to avoid making too many assumptions about how the sig sum will be + used. If a return_annotation is needed (say, for composition, the last + return_annotation summed), one can subclass Sig and overwrite __add__ - POSITION_ONLY and KEYWORD_ONLY kinds will be replaced by - POSITIONAL_OR_KEYWORD kind. - This is to simplify the interface and code. + POSITIONAL_OR_KEYWORD kind. This is to simplify the interface and code. If the user really wants to maintain those kinds, they can replace them back after the fact. @@ -2366,7 +2368,6 @@ def _chain_params_of_signatures(*sigs): ... ) ... ) '[, , , , ]' - """ already_merged_names = set() for s in sigs: @@ -2467,16 +2468,17 @@ def map_arguments( will extract their needs from it. That's where `Sig.map_arguments_from_variadics(*args, **kwargs)` is needed. + :param args: The args the function will be called with. :param kwargs: The kwargs the function will be called with. :param apply_defaults: (bool) Whether to apply signature defaults to the - non-specified argument names + non-specified argument names :param allow_partial: (bool) True iff you want to allow partial signature - fulfillment. + fulfillment. :param allow_excess: (bool) Set to True iff you want to allow extra kwargs - items to be ignored. + items to be ignored. :param ignore_kind: (bool) Set to True iff you want to ignore the position and - keyword only kinds, + keyword only kinds, in order to be able to accept args and kwargs in such a way that there can be cross-over (args that are supposed to be keyword only, and kwargs that are supposed @@ -2618,19 +2620,20 @@ def mk_args_and_kwargs( ignore_kind=False, args_limit: int | None = 0, ) -> tuple[tuple, dict]: - """Extract args and kwargs such that func(*args, **kwargs) can be called, + """Extract args and kwargs such that ``func(*args, **kwargs)`` can be called, where func has instance's signature. :param arguments: The {param_name: arg_val,...} dict to process :param args_limit: How "far" in the params should args (positional arguments) be searched for. + - args_limit==0: Take the minimum number possible of args (positional - arguments). Only those that are position only or before a var-positional. + arguments). Only those that are position only or before a var-positional. - args_limit is None: Take the maximum number of args (positional arguments). - The only kwargs (keyword arguments) you should have are keyword-only - and var-keyword arguments. + The only kwargs (keyword arguments) you should have are keyword-only + and var-keyword arguments. - args_limit positive integer: Take the args_limit first argument names - (of signature) as args, and the rest as kwargs. + (of signature) as args, and the rest as kwargs. >>> def foo(w, /, x: float, y=1, *, z: int = 1): ... return ((w + x) * y) ** z @@ -2660,11 +2663,12 @@ def mk_args_and_kwargs( The `args_limit` begs explanation. Consider the signature of `def foo(w, /, x: float, y=1, *, z: int = 1): ...` for instance. We could call the function with the following (args, kwargs) pairs: + - ((1,), {'x': 2, 'y': 3, 'z': 4}) - ((1, 2), {'y': 3, 'z': 4}) - ((1, 2, 3), {'z': 4}) - The two other combinations (empty args or empty kwargs) are not valid - because of the / and * constraints. + The two other combinations (empty args or empty kwargs) are not valid + because of the / and * constraints. But when asked for an (args, kwargs) pair, which of the three valid options should be returned? This is what the `args_limit` argument controls. @@ -3063,8 +3067,6 @@ def source_arguments( ... 4, x=3, y=2, extra="keywords", are="ignored", _apply_defaults=True ... ) {'w': 4, 'x': 3, 'y': 2, 'z': 'ZZ'} - - """ return self.map_arguments( args, @@ -3194,7 +3196,6 @@ def inject_into_keyword_variadic(self): >>> Sig(sauce) - """ return replace_kwargs_using(self) @@ -3208,7 +3209,6 @@ def _fill_defaults_and_annotations(sig1: Sig, sig2: Sig): ... Sig('(a, /, b: str, *, c=3)'), Sig('(a: float, b: int = 2, c=300)') ... ) - """ def filled_properties_of_sig1(): @@ -3304,6 +3304,7 @@ def mk_sig_from_args(*args_without_default, **args_with_defaults): def _remove_variadics_from_sig(sig, ch_variadic_keyword_to_keyword=True): """Remove variadics from signature + >>> def foo(a, *args, bar, **kwargs): ... return f"{a=}, {args=}, {bar=}, {kwargs=}" >>> sig = Sig(foo) @@ -3330,7 +3331,7 @@ def _remove_variadics_from_sig(sig, ch_variadic_keyword_to_keyword=True): If you only want the variadic positional to be handled, but leave leave any - VARIADIC_KEYWORD kinds (**kwargs) alone, you can do so by setting + VARIADIC_KEYWORD kinds (``**kwargs``) alone, you can do so by setting `ch_variadic_keyword_to_keyword=False`. >>> def foo(a, *args, bar=None, **kwargs): @@ -3369,10 +3370,11 @@ def call_forgivingly(func, *args, **kwargs): Call function on given args and kwargs, but only taking what the function needs (not choking if they're extras variables) - Tip: If you into trouble because your kwargs has a 'func' key, - (which would then clash with the ``func`` param of call_forgivingly), then - use `_call_forgivingly` instead, specifying args and kwargs as tuple and - dict. + Tip: + If you into trouble because your kwargs has a 'func' key, + (which would then clash with the ``func`` param of call_forgivingly), then + use `_call_forgivingly` instead, specifying args and kwargs as tuple and + dict. >>> def foo(a, b: int = 0, c=None) -> int: ... return "foo", (a, b, c) @@ -3390,20 +3392,16 @@ def call_forgivingly(func, *args, **kwargs): ... return x, args1, y, kwargs1 >>> call_forgivingly(bar, 1, 2, 3, y=4, z=5) (1, (2, 3), 4, {'z': 5}) - + """ + # TODO: Examples that don't yet behave as one might want: # >>> def bar(x, y=1, **kwargs1): # ... return x, y, kwargs1 # >>> call_forgivingly(bar, 1, 2, 3, y=4, z=5) # (1, 4, {'z': 5}) - - # >>> call_forgivingly(bar, 1, 2, 3, y=4, z=5) - # >>> def bar(x, *args1, y=1): # ... return x, args1, y # >>> call_forgivingly(bar, 1, 2, 3, y=4, z=5) # (1, (2, 3), {'z': 5}) - - """ return _call_forgivingly(func, args, kwargs) @@ -3499,7 +3497,6 @@ def call_somewhat_forgivingly( Traceback (most recent call last): ... TypeError: got an unexpected keyword argument 'this_argname' - """ enforce_sig = Sig(enforce_sig or func) # Validate that args and kwargs are compatible with enforce_sig @@ -3534,8 +3531,8 @@ def kind_forgiving_func(func, kinds_modifier=convert_to_PK): >>> isinstance_of_str(42) False - See also: ``i2.signatures.all_pk_signature`` - + See also: + ``i2.signatures.all_pk_signature`` """ sig = Sig(func) kinds_modif = kinds_modifier(sig.kinds) @@ -3566,10 +3563,12 @@ def use_interface(interface_sig): `g` should do, but doesn't use `a`, and doesn't even have it in it's arguments. The solution to this is to _adapt_ `f` to the `g` interface: - ``` - def my_g(a, b): - return f(a) - ``` + + .. code-block:: python + + def my_g(a, b): + return f(a) + and use `my_g`. >>> f = lambda a: a * 11 @@ -3637,7 +3636,6 @@ def has_signature(obj, robust=False): If robust is set to True, `has_signature` will use `Sig` to get the signature, so will return True in most cases. - """ if robust: return bool(Sig.sig_or_none(obj)) @@ -3671,7 +3669,7 @@ def all_pk_signature(callable_or_signature: Callable | Signature): >>> all_pk_signature(signature(foo)) - But note that the variadic arguments *args and **kwargs remain variadic: + But note that the variadic arguments ``*args`` and ``**kwargs`` remain variadic: >>> all_pk_signature(signature(bar)) @@ -3685,8 +3683,8 @@ def all_pk_signature(callable_or_signature: Callable | Signature): >>> sig.name 'bar' - See also: ``i2.signatures.kind_forgiving_func`` - + See also: + ``i2.signatures.kind_forgiving_func`` """ if isinstance(callable_or_signature, Signature): @@ -3759,13 +3757,13 @@ def _func(*args, **kwargs): def ch_variadics_to_non_variadic_kind(func, *, ch_variadic_keyword_to_keyword=True): - """A decorator that will change a VAR_POSITIONAL (*args) argument to a tuple (args) + """A decorator that will change a VAR_POSITIONAL (``*args``) argument to a tuple (args) argument of the same name. Essentially, given a `func(a, *b, c, **d)` function want to get a `new_func(a, b=(), c=None, d={})` that has the same functionality (in fact, calls the original `func` function behind the scenes), but without - where the variadic arguments *b and **d are replaced with a `b` expecting an + where the variadic arguments ``*b`` and ``**d`` are replaced with a `b` expecting an iterable (e.g. tuple/list) and `d` expecting a `dict` to contain the desired inputs. @@ -3817,7 +3815,7 @@ def ch_variadics_to_non_variadic_kind(func, *, ch_variadic_keyword_to_keyword=Tr True If you only want the variadic positional to be handled, but leave leave any - VARIADIC_KEYWORD kinds (**kwargs) alone, you can do so by setting + VARIADIC_KEYWORD kinds (``**kwargs``) alone, you can do so by setting `ch_variadic_keyword_to_keyword=False`. If you'll need to use `ch_variadics_to_non_variadic_kind` in such a way repeatedly, we suggest you use `functools.partial` to not have to specify this @@ -3834,10 +3832,6 @@ def ch_variadics_to_non_variadic_kind(func, *, ch_variadic_keyword_to_keyword=Tr >>> foo(1, (2, 3), bar=4, hello="world") "a=1, args=(2, 3), bar=4, kwargs={'hello': 'world'}" - - - - """ if func is None: return partial( @@ -3961,7 +3955,8 @@ def ch_func_to_all_pk(func): >>> gg = ch_func_to_all_pk(g) >>> print(Sig(gg)) (x, y=1, args=(), **kwargs) - + """ + # TODO: Make this work (currently raises "non-default argument follows default argument"): # >>> def h(x, *y, z): # ... print(f"{x=}, {y=}, {z=}") # >>> h(1, 2, 3, z=4) @@ -3969,7 +3964,6 @@ def ch_func_to_all_pk(func): # >>> hh = ch_func_to_all_pk(h) # >>> hh(1, (2, 3), z=4) # x=1, y=(2, 3), z=4 - """ # _func = tuple_the_args(func) # sig = Sig(_func) @@ -4069,7 +4063,8 @@ def common_and_diff_argnames(func1: callable, func2: callable) -> dict: func1: First function func2: Second function - Returns: A dict with fields 'common', 'func1_not_func2', and 'func2_not_func1' + Returns: + A dict with fields 'common', 'func1_not_func2', and 'func2_not_func1' >>> def f(t, h, i, n, k): ... ... @@ -4109,6 +4104,7 @@ def set_signature_of_func( signature: A list of parameter specifications. This could be an inspect.Parameter object or anything that the mk_param function can resolve into an inspect.Parameter object. + return_annotation: Passed on to inspect.Signature. __validate_parameters__: Passed on to inspect.Signature. @@ -4142,7 +4138,6 @@ def set_signature_of_func( ... ) >>> inspect.signature(foo) str> - """ sig = Sig( parameters, @@ -4186,10 +4181,10 @@ def sig_to_dataclass( >>> K = sig_to_dataclass(Sig(foo), cls_name='K') >>> K = sig_to_dataclass(Sig(foo).params, cls_name='K') - Note: ``cls_name`` is not required (we'll try to figure out a good default for you), - but it's advised to only use this convenience in extreme mode. - Choosing your own name might make for a safer future if you're reusing your class. - + Note: + ``cls_name`` is not required (we'll try to figure out a good default for you), + but it's advised to only use this convenience in extreme mode. + Choosing your own name might make for a safer future if you're reusing your class. """ from dataclasses import make_dataclass @@ -4257,7 +4252,6 @@ def replace_kwargs_using(sig: SignatureAble): ... return b * c + apple(a, **sauce_kwargs) >>> Sig(sauce) - """ def decorator(targ_func): @@ -4327,7 +4321,6 @@ def _robust_signature_of_callable(callable_obj: Callable) -> Signature: ... slice ... ) # doesn't have one, so will return a blanket one - """ # First check if we have a custom signature for this type/object # This is important for operator instances that might have generic signatures in Python 3.12+ @@ -4393,7 +4386,6 @@ def resolve_function(obj: T) -> T | Callable: True >>> isinstance(inspect.getsource(resolve_function(C.partial_func)), str) True - """ if isinstance(obj, cached_property): return obj.func @@ -4446,14 +4438,14 @@ def filter(function, iterable, /): """filter(function or None, iterable) --> filter object""" def map(func, iterable, /, *iterables): - """map(func, *iterables) --> map object""" + """``map(func, *iterables) --> map object``""" def print(*value, sep=" ", end="\n", file=sys.stdout, flush=False): - """print(value, ..., sep=' ', end='\n', file=sys.stdout, flush=False)""" + r"""``print(value, ..., sep=' ', end='\n', file=sys.stdout, flush=False)``""" def zip(*iterables): """ - zip(*iterables) --> A zip object yielding tuples until an input is exhausted. + ``zip(*iterables)`` --> A zip object yielding tuples until an input is exhausted. """ def bool(x: Any, /) -> bool: ... @@ -4832,7 +4824,6 @@ def param_comparator( True See https://github.com/i2mint/i2/issues/50#issuecomment-1381686812 for discussion. - """ return aggreg( ( @@ -4860,7 +4851,6 @@ def dflt1_is_empty_or_dflt2_is_not(dflt1, dflt2): without specifying p, but func2 couldn't. So to avoid this situation, we use dflt1_is_empty_or_dflt2_is_not as the default - """ return dflt1 is empty or dflt2 is not empty @@ -4960,14 +4950,14 @@ def defaults_are_the_same_when_not_empty(dflt1, dflt2): """ Check if two defaults are the same when they are not empty. - # >>> defaults_are_the_same_when_not_empty(1, 1) - # True - # >>> defaults_are_the_same_when_not_empty(1, 2) - # False - # >>> defaults_are_the_same_when_not_empty(1, None) - # False - # >>> defaults_are_the_same_when_not_empty(1, Parameter.empty) - # True + >>> defaults_are_the_same_when_not_empty(1, 1) + True + >>> defaults_are_the_same_when_not_empty(1, 2) + False + >>> defaults_are_the_same_when_not_empty(1, None) + False + >>> defaults_are_the_same_when_not_empty(1, Parameter.empty) + True """ return dflt1 is empty or dflt2 is empty or dflt1 == dflt2 @@ -5232,7 +5222,6 @@ class SigPair: >>> SigPair(three, pigs).are_call_compatible() False - """ sig1: Callable | Sig diff --git a/dol/sources.py b/dol/sources.py index f7f39324..6bc31e3b 100644 --- a/dol/sources.py +++ b/dol/sources.py @@ -1,5 +1,22 @@ -""" -This module contains key-value views of disparate sources. +"""Key-value views of disparate sources. + +Readers and persisters over things that are not stores to begin with: several stores +at once (fan-out and cascades), sequences, functions, and the attributes of objects. + +Main entry points: + +- ``FanoutReader``, ``FanoutPersister``: one key, read from (written to) several stores +- ``CascadedStores``: write to all stores, read from the first one that has the key +- ``SequenceKvReader``: an iterable of elements, keyed by a key function (index by default) +- ``FuncReader``: functions as a store, keyed by name +- ``Attrs``: the attributes of an object as a (recursive) reader + + >>> from dol.sources import FuncReader + >>> def foo(): + ... return 'bar' + >>> r = FuncReader([foo]) + >>> list(r), r['foo'] + (['foo'], 'bar') """ from typing import Union, Any @@ -124,9 +141,9 @@ class FanoutReader(KvReader): That is, when a key is requested, the key is passed to all the stores, and results accumulated in a dict that is then returned. - param stores: A mapping of store keys to stores. - param default: The value to return if the key is not in any of the stores. - param get_existing_values_only: If True, only return values for stores that contain + :param stores: A mapping of store keys to stores. + :param default: The value to return if the key is not in any of the stores. + :param get_existing_values_only: If True, only return values for stores that contain the key. Let's define the following sub-stores: @@ -202,9 +219,9 @@ def from_variadics(cls, *args, **kwargs): """A way to create a fan-out store from a mix of args and kwargs, instead of a single dict. - param args: sub-stores used to fan-out the data. These stores will be + :param args: sub-stores used to fan-out the data. These stores will be represented by their index in the tuple. - param kwargs: sub-stores used to fan-out the data. These stores will be + :param kwargs: sub-stores used to fan-out the data. These stores will be represented by their name in the dict. __init__ arguments can also be passed as kwargs (i.e. `default`, `get_existing_values_only`, and any other subclass specific arguments). @@ -295,13 +312,13 @@ class FanoutPersister(FanoutReader, KvPersister): """ A fanout persister is a fanout reader that can also set and delete items. - param stores: A mapping of store keys to stores. - param default: The value to return if the key is not in any of the stores. - param get_existing_values_only: If True, only return values for stores that contain + :param stores: A mapping of store keys to stores. + :param default: The value to return if the key is not in any of the stores. + :param get_existing_values_only: If True, only return values for stores that contain the key. - param need_to_set_all_stores: If True, all stores must be set when setting a value. + :param need_to_set_all_stores: If True, all stores must be set when setting a value. If False, only the stores that are set will be updated. - param ignore_non_existing_store_keys: If True, ignore store keys from the value that + :param ignore_non_existing_store_keys: If True, ignore store keys from the value that are not in the persister. If False, a ValueError is raised. Let's create a persister from in-memory stores: @@ -521,8 +538,6 @@ class CascadedStores(FanoutPersister): >>> remote {'f': 42} - - """ # Note: Need to overwrite FanoutPersister's getitem to not read values from all stores @@ -701,7 +716,6 @@ class SequenceKvReader(KvReader): Traceback (most recent call last): ... sources.NotUnique: iterator had more than one element - """ def __init__( @@ -800,7 +814,6 @@ class FuncReader(KvReader): ['FU', 'Pie'] >>> s['FU'] 'bar' - """ def __init__(self, funcs: Mapping[str, Callable] | Iterable[Callable]): @@ -910,7 +923,6 @@ class ObjReader: >>> 'therefore should contain what I just said' in pr[file_where_this_code_is] True - """ def __init__(self, _obj_of_key: Callable): @@ -940,8 +952,9 @@ class Attrs(ObjReader): """A simple recursive KvReader for the attributes of a python object. Keys are attr names, values are Attrs(attr_val) instances. - Note: A more significant version of Attrs, along with many tools based on it, - was moved to pypi package: guide. + Note: + A more significant version of Attrs, along with many tools based on it, + was moved to pypi package: guide. pip install guide diff --git a/dol/tests/base_test.py b/dol/tests/base_test.py index 966aa856..0a581c4a 100644 --- a/dol/tests/base_test.py +++ b/dol/tests/base_test.py @@ -149,7 +149,6 @@ def test_wrap_kvs_vs_class_and_static_methods(): See issue "dol.base.Store.wrap breaks unbound method calls": https://github.com/i2mint/dol/issues/17 - """ @Store.wrap diff --git a/dol/tests/test_interface_wrap.py b/dol/tests/test_interface_wrap.py index e52cd4c9..d8a0e823 100644 --- a/dol/tests/test_interface_wrap.py +++ b/dol/tests/test_interface_wrap.py @@ -443,7 +443,7 @@ def find(self, k): def test_var_positional_role_maps_elementwise(): - """*keys: KT must encode each element, not the tuple (was a silent bug).""" + """``*keys: KT`` must encode each element, not the tuple (was a silent bug).""" class VarSpec(Protocol[KT]): def delete(self, *keys: KT) -> None: ... diff --git a/dol/tools.py b/dol/tools.py index e262257f..a7c4ae1c 100644 --- a/dol/tools.py +++ b/dol/tools.py @@ -1,5 +1,24 @@ -""" -Various tools to add functionality to stores +"""Various tools to add functionality to stores. + +Main entry points: + +- ``store_aggregate``: aggregate a store's items into one object (a Markdown text by default) +- ``confirm_overwrite``: a ``wrap_kvs`` preset that asks before overwriting a value +- ``Forest``: a key-value tree view of nested objects + + >>> from dol.tools import store_aggregate + >>> print(store_aggregate({'a': 'x', 'b': 'y'})) + ## a + + x + + + + ## b + + y + + """ from dol.trans import store_decorator @@ -47,7 +66,6 @@ def confirm_overwrite( If you want to overwrite it with alligator, confirm by typing alligator here: And we'll have to type `alligator` and press RETURN to make the write go through. - """ if (existing_v := mapping.get(k, NoSuchKey)) is not NoSuchKey and existing_v != v: user_input = input(user_input_msg.format(k=k, v=v, existing_v=existing_v)) @@ -137,22 +155,22 @@ def store_aggregate( Args: content_store (Union[Mapping[KT, VT], str]): Path to the folder or dol store to read from. - kv_to_item (Callable[[KT, VT], Item]): - Function to convert key-value pairs to an Item (usually a string). - aggregator (Callable[[Iterable[Item]], Aggregate]): - The function that will aggregate the items that `kv_to_item` produces. + kv_to_item (Callable[[KT, VT], Item]): Function to convert key-value pairs to an Item (usually a string). + + aggregator (Callable[[Iterable[Item]], Aggregate]): The function that will aggregate the items that `kv_to_item` produces. Defaults to '\n\n'.join. - egress (Union[Callable[[Aggregate], Any], str]): - The function that will be called on the aggregate before returning it. + + egress (Union[Callable[[Aggregate], Any], str]): The function that will be called on the aggregate before returning it. Defaults to identity. Note that if you provide a string, the function will save the aggregate text to a file, assuming it is indeed text. - key_filter (Optional[Callable[[KT], bool]]): - Optional filter for keys. Defaults to None (no filtering). - value_filter (Optional[Callable[[VT], bool]]): - Optional filter for values. Defaults to None (no filtering). - kv_filter (Optional[Callable[[Tuple[KT, VT]], bool]]): - Optional filter for key-value pairs. Defaults to None (no filtering). + + key_filter (Optional[Callable[[KT], bool]]): Optional filter for keys. Defaults to None (no filtering). + + value_filter (Optional[Callable[[VT], bool]]): Optional filter for values. Defaults to None (no filtering). + + kv_filter (Optional[Callable[[Tuple[KT, VT]], bool]]): Optional filter for key-value pairs. Defaults to None (no filtering). + local_store_factory (Callable[[str], Mapping[KT, VT]]): Factory function for the local store, used only if `content_store` is an existing folder path. Defaults to Latin1TextFiles. @@ -500,7 +518,6 @@ class Forest(KvReader): True >>> list(fff) ['granny', 'fuji'] - """ def __init__( diff --git a/dol/trans.py b/dol/trans.py index efaed741..2395e21b 100644 --- a/dol/trans.py +++ b/dol/trans.py @@ -1,4 +1,26 @@ -"""Transformation/wrapping tools""" +"""Tools to wrap stores with key/value transforms, filters, caches and other layers. + +A wrap leaves the backend untouched and builds a new class (or instance) around it. +The decorators built with ``store_decorator`` (``wrap_kvs``, ``filt_iter``, +``cached_keys``, ``add_path_access``, ...) can be applied to a class, to an instance, +or used as a factory (``deco(**params)(store)``). + +Main entry points: + +- ``wrap_kvs``: key/value transforms (``key_of_id``, ``obj_of_data``, codecs, ...) +- ``filt_iter``: restrict a store to a subset of its keys +- ``cached_keys``: cache the key listing of a slow store +- ``add_path_access``: read/write nested stores through key paths +- ``kv_wrap``: wrap using an object that holds ``_id_of_key``/``_obj_of_data``-style methods + + >>> from dol.trans import wrap_kvs + >>> s = wrap_kvs({}, key_of_id=str.upper, id_of_key=str.lower, obj_of_data=int, data_of_obj=str) + >>> s['a'] = 1 + >>> s.store # the backend holds the transformed key and value + {'a': '1'} + >>> list(s), s['A'] + (['A'], 1) +""" from functools import wraps, partial, reduce import types @@ -46,9 +68,10 @@ def double_up_as_factory(decorator_func): That is, from a decorator that is defined do ``wrapped_func = decorator(func, **params)``, make it also be able to do ``wrapped_func = decorator(**params)(func)``. - Note: You'll only be able to do this if all but the first argument are keyword-only, - and the first argument (the function to decorate) has a default of ``None`` (this is for your own good). - This is validated before making the "double up as factory" decorator. + Note: + You'll only be able to do this if all but the first argument are keyword-only, + and the first argument (the function to decorate) has a default of ``None`` (this is for your own good). + This is validated before making the "double up as factory" decorator. >>> @double_up_as_factory ... def decorator(func=None, *, multiplier=2): @@ -92,7 +115,6 @@ def double_up_as_factory(decorator_func): Traceback (most recent call last): ... AssertionError: All arguments (besides the first) need to be keyword-only - """ def validate_decorator_func(decorator_func): @@ -143,6 +165,7 @@ def store_decorator(func): ``store_decorator`` takes that ``func`` and provides an enhanced class decorator specialized for stores. Namely it will: + - Add ``__module__``, ``__qualname__``, ``__name__`` and ``__doc__`` arguments to it - Copy the aforementioned arguments to the decorated class, or copy the attributes of the original if not specified. - Output a decorator that can be used in four different ways: a class/instance decorator/factory. @@ -150,6 +173,7 @@ def store_decorator(func): By class/instance decorator/factory we mean that if ``A`` is a class, ``a`` an instance of it, and ``deco`` a decorator obtained with ``store_decorator(func)``, we can use ``deco`` to + - class decorator: decorate a class - class decorator factory: make a function that decorates classes - instance decorator: decorate an instance of a store @@ -224,7 +248,8 @@ def store_decorator(func): >>> b = deco(a, x=42); assert b.x == 42 # b has an x and it's 42 >>> b = deco(x=42)(a); assert b.x == 42; # b has an x and it's 42 - WARNING: Note though that the type of ``b`` is not the same type as ``a`` + WARNING: + Note though that the type of ``b`` is not the same type as ``a`` >>> isinstance(b, a.__class__) False @@ -314,7 +339,6 @@ def store_decorator(func): >>> wd = remove_deletion(d, msg='No way. I do not trust you!!') >>> assert wd == d # same as far as dict comparison goes >>> assert wd.__delitem__('x') == 'No way. I do not trust you!!' - """ # wrapper_assignments = ('__module__', '__qualname__', '__name__', '__doc__', '__annotations__') @@ -387,12 +411,14 @@ def wrapper(store=None, **kwargs): def ensure_set(x): + """A set from ``x``, treating a string as a single element.""" if isinstance(x, str): x = [x] return set(x) def get_class_name(cls, dflt_name=None): + """The ``__qualname__`` of ``cls`` (or of its class), else ``dflt_name``; raises ``ValueError`` if there is neither.""" name = getattr(cls, "__qualname__", None) if name is None: name = getattr(getattr(cls, "__class__", object), "__qualname__", None) @@ -405,6 +431,7 @@ def get_class_name(cls, dflt_name=None): def store_wrap(obj): + """Wrap a class or an instance in a ``Store`` (a class gets a ``Store`` subclass whose ``__init__`` builds the wrapped instance).""" if isinstance(obj, type): @wraps(type(obj), updated=()) # added this: test @@ -557,6 +584,7 @@ class FirstArgIsMapping(LiteralVal): def transparent_key_method(self, k): + """Return the key as is (the default ``getitem`` of ``mk_kv_reader_from_kv_collection``).""" return k @@ -572,7 +600,8 @@ def mk_kv_reader_from_kv_collection( By default, getitem will be transparent_key_method, returning the key as is. This default is useful when you want to delegate the actual getting to a _obj_of_data wrapper. - Returns: A KvReader class that subclasses the input kv_collection + Returns: + A KvReader class that subclasses the input kv_collection """ name = name or kv_collection.__qualname__ + "Reader" @@ -581,6 +610,7 @@ def mk_kv_reader_from_kv_collection( def raise_disabled_error(functionality): + """Make a function that raises ``ValueError(' is disabled')`` whenever called.""" def disabled_function(*args, **kwargs): raise ValueError(f"{functionality} is disabled") @@ -588,29 +618,50 @@ def disabled_function(*args, **kwargs): def disable_delitem(o): + """Replace ``o.__delitem__`` (if any) with a function raising ``ValueError``. + + Meant for classes: on an instance, ``del o[k]`` still uses the type's method. + """ if hasattr(o, "__delitem__"): o.__delitem__ = raise_disabled_error("deletion") return o def disable_setitem(o): + """Replace ``o.__setitem__`` (if any) with a function raising ``ValueError``. + + Meant for classes: on an instance, ``o[k] = v`` still uses the type's method. + """ if hasattr(o, "__setitem__"): o.__setitem__ = raise_disabled_error("writing") return o def mk_read_only(o): + """Disable ``__setitem__`` and ``__delitem__`` on ``o`` (a store class, typically). + + >>> class D(dict): + ... pass + >>> D = mk_read_only(D) + >>> D()['a'] = 1 + Traceback (most recent call last): + ... + ValueError: writing is disabled + """ return disable_delitem(disable_setitem(o)) def is_iterable(x): + """Whether ``x`` is an ``Iterable``.""" return isinstance(x, Iterable) def add_ipython_key_completions(store): """Add tab completion that shows you the keys of the store. - Note: ipython already adds local path listing automatically, - so you'll still get those along with your valid store keys. + + Note: + ipython already adds local path listing automatically, + so you'll still get those along with your valid store keys. """ def _ipython_key_completions_(self): @@ -632,6 +683,8 @@ def _ipython_key_completions_(self): def disallow_overwrites(store, *, error_msg=None, disable_deletes=True): + """Intended to make a store class's ``__setitem__`` raise ``OverWritesNotAllowedError`` on existing keys; + currently a no-op that returns ``None`` (the override is never attached). Use ``OverWritesNotAllowedMixin``.""" assert isinstance(store, type), "store needs to be a type" if hasattr(store, "__setitem__"): @@ -648,7 +701,9 @@ def __setitem__(self, k, v): class OverWritesNotAllowedMixin: """Mixin for only allowing a write to a key if they key doesn't already exist. - Note: Should be before the persister in the MRO. + + Note: + Should be before the persister in the MRO. >>> class TestPersister(OverWritesNotAllowedMixin, dict): ... pass @@ -780,7 +835,6 @@ def insert_hash_method( ... pass >>> hash(F({1: 2, 3: 4})) 6 - """ return _wrap_store(_insert_hash_method, locals()) @@ -834,17 +888,20 @@ def cached_keys( It is assumed, if you're using the cached_keys transformation, that you're dealing with static data (or data that can be considered static for the life of the store -- for example, when conducting analytics). If you ever need to refresh the cache during the life of the store, you can to delete _keys_cache like this: - ``` - del your_store._keys_cache - ``` + + .. code-block:: python + + del your_store._keys_cache + Once you do that, the next time you try to ask something about the contents of the store, it will actually do a live query again, as for the first time. - Note: The default keys_cache is list though in many cases, you'd probably should use set, or an explicitly - computer set instead. The reason list is used as the default is because (1) we didn't want to assume that - order did not matter (maybe it does to you) and (2) we didn't want to assume that your keys were hashable. - That said, if you're keys are hashable, and order does not matter, use set. That'll give you two things: - (a) your `key in store` checks will be faster (O(1) instead of O(n)) and (b) you'll enforce unicity of keys. + Note: + The default keys_cache is list though in many cases, you'd probably should use set, or an explicitly + computer set instead. The reason list is used as the default is because (1) we didn't want to assume that + order did not matter (maybe it does to you) and (2) we didn't want to assume that your keys were hashable. + That said, if you're keys are hashable, and order does not matter, use set. That'll give you two things: + (a) your `key in store` checks will be faster (O(1) instead of O(n)) and (b) you'll enforce unicity of keys. Know also that if you precompute the keys you want to cache with a container that has an update method (by default `update`) your cache updates will be faster and if the container you use has @@ -855,24 +912,23 @@ def cached_keys( keys_cache: An explicit collection of keys iter_to_container: The function that will be applied to existing __iter__() and assigned to cache. The default is list. Another useful one is the sorted function. - cache_update_method: Name of the keys_cache update method to use, if it is an attribute of keys_cache. - Note that this cache_update_method will be used only - if keys_cache is an explicit iterable and has that attribute - if keys_cache is a callable and has that attribute. - The default None + cache_update_method: Name of the keys_cache update method to use, if it is an + attribute of keys_cache (whether keys_cache is an explicit iterable or a + callable). Default ``'update'``. name: The name of the new class Returns: - If store is: - None: Will return a decorator that can be applied to a store - a store class: Will return a wrapped class that caches it's keys - a store instance: Will return a wrapped instance that caches it's keys + If store is None, a decorator that can be applied to a store; if store is a + class, a wrapped class that caches its keys; if store is an instance, a + wrapped instance that caches its keys. The instances of such key-cached classes have some extra attributes: - _explicit_keys: The actual cache. An iterable container - update_keys_cache: Is called if a user uses the instance to mutate the store (i.e. write or delete). + ``_keys_cache`` (the actual cache), ``_explicit_keys`` (whether the cache was + given explicitly) and ``update_keys_cache`` (called on ``__setitem__`` and + ``update``). You have two ways of caching keys: + - By providing the explicit list of keys you want cache (and use) - By providing a callable that will iterate through your store and collect an explicit list of keys @@ -1252,13 +1308,13 @@ def catch_and_cache_error_keys( You don't like it? Neither do I. But - It's not a completely outrageous behavior -- if you're talking to live data, it - often happens that you get more, or less, from one second to another. + often happens that you get more, or less, from one second to another. - This store isn't meant to be long living, but rather meant to solve the problem of - skiping items that are problematic (for example, malformatted files), - with a trace of what was skipped and what's valid (in case we need to iterate - again and don't want to bear the hit of requesting values for keys we already - know are problematic. + skiping items that are problematic (for example, malformatted files), + with a trace of what was skipped and what's valid (in case we need to iterate + again and don't want to bear the hit of requesting values for keys we already + know are problematic. Here's a little peep of what is happening under the hood. Meet ``_keys_cache`` and ``_error_keys`` sets (yes, unordered -- so know it) that are meant @@ -1300,7 +1356,6 @@ def catch_and_cache_error_keys( >>> sorted(s.values()) # sorting to get consistent output Error with black key: "Nope, that's from the black list!" [13, 20] - """ assert isinstance(store, type), ( @@ -1405,6 +1460,7 @@ def __contains__(self, k): def iterate_values_and_accumulate_non_error_keys( store, cache_keys_here: list, errors_caught=Exception, error_callback=None ): + """Yield the values of ``store``, appending to ``cache_keys_here`` the keys whose value was fetched without error.""" for k in store: try: v = store[k] @@ -1420,6 +1476,7 @@ def iterate_values_and_accumulate_non_error_keys( def take_everything(key): + """Key filter that accepts every key.""" return True @@ -1439,12 +1496,13 @@ def filt_iter( """Make a wrapper that will transform a store (class or instance thereof) into a sub-store (i.e. subset of keys). Args: - filt: A callable or iterable: - callable: Boolean filter function. A func taking a key and and returns True iff the key should be included. - iterable: The collection of keys you want to filter "in" + filt: A callable or iterable. If a callable, a boolean filter function taking + a key and returning True iff the key should be included. If an iterable, + the collection of keys you want to filter "in". name: The name to give the wrapped class - Returns: A wrapper (that then needs to be applied to a store instance or class. + Returns: + A wrapper (that then needs to be applied to a store instance or class. >>> filtered_dict = filt_iter(filt=lambda k: (len(k) % 2) == 1)(dict) # keep only odd length keys >>> @@ -1602,7 +1660,6 @@ def filter_suffixes(suffixes): True >>> is_text("image.jpg") False - """ if isinstance(suffixes, str): suffixes = [suffixes] @@ -1624,7 +1681,6 @@ def filter_prefixes(prefixes): True >>> is_test_or_report("image.jpg") False - """ if isinstance(prefixes, str): prefixes = [prefixes] @@ -1632,6 +1688,7 @@ def filter_prefixes(prefixes): class FiltIter: + """Namespace of ``filt_iter`` factories (``regex``, ``suffixes``, ...); not meant to be instantiated.""" def __init__(self, *args, **kwargs): raise ValueError( "This class is not meant to be instantiated, but only act as a collection " @@ -1707,7 +1764,8 @@ def kv_wrap_persister_cls(persister_cls, name=None): Args: persister_cls: The persister class to wrap - Returns: A Store wrapping the persister (see dol.base) + Returns: + A Store wrapping the persister (see dol.base) >>> A = kv_wrap_persister_cls(dict) >>> a = A() @@ -1818,7 +1876,8 @@ def _wrap_outcoming( trans_func: The transformation function. wrap_arg_idx: The index of the - Returns: Nothing. It transforms the class in-place + Returns: + Nothing. It transforms the class in-place >>> from dol.trans import store_wrap >>> S = store_wrap(dict) @@ -1921,7 +1980,8 @@ def wrap_kvs( ): r"""Make a Store that is wrapped with the given key/val transformers. - Naming convention: + Naming convention:: + Morphemes: key: outer key _id: inner key @@ -1946,7 +2006,8 @@ def wrap_kvs( The function is called with both `k` and `v` as inputs, and should output a transformed value. The intent use is to do ingoing value transformations conditioned on the key. For example, you may want to serialize an object depending on if you're writing to a - '.csv', or '.json', or '.pickle' file. + '.csv', or '.json', or '.pickle' file. + Forms are `preset(k, obj)` or `preset(self, k, obj)` postget: A function that is called after the value `v` for a key `k` is be `__getitem__`. The function is called with both `k` and `v` as inputs, and should output a transformed value. @@ -1955,7 +2016,8 @@ def wrap_kvs( For example, you may want to deserialize the bytes of a '.csv', or '.json', or '.pickle' in different ways. Forms are `obj = postget(k, data)` or `obj = postget(self, k, data)` - Returns: A key and/or value transformed wrapped (or wrapper) class (or instance). + Returns: + A key and/or value transformed wrapped (or wrapper) class (or instance). >>> def key_of_id(_id): ... return _id.upper() @@ -2048,8 +2110,8 @@ def wrap_kvs( >>> d['foo.pkl'] [['a', 'b', 'c'], ['d', 'e', 'f']] - # TODO: Add tests for outcoming_key_methods etc. """ + # TODO: Add tests for outcoming_key_methods etc. # kwargs = dict( # locals(), wrapper=wrapper or Store, name=store.__qualname__ + "Wrapped" # ) @@ -2067,7 +2129,8 @@ def _handle_codecs(kwargs: dict): """Handle the key_codec and data_codec kwargs, converting them to key_of_id and obj_of_data. - Warning: Mutates kwargs in place. + Warning: + Mutates kwargs in place. >>> kwargs = {'value_decoder': int, 'value_encoder': str} >>> _handle_codecs(kwargs) @@ -2078,7 +2141,6 @@ def _handle_codecs(kwargs: dict): >>> _handle_codecs(kwargs) >>> assert kwargs['key_of_id'] == int >>> assert kwargs['id_of_key'] == str - """ if key_codec := kwargs.get("key_codec", None): if kwargs.get("key_of_id", None): @@ -2168,7 +2230,8 @@ def _handle_codecs(kwargs: dict): def add_decoder(store_cls=None, *, decoder: Callable = None, name=None): """Add a decoder layer to a store. - Note: This is a convenience function for ``wrap_kvs(..., obj_of_data=decoder)``. + Note: + This is a convenience function for ``wrap_kvs(..., obj_of_data=decoder)``. >>> s = {'a': "42"} >>> ss = add_decoder(s, decoder=int) @@ -2278,8 +2341,9 @@ def _kv_wrap_outcoming_keys(trans_func): Use this when you wouldn't use the keys in their original format, or when you want to extract information from it. - Warning: If you haven't also wrapped incoming keys with a corresponding inverse transformation, - you won't be able to use the outcoming keys to fetch data. + Warning: + If you haven't also wrapped incoming keys with a corresponding inverse transformation, + you won't be able to use the outcoming keys to fetch data. >>> from collections import UserDict >>> S = kv_wrap.outcoming_keys(lambda x: x[5:])(UserDict) @@ -2289,10 +2353,10 @@ def _kv_wrap_outcoming_keys(trans_func): >>> list(s.keys()) ['foo', 'bar'] + """ # TODO: Asymmetric key trans breaks getting items (therefore items()). Resolve (remove items() for asym keys?) # >>> list(s.items()) # [('foo', 10), ('bar', 'xo')] - """ def wrapper(o, name=None): name = ( @@ -2312,8 +2376,9 @@ def _kv_wrap_ingoing_keys(trans_func): (because you shouldn't) 'manually' extract that information and construct the key manually every time you need to write something or fetch some existing data. - Warning: If you haven't also wrapped outcoming keys with a corresponding inverse transformation, - you won't be able to use the incoming keys to fetch data. + Warning: + If you haven't also wrapped outcoming keys with a corresponding inverse transformation, + you won't be able to use the incoming keys to fetch data. >>> from collections import UserDict >>> S = kv_wrap.ingoing_keys(lambda x: 'root/' + x)(UserDict) @@ -2325,10 +2390,10 @@ def _kv_wrap_ingoing_keys(trans_func): >>> list(s.keys()) ['root/foo', 'root/bar'] + """ # TODO: Asymmetric key trans breaks getting items (therefore items()). Resolve (remove items() for asym keys?) # >>> list(s.items()) # [('root/foo', 10), ('root/bar', 'xo')] - """ def wrapper(o, name=None): name = ( @@ -2349,7 +2414,8 @@ def _kv_wrap_outcoming_vals(trans_func): but you want it to be interpreted as a JSON formatted text and get a dict instead. Both of these are de-serialization layers, or out-coming value transformations. - Warning: If it matters, make sure you also wrapped with a corresponding inverse serialization. + Warning: + If it matters, make sure you also wrapped with a corresponding inverse serialization. >>> from collections import UserDict >>> S = kv_wrap.outcoming_vals(lambda x: x * 2)(UserDict) @@ -2377,7 +2443,8 @@ def _kv_wrap_ingoing_vals(trans_func): For example, say you have a list of audio samples, and you want to save these in a WAV format. - Warning: If it matters, make sure you also wrapped with a corresponding inverse de-serialization. + Warning: + If it matters, make sure you also wrapped with a corresponding inverse de-serialization. >>> from collections import UserDict >>> S = kv_wrap.ingoing_vals(lambda x: x * 2)(UserDict) @@ -2439,6 +2506,7 @@ def mk_trans_obj(**kwargs): class SimpleDelegator: + """Forward attribute access (and calls) to the wrapped ``obj``.""" def __init__(self, obj): self._obj = obj @@ -2492,6 +2560,8 @@ def mk_confirm_overwrite_preset( """ def confirm_overwrite(self, k, v): + """``preset`` that returns ``v`` unless ``k`` already holds a different value + and the user does not confirm; then the existing value is kept.""" _sentinel = object() existing = self.get(k, _sentinel) if existing is not _sentinel and existing != v: @@ -2522,7 +2592,6 @@ def add_aliases(obj, **aliases): See also, and not to be confused with ``insert_aliases``, which adds aliases to dunder mapping methods (like ``__iter__``, ``__getitem__``) etc. - """ if not aliases: return obj @@ -2568,12 +2637,10 @@ def kv_wrap(trans_obj): >>> d['d', 'e'] 2 - ``kv_wrap`` also has convenience attributes: - ``outcoming_keys``, ``ingoing_keys``, ``outcoming_vals``, ``ingoing_vals``, - and ``val_reads_wrt_to_keys`` + ``kv_wrap`` also has convenience attributes (``outcoming_keys``, ``ingoing_keys``, + ``outcoming_vals``, ``ingoing_vals``, and ``val_reads_wrt_to_keys``) which will only add a single specific wrapper (specified as a function), when that's what you need. - """ key_of_id = getattr(trans_obj, "_key_of_id", None) @@ -2616,13 +2683,17 @@ def mk_wrapper(wrap_cls): You have a wrapper class and you want to make a wrapper out of it, that is, a decorator factory with which you can make wrappers, like this: - ``` - wrapper = mk_wrapper(wrap_cls) - ``` + + .. code-block:: python + + wrapper = mk_wrapper(wrap_cls) + that you can then use to transform stores like thiis: - ``` - MyStore = wrapper(**wrapper_kwargs)(StoreYouWantToTransform) - ``` + + .. code-block:: python + + MyStore = wrapper(**wrapper_kwargs)(StoreYouWantToTransform) + :param wrap_cls: :return: @@ -2707,6 +2778,7 @@ def _conditional_data_trans(v, condition, data_trans): @store_decorator def conditional_data_trans(store=None, *, condition, data_trans): + """Wrap ``store`` so that ``data_trans`` is applied to the read values satisfying ``condition`` (others pass through).""" _data_trans = partial( _conditional_data_trans, condition=condition, data_trans=data_trans ) @@ -2727,7 +2799,7 @@ def add_path_get(store=None, *, name=None, path_type: type = tuple): See issue: https://github.com/i2mint/dol/issues/10.) Say you have some nested stores. - You know... like a `ZipFileReader` store whose values are `ZipReader`s, + You know... like a `ZipFileReader` store whose values are `ZipReader` instances, whose values are bytes of the zipped files (and you can go on... whose (json) values are...). @@ -2752,6 +2824,7 @@ def add_path_get(store=None, *, name=None, path_type: type = tuple): Args: store: The store (class or instance) you're wrapping. If not specified, the function will return a decorator. + name: The name to give the class (not applicable to instance wrapping) path_type: The type that paths are expressed as. Needs to be an Iterable type. By default, a tuple. @@ -2765,7 +2838,7 @@ def add_path_get(store=None, *, name=None, path_type: type = tuple): .. seealso:: - ``KeyPath`` in :doc:`paths` + ``KeyPath`` in ``dol.paths`` Wrapping an instance @@ -2861,7 +2934,7 @@ def add_path_access( forms like ``'a.b.c'``, ``'a/b/c'``, etc. Say you have some nested stores. - You know... like a `ZipFileReader` store whose values are `ZipReader`s, + You know... like a `ZipFileReader` store whose values are `ZipReader` instances, whose values are bytes of the zipped files (and you can go on... whose (json) values are...). @@ -2896,6 +2969,7 @@ def add_path_access( Args: store: The store (class or instance) you're wrapping. If not specified, the function will return a decorator. + name: The name to give the class (not applicable to instance wrapping) path_type: The type that paths are expressed as. Needs to be an Iterable type. By default, a tuple. @@ -2909,7 +2983,7 @@ def add_path_access( .. seealso:: - ``KeyPath`` in :doc:`paths` + ``KeyPath`` in ``dol.paths`` Wrapping a class @@ -2959,7 +3033,8 @@ def add_path_access( >>> s {'a': {'b': {}}} - Note: The add_path_access doesn't carry on to values. + Note: + The add_path_access doesn't carry on to values. >>> s = add_path_access({'a': {'b': {'c': 42}}}) >>> s['a', 'b', 'c'] @@ -2991,7 +3066,6 @@ def add_path_access( >>> # But now this works: >>> s['a']['b', 'c'] 42 - """ store_cls = kv_wrap_persister_cls(store, name=name) store_cls = add_path_get(store_cls, name=name, path_type=path_type) @@ -3099,7 +3173,7 @@ def autoviv(store=None, **kwargs): @store_decorator def flatten(store=None, *, levels=None, cache_keys=False): """ - Flatten a nested store. + Give a nested store a flat view whose keys are the ``(a, b, c)`` key paths. Say you have a store that has three levels (or more), that is, that you can always ask for the value ``store[a][b][c]`` if ``a`` is a valid key of ``store``, @@ -3127,8 +3201,9 @@ def flatten(store=None, *, levels=None, cache_keys=False): cases. The only reason it is not the default is because if you have millions of keys, but little memory, that's not what you might want. - Note: Flattening just provides a wrapper giving you a "flattened view". It doesn't - change the store itself, or it's contents. + Note: + Flattening just provides a wrapper giving you a "flattened view". It doesn't + change the store itself, or it's contents. :param store: The store instance or class to be wrapped :param levels: The number of nested levels to flatten @@ -3235,6 +3310,7 @@ def mk_level_walk_filt(levels): def leveled_paths_walk(m, levels): + """Yield the key paths of ``m``, down to ``levels`` levels.""" yield from kv_walk( m, leaf_yield=lambda p, k, v: p, walk_filt=mk_level_walk_filt(levels) ) @@ -3253,8 +3329,9 @@ def insert_aliases( If store is a class, you'll get a copy of the class with those methods added. If store is an instance, the methods will be added in place (no copy will be made). - Note: If an operation (write, read, delete, list, count) is not specified, no alias will be created for - that operation. + Note: + If an operation (write, read, delete, list, count) is not specified, no alias will be created for + that operation. IMPORTANT NOTE: The signatures of the methods the aliases will point to will not change. We say this because, you can call the write method "dump", but you'll have to use it as @@ -3272,7 +3349,8 @@ def insert_aliases( list: Desired method name for __iter__ count: Desired method name for __len__ - Returns: A store with the desired aliases. + Returns: + A store with the desired aliases. >>> # Example of extending a class >>> mydict = insert_aliases(dict, write='dump', read='load', delete='rm', list='peek', count='size') @@ -3321,7 +3399,8 @@ def insert_load_dump_aliases(store=None, *, delete=None, list=None, count=None): list: Desired method name for __iter__ count: Desired method name for __len__ - Returns: A store with the desired aliases. + Returns: + A store with the desired aliases. >>> mydict = insert_load_dump_aliases(dict) >>> s = mydict() @@ -3357,7 +3436,6 @@ def constant_output(return_val=None, *args, **kwargs): >>> always_true = partial(constant_output, True) >>> always_true('regardless', 'of', the='input', will='return True') True - """ return return_val @@ -3371,6 +3449,7 @@ def condition_function_call( constant_output, None ), ): + """Decorator: call ``func`` only when ``condition(*args, **kwargs)`` holds, else ``callback_if_condition_not_met``.""" @wraps(func) def wrapped_func(*args, **kwargs): if condition(*args, **kwargs): @@ -3514,7 +3593,8 @@ def add_missing_key_handling( ): """Overrides the ``__missing__`` method of a store with a custom callback. - Note: The callback must have two arguments: the store and the key. + Note: + The callback must have two arguments: the store and the key. Args: store: The store class to wrap. @@ -3560,6 +3640,7 @@ class StoreWithMissingKeyCallback(store): def ignore_if_error(store=None, *, errors=(KeyError,)): + """Wrap ``store`` so that ``__getitem__`` errors in ``errors`` return ``None`` instead of raising.""" def _ignore(store, k): pass @@ -3574,6 +3655,7 @@ def warn_and_ignore_if_error( errors=(KeyError,), warn_msg="Ignoring error in __getitem__ for key {k}: {e}", ): + """Like ``ignore_if_error``, but also emit a warning (``warn_msg``) for each ignored error.""" def _warn(store, k): import sys @@ -3587,6 +3669,7 @@ def _warn(store, k): def return_default_if_error(store=None, *, default=None, errors=(KeyError,)): + """Wrap ``store`` so that ``__getitem__`` errors in ``errors`` return ``default`` instead of raising.""" def _default(store, k): return default @@ -3606,6 +3689,7 @@ def _default(store, k): # TODO: Want a way to specify Encoded type and Decoded type @dataclass class Codec(Generic[DecodedType, EncodedType]): + """An ``encoder``/``decoder`` pair; iterates as ``(encoder, decoder)`` and composes with ``compose_with``.""" encoder: Callable[[DecodedType], EncodedType] decoder: Callable[[EncodedType], DecodedType] @@ -3634,16 +3718,19 @@ def invert(self): class ValueCodec(*_CodecT): + """A ``Codec`` that, called on a store, wraps its values (``data_of_obj``/``obj_of_data``).""" def __call__(self, obj): return wrap_kvs(obj, data_of_obj=self.encoder, obj_of_data=self.decoder) class KeyCodec(*_CodecT): + """A ``Codec`` that, called on a store, wraps its keys (``id_of_key``/``key_of_id``).""" def __call__(self, obj): return wrap_kvs(obj, id_of_key=self.encoder, key_of_id=self.decoder) class KeyValueCodec(*_CodecT): + """A ``Codec`` that, called on a store, wraps values with key context (``preset``/``postget``).""" def __call__(self, obj): return wrap_kvs(obj, preset=self.encoder, postget=self.decoder) @@ -3654,6 +3741,7 @@ def __call__(self, obj): def _affix_encoder(string: str, prefix: str = "", suffix: str = ""): """Affix a prefix and suffix to a string + >>> _affix_encoder('name', prefix='/folder/', suffix='.txt') '/folder/name.txt' """ @@ -3662,6 +3750,7 @@ def _affix_encoder(string: str, prefix: str = "", suffix: str = ""): def _affix_decoder(string: str, prefix: str = "", suffix: str = ""): """Remove prefix and suffix from string + >>> _affix_decoder('/folder/name.txt', prefix='/folder/', suffix='.txt') 'name' """ @@ -3688,7 +3777,8 @@ def affix_key_codec(prefix: str = "", suffix: str = ""): def redirect_getattr_to_getitem(cls=None, *, keys_have_priority_over_attributes=False): """A mapping decorator that redirects attribute access to __getitem__. - Warning: This decorator will make your class un-pickleable. + Warning: + This decorator will make your class un-pickleable. :param keys_have_priority_over_attributes: If True, keys will have priority over existing attributes. @@ -3702,7 +3792,6 @@ def redirect_getattr_to_getitem(cls=None, *, keys_have_priority_over_attributes= 2 >>> list(d) ['a', 'b'] - """ class RidirectGetattrToGetitem(cls): diff --git a/dol/trash.py b/dol/trash.py index ab57e1cb..dccc712f 100644 --- a/dol/trash.py +++ b/dol/trash.py @@ -4,13 +4,13 @@ moving files to trash/recycle bin instead of permanent deletion. Available deletion strategies: + - default_delete_func: Safe trash with warning on fallback to os.remove - permanent_delete: Direct os.remove (no warnings) - trash_only: Error if trash unavailable -Usage: - Configure deletion behavior when creating file stores by passing the - delete_func parameter or setting _delete_func class attribute. +Configure deletion behavior when creating file stores by passing the +``delete_func`` parameter or setting the ``_delete_func`` class attribute. """ import os @@ -31,6 +31,7 @@ def get_platform_trash_func() -> Optional[DeleteFunc]: Returns None if no trash function is available. Priority order: + 1. send2trash library (if installed) 2. Platform-specific implementation (macOS, Windows, Linux) 3. None (will fall back to os.remove) diff --git a/dol/util.py b/dol/util.py index 324501a9..fac7a5ee 100644 --- a/dol/util.py +++ b/dol/util.py @@ -1,4 +1,17 @@ -"""General util objects""" +"""General util objects: function composition, grouping, partial classes, file helpers. + +Main entry points: + +- ``Pipe``: compose functions left to right +- ``partialclass``: ``functools.partial`` for classes +- ``groupby``, ``regroupby``, ``igroupby``: group items by a key function +- ``chain_get``: first value found for a sequence of keys +- ``written_bytes``, ``read_from_bytes``: turn file-writing/reading functions into bytes codecs + + >>> from dol.util import Pipe + >>> Pipe(lambda x: x + 1, str)(1) + '2' +""" import os import shutil @@ -66,6 +79,7 @@ def non_colliding_key( collision_handler: Function taking (key, attempt_number) and returning a modified key. For strings, defaults to appending " (N)" suffix before extension. For other types, must be provided. + max_attempts: Maximum number of transformation attempts Returns: @@ -83,7 +97,6 @@ def non_colliding_key( 'file (2).txt' >>> non_colliding_key(42, {42}, collision_handler=lambda k, n: k + n) 43 - """ if key not in exclude: return key @@ -139,6 +152,7 @@ def safe_compile(path, normalize_path=True): re.Pattern: A compiled regular expression object for the given path. Examples: + >>> import re >>> isinstance(safe_compile("/fun/paths/are/awesome"), re.Pattern) True @@ -219,6 +233,7 @@ def is_unbound_method(obj): True if obj is an unbound method, False otherwise. Examples: + >>> import sys >>> import types >>> def function(): @@ -282,7 +297,6 @@ def add_as_attribute_of(obj, name=None): In reality, any object that has a ``__name__`` can be added to the attribute of ``obj``, but the intention is to add helper functions to main "container" functions. - """ def _decorator(f): @@ -299,8 +313,9 @@ def chain_get(d: Mapping, keys, default=None): """ Returns the ``d[key]`` value for the first ``key`` in ``keys`` that is in ``d``, and default if none are found - Note: Think of ``collections.ChainMap`` where you can look for a single key in a sequence of maps until we find it. - Here we look for a sequence of keys in a single map, stopping as soon as we find a key that the map has. + Note: + Think of ``collections.ChainMap`` where you can look for a single key in a sequence of maps until we find it. + Here we look for a sequence of keys in a single map, stopping as soon as we find a key that the map has. >>> d = {'here': '&', 'there': 'and', 'every': 'where'} >>> chain_get(d, ['not there', 'not there either', 'there', 'every']) @@ -316,7 +331,6 @@ def chain_get(d: Mapping, keys, default=None): >>> chain_get(d, ('none', 'of', 'these'), default='Not Found') 'Not Found' - """ for key in keys: if key in d: @@ -332,7 +346,6 @@ class LiteralVal: 42 >>> t() 42 - """ def __init__(self, val): @@ -373,7 +386,6 @@ def decorate_callables(decorator, cls=None): 'dry' >>> a.big() # doctest: +SKIP 'small' - """ if cls is None: return partial(decorate_callables, decorator) @@ -495,6 +507,7 @@ def not_a_mac_junk_path(path: str): def inject_method(obj, method_function, method_name=None): """ method_function could be: + * a function * a {method_name: function, ...} dict (for multiple injections) * a list of functions or (function, method_name) pairs @@ -533,7 +546,6 @@ def _disabled_clear_method(self): del self[k] except KeyError: pass - """ raise NotImplementedError(f"Instance of {type(self)}: {self.clear.__doc__}") @@ -603,7 +615,6 @@ def truncate_string_with_marker( '12---90' >>> truncate_string('supercalifragilisticexpialidocious') 'su---us' - """ middle_marker_len = len(middle_marker) if len(s) <= left_limit + right_limit: @@ -660,6 +671,7 @@ class Pipe: 3 Notes: + - Pipe instances don't have a __name__ etc. So some expectations of normal functions are not met. - Pipe instance are pickalable (as long as the functions that compose them are) @@ -685,7 +697,6 @@ class Pipe: 'map_and_sum' >>> f.__doc__ 'Apply func and add' - """ funcs = () @@ -804,7 +815,7 @@ def flatten_pipe(pipe): def partialclass(cls, *args, **kwargs): - """What partial(cls, *args, **kwargs) does, but returning a class instead of an object. + """What ``partial(cls, *args, **kwargs)`` does, but returning a class instead of an object. :param cls: Class to get the partial of :param kwargs: The kwargs to fix @@ -858,14 +869,12 @@ def partialclass(cls, *args, **kwargs): ... TypeError: __init__() got multiple values for argument 'a' - On the other hand, you can use *args to specify the fixtures: + On the other hand, you can use ``*args`` to specify the fixtures: >>> AA = partialclass(A, 22) >>> assert str(AA()) == 'A(a=22, b=1)' >>> assert str(signature(AA)) == '(b=1)' >>> assert str(AA(3)) == 'A(a=22, b=3)' - - """ assert isinstance(cls, type), f"cls should be a type, was a {type(cls)}: {cls}" @@ -1074,7 +1083,6 @@ def format_invocation(name="", args=(), kwargs=None): a_func(1) >>> print(format_invocation('kw_func', kwargs=[('a', 1), ('b', 2)])) kw_func(a=1, b=2) - """ kwargs = kwargs or {} a_text = ", ".join([repr(a) for a in args]) @@ -1099,7 +1107,7 @@ def groupby( group_factory=list, ) -> dict: """Groups items according to group keys updated from those items through the given - (item_to_)key function. + ``key`` function (mapping an item to its group key). Args: items: iterable of items @@ -1110,9 +1118,11 @@ def groupby( group_items.append(x) will be called to add x to that collection The default is `list` - Returns: A dict of {group_key: items_in_that_group, ...} + Returns: + A dict of {group_key: items_in_that_group, ...} - See Also: regroupby, itertools.groupby, and dol.source.SequenceKvReader + See Also: + regroupby, itertools.groupby, and dol.source.SequenceKvReader >>> groupby(range(11), key=lambda x: x % 3) {0: [0, 3, 6, 9], 1: [1, 4, 7, 10], 2: [2, 5, 8]} @@ -1142,10 +1152,13 @@ def groupby( def regroupby(items, *key_funcs, **named_key_funcs): """Recursive groupby. Applies the groupby function recursively, using a sequence of key functions. - Note: The named_key_funcs argument names don't have any external effect. + Note: + The named_key_funcs argument names don't have any external effect. + They just give a name to the key function, for code reading clarity purposes. - See Also: groupby, itertools.groupby, and dol.source.SequenceKvReader + See Also: + groupby, itertools.groupby, and dol.source.SequenceKvReader >>> # group by how big the number is, then by it's mod 3 value >>> # note that named_key_funcs argument names doesn't have any external effect (but give a name to the function) @@ -1193,7 +1206,7 @@ def igroupby( grouper_mapping=defaultdict, ): """The generator version of dol groupby. - Groups items according to group keys updated from those items through the given (item_to_)key function, + Groups items according to group keys updated from those items through the given ``key`` function (mapping an item to its group key), yielding the groups according to a logic defined by ``group_release_cond`` Args: @@ -1213,7 +1226,8 @@ def igroupby( items in the grouping "cache". ``release_remainding`` is a boolean that indicates whether the contents of this cache should be released or not. - Yields: ``(group_key, items_in_that_group)`` pairs + Yields: + ``(group_key, items_in_that_group)`` pairs The following will group numbers according to their parity (0 for even, 1 for odd), @@ -1264,7 +1278,6 @@ def igroupby( >>> kws.update(key=lambda w: ['words', 'stopwords'][int(w in stopwords)]) >>> assert (dict(igroupby(**kws)) == groupby(**kws) ... == {'stopwords': ['the', 'in', 'a'], 'words': ['fox', 'is', 'box']}) - """ groups = grouper_mapping(group_factory) @@ -1336,8 +1349,11 @@ def fill_with_dflts(d, dflt_dict=None): See examples to know how to use it. - ATTENTION: A shallow copy of the dict is made. Know how that affects you (or not). - ATTENTION: This is not recursive: It won't be filling any nested fields with defaults. + ATTENTION: + A shallow copy of the dict is made. Know how that affects you (or not). + + ATTENTION: + This is not recursive: It won't be filling any nested fields with defaults. Args: d: The dict you want to "fill" @@ -1799,8 +1815,9 @@ def written_bytes( This is the write version of the `read_from_bytes` function of the same module. - Note: If obj is not given, `write_bytes` will return a "bytes writer" function that - takes obj as the first argument, and uses the file_writer to write the bytes. + Note: + If obj is not given, `write_bytes` will return a "bytes writer" function that + takes obj as the first argument, and uses the file_writer to write the bytes. :param file_writer: A function that writes an object to a file-like object. :param obj: The object to write. @@ -1823,21 +1840,15 @@ def written_bytes( Here's another example with pandas DataFrame.to_parquet: - >>> import pandas as pd # doctest: +SKIP - >>> df = pd.DataFrame({ # doctest: +SKIP - ... 'column1': [1, 2, 3], - ... 'column2': ['A', 'B', 'C'] - ... }) - - Get a function that converts DataFrame to Parquet bytes - - df_to_parquet_bytes = written_bytes(pd.DataFrame.to_parquet) - - # Get the bytes of the DataFrame in Parquet format - parquet_bytes = df_to_parquet_bytes(df) - all(pd.read_parquet(io.BytesIO(parquet_bytes)) == df) - + .. code-block:: python + import pandas as pd + df = pd.DataFrame({'column1': [1, 2, 3], 'column2': ['A', 'B', 'C']}) + # Get a function that converts DataFrame to Parquet bytes + df_to_parquet_bytes = written_bytes(pd.DataFrame.to_parquet) + # Get the bytes of the DataFrame in Parquet format + parquet_bytes = df_to_parquet_bytes(df) + all(pd.read_parquet(io.BytesIO(parquet_bytes)) == df) """ if obj is None: return partial( @@ -1902,8 +1913,9 @@ def read_from_bytes( This is the read version of the `written_bytes` function of the same module. - Note: If obj is not given, read_from_bytes will return a "bytes reader" function that - takes obj as the first argument, and uses the file_reader to read the bytes. + Note: + If obj is not given, read_from_bytes will return a "bytes reader" function that + takes obj as the first argument, and uses the file_reader to read the bytes. :param file_reader: A function that reads from a file-like object. :param obj: The bytes to read. @@ -1983,7 +1995,7 @@ def written_key( If a string starting with '*', the '*' is replaced with a unique temporary filename. If a string that has a '*' somewhere in the middle, what's on the left of if is used as a directory and the '*' is replaced with a unique temporary filename. For example - '/tmp/*_file.ext' would be replaced with '/tmp/oiu8fj9873_file.ext'. + ``'/tmp/*_file.ext'`` would be replaced with ``'/tmp/oiu8fj9873_file.ext'``. If a callable, it will be called with obj as input to get the key. One use case is to use a function that generates a key based on the object. :param obj_arg_position_in_writer: Position of the object argument in writer function (0 or 1). @@ -2066,7 +2078,6 @@ def written_key( '/var/folders/mc/c070wfh51kxd9lft8dl74q1r0000gn/T/tmp8yaczd8b.json' >>> json.loads(open(filepath).read()) {'a': 1, 'b': 2} - """ if obj is None: return partial( @@ -2204,7 +2215,7 @@ class AttributeMapping(SimpleNamespace, Mapping[str, Any]): Useful when you want mapping interface but don't need mutation. - Examples: + .. rubric:: Examples >>> ns = AttributeMapping(x=10, y=20) >>> ns.x @@ -2245,7 +2256,7 @@ class AttributeMutableMapping(AttributeMapping, MutableMapping[str, Any]): Extends AttributeMapping with mutation capabilities, ensuring proper error handling and protocol compliance. - Examples: + .. rubric:: Examples >>> ns = AttributeMutableMapping(apple=1, banana=2) >>> ns.apple diff --git a/dol/zipfiledol.py b/dol/zipfiledol.py index 4bbe8b50..92d0918d 100644 --- a/dol/zipfiledol.py +++ b/dol/zipfiledol.py @@ -1,5 +1,16 @@ -""" -Data object layers and other utils to work with zip files. +"""Data object layers and other utils to work with zip files. + +Main entry points: + +- ``FilesOfZip``: read-only bytes of the files in a zip archive +- ``ZipReader``: same, but browsing folders as nested readers +- ``ZipFiles``: read-write-delete access to files in a zip archive +- ``FlatZipFilesReader``: the union of the contents of several zip files +- ``zip_compress``, ``zip_decompress``: single-file zip bytes helpers + + >>> from dol.zipfiledol import zip_compress, zip_decompress + >>> zip_decompress(zip_compress(b'hello')) + b'hello' """ import os @@ -177,7 +188,6 @@ def to_zip_file( :param zip_filepath: zip filepath to save the zipped input to :param filename: The name/path of the zip entry we want to save to :param encoding: In case the input is str, the encoding to use to convert to bytes - """ z = ZipFiles( zip_filepath, @@ -277,14 +287,15 @@ class ZipReader(KvReader): When a file, the value return is bytes, as usual. When a directory, the value returned is a ``ZipReader`` itself, with all params the same, - except for the ``prefix`` - which serves `to specify the subfolder (that is, ``prefix`` acts as a filter). + except for the ``prefix``, which serves to specify the subfolder (that is, + ``prefix`` acts as a filter). - Note: If you get data zipped by a mac, you might get some junk along with it. - Namely `__MACOSX` folders `.DS_Store` files. I won't rant about it, since others have. - But you might find it useful to remove them from view. One choice is to use - `dol.trans.filt_iter` - to get a filtered view of the zips contents. In most cases, this should do the job: + Note: + If you get data zipped by a mac, you might get some junk along with it. + Namely `__MACOSX` folders `.DS_Store` files. I won't rant about it, since others have. + But you might find it useful to remove them from view. One choice is to use + `dol.trans.filt_iter` + to get a filtered view of the zips contents. In most cases, this should do the job: .. code-block:: @@ -300,7 +311,7 @@ class ZipReader(KvReader): zip -d filename.zip \*/.DS_Store - Examples: + .. rubric:: Examples .. code-block:: @@ -476,7 +487,6 @@ class FileStreamsOfZip(FilesOfZip): z = FileStreamsOfZip(rootdir) with z[relpath] as fp: ... # do stuff with fp, like fp.readlines() or such... - """ def __getitem__(self, k): @@ -571,7 +581,6 @@ class FlatZipFilesReader(FlatReader, ZipFilesReader): Well, one solution, provided through FlatZipFilesReader, is to not unzip at all, but instead, give you a store that provides you a view "as if you unzipped and merged". - """ __init__ = ZipFilesReader.__init__ @@ -704,7 +713,7 @@ class ZipFiles(KvPersister): makes for a not so efficient store, out of the box. I advise using one of the zip readers if all you need to do is read, or subclassing or - wrapping ZipFiles with caching layers if it is appropriate to you. + wrapping ZipFiles with caching layers if it is appropriate to you. Let's verify that a ZipFiles can indeed write data. First, we'll set things up! @@ -740,7 +749,6 @@ class ZipFiles(KvPersister): And indeed we have a zip file now: >>> assert os.path.isfile(test_zipfile) - """ _zipfile_init_kw = dict( @@ -901,14 +909,14 @@ def remove_some_entries_from_zip( presented with the keys first, and asked permission to delete. :return: The ZipFiles (in case you want to do further work with it) - Tip: If you want to delete with no questions asked, use currying: + Tip: + If you want to delete with no questions asked, use currying: >>> from functools import partial >>> rm_keys_without_asking = partial( ... remove_some_entries_from_zip, ... ask_before_before_deleting=False ... ) - """ z = zip_source if not isinstance(z, Mapping): @@ -969,6 +977,11 @@ def is_a_mac_junk_path(path): def tar_compress(data_bytes, file_name="data.bin"): + """Bytes of an (uncompressed) tar archive holding ``data_bytes`` as a single file. + + >>> tar_decompress(tar_compress(b'hello', file_name='x.bin')) + b'hello' + """ import tarfile import io @@ -982,6 +995,7 @@ def tar_compress(data_bytes, file_name="data.bin"): def tar_decompress(tar_bytes): + """Bytes of the first file found in the tar archive ``tar_bytes`` (None if none).""" import tarfile import io