diff options
| author | Jannis Leidel <jannis@leidel.info> | 2019-10-18 15:57:13 +0200 |
|---|---|---|
| committer | GitHub <noreply@github.com> | 2019-10-18 15:57:13 +0200 |
| commit | f6bf14afd22d8e5b706670590cc95f29d4483434 (patch) | |
| tree | a65fbc8b6fd1112222beb6f66c31082ffe231460 /src | |
| parent | f3d02aa3b088e2f9044fe5e4869e8c8cb91d2cdc (diff) | |
| download | tablib-f6bf14afd22d8e5b706670590cc95f29d4483434.tar.gz | |
Add project release config and cleanup project setup. (#398)
* Add project release config and use Travis build stages.
Refs #378.
* Restructure project to use src/ and tests/ directories.
* Fix testing.
* Remove eggs.
* More fixes.
- isort and flake8 config
- manifest template update
- tox ini extension
- docs build fixes
- docs content fixes
* Docs and license cleanup.
Diffstat (limited to 'src')
33 files changed, 5795 insertions, 0 deletions
diff --git a/src/tablib/__init__.py b/src/tablib/__init__.py new file mode 100644 index 0000000..fafb569 --- /dev/null +++ b/src/tablib/__init__.py @@ -0,0 +1,13 @@ +""" Tablib. """ +from pkg_resources import get_distribution, DistributionNotFound + +from tablib.core import ( + Databook, Dataset, detect_format, import_set, import_book, + InvalidDatasetType, InvalidDimensions, UnsupportedFormat +) + +try: + __version__ = get_distribution(__name__).version +except DistributionNotFound: + # package is not installed + __version__ = None diff --git a/src/tablib/compat.py b/src/tablib/compat.py new file mode 100644 index 0000000..1582173 --- /dev/null +++ b/src/tablib/compat.py @@ -0,0 +1,36 @@ +# -*- coding: utf-8 -*- + +""" +tablib.compat +~~~~~~~~~~~~~ + +Tablib compatiblity module. + +""" + +import sys + +is_py3 = (sys.version_info[0] > 2) + + +if is_py3: + from io import StringIO + from statistics import median + from itertools import zip_longest as izip_longest + import csv + import tablib.packages.dbfpy3 as dbfpy + + unicode = str + xrange = range + +else: + from StringIO import StringIO + from tablib.packages.statistics import median + from itertools import izip_longest + from backports import csv + import tablib.packages.dbfpy as dbfpy + + unicode = unicode + xrange = xrange + +from MarkupPy import markup # Kept temporarily to avoid breaking existing imports diff --git a/src/tablib/core.py b/src/tablib/core.py new file mode 100644 index 0000000..65dd901 --- /dev/null +++ b/src/tablib/core.py @@ -0,0 +1,1160 @@ +# -*- coding: utf-8 -*- +""" + tablib.core + ~~~~~~~~~~~ + + This module implements the central Tablib objects. + + :copyright: (c) 2016 by Kenneth Reitz. 2019 Jazzband. + :license: MIT, see LICENSE for more details. +""" + +from collections import OrderedDict +from copy import copy +from operator import itemgetter + +from tablib import formats + +from tablib.compat import unicode + + +__title__ = 'tablib' +__author__ = 'Kenneth Reitz' +__license__ = 'MIT' +__copyright__ = 'Copyright 2017 Kenneth Reitz. 2019 Jazzband.' +__docformat__ = 'restructuredtext' + + +class Row(object): + """Internal Row object. Mainly used for filtering.""" + + __slots__ = ['_row', 'tags'] + + def __init__(self, row=list(), tags=list()): + self._row = list(row) + self.tags = list(tags) + + def __iter__(self): + return (col for col in self._row) + + def __len__(self): + return len(self._row) + + def __repr__(self): + return repr(self._row) + + def __getslice__(self, i, j): + return self._row[i:j] + + def __getitem__(self, i): + return self._row[i] + + def __setitem__(self, i, value): + self._row[i] = value + + def __delitem__(self, i): + del self._row[i] + + def __getstate__(self): + + slots = dict() + + for slot in self.__slots__: + attribute = getattr(self, slot) + slots[slot] = attribute + + return slots + + def __setstate__(self, state): + for (k, v) in list(state.items()): setattr(self, k, v) + + def rpush(self, value): + self.insert(0, value) + + def lpush(self, value): + self.insert(len(value), value) + + def append(self, value): + self.rpush(value) + + def insert(self, index, value): + self._row.insert(index, value) + + def __contains__(self, item): + return (item in self._row) + + @property + def tuple(self): + """Tuple representation of :class:`Row`.""" + return tuple(self._row) + + @property + def list(self): + """List representation of :class:`Row`.""" + return list(self._row) + + def has_tag(self, tag): + """Returns true if current row contains tag.""" + + if tag == None: + return False + elif isinstance(tag, str): + return (tag in self.tags) + else: + return bool(len(set(tag) & set(self.tags))) + + +class Dataset(object): + """The :class:`Dataset` object is the heart of Tablib. It provides all core + functionality. + + Usually you create a :class:`Dataset` instance in your main module, and append + rows as you collect data. :: + + data = tablib.Dataset() + data.headers = ('name', 'age') + + for (name, age) in some_collector(): + data.append((name, age)) + + + Setting columns is similar. The column data length must equal the + current height of the data and headers must be set :: + + data = tablib.Dataset() + data.headers = ('first_name', 'last_name') + + data.append(('John', 'Adams')) + data.append(('George', 'Washington')) + + data.append_col((90, 67), header='age') + + + You can also set rows and headers upon instantiation. This is useful if + dealing with dozens or hundreds of :class:`Dataset` objects. :: + + headers = ('first_name', 'last_name') + data = [('John', 'Adams'), ('George', 'Washington')] + + data = tablib.Dataset(*data, headers=headers) + + :param \\*args: (optional) list of rows to populate Dataset + :param headers: (optional) list strings for Dataset header row + :param title: (optional) string to use as title of the Dataset + + + .. admonition:: Format Attributes Definition + + If you look at the code, the various output/import formats are not + defined within the :class:`Dataset` object. To add support for a new format, see + :ref:`Adding New Formats <newformats>`. + + """ + + _formats = {} + + def __init__(self, *args, **kwargs): + self._data = list(Row(arg) for arg in args) + self.__headers = None + + # ('title', index) tuples + self._separators = [] + + # (column, callback) tuples + self._formatters = [] + + self.headers = kwargs.get('headers') + + self.title = kwargs.get('title') + + self._register_formats() + + def __len__(self): + return self.height + + def __getitem__(self, key): + if isinstance(key, (str, unicode)): + if key in self.headers: + pos = self.headers.index(key) # get 'key' index from each data + return [row[pos] for row in self._data] + else: + raise KeyError + else: + _results = self._data[key] + if isinstance(_results, Row): + return _results.tuple + else: + return [result.tuple for result in _results] + + def __setitem__(self, key, value): + self._validate(value) + self._data[key] = Row(value) + + def __delitem__(self, key): + if isinstance(key, (str, unicode)): + + if key in self.headers: + + pos = self.headers.index(key) + del self.headers[pos] + + for i, row in enumerate(self._data): + + del row[pos] + self._data[i] = row + else: + raise KeyError + else: + del self._data[key] + + def __repr__(self): + try: + return '<%s dataset>' % (self.title.lower()) + except AttributeError: + return '<dataset object>' + + def __unicode__(self): + result = [] + + # Add unicode representation of headers. + if self.__headers: + result.append([unicode(h) for h in self.__headers]) + + # Add unicode representation of rows. + result.extend(list(map(unicode, row)) for row in self._data) + + lens = [list(map(len, row)) for row in result] + field_lens = list(map(max, zip(*lens))) + + # delimiter between header and data + if self.__headers: + result.insert(1, ['-' * length for length in field_lens]) + + format_string = '|'.join('{%s:%s}' % item for item in enumerate(field_lens)) + + return '\n'.join(format_string.format(*row) for row in result) + + def __str__(self): + return self.__unicode__() + + # --------- + # Internals + # --------- + + @classmethod + def _register_formats(cls): + """Adds format properties.""" + for fmt in formats.available: + try: + try: + setattr(cls, fmt.title, property(fmt.export_set, fmt.import_set)) + setattr(cls, 'get_%s' % fmt.title, fmt.export_set) + setattr(cls, 'set_%s' % fmt.title, fmt.import_set) + cls._formats[fmt.title] = (fmt.export_set, fmt.import_set) + except AttributeError: + setattr(cls, fmt.title, property(fmt.export_set)) + setattr(cls, 'get_%s' % fmt.title, fmt.export_set) + cls._formats[fmt.title] = (fmt.export_set, None) + + except AttributeError: + cls._formats[fmt.title] = (None, None) + + def _validate(self, row=None, col=None, safety=False): + """Assures size of every row in dataset is of proper proportions.""" + if row: + is_valid = (len(row) == self.width) if self.width else True + elif col: + if len(col) < 1: + is_valid = True + else: + is_valid = (len(col) == self.height) if self.height else True + else: + is_valid = all((len(x) == self.width for x in self._data)) + + if is_valid: + return True + else: + if not safety: + raise InvalidDimensions + return False + + def _package(self, dicts=True, ordered=True): + """Packages Dataset into lists of dictionaries for transmission.""" + # TODO: Dicts default to false? + + _data = list(self._data) + + if ordered: + dict_pack = OrderedDict + else: + dict_pack = dict + + # Execute formatters + if self._formatters: + for row_i, row in enumerate(_data): + for col, callback in self._formatters: + try: + if col is None: + for j, c in enumerate(row): + _data[row_i][j] = callback(c) + else: + _data[row_i][col] = callback(row[col]) + except IndexError: + raise InvalidDatasetIndex + + if self.headers: + if dicts: + data = [dict_pack(list(zip(self.headers, data_row))) for data_row in _data] + else: + data = [list(self.headers)] + list(_data) + else: + data = [list(row) for row in _data] + + return data + + def _get_headers(self): + """An *optional* list of strings to be used for header rows and attribute names. + + This must be set manually. The given list length must equal :class:`Dataset.width`. + + """ + return self.__headers + + def _set_headers(self, collection): + """Validating headers setter.""" + self._validate(collection) + if collection: + try: + self.__headers = list(collection) + except TypeError: + raise TypeError + else: + self.__headers = None + + headers = property(_get_headers, _set_headers) + + def _get_dict(self): + """A native Python representation of the :class:`Dataset` object. If headers have + been set, a list of Python dictionaries will be returned. If no headers have been set, + a list of tuples (rows) will be returned instead. + + A dataset object can also be imported by setting the `Dataset.dict` attribute: :: + + data = tablib.Dataset() + data.dict = [{'age': 90, 'first_name': 'Kenneth', 'last_name': 'Reitz'}] + + """ + return self._package() + + def _set_dict(self, pickle): + """A native Python representation of the Dataset object. If headers have been + set, a list of Python dictionaries will be returned. If no headers have been + set, a list of tuples (rows) will be returned instead. + + A dataset object can also be imported by setting the :class:`Dataset.dict` attribute. :: + + data = tablib.Dataset() + data.dict = [{'age': 90, 'first_name': 'Kenneth', 'last_name': 'Reitz'}] + + """ + + if not len(pickle): + return + + # if list of rows + if isinstance(pickle[0], list): + self.wipe() + for row in pickle: + self.append(Row(row)) + + # if list of objects + elif isinstance(pickle[0], dict): + self.wipe() + self.headers = list(pickle[0].keys()) + for row in pickle: + self.append(Row(list(row.values()))) + else: + raise UnsupportedFormat + + dict = property(_get_dict, _set_dict) + + def _clean_col(self, col): + """Prepares the given column for insert/append.""" + + col = list(col) + + if self.headers: + header = [col.pop(0)] + else: + header = [] + + if len(col) == 1 and hasattr(col[0], '__call__'): + + col = list(map(col[0], self._data)) + col = tuple(header + col) + + return col + + @property + def height(self): + """The number of rows currently in the :class:`Dataset`. + Cannot be directly modified. + """ + return len(self._data) + + @property + def width(self): + """The number of columns currently in the :class:`Dataset`. + Cannot be directly modified. + """ + + try: + return len(self._data[0]) + except IndexError: + try: + return len(self.headers) + except TypeError: + return 0 + + def load(self, in_stream, format=None, **kwargs): + """ + Import `in_stream` to the :class:`Dataset` object using the `format`. + + :param \\*\\*kwargs: (optional) custom configuration to the format `import_set`. + """ + + if not format: + format = detect_format(in_stream) + + export_set, import_set = self._formats.get(format, (None, None)) + if not import_set: + raise UnsupportedFormat('Format {0} cannot be imported.'.format(format)) + + import_set(self, in_stream, **kwargs) + return self + + def export(self, format, **kwargs): + """ + Export :class:`Dataset` object to `format`. + + :param \\*\\*kwargs: (optional) custom configuration to the format `export_set`. + """ + export_set, import_set = self._formats.get(format, (None, None)) + if not export_set: + raise UnsupportedFormat('Format {0} cannot be exported.'.format(format)) + + return export_set(self, **kwargs) + + # ------- + # Formats + # ------- + + @property + def xls(): + """A Legacy Excel Spreadsheet representation of the :class:`Dataset` object, with :ref:`separators`. Cannot be set. + + .. note:: + + XLS files are limited to a maximum of 65,000 rows. Use :class:`Dataset.xlsx` to avoid this limitation. + + .. admonition:: Binary Warning + + :class:`Dataset.xls` contains binary data, so make sure to write in binary mode:: + + with open('output.xls', 'wb') as f: + f.write(data.xls) + """ + pass + + @property + def xlsx(): + """An Excel '07+ Spreadsheet representation of the :class:`Dataset` object, with :ref:`separators`. Cannot be set. + + .. admonition:: Binary Warning + + :class:`Dataset.xlsx` contains binary data, so make sure to write in binary mode:: + + with open('output.xlsx', 'wb') as f: + f.write(data.xlsx) + """ + pass + + @property + def ods(): + """An OpenDocument Spreadsheet representation of the :class:`Dataset` object, with :ref:`separators`. Cannot be set. + + .. admonition:: Binary Warning + + :class:`Dataset.ods` contains binary data, so make sure to write in binary mode:: + + with open('output.ods', 'wb') as f: + f.write(data.ods) + """ + pass + + @property + def csv(): + """A CSV representation of the :class:`Dataset` object. The top row will contain + headers, if they have been set. Otherwise, the top row will contain + the first row of the dataset. + + A dataset object can also be imported by setting the :class:`Dataset.csv` attribute. :: + + data = tablib.Dataset() + data.csv = 'age, first_name, last_name\\n90, John, Adams' + + Import assumes (for now) that headers exist. + + .. admonition:: Binary Warning for Python 2 + + :class:`Dataset.csv` uses \\r\\n line endings by default so, in Python 2, make + sure to write in binary mode:: + + with open('output.csv', 'wb') as f: + f.write(data.csv) + + If you do not do this, and you export the file on Windows, your + CSV file will open in Excel with a blank line between each row. + + .. admonition:: Line endings for Python 3 + + :class:`Dataset.csv` uses \\r\\n line endings by default so, in Python 3, make + sure to include newline='' otherwise you will get a blank line between each row + when you open the file in Excel:: + + with open('output.csv', 'w', newline='') as f: + f.write(data.csv) + + If you do not do this, and you export the file on Windows, your + CSV file will open in Excel with a blank line between each row. + """ + pass + + @property + def tsv(): + """A TSV representation of the :class:`Dataset` object. The top row will contain + headers, if they have been set. Otherwise, the top row will contain + the first row of the dataset. + + A dataset object can also be imported by setting the :class:`Dataset.tsv` attribute. :: + + data = tablib.Dataset() + data.tsv = 'age\tfirst_name\tlast_name\\n90\tJohn\tAdams' + + Import assumes (for now) that headers exist. + """ + pass + + @property + def yaml(): + """A YAML representation of the :class:`Dataset` object. If headers have been + set, a YAML list of objects will be returned. If no headers have + been set, a YAML list of lists (rows) will be returned instead. + + A dataset object can also be imported by setting the :class:`Dataset.yaml` attribute: :: + + data = tablib.Dataset() + data.yaml = '- {age: 90, first_name: John, last_name: Adams}' + + Import assumes (for now) that headers exist. + """ + pass + + @property + def df(): + """A DataFrame representation of the :class:`Dataset` object. + + A dataset object can also be imported by setting the :class:`Dataset.df` attribute: :: + + data = tablib.Dataset() + data.df = DataFrame(np.random.randn(6,4)) + + Import assumes (for now) that headers exist. + """ + pass + + @property + def json(): + """A JSON representation of the :class:`Dataset` object. If headers have been + set, a JSON list of objects will be returned. If no headers have + been set, a JSON list of lists (rows) will be returned instead. + + A dataset object can also be imported by setting the :class:`Dataset.json` attribute: :: + + data = tablib.Dataset() + data.json = '[{"age": 90, "first_name": "John", "last_name": "Adams"}]' + + Import assumes (for now) that headers exist. + """ + pass + + @property + def html(): + """A HTML table representation of the :class:`Dataset` object. If + headers have been set, they will be used as table headers. + + ..notice:: This method can be used for export only. + """ + pass + + @property + def dbf(): + """A dBASE representation of the :class:`Dataset` object. + + A dataset object can also be imported by setting the + :class:`Dataset.dbf` attribute. :: + + # To import data from an existing DBF file: + data = tablib.Dataset() + data.dbf = open('existing_table.dbf', mode='rb').read() + + # to import data from an ASCII-encoded bytestring: + data = tablib.Dataset() + data.dbf = '<bytestring of tabular data>' + + .. admonition:: Binary Warning + + :class:`Dataset.dbf` contains binary data, so make sure to write in binary mode:: + + with open('output.dbf', 'wb') as f: + f.write(data.dbf) + """ + pass + + @property + def latex(): + """A LaTeX booktabs representation of the :class:`Dataset` object. If a + title has been set, it will be exported as the table caption. + + .. note:: This method can be used for export only. + """ + pass + + @property + def jira(): + """A Jira table representation of the :class:`Dataset` object. + + .. note:: This method can be used for export only. + """ + pass + + # ---- + # Rows + # ---- + + def insert(self, index, row, tags=list()): + """Inserts a row to the :class:`Dataset` at the given index. + + Rows inserted must be the correct size (height or width). + + The default behaviour is to insert the given row to the :class:`Dataset` + object at the given index. + """ + + self._validate(row) + self._data.insert(index, Row(row, tags=tags)) + + def rpush(self, row, tags=list()): + """Adds a row to the end of the :class:`Dataset`. + See :class:`Dataset.insert` for additional documentation. + """ + + self.insert(self.height, row=row, tags=tags) + + def lpush(self, row, tags=list()): + """Adds a row to the top of the :class:`Dataset`. + See :class:`Dataset.insert` for additional documentation. + """ + + self.insert(0, row=row, tags=tags) + + def append(self, row, tags=list()): + """Adds a row to the :class:`Dataset`. + See :class:`Dataset.insert` for additional documentation. + """ + + self.rpush(row, tags) + + def extend(self, rows, tags=list()): + """Adds a list of rows to the :class:`Dataset` using + :class:`Dataset.append` + """ + + for row in rows: + self.append(row, tags) + + def lpop(self): + """Removes and returns the first row of the :class:`Dataset`.""" + + cache = self[0] + del self[0] + + return cache + + def rpop(self): + """Removes and returns the last row of the :class:`Dataset`.""" + + cache = self[-1] + del self[-1] + + return cache + + def pop(self): + """Removes and returns the last row of the :class:`Dataset`.""" + + return self.rpop() + + # ------- + # Columns + # ------- + + def insert_col(self, index, col=None, header=None): + """Inserts a column to the :class:`Dataset` at the given index. + + Columns inserted must be the correct height. + + You can also insert a column of a single callable object, which will + add a new column with the return values of the callable each as an + item in the column. :: + + data.append_col(col=random.randint) + + If inserting a column, and :class:`Dataset.headers` is set, the + header attribute must be set, and will be considered the header for + that row. + + See :ref:`dyncols` for an in-depth example. + + .. versionchanged:: 0.9.0 + If inserting a column, and :class:`Dataset.headers` is set, the + header attribute must be set, and will be considered the header for + that row. + + .. versionadded:: 0.9.0 + If inserting a row, you can add :ref:`tags <tags>` to the row you are inserting. + This gives you the ability to :class:`filter <Dataset.filter>` your + :class:`Dataset` later. + + """ + + if col is None: + col = [] + + # Callable Columns... + if hasattr(col, '__call__'): + col = list(map(col, self._data)) + + col = self._clean_col(col) + self._validate(col=col) + + if self.headers: + # pop the first item off, add to headers + if not header: + raise HeadersNeeded() + + # corner case - if header is set without data + elif header and self.height == 0 and len(col): + raise InvalidDimensions + + self.headers.insert(index, header) + + if self.height and self.width: + + for i, row in enumerate(self._data): + + row.insert(index, col[i]) + self._data[i] = row + else: + self._data = [Row([row]) for row in col] + + def rpush_col(self, col, header=None): + """Adds a column to the end of the :class:`Dataset`. + See :class:`Dataset.insert` for additional documentation. + """ + + self.insert_col(self.width, col, header=header) + + def lpush_col(self, col, header=None): + """Adds a column to the top of the :class:`Dataset`. + See :class:`Dataset.insert` for additional documentation. + """ + + self.insert_col(0, col, header=header) + + def insert_separator(self, index, text='-'): + """Adds a separator to :class:`Dataset` at given index.""" + + sep = (index, text) + self._separators.append(sep) + + def append_separator(self, text='-'): + """Adds a :ref:`separator <separators>` to the :class:`Dataset`.""" + + # change offsets if headers are or aren't defined + if not self.headers: + index = self.height if self.height else 0 + else: + index = (self.height + 1) if self.height else 1 + + self.insert_separator(index, text) + + def append_col(self, col, header=None): + """Adds a column to the :class:`Dataset`. + See :class:`Dataset.insert_col` for additional documentation. + """ + + self.rpush_col(col, header) + + def get_col(self, index): + """Returns the column from the :class:`Dataset` at the given index.""" + + return [row[index] for row in self._data] + + # ---- + # Misc + # ---- + + def add_formatter(self, col, handler): + """Adds a formatter to the :class:`Dataset`. + + .. versionadded:: 0.9.5 + + :param col: column to. Accepts index int or header str. + :param handler: reference to callback function to execute against + each cell value. + """ + + if isinstance(col, unicode): + if col in self.headers: + col = self.headers.index(col) # get 'key' index from each data + else: + raise KeyError + + if not col > self.width: + self._formatters.append((col, handler)) + else: + raise InvalidDatasetIndex + + return True + + def filter(self, tag): + """Returns a new instance of the :class:`Dataset`, excluding any rows + that do not contain the given :ref:`tags <tags>`. + """ + _dset = copy(self) + _dset._data = [row for row in _dset._data if row.has_tag(tag)] + + return _dset + + def sort(self, col, reverse=False): + """Sort a :class:`Dataset` by a specific column, given string (for + header) or integer (for column index). The order can be reversed by + setting ``reverse`` to ``True``. + + Returns a new :class:`Dataset` instance where columns have been + sorted. + """ + + if isinstance(col, (str, unicode)): + + if not self.headers: + raise HeadersNeeded + + _sorted = sorted(self.dict, key=itemgetter(col), reverse=reverse) + _dset = Dataset(headers=self.headers, title=self.title) + + for item in _sorted: + row = [item[key] for key in self.headers] + _dset.append(row=row) + + else: + if self.headers: + col = self.headers[col] + + _sorted = sorted(self.dict, key=itemgetter(col), reverse=reverse) + _dset = Dataset(headers=self.headers, title=self.title) + + for item in _sorted: + if self.headers: + row = [item[key] for key in self.headers] + else: + row = item + _dset.append(row=row) + + return _dset + + def transpose(self): + """Transpose a :class:`Dataset`, turning rows into columns and vice + versa, returning a new ``Dataset`` instance. The first row of the + original instance becomes the new header row.""" + + # Don't transpose if there is no data + if not self: + return + + _dset = Dataset() + # The first element of the headers stays in the headers, + # it is our "hinge" on which we rotate the data + new_headers = [self.headers[0]] + self[self.headers[0]] + + _dset.headers = new_headers + for index, column in enumerate(self.headers): + + if column == self.headers[0]: + # It's in the headers, so skip it + continue + + # Adding the column name as now they're a regular column + # Use `get_col(index)` in case there are repeated values + row_data = [column] + self.get_col(index) + row_data = Row(row_data) + _dset.append(row=row_data) + return _dset + + def stack(self, other): + """Stack two :class:`Dataset` instances together by + joining at the row level, and return new combined + ``Dataset`` instance.""" + + if not isinstance(other, Dataset): + return + + if self.width != other.width: + raise InvalidDimensions + + # Copy the source data + _dset = copy(self) + + rows_to_stack = [row for row in _dset._data] + other_rows = [row for row in other._data] + + rows_to_stack.extend(other_rows) + _dset._data = rows_to_stack + + return _dset + + def stack_cols(self, other): + """Stack two :class:`Dataset` instances together by + joining at the column level, and return a new + combined ``Dataset`` instance. If either ``Dataset`` + has headers set, than the other must as well.""" + + if not isinstance(other, Dataset): + return + + if self.headers or other.headers: + if not self.headers or not other.headers: + raise HeadersNeeded + + if self.height != other.height: + raise InvalidDimensions + + try: + new_headers = self.headers + other.headers + except TypeError: + new_headers = None + + _dset = Dataset() + + for column in self.headers: + _dset.append_col(col=self[column]) + + for column in other.headers: + _dset.append_col(col=other[column]) + + _dset.headers = new_headers + + return _dset + + def remove_duplicates(self): + """Removes all duplicate rows from the :class:`Dataset` object + while maintaining the original order.""" + seen = set() + self._data[:] = [row for row in self._data if not (tuple(row) in seen or seen.add(tuple(row)))] + + def wipe(self): + """Removes all content and headers from the :class:`Dataset` object.""" + self._data = list() + self.__headers = None + + def subset(self, rows=None, cols=None): + """Returns a new instance of the :class:`Dataset`, + including only specified rows and columns. + """ + + # Don't return if no data + if not self: + return + + if rows is None: + rows = list(range(self.height)) + + if cols is None: + cols = list(self.headers) + + #filter out impossible rows and columns + rows = [row for row in rows if row in range(self.height)] + cols = [header for header in cols if header in self.headers] + + _dset = Dataset() + + #filtering rows and columns + _dset.headers = list(cols) + + _dset._data = [] + for row_no, row in enumerate(self._data): + data_row = [] + for key in _dset.headers: + if key in self.headers: + pos = self.headers.index(key) + data_row.append(row[pos]) + else: + raise KeyError + + if row_no in rows: + _dset.append(row=Row(data_row)) + + return _dset + + +class Databook(object): + """A book of :class:`Dataset` objects. + """ + + _formats = {} + + def __init__(self, sets=None): + + if sets is None: + self._datasets = list() + else: + self._datasets = sets + + self._register_formats() + + def __repr__(self): + try: + return '<%s databook>' % (self.title.lower()) + except AttributeError: + return '<databook object>' + + def wipe(self): + """Removes all :class:`Dataset` objects from the :class:`Databook`.""" + self._datasets = [] + + @classmethod + def _register_formats(cls): + """Adds format properties.""" + for fmt in formats.available: + try: + try: + setattr(cls, fmt.title, property(fmt.export_book, fmt.import_book)) + cls._formats[fmt.title] = (fmt.export_book, fmt.import_book) + except AttributeError: + setattr(cls, fmt.title, property(fmt.export_book)) + cls._formats[fmt.title] = (fmt.export_book, None) + + except AttributeError: + cls._formats[fmt.title] = (None, None) + + def sheets(self): + return self._datasets + + def add_sheet(self, dataset): + """Adds given :class:`Dataset` to the :class:`Databook`.""" + if isinstance(dataset, Dataset): + self._datasets.append(dataset) + else: + raise InvalidDatasetType + + def _package(self, ordered=True): + """Packages :class:`Databook` for delivery.""" + collector = [] + + if ordered: + dict_pack = OrderedDict + else: + dict_pack = dict + + for dset in self._datasets: + collector.append(dict_pack( + title = dset.title, + data = dset._package(ordered=ordered) + )) + return collector + + @property + def size(self): + """The number of the :class:`Dataset` objects within :class:`Databook`.""" + return len(self._datasets) + + def load(self, in_stream, format, **kwargs): + """ + Import `in_stream` to the :class:`Databook` object using the `format`. + + :param \\*\\*kwargs: (optional) custom configuration to the format `import_book`. + """ + + if not format: + format = detect_format(in_stream) + + export_book, import_book = self._formats.get(format, (None, None)) + if not import_book: + raise UnsupportedFormat('Format {0} cannot be loaded.'.format(format)) + + import_book(self, in_stream, **kwargs) + return self + + def export(self, format, **kwargs): + """ + Export :class:`Databook` object to `format`. + + :param \\*\\*kwargs: (optional) custom configuration to the format `export_book`. + """ + export_book, import_book = self._formats.get(format, (None, None)) + if not export_book: + raise UnsupportedFormat('Format {0} cannot be exported.'.format(format)) + + return export_book(self, **kwargs) + + +def detect_format(stream): + """Return format name of given stream.""" + for fmt in formats.available: + try: + if fmt.detect(stream): + return fmt.title + except AttributeError: + pass + + +def import_set(stream, format=None, **kwargs): + """Return dataset of given stream.""" + + return Dataset().load(stream, format, **kwargs) + + +def import_book(stream, format=None, **kwargs): + """Return dataset of given stream.""" + + return Databook().load(stream, format, **kwargs) + + +class InvalidDatasetType(Exception): + "Only Datasets can be added to a DataBook" + + +class InvalidDimensions(Exception): + "Invalid size" + + +class InvalidDatasetIndex(Exception): + "Outside of Dataset size" + + +class HeadersNeeded(Exception): + "Header parameter must be given when appending a column in this Dataset." + + +class UnsupportedFormat(NotImplementedError): + "Format is not supported" diff --git a/src/tablib/formats/__init__.py b/src/tablib/formats/__init__.py new file mode 100644 index 0000000..52cf472 --- /dev/null +++ b/src/tablib/formats/__init__.py @@ -0,0 +1,21 @@ +# -*- coding: utf-8 -*- + +""" Tablib - formats +""" + +from . import _csv as csv +from . import _json as json +from . import _xls as xls +from . import _yaml as yaml +from . import _tsv as tsv +from . import _html as html +from . import _xlsx as xlsx +from . import _ods as ods +from . import _dbf as dbf +from . import _latex as latex +from . import _df as df +from . import _rst as rst +from . import _jira as jira + +# xlsx before as xls (xlrd) can also read xlsx +available = (json, xlsx, xls, yaml, csv, dbf, tsv, html, jira, latex, ods, df, rst) diff --git a/src/tablib/formats/_csv.py b/src/tablib/formats/_csv.py new file mode 100644 index 0000000..5c03d6f --- /dev/null +++ b/src/tablib/formats/_csv.py @@ -0,0 +1,59 @@ +# -*- coding: utf-8 -*- + +""" Tablib - *SV Support. +""" + +from tablib.compat import csv, StringIO, unicode + + +title = 'csv' +extensions = ('csv',) + + +DEFAULT_DELIMITER = unicode(',') + + +def export_stream_set(dataset, **kwargs): + """Returns CSV representation of Dataset as file-like.""" + stream = StringIO() + + kwargs.setdefault('delimiter', DEFAULT_DELIMITER) + + _csv = csv.writer(stream, **kwargs) + + for row in dataset._package(dicts=False): + _csv.writerow(row) + + stream.seek(0) + return stream + + +def export_set(dataset, **kwargs): + """Returns CSV representation of Dataset.""" + stream = export_stream_set(dataset, **kwargs) + return stream.getvalue() + + +def import_set(dset, in_stream, headers=True, **kwargs): + """Returns dataset from CSV stream.""" + + dset.wipe() + + kwargs.setdefault('delimiter', DEFAULT_DELIMITER) + + rows = csv.reader(StringIO(in_stream), **kwargs) + for i, row in enumerate(rows): + + if (i == 0) and (headers): + dset.headers = row + elif row: + dset.append(row) + + +def detect(stream, delimiter=DEFAULT_DELIMITER): + """Returns True if given stream is valid CSV.""" + try: + csv.Sniffer().sniff(stream, delimiters=delimiter) + return True + except Exception: + return False diff --git a/src/tablib/formats/_dbf.py b/src/tablib/formats/_dbf.py new file mode 100644 index 0000000..0d1c87b --- /dev/null +++ b/src/tablib/formats/_dbf.py @@ -0,0 +1,87 @@ +# -*- coding: utf-8 -*- + +""" Tablib - DBF Support. +""" +import tempfile +import struct +import os + +from tablib.compat import StringIO +from tablib.compat import dbfpy +from tablib.compat import is_py3 + +if is_py3: + from tablib.packages.dbfpy3 import dbf + from tablib.packages.dbfpy3 import dbfnew + from tablib.packages.dbfpy3 import record as dbfrecord + import io +else: + from tablib.packages.dbfpy import dbf + from tablib.packages.dbfpy import dbfnew + from tablib.packages.dbfpy import record as dbfrecord + + +title = 'dbf' +extensions = ('csv',) + +DEFAULT_ENCODING = 'utf-8' + +def export_set(dataset): + """Returns DBF representation of a Dataset""" + new_dbf = dbfnew.dbf_new() + temp_file, temp_uri = tempfile.mkstemp() + + # create the appropriate fields based on the contents of the first row + first_row = dataset[0] + for fieldname, field_value in zip(dataset.headers, first_row): + if type(field_value) in [int, float]: + new_dbf.add_field(fieldname, 'N', 10, 8) + else: + new_dbf.add_field(fieldname, 'C', 80) + + new_dbf.write(temp_uri) + + dbf_file = dbf.Dbf(temp_uri, readOnly=0) + for row in dataset: + record = dbfrecord.DbfRecord(dbf_file) + for fieldname, field_value in zip(dataset.headers, row): + record[fieldname] = field_value + record.store() + + dbf_file.close() + dbf_stream = open(temp_uri, 'rb') + if is_py3: + stream = io.BytesIO(dbf_stream.read()) + else: + stream = StringIO(dbf_stream.read()) + dbf_stream.close() + os.close(temp_file) + os.remove(temp_uri) + return stream.getvalue() + +def import_set(dset, in_stream, headers=True): + """Returns a dataset from a DBF stream.""" + + dset.wipe() + if is_py3: + _dbf = dbf.Dbf(io.BytesIO(in_stream)) + else: + _dbf = dbf.Dbf(StringIO(in_stream)) + dset.headers = _dbf.fieldNames + for record in range(_dbf.recordCount): + row = [_dbf[record][f] for f in _dbf.fieldNames] + dset.append(row) + +def detect(stream): + """Returns True if the given stream is valid DBF""" + #_dbf = dbf.Table(StringIO(stream)) + try: + if is_py3: + if type(stream) is not bytes: + stream = bytes(stream, 'utf-8') + _dbf = dbf.Dbf(io.BytesIO(stream), readOnly=True) + else: + _dbf = dbf.Dbf(StringIO(stream), readOnly=True) + return True + except Exception: + return False diff --git a/src/tablib/formats/_df.py b/src/tablib/formats/_df.py new file mode 100644 index 0000000..0c7ebec --- /dev/null +++ b/src/tablib/formats/_df.py @@ -0,0 +1,43 @@ +""" Tablib - DataFrame Support. +""" + +import sys +from io import BytesIO + +try: + from pandas import DataFrame +except ImportError: + DataFrame = None + +import tablib + +from tablib.compat import unicode + +title = 'df' +extensions = ('df', ) + +def detect(stream): + """Returns True if given stream is a DataFrame.""" + if DataFrame is None: + return False + try: + DataFrame(stream) + return True + except ValueError: + return False + + +def export_set(dset, index=None): + """Returns DataFrame representation of DataBook.""" + if DataFrame is None: + raise NotImplementedError( + 'DataFrame Format requires `pandas` to be installed.' + ' Try `pip install tablib[pandas]`.') + dataframe = DataFrame(dset.dict, columns=dset.headers) + return dataframe + + +def import_set(dset, in_stream): + """Returns dataset from DataFrame.""" + dset.wipe() + dset.dict = in_stream.to_dict(orient='records') diff --git a/src/tablib/formats/_html.py b/src/tablib/formats/_html.py new file mode 100644 index 0000000..ac72bd6 --- /dev/null +++ b/src/tablib/formats/_html.py @@ -0,0 +1,65 @@ +# -*- coding: utf-8 -*- + +""" Tablib - HTML export support. +""" + +import codecs +import sys +from io import BytesIO + +from MarkupPy import markup +import tablib +from tablib.compat import unicode + +BOOK_ENDINGS = 'h3' + +title = 'html' +extensions = ('html', ) + + +def export_set(dataset): + """HTML representation of a Dataset.""" + + stream = BytesIO() + + page = markup.page() + page.table.open() + + if dataset.headers is not None: + new_header = [item if item is not None else '' for item in dataset.headers] + + page.thead.open() + headers = markup.oneliner.th(new_header) + page.tr(headers) + page.thead.close() + + for row in dataset: + new_row = [item if item is not None else '' for item in row] + + html_row = markup.oneliner.td(new_row) + page.tr(html_row) + + page.table.close() + + # Allow unicode characters in output + wrapper = codecs.getwriter("utf8")(stream) + wrapper.writelines(unicode(page)) + + return stream.getvalue().decode('utf-8') + + +def export_book(databook): + """HTML representation of a Databook.""" + + stream = BytesIO() + + # Allow unicode characters in output + wrapper = codecs.getwriter("utf8")(stream) + + for i, dset in enumerate(databook._datasets): + title = (dset.title if dset.title else 'Set %s' % (i)) + wrapper.write('<%s>%s</%s>\n' % (BOOK_ENDINGS, title, BOOK_ENDINGS)) + wrapper.write(dset.html) + wrapper.write('\n') + + return stream.getvalue().decode('utf-8') diff --git a/src/tablib/formats/_jira.py b/src/tablib/formats/_jira.py new file mode 100644 index 0000000..55fce52 --- /dev/null +++ b/src/tablib/formats/_jira.py @@ -0,0 +1,39 @@ +# -*- coding: utf-8 -*- + +"""Tablib - Jira table export support. + + Generates a Jira table from the dataset. +""" +from tablib.compat import unicode + +title = 'jira' + + +def export_set(dataset): + """Formats the dataset according to the Jira table syntax: + + ||heading 1||heading 2||heading 3|| + |col A1|col A2|col A3| + |col B1|col B2|col B3| + + :param dataset: dataset to serialize + :type dataset: tablib.core.Dataset + """ + + header = _get_header(dataset.headers) if dataset.headers else '' + body = _get_body(dataset) + return '%s\n%s' % (header, body) if header else body + + +def _get_body(dataset): + return '\n'.join([_serialize_row(row) for row in dataset]) + + +def _get_header(headers): + return _serialize_row(headers, delimiter='||') + + +def _serialize_row(row, delimiter='|'): + return '%s%s%s' % (delimiter, + delimiter.join([unicode(item) if item else ' ' for item in row]), + delimiter) diff --git a/src/tablib/formats/_json.py b/src/tablib/formats/_json.py new file mode 100644 index 0000000..009fb3c --- /dev/null +++ b/src/tablib/formats/_json.py @@ -0,0 +1,59 @@ +# -*- coding: utf-8 -*- + +""" Tablib - JSON Support +""" +import decimal +import json +from uuid import UUID + +import tablib + + +title = 'json' +extensions = ('json', 'jsn') + + +def serialize_objects_handler(obj): + if isinstance(obj, (decimal.Decimal, UUID)): + return str(obj) + elif hasattr(obj, 'isoformat'): + return obj.isoformat() + else: + return obj + + +def export_set(dataset): + """Returns JSON representation of Dataset.""" + return json.dumps(dataset.dict, default=serialize_objects_handler) + + +def export_book(databook): + """Returns JSON representation of Databook.""" + return json.dumps(databook._package(), default=serialize_objects_handler) + + +def import_set(dset, in_stream): + """Returns dataset from JSON stream.""" + + dset.wipe() + dset.dict = json.loads(in_stream) + + +def import_book(dbook, in_stream): + """Returns databook from JSON stream.""" + + dbook.wipe() + for sheet in json.loads(in_stream): + data = tablib.Dataset() + data.title = sheet['title'] + data.dict = sheet['data'] + dbook.add_sheet(data) + + +def detect(stream): + """Returns True if given stream is valid JSON.""" + try: + json.loads(stream) + return True + except (TypeError, ValueError): + return False diff --git a/src/tablib/formats/_latex.py b/src/tablib/formats/_latex.py new file mode 100644 index 0000000..44ee101 --- /dev/null +++ b/src/tablib/formats/_latex.py @@ -0,0 +1,134 @@ +# -*- coding: utf-8 -*- + +"""Tablib - LaTeX table export support. + + Generates a LaTeX booktabs-style table from the dataset. +""" +import re + +from tablib.compat import unicode + +title = 'latex' +extensions = ('tex',) + +TABLE_TEMPLATE = """\ +%% Note: add \\usepackage{booktabs} to your preamble +%% +\\begin{table}[!htbp] + \\centering + %(CAPTION)s + \\begin{tabular}{%(COLSPEC)s} + \\toprule +%(HEADER)s + %(MIDRULE)s +%(BODY)s + \\bottomrule + \\end{tabular} +\\end{table} +""" + +TEX_RESERVED_SYMBOLS_MAP = dict([ + ('\\', '\\textbackslash{}'), + ('{', '\\{'), + ('}', '\\}'), + ('$', '\\$'), + ('&', '\\&'), + ('#', '\\#'), + ('^', '\\textasciicircum{}'), + ('_', '\\_'), + ('~', '\\textasciitilde{}'), + ('%', '\\%'), +]) + +TEX_RESERVED_SYMBOLS_RE = re.compile( + '(%s)' % '|'.join(map(re.escape, TEX_RESERVED_SYMBOLS_MAP.keys()))) + + +def export_set(dataset): + """Returns LaTeX representation of dataset + + :param dataset: dataset to serialize + :type dataset: tablib.core.Dataset + """ + + caption = '\\caption{%s}' % dataset.title if dataset.title else '%' + colspec = _colspec(dataset.width) + header = _serialize_row(dataset.headers) if dataset.headers else '' + midrule = _midrule(dataset.width) + body = '\n'.join([_serialize_row(row) for row in dataset]) + return TABLE_TEMPLATE % dict(CAPTION=caption, COLSPEC=colspec, + HEADER=header, MIDRULE=midrule, BODY=body) + + +def _colspec(dataset_width): + """Generates the column specification for the LaTeX `tabular` environment + based on the dataset width. + + The first column is justified to the left, all further columns are aligned + to the right. + + .. note:: This is only a heuristic and most probably has to be fine-tuned + post export. Column alignment should depend on the data type, e.g., textual + content should usually be aligned to the left while numeric content almost + always should be aligned to the right. + + :param dataset_width: width of the dataset + """ + + spec = 'l' + for _ in range(1, dataset_width): + spec += 'r' + return spec + + +def _midrule(dataset_width): + """Generates the table `midrule`, which may be composed of several + `cmidrules`. + + :param dataset_width: width of the dataset to serialize + """ + + if not dataset_width or dataset_width == 1: + return '\\midrule' + return ' '.join([_cmidrule(colindex, dataset_width) for colindex in + range(1, dataset_width + 1)]) + + +def _cmidrule(colindex, dataset_width): + """Generates the `cmidrule` for a single column with appropriate trimming + based on the column position. + + :param colindex: Column index + :param dataset_width: width of the dataset + """ + + rule = '\\cmidrule(%s){%d-%d}' + if colindex == 1: + # Rule of first column is trimmed on the right + return rule % ('r', colindex, colindex) + if colindex == dataset_width: + # Rule of last column is trimmed on the left + return rule % ('l', colindex, colindex) + # Inner columns are trimmed on the left and right + return rule % ('lr', colindex, colindex) + + +def _serialize_row(row): + """Returns string representation of a single row. + + :param row: single dataset row + """ + + new_row = [_escape_tex_reserved_symbols(unicode(item)) if item else '' for + item in row] + return 6 * ' ' + ' & '.join(new_row) + ' \\\\' + + +def _escape_tex_reserved_symbols(input): + """Escapes all TeX reserved symbols ('_', '~', etc.) in a string. + + :param input: String to escape + """ + def replace(match): + return TEX_RESERVED_SYMBOLS_MAP[match.group()] + return TEX_RESERVED_SYMBOLS_RE.sub(replace, input) diff --git a/src/tablib/formats/_ods.py b/src/tablib/formats/_ods.py new file mode 100644 index 0000000..dbf57c4 --- /dev/null +++ b/src/tablib/formats/_ods.py @@ -0,0 +1,104 @@ +# -*- coding: utf-8 -*- + +""" Tablib - ODF Support. +""" + +from io import BytesIO +from odf import opendocument, style, table, text +from tablib.compat import unicode + +title = 'ods' +extensions = ('ods',) + +bold = style.Style(name="bold", family="paragraph") +bold.addElement(style.TextProperties(fontweight="bold", fontweightasian="bold", fontweightcomplex="bold")) + +def export_set(dataset): + """Returns ODF representation of Dataset.""" + + wb = opendocument.OpenDocumentSpreadsheet() + wb.automaticstyles.addElement(bold) + + ws = table.Table(name=dataset.title if dataset.title else 'Tablib Dataset') + wb.spreadsheet.addElement(ws) + dset_sheet(dataset, ws) + + stream = BytesIO() + wb.save(stream) + return stream.getvalue() + + +def export_book(databook): + """Returns ODF representation of DataBook.""" + + wb = opendocument.OpenDocumentSpreadsheet() + wb.automaticstyles.addElement(bold) + + for i, dset in enumerate(databook._datasets): + ws = table.Table(name=dset.title if dset.title else 'Sheet%s' % (i)) + wb.spreadsheet.addElement(ws) + dset_sheet(dset, ws) + + stream = BytesIO() + wb.save(stream) + return stream.getvalue() + + +def dset_sheet(dataset, ws): + """Completes given worksheet from given Dataset.""" + _package = dataset._package(dicts=False) + + for i, sep in enumerate(dataset._separators): + _offset = i + _package.insert((sep[0] + _offset), (sep[1],)) + + for i, row in enumerate(_package): + row_number = i + 1 + odf_row = table.TableRow(stylename=bold, defaultcellstylename='bold') + for j, col in enumerate(row): + try: + col = unicode(col, errors='ignore') + except TypeError: + ## col is already unicode + pass + ws.addElement(table.TableColumn()) + + # bold headers + if (row_number == 1) and dataset.headers: + odf_row.setAttribute('stylename', bold) + ws.addElement(odf_row) + cell = table.TableCell() + p = text.P() + p.addElement(text.Span(text=col, stylename=bold)) + cell.addElement(p) + odf_row.addElement(cell) + + # wrap the rest + else: + try: + if '\n' in col: + ws.addElement(odf_row) + cell = table.TableCell() + cell.addElement(text.P(text=col)) + odf_row.addElement(cell) + else: + ws.addElement(odf_row) + cell = table.TableCell() + cell.addElement(text.P(text=col)) + odf_row.addElement(cell) + except TypeError: + ws.addElement(odf_row) + cell = table.TableCell() + cell.addElement(text.P(text=col)) + odf_row.addElement(cell) + + +def detect(stream): + if isinstance(stream, bytes): + # load expects a file-like object. + stream = BytesIO(stream) + try: + opendocument.load(stream) + return True + except Exception: + return False diff --git a/src/tablib/formats/_rst.py b/src/tablib/formats/_rst.py new file mode 100644 index 0000000..4b53ad7 --- /dev/null +++ b/src/tablib/formats/_rst.py @@ -0,0 +1,273 @@ +# -*- coding: utf-8 -*- + +""" Tablib - reStructuredText Support +""" +from __future__ import division +from __future__ import print_function +from __future__ import unicode_literals + +from textwrap import TextWrapper + +from tablib.compat import ( + median, + unicode, + izip_longest, +) + + +title = 'rst' +extensions = ('rst',) + + +MAX_TABLE_WIDTH = 80 # Roughly. It may be wider to avoid breaking words. + + +JUSTIFY_LEFT = 'left' +JUSTIFY_CENTER = 'center' +JUSTIFY_RIGHT = 'right' +JUSTIFY_VALUES = (JUSTIFY_LEFT, JUSTIFY_CENTER, JUSTIFY_RIGHT) + + +def to_unicode(value): + if isinstance(value, bytes): + return value.decode('utf-8') + return unicode(value) + + +def _max_word_len(text): + """ + Return the length of the longest word in `text`. + + + >>> _max_word_len('Python Module for Tabular Datasets') + 8 + + """ + return max((len(word) for word in text.split())) + + +def _get_column_string_lengths(dataset): + """ + Returns a list of string lengths of each column, and a list of + maximum word lengths. + """ + if dataset.headers: + column_lengths = [[len(h)] for h in dataset.headers] + word_lens = [_max_word_len(h) for h in dataset.headers] + else: + column_lengths = [[] for _ in range(dataset.width)] + word_lens = [0 for _ in range(dataset.width)] + for row in dataset.dict: + values = iter(row.values() if hasattr(row, 'values') else row) + for i, val in enumerate(values): + text = to_unicode(val) + column_lengths[i].append(len(text)) + word_lens[i] = max(word_lens[i], _max_word_len(text)) + return column_lengths, word_lens + + +def _row_to_lines(values, widths, wrapper, sep='|', justify=JUSTIFY_LEFT): + """ + Returns a table row of wrapped values as a list of lines + """ + if justify not in JUSTIFY_VALUES: + raise ValueError('Value of "justify" must be one of "{}"'.format( + '", "'.join(JUSTIFY_VALUES) + )) + if justify == JUSTIFY_LEFT: + just = lambda text, width: text.ljust(width) + elif justify == JUSTIFY_CENTER: + just = lambda text, width: text.center(width) + else: + just = lambda text, width: text.rjust(width) + lpad = sep + ' ' if sep else '' + rpad = ' ' + sep if sep else '' + pad = ' ' + sep + ' ' + cells = [] + for value, width in zip(values, widths): + wrapper.width = width + text = to_unicode(value) + cell = wrapper.wrap(text) + cells.append(cell) + lines = izip_longest(*cells, fillvalue='') + lines = ( + (just(cell_line, widths[i]) for i, cell_line in enumerate(line)) + for line in lines + ) + lines = [''.join((lpad, pad.join(line), rpad)) for line in lines] + return lines + + +def _get_column_widths(dataset, max_table_width=MAX_TABLE_WIDTH, pad_len=3): + """ + Returns a list of column widths proportional to the median length + of the text in their cells. + """ + str_lens, word_lens = _get_column_string_lengths(dataset) + median_lens = [int(median(lens)) for lens in str_lens] + total = sum(median_lens) + if total > max_table_width - (pad_len * len(median_lens)): + column_widths = (max_table_width * l // total for l in median_lens) + else: + column_widths = (l for l in median_lens) + # Allow for separator and padding: + column_widths = (w - pad_len if w > pad_len else w for w in column_widths) + # Rather widen table than break words: + column_widths = [max(w, l) for w, l in zip(column_widths, word_lens)] + return column_widths + + +def export_set_as_simple_table(dataset, column_widths=None): + """ + Returns reStructuredText grid table representation of dataset. + """ + lines = [] + wrapper = TextWrapper() + if column_widths is None: + column_widths = _get_column_widths(dataset, pad_len=2) + border = ' '.join(['=' * w for w in column_widths]) + + lines.append(border) + if dataset.headers: + lines.extend(_row_to_lines( + dataset.headers, + column_widths, + wrapper, + sep='', + justify=JUSTIFY_CENTER, + )) + lines.append(border) + for row in dataset.dict: + values = iter(row.values() if hasattr(row, 'values') else row) + lines.extend(_row_to_lines(values, column_widths, wrapper, '')) + lines.append(border) + return '\n'.join(lines) + + +def export_set_as_grid_table(dataset, column_widths=None): + """ + Returns reStructuredText grid table representation of dataset. + + + >>> from tablib import Dataset + >>> from tablib.formats import rst + >>> bits = ((0, 0), (1, 0), (0, 1), (1, 1)) + >>> data = Dataset() + >>> data.headers = ['A', 'B', 'A and B'] + >>> for a, b in bits: + ... data.append([bool(a), bool(b), bool(a * b)]) + >>> print(rst.export_set(data, force_grid=True)) + +-------+-------+-------+ + | A | B | A and | + | | | B | + +=======+=======+=======+ + | False | False | False | + +-------+-------+-------+ + | True | False | False | + +-------+-------+-------+ + | False | True | False | + +-------+-------+-------+ + | True | True | True | + +-------+-------+-------+ + + """ + lines = [] + wrapper = TextWrapper() + if column_widths is None: + column_widths = _get_column_widths(dataset) + header_sep = '+=' + '=+='.join(['=' * w for w in column_widths]) + '=+' + row_sep = '+-' + '-+-'.join(['-' * w for w in column_widths]) + '-+' + + lines.append(row_sep) + if dataset.headers: + lines.extend(_row_to_lines( + dataset.headers, + column_widths, + wrapper, + justify=JUSTIFY_CENTER, + )) + lines.append(header_sep) + for row in dataset.dict: + values = iter(row.values() if hasattr(row, 'values') else row) + lines.extend(_row_to_lines(values, column_widths, wrapper)) + lines.append(row_sep) + return '\n'.join(lines) + + +def _use_simple_table(head0, col0, width0): + """ + Use a simple table if the text in the first column is never wrapped + + + >>> _use_simple_table('menu', ['egg', 'bacon'], 10) + True + >>> _use_simple_table(None, ['lobster thermidor', 'spam'], 10) + False + + """ + if head0 is not None: + head0 = to_unicode(head0) + if len(head0) > width0: + return False + for cell in col0: + cell = to_unicode(cell) + if len(cell) > width0: + return False + return True + + +def export_set(dataset, **kwargs): + """ + Returns reStructuredText table representation of dataset. + + Returns a simple table if the text in the first column is never + wrapped, otherwise returns a grid table. + + + >>> from tablib import Dataset + >>> bits = ((0, 0), (1, 0), (0, 1), (1, 1)) + >>> data = Dataset() + >>> data.headers = ['A', 'B', 'A and B'] + >>> for a, b in bits: + ... data.append([bool(a), bool(b), bool(a * b)]) + >>> table = data.rst + >>> table.split('\\n') == [ + ... '===== ===== =====', + ... ' A B A and', + ... ' B ', + ... '===== ===== =====', + ... 'False False False', + ... 'True False False', + ... 'False True False', + ... 'True True True ', + ... '===== ===== =====', + ... ] + True + + """ + if not dataset.dict: + return '' + force_grid = kwargs.get('force_grid', False) + max_table_width = kwargs.get('max_table_width', MAX_TABLE_WIDTH) + column_widths = _get_column_widths(dataset, max_table_width) + + use_simple_table = _use_simple_table( + dataset.headers[0] if dataset.headers else None, + dataset.get_col(0), + column_widths[0], + ) + if use_simple_table and not force_grid: + return export_set_as_simple_table(dataset, column_widths) + else: + return export_set_as_grid_table(dataset, column_widths) + + +def export_book(databook): + """ + reStructuredText representation of a Databook. + + Tables are separated by a blank line. All tables use the grid + format. + """ + return '\n\n'.join(export_set(dataset, force_grid=True) + for dataset in databook._datasets) diff --git a/src/tablib/formats/_tsv.py b/src/tablib/formats/_tsv.py new file mode 100644 index 0000000..1c6d6a1 --- /dev/null +++ b/src/tablib/formats/_tsv.py @@ -0,0 +1,30 @@ +# -*- coding: utf-8 -*- + +""" Tablib - TSV (Tab Separated Values) Support. +""" + +from tablib.compat import unicode +from tablib.formats._csv import ( + export_set as export_set_wrapper, + import_set as import_set_wrapper, + detect as detect_wrapper, +) + +title = 'tsv' +extensions = ('tsv',) + +DELIMITER = unicode('\t') + +def export_set(dataset): + """Returns TSV representation of Dataset.""" + return export_set_wrapper(dataset, delimiter=DELIMITER) + + +def import_set(dset, in_stream, headers=True): + """Returns dataset from TSV stream.""" + return import_set_wrapper(dset, in_stream, headers=headers, delimiter=DELIMITER) + + +def detect(stream): + """Returns True if given stream is valid TSV.""" + return detect_wrapper(stream, delimiter=DELIMITER) diff --git a/src/tablib/formats/_xls.py b/src/tablib/formats/_xls.py new file mode 100644 index 0000000..88e8636 --- /dev/null +++ b/src/tablib/formats/_xls.py @@ -0,0 +1,137 @@ +# -*- coding: utf-8 -*- + +""" Tablib - XLS Support. +""" + +import sys +from io import BytesIO + +from tablib.compat import xrange +import tablib +import xlrd +import xlwt +from xlrd.biffh import XLRDError + +title = 'xls' +extensions = ('xls',) + +# special styles +wrap = xlwt.easyxf("alignment: wrap on") +bold = xlwt.easyxf("font: bold on") + + +def detect(stream): + """Returns True if given stream is a readable excel file.""" + try: + xlrd.open_workbook(file_contents=stream) + return True + except Exception: + pass + try: + xlrd.open_workbook(file_contents=stream.read()) + return True + except Exception: + pass + try: + xlrd.open_workbook(filename=stream) + return True + except Exception: + return False + + +def export_set(dataset): + """Returns XLS representation of Dataset.""" + + wb = xlwt.Workbook(encoding='utf8') + ws = wb.add_sheet(dataset.title if dataset.title else 'Tablib Dataset') + + dset_sheet(dataset, ws) + + stream = BytesIO() + wb.save(stream) + return stream.getvalue() + + +def export_book(databook): + """Returns XLS representation of DataBook.""" + + wb = xlwt.Workbook(encoding='utf8') + + for i, dset in enumerate(databook._datasets): + ws = wb.add_sheet(dset.title if dset.title else 'Sheet%s' % (i)) + + dset_sheet(dset, ws) + + stream = BytesIO() + wb.save(stream) + return stream.getvalue() + + +def import_set(dset, in_stream, headers=True): + """Returns databook from XLS stream.""" + + dset.wipe() + + xls_book = xlrd.open_workbook(file_contents=in_stream) + sheet = xls_book.sheet_by_index(0) + + dset.title = sheet.name + + for i in xrange(sheet.nrows): + if (i == 0) and (headers): + dset.headers = sheet.row_values(0) + else: + dset.append(sheet.row_values(i)) + +def import_book(dbook, in_stream, headers=True): + """Returns databook from XLS stream.""" + + dbook.wipe() + + xls_book = xlrd.open_workbook(file_contents=in_stream) + + for sheet in xls_book.sheets(): + data = tablib.Dataset() + data.title = sheet.name + + for i in xrange(sheet.nrows): + if (i == 0) and (headers): + data.headers = sheet.row_values(0) + else: + data.append(sheet.row_values(i)) + + dbook.add_sheet(data) + + +def dset_sheet(dataset, ws): + """Completes given worksheet from given Dataset.""" + _package = dataset._package(dicts=False) + + for i, sep in enumerate(dataset._separators): + _offset = i + _package.insert((sep[0] + _offset), (sep[1],)) + + for i, row in enumerate(_package): + for j, col in enumerate(row): + + # bold headers + if (i == 0) and dataset.headers: + ws.write(i, j, col, bold) + + # frozen header row + ws.panes_frozen = True + ws.horz_split_pos = 1 + + # bold separators + elif len(row) < dataset.width: + ws.write(i, j, col, bold) + + # wrap the rest + else: + try: + if '\n' in col: + ws.write(i, j, col, wrap) + else: + ws.write(i, j, col) + except TypeError: + ws.write(i, j, col) diff --git a/src/tablib/formats/_xlsx.py b/src/tablib/formats/_xlsx.py new file mode 100644 index 0000000..f8f21c2 --- /dev/null +++ b/src/tablib/formats/_xlsx.py @@ -0,0 +1,144 @@ +# -*- coding: utf-8 -*- + +""" Tablib - XLSX Support. +""" + +import sys +from io import BytesIO + +import openpyxl +import tablib + +Workbook = openpyxl.workbook.Workbook +ExcelWriter = openpyxl.writer.excel.ExcelWriter +get_column_letter = openpyxl.utils.get_column_letter + +from tablib.compat import unicode + + +title = 'xlsx' +extensions = ('xlsx',) + + +def detect(stream): + """Returns True if given stream is a readable excel file.""" + if isinstance(stream, bytes): + # load_workbook expects a file-like object. + stream = BytesIO(stream) + try: + openpyxl.reader.excel.load_workbook(stream, read_only=True) + return True + except Exception: + return False + +def export_set(dataset, freeze_panes=True): + """Returns XLSX representation of Dataset.""" + + wb = Workbook() + ws = wb.worksheets[0] + ws.title = dataset.title if dataset.title else 'Tablib Dataset' + + dset_sheet(dataset, ws, freeze_panes=freeze_panes) + + stream = BytesIO() + wb.save(stream) + return stream.getvalue() + + +def export_book(databook, freeze_panes=True): + """Returns XLSX representation of DataBook.""" + + wb = Workbook() + for sheet in wb.worksheets: + wb.remove(sheet) + for i, dset in enumerate(databook._datasets): + ws = wb.create_sheet() + ws.title = dset.title if dset.title else 'Sheet%s' % (i) + + dset_sheet(dset, ws, freeze_panes=freeze_panes) + + stream = BytesIO() + wb.save(stream) + return stream.getvalue() + + +def import_set(dset, in_stream, headers=True): + """Returns databook from XLS stream.""" + + dset.wipe() + + xls_book = openpyxl.reader.excel.load_workbook(BytesIO(in_stream), read_only=True) + sheet = xls_book.active + + dset.title = sheet.title + + for i, row in enumerate(sheet.rows): + row_vals = [c.value for c in row] + if (i == 0) and (headers): + dset.headers = row_vals + else: + dset.append(row_vals) + + +def import_book(dbook, in_stream, headers=True): + """Returns databook from XLS stream.""" + + dbook.wipe() + + xls_book = openpyxl.reader.excel.load_workbook(BytesIO(in_stream), read_only=True) + + for sheet in xls_book.worksheets: + data = tablib.Dataset() + data.title = sheet.title + + for i, row in enumerate(sheet.rows): + row_vals = [c.value for c in row] + if (i == 0) and (headers): + data.headers = row_vals + else: + data.append(row_vals) + + dbook.add_sheet(data) + + +def dset_sheet(dataset, ws, freeze_panes=True): + """Completes given worksheet from given Dataset.""" + _package = dataset._package(dicts=False) + + for i, sep in enumerate(dataset._separators): + _offset = i + _package.insert((sep[0] + _offset), (sep[1],)) + + bold = openpyxl.styles.Font(bold=True) + wrap_text = openpyxl.styles.Alignment(wrap_text=True) + + for i, row in enumerate(_package): + row_number = i + 1 + for j, col in enumerate(row): + col_idx = get_column_letter(j + 1) + cell = ws['%s%s' % (col_idx, row_number)] + + # bold headers + if (row_number == 1) and dataset.headers: + cell.font = bold + if freeze_panes: + # Export Freeze only after first Line + ws.freeze_panes = 'A2' + + # bold separators + elif len(row) < dataset.width: + cell.font = bold + + # wrap the rest + else: + try: + str_col_value = unicode(col) + except TypeError: + str_col_value = '' + if '\n' in str_col_value: + cell.alignment = wrap_text + + try: + cell.value = col + except (ValueError, TypeError): + cell.value = unicode(col) diff --git a/src/tablib/formats/_yaml.py b/src/tablib/formats/_yaml.py new file mode 100644 index 0000000..3d17baf --- /dev/null +++ b/src/tablib/formats/_yaml.py @@ -0,0 +1,53 @@ +# -*- coding: utf-8 -*- + +""" Tablib - YAML Support. +""" + +import tablib +import yaml + +title = 'yaml' +extensions = ('yaml', 'yml') + + +def export_set(dataset): + """Returns YAML representation of Dataset.""" + + return yaml.safe_dump(dataset._package(ordered=False)) + + +def export_book(databook): + """Returns YAML representation of Databook.""" + return yaml.safe_dump(databook._package(ordered=False)) + + +def import_set(dset, in_stream): + """Returns dataset from YAML stream.""" + + dset.wipe() + dset.dict = yaml.safe_load(in_stream) + + +def import_book(dbook, in_stream): + """Returns databook from YAML stream.""" + + dbook.wipe() + + for sheet in yaml.safe_load(in_stream): + data = tablib.Dataset() + data.title = sheet['title'] + data.dict = sheet['data'] + dbook.add_sheet(data) + + +def detect(stream): + """Returns True if given stream is valid YAML.""" + try: + _yaml = yaml.safe_load(stream) + if isinstance(_yaml, (list, tuple, dict)): + return True + else: + return False + except (yaml.parser.ParserError, yaml.reader.ReaderError, + yaml.scanner.ScannerError): + return False diff --git a/src/tablib/packages/__init__.py b/src/tablib/packages/__init__.py new file mode 100644 index 0000000..e69de29 --- /dev/null +++ b/src/tablib/packages/__init__.py diff --git a/src/tablib/packages/dbfpy/__init__.py b/src/tablib/packages/dbfpy/__init__.py new file mode 100644 index 0000000..e69de29 --- /dev/null +++ b/src/tablib/packages/dbfpy/__init__.py diff --git a/src/tablib/packages/dbfpy/dbf.py b/src/tablib/packages/dbfpy/dbf.py new file mode 100644 index 0000000..8147d0e --- /dev/null +++ b/src/tablib/packages/dbfpy/dbf.py @@ -0,0 +1,297 @@ +#! /usr/bin/env python +"""DBF accessing helpers. + +FIXME: more documentation needed + +Examples: + + Create new table, setup structure, add records: + + dbf = Dbf(filename, new=True) + dbf.addField( + ("NAME", "C", 15), + ("SURNAME", "C", 25), + ("INITIALS", "C", 10), + ("BIRTHDATE", "D"), + ) + for (n, s, i, b) in ( + ("John", "Miller", "YC", (1980, 10, 11)), + ("Andy", "Larkin", "", (1980, 4, 11)), + ): + rec = dbf.newRecord() + rec["NAME"] = n + rec["SURNAME"] = s + rec["INITIALS"] = i + rec["BIRTHDATE"] = b + rec.store() + dbf.close() + + Open existed dbf, read some data: + + dbf = Dbf(filename, True) + for rec in dbf: + for fldName in dbf.fieldNames: + print('%s:\t %s (%s)' % (fldName, rec[fldName], + type(rec[fldName]))) + print() + dbf.close() + +""" +"""History (most recent first): +11-feb-2007 [als] export INVALID_VALUE; + Dbf: added .ignoreErrors, .INVALID_VALUE +04-jul-2006 [als] added export declaration +20-dec-2005 [yc] removed fromStream and newDbf methods: + use argument of __init__ call must be used instead; + added class fields pointing to the header and + record classes. +17-dec-2005 [yc] split to several modules; reimplemented +13-dec-2005 [yc] adapted to the changes of the `strutil` module. +13-sep-2002 [als] support FoxPro Timestamp datatype +15-nov-1999 [jjk] documentation updates, add demo +24-aug-1998 [jjk] add some encodeValue methods (not tested), other tweaks +08-jun-1998 [jjk] fix problems, add more features +20-feb-1998 [jjk] fix problems, add more features +19-feb-1998 [jjk] add create/write capabilities +18-feb-1998 [jjk] from dbfload.py +""" + +__version__ = "$Revision: 1.7 $"[11:-2] +__date__ = "$Date: 2007/02/11 09:23:13 $"[7:-2] +__author__ = "Jeff Kunce <kuncej@mail.conservation.state.mo.us>" + +__all__ = ["Dbf"] + +from . import header +from . import record +from utils import INVALID_VALUE + + +class Dbf(object): + """DBF accessor. + + FIXME: + docs and examples needed (dont' forget to tell + about problems adding new fields on the fly) + + Implementation notes: + ``_new`` field is used to indicate whether this is + a new data table. `addField` could be used only for + the new tables! If at least one record was appended + to the table it's structure couldn't be changed. + + """ + + __slots__ = ("name", "header", "stream", + "_changed", "_new", "_ignore_errors") + + HeaderClass = header.DbfHeader + RecordClass = record.DbfRecord + INVALID_VALUE = INVALID_VALUE + + # initialization and creation helpers + + def __init__(self, f, readOnly=False, new=False, ignoreErrors=False): + """Initialize instance. + + Arguments: + f: + Filename or file-like object. + new: + True if new data table must be created. Assume + data table exists if this argument is False. + readOnly: + if ``f`` argument is a string file will + be opend in read-only mode; in other cases + this argument is ignored. This argument is ignored + even if ``new`` argument is True. + headerObj: + `header.DbfHeader` instance or None. If this argument + is None, new empty header will be used with the + all fields set by default. + ignoreErrors: + if set, failing field value conversion will return + ``INVALID_VALUE`` instead of raising conversion error. + + """ + if isinstance(f, basestring): + # a filename + self.name = f + if new: + # new table (table file must be + # created or opened and truncated) + self.stream = file(f, "w+b") + else: + # tabe file must exist + self.stream = file(f, ("r+b", "rb")[bool(readOnly)]) + else: + # a stream + self.name = getattr(f, "name", "") + self.stream = f + if new: + # if this is a new table, header will be empty + self.header = self.HeaderClass() + else: + # or instantiated using stream + self.header = self.HeaderClass.fromStream(self.stream) + self.ignoreErrors = ignoreErrors + self._new = bool(new) + self._changed = False + + # properties + + closed = property(lambda self: self.stream.closed) + recordCount = property(lambda self: self.header.recordCount) + fieldNames = property( + lambda self: [_fld.name for _fld in self.header.fields]) + fieldDefs = property(lambda self: self.header.fields) + changed = property(lambda self: self._changed or self.header.changed) + + def ignoreErrors(self, value): + """Update `ignoreErrors` flag on the header object and self""" + self.header.ignoreErrors = self._ignore_errors = bool(value) + + ignoreErrors = property( + lambda self: self._ignore_errors, + ignoreErrors, + doc="""Error processing mode for DBF field value conversion + + if set, failing field value conversion will return + ``INVALID_VALUE`` instead of raising conversion error. + + """) + + # protected methods + + def _fixIndex(self, index): + """Return fixed index. + + This method fails if index isn't a numeric object + (long or int). Or index isn't in a valid range + (less or equal to the number of records in the db). + + If ``index`` is a negative number, it will be + treated as a negative indexes for list objects. + + Return: + Return value is numeric object maning valid index. + + """ + if not isinstance(index, (int, long)): + raise TypeError("Index must be a numeric object") + if index < 0: + # index from the right side + # fix it to the left-side index + index += len(self) + 1 + if index >= len(self): + raise IndexError("Record index out of range") + return index + + # iterface methods + + def close(self): + self.flush() + self.stream.close() + + def flush(self): + """Flush data to the associated stream.""" + if self.changed: + self.header.setCurrentDate() + self.header.write(self.stream) + self.stream.flush() + self._changed = False + + def indexOfFieldName(self, name): + """Index of field named ``name``.""" + # FIXME: move this to header class + return self.header.fields.index(name) + + def newRecord(self): + """Return new record, which belong to this table.""" + return self.RecordClass(self) + + def append(self, record): + """Append ``record`` to the database.""" + record.index = self.header.recordCount + record._write() + self.header.recordCount += 1 + self._changed = True + self._new = False + + def addField(self, *defs): + """Add field definitions. + + For more information see `header.DbfHeader.addField`. + + """ + if self._new: + self.header.addField(*defs) + else: + raise TypeError("At least one record was added, " + "structure can't be changed") + + # 'magic' methods (representation and sequence interface) + + def __repr__(self): + return "Dbf stream '%s'\n" % self.stream + repr(self.header) + + def __len__(self): + """Return number of records.""" + return self.recordCount + + def __getitem__(self, index): + """Return `DbfRecord` instance.""" + return self.RecordClass.fromStream(self, self._fixIndex(index)) + + def __setitem__(self, index, record): + """Write `DbfRecord` instance to the stream.""" + record.index = self._fixIndex(index) + record._write() + self._changed = True + self._new = False + + # def __del__(self): + # """Flush stream upon deletion of the object.""" + # self.flush() + + +def demo_read(filename): + _dbf = Dbf(filename, True) + for _rec in _dbf: + print() + print(repr(_rec)) + _dbf.close() + + +def demo_create(filename): + _dbf = Dbf(filename, new=True) + _dbf.addField( + ("NAME", "C", 15), + ("SURNAME", "C", 25), + ("INITIALS", "C", 10), + ("BIRTHDATE", "D"), + ) + for (_n, _s, _i, _b) in ( + ("John", "Miller", "YC", (1981, 1, 2)), + ("Andy", "Larkin", "AL", (1982, 3, 4)), + ("Bill", "Clinth", "", (1983, 5, 6)), + ("Bobb", "McNail", "", (1984, 7, 8)), + ): + _rec = _dbf.newRecord() + _rec["NAME"] = _n + _rec["SURNAME"] = _s + _rec["INITIALS"] = _i + _rec["BIRTHDATE"] = _b + _rec.store() + print(repr(_dbf)) + _dbf.close() + + +if __name__ == '__main__': + import sys + + _name = len(sys.argv) > 1 and sys.argv[1] or "county.dbf" + demo_create(_name) + demo_read(_name) + + # vim: set et sw=4 sts=4 : diff --git a/src/tablib/packages/dbfpy/dbfnew.py b/src/tablib/packages/dbfpy/dbfnew.py new file mode 100644 index 0000000..008d307 --- /dev/null +++ b/src/tablib/packages/dbfpy/dbfnew.py @@ -0,0 +1,189 @@ +#!/usr/bin/python +""".DBF creation helpers. + +Note: this is a legacy interface. New code should use Dbf class + for table creation (see examples in dbf.py) + +TODO: + - handle Memo fields. + - check length of the fields accoring to the + `http://www.clicketyclick.dk/databases/xbase/format/data_types.html` + +""" +"""History (most recent first) +04-jul-2006 [als] added export declaration; + updated for dbfpy 2.0 +15-dec-2005 [yc] define dbf_new.__slots__ +14-dec-2005 [yc] added vim modeline; retab'd; added doc-strings; + dbf_new now is a new class (inherited from object) +??-jun-2000 [--] added by Hans Fiby +""" + +__version__ = "$Revision: 1.4 $"[11:-2] +__date__ = "$Date: 2006/07/04 08:18:18 $"[7:-2] + +__all__ = ["dbf_new"] + +from dbf import * +from fields import * +from header import * +from record import * + + +class _FieldDefinition(object): + """Field definition. + + This is a simple structure, which contains ``name``, ``type``, + ``len``, ``dec`` and ``cls`` fields. + + Objects also implement get/setitem magic functions, so fields + could be accessed via sequence iterface, where 'name' has + index 0, 'type' index 1, 'len' index 2, 'dec' index 3 and + 'cls' could be located at index 4. + + """ + + __slots__ = "name", "type", "len", "dec", "cls" + + # WARNING: be attentive - dictionaries are mutable! + FLD_TYPES = { + # type: (cls, len) + "C": (DbfCharacterFieldDef, None), + "N": (DbfNumericFieldDef, None), + "L": (DbfLogicalFieldDef, 1), + # FIXME: support memos + # "M": (DbfMemoFieldDef), + "D": (DbfDateFieldDef, 8), + # FIXME: I'm not sure length should be 14 characters! + # but temporary I use it, cuz date is 8 characters + # and time 6 (hhmmss) + "T": (DbfDateTimeFieldDef, 14), + } + + def __init__(self, name, type, len=None, dec=0): + _cls, _len = self.FLD_TYPES[type] + if _len is None: + if len is None: + raise ValueError("Field length must be defined") + _len = len + self.name = name + self.type = type + self.len = _len + self.dec = dec + self.cls = _cls + + def getDbfField(self): + "Return `DbfFieldDef` instance from the current definition." + return self.cls(self.name, self.len, self.dec) + + def appendToHeader(self, dbfh): + """Create a `DbfFieldDef` instance and append it to the dbf header. + + Arguments: + dbfh: `DbfHeader` instance. + + """ + _dbff = self.getDbfField() + dbfh.addField(_dbff) + + +class dbf_new(object): + """New .DBF creation helper. + + Example Usage: + + dbfn = dbf_new() + dbfn.add_field("name",'C',80) + dbfn.add_field("price",'N',10,2) + dbfn.add_field("date",'D',8) + dbfn.write("tst.dbf") + + Note: + This module cannot handle Memo-fields, + they are special. + + """ + + __slots__ = ("fields",) + + FieldDefinitionClass = _FieldDefinition + + def __init__(self): + self.fields = [] + + def add_field(self, name, typ, len, dec=0): + """Add field definition. + + Arguments: + name: + field name (str object). field name must not + contain ASCII NULs and it's length shouldn't + exceed 10 characters. + typ: + type of the field. this must be a single character + from the "CNLMDT" set meaning character, numeric, + logical, memo, date and date/time respectively. + len: + length of the field. this argument is used only for + the character and numeric fields. all other fields + have fixed length. + FIXME: use None as a default for this argument? + dec: + decimal precision. used only for the numric fields. + + """ + self.fields.append(self.FieldDefinitionClass(name, typ, len, dec)) + + def write(self, filename): + """Create empty .DBF file using current structure.""" + _dbfh = DbfHeader() + _dbfh.setCurrentDate() + for _fldDef in self.fields: + _fldDef.appendToHeader(_dbfh) + _dbfStream = file(filename, "wb") + _dbfh.write(_dbfStream) + _dbfStream.close() + + def write_stream(self, stream): + _dbfh = DbfHeader() + _dbfh.setCurrentDate() + for _fldDef in self.fields: + _fldDef.appendToHeader(_dbfh) + _dbfh.write(stream) + + +if __name__ == '__main__': + # create a new DBF-File + dbfn = dbf_new() + dbfn.add_field("name", 'C', 80) + dbfn.add_field("price", 'N', 10, 2) + dbfn.add_field("date", 'D', 8) + dbfn.write("tst.dbf") + # test new dbf + print("*** created tst.dbf: ***") + dbft = Dbf('tst.dbf', readOnly=0) + print(repr(dbft)) + # add a record + rec = DbfRecord(dbft) + rec['name'] = 'something' + rec['price'] = 10.5 + rec['date'] = (2000, 1, 12) + rec.store() + # add another record + rec = DbfRecord(dbft) + rec['name'] = 'foo and bar' + rec['price'] = 12234 + rec['date'] = (1992, 7, 15) + rec.store() + + # show the records + print("*** inserted 2 records into tst.dbf: ***") + print(repr(dbft)) + for i1 in range(len(dbft)): + rec = dbft[i1] + for fldName in dbft.fieldNames: + print('%s:\t %s' % (fldName, rec[fldName])) + print() + dbft.close() + + # vim: set et sts=4 sw=4 : diff --git a/src/tablib/packages/dbfpy/fields.py b/src/tablib/packages/dbfpy/fields.py new file mode 100644 index 0000000..69cd436 --- /dev/null +++ b/src/tablib/packages/dbfpy/fields.py @@ -0,0 +1,466 @@ +"""DBF fields definitions. + +TODO: + - make memos work +""" +"""History (most recent first): +26-may-2009 [als] DbfNumericFieldDef.decodeValue: strip zero bytes +05-feb-2009 [als] DbfDateFieldDef.encodeValue: empty arg produces empty date +16-sep-2008 [als] DbfNumericFieldDef decoding looks for decimal point + in the value to select float or integer return type +13-mar-2008 [als] check field name length in constructor +11-feb-2007 [als] handle value conversion errors +10-feb-2007 [als] DbfFieldDef: added .rawFromRecord() +01-dec-2006 [als] Timestamp columns use None for empty values +31-oct-2006 [als] support field types 'F' (float), 'I' (integer) + and 'Y' (currency); + automate export and registration of field classes +04-jul-2006 [als] added export declaration +10-mar-2006 [als] decode empty values for Date and Logical fields; + show field name in errors +10-mar-2006 [als] fix Numeric value decoding: according to spec, + value always is string representation of the number; + ensure that encoded Numeric value fits into the field +20-dec-2005 [yc] use field names in upper case +15-dec-2005 [yc] field definitions moved from `dbf`. +""" + +__version__ = "$Revision: 1.14 $"[11:-2] +__date__ = "$Date: 2009/05/26 05:16:51 $"[7:-2] + +__all__ = ["lookupFor",] # field classes added at the end of the module + +import datetime +import struct +import sys + +from . import utils + +## abstract definitions + +class DbfFieldDef(object): + """Abstract field definition. + + Child classes must override ``type`` class attribute to provide datatype + infromation of the field definition. For more info about types visit + `http://www.clicketyclick.dk/databases/xbase/format/data_types.html` + + Also child classes must override ``defaultValue`` field to provide + default value for the field value. + + If child class has fixed length ``length`` class attribute must be + overriden and set to the valid value. None value means, that field + isn't of fixed length. + + Note: ``name`` field must not be changed after instantiation. + + """ + + __slots__ = ("name", "length", "decimalCount", + "start", "end", "ignoreErrors") + + # length of the field, None in case of variable-length field, + # or a number if this field is a fixed-length field + length = None + + # field type. for more information about fields types visit + # `http://www.clicketyclick.dk/databases/xbase/format/data_types.html` + # must be overriden in child classes + typeCode = None + + # default value for the field. this field must be + # overriden in child classes + defaultValue = None + + def __init__(self, name, length=None, decimalCount=None, + start=None, stop=None, ignoreErrors=False, + ): + """Initialize instance.""" + assert self.typeCode is not None, "Type code must be overriden" + assert self.defaultValue is not None, "Default value must be overriden" + ## fix arguments + if len(name) >10: + raise ValueError("Field name \"%s\" is too long" % name) + name = str(name).upper() + if self.__class__.length is None: + if length is None: + raise ValueError("[%s] Length isn't specified" % name) + length = int(length) + if length <= 0: + raise ValueError("[%s] Length must be a positive integer" + % name) + else: + length = self.length + if decimalCount is None: + decimalCount = 0 + ## set fields + self.name = name + # FIXME: validate length according to the specification at + # http://www.clicketyclick.dk/databases/xbase/format/data_types.html + self.length = length + self.decimalCount = decimalCount + self.ignoreErrors = ignoreErrors + self.start = start + self.end = stop + + def __cmp__(self, other): + return cmp(self.name, str(other).upper()) + + def __hash__(self): + return hash(self.name) + + def fromString(cls, string, start, ignoreErrors=False): + """Decode dbf field definition from the string data. + + Arguments: + string: + a string, dbf definition is decoded from. length of + the string must be 32 bytes. + start: + position in the database file. + ignoreErrors: + initial error processing mode for the new field (boolean) + + """ + assert len(string) == 32 + _length = ord(string[16]) + return cls(utils.unzfill(string)[:11], _length, ord(string[17]), + start, start + _length, ignoreErrors=ignoreErrors) + fromString = classmethod(fromString) + + def toString(self): + """Return encoded field definition. + + Return: + Return value is a string object containing encoded + definition of this field. + + """ + if sys.version_info < (2, 4): + # earlier versions did not support padding character + _name = self.name[:11] + "\0" * (11 - len(self.name)) + else: + _name = self.name.ljust(11, '\0') + return ( + _name + + self.typeCode + + #data address + chr(0) * 4 + + chr(self.length) + + chr(self.decimalCount) + + chr(0) * 14 + ) + + def __repr__(self): + return "%-10s %1s %3d %3d" % self.fieldInfo() + + def fieldInfo(self): + """Return field information. + + Return: + Return value is a (name, type, length, decimals) tuple. + + """ + return (self.name, self.typeCode, self.length, self.decimalCount) + + def rawFromRecord(self, record): + """Return a "raw" field value from the record string.""" + return record[self.start:self.end] + + def decodeFromRecord(self, record): + """Return decoded field value from the record string.""" + try: + return self.decodeValue(self.rawFromRecord(record)) + except: + if self.ignoreErrors: + return utils.INVALID_VALUE + else: + raise + + def decodeValue(self, value): + """Return decoded value from string value. + + This method shouldn't be used publicly. It's called from the + `decodeFromRecord` method. + + This is an abstract method and it must be overridden in child classes. + """ + raise NotImplementedError + + def encodeValue(self, value): + """Return str object containing encoded field value. + + This is an abstract method and it must be overriden in child classes. + """ + raise NotImplementedError + +## real classes + +class DbfCharacterFieldDef(DbfFieldDef): + """Definition of the character field.""" + + typeCode = "C" + defaultValue = "" + + def decodeValue(self, value): + """Return string object. + + Return value is a ``value`` argument with stripped right spaces. + + """ + return value.rstrip(" ") + + def encodeValue(self, value): + """Return raw data string encoded from a ``value``.""" + return str(value)[:self.length].ljust(self.length) + + +class DbfNumericFieldDef(DbfFieldDef): + """Definition of the numeric field.""" + + typeCode = "N" + # XXX: now I'm not sure it was a good idea to make a class field + # `defaultValue` instead of a generic method as it was implemented + # previously -- it's ok with all types except number, cuz + # if self.decimalCount is 0, we should return 0 and 0.0 otherwise. + defaultValue = 0 + + def decodeValue(self, value): + """Return a number decoded from ``value``. + + If decimals is zero, value will be decoded as an integer; + or as a float otherwise. + + Return: + Return value is a int (long) or float instance. + + """ + value = value.strip(" \0") + if "." in value: + # a float (has decimal separator) + return float(value) + elif value: + # must be an integer + return int(value) + else: + return 0 + + def encodeValue(self, value): + """Return string containing encoded ``value``.""" + _rv = ("%*.*f" % (self.length, self.decimalCount, value)) + if len(_rv) > self.length: + _ppos = _rv.find(".") + if 0 <= _ppos <= self.length: + _rv = _rv[:self.length] + else: + raise ValueError("[%s] Numeric overflow: %s (field width: %i)" + % (self.name, _rv, self.length)) + return _rv + +class DbfFloatFieldDef(DbfNumericFieldDef): + """Definition of the float field - same as numeric.""" + + typeCode = "F" + +class DbfIntegerFieldDef(DbfFieldDef): + """Definition of the integer field.""" + + typeCode = "I" + length = 4 + defaultValue = 0 + + def decodeValue(self, value): + """Return an integer number decoded from ``value``.""" + return struct.unpack("<i", value)[0] + + def encodeValue(self, value): + """Return string containing encoded ``value``.""" + return struct.pack("<i", int(value)) + +class DbfCurrencyFieldDef(DbfFieldDef): + """Definition of the currency field.""" + + typeCode = "Y" + length = 8 + defaultValue = 0.0 + + def decodeValue(self, value): + """Return float number decoded from ``value``.""" + return struct.unpack("<q", value)[0] / 10000. + + def encodeValue(self, value): + """Return string containing encoded ``value``.""" + return struct.pack("<q", round(value * 10000)) + +class DbfLogicalFieldDef(DbfFieldDef): + """Definition of the logical field.""" + + typeCode = "L" + defaultValue = -1 + length = 1 + + def decodeValue(self, value): + """Return True, False or -1 decoded from ``value``.""" + # Note: value always is 1-char string + if value == "?": + return -1 + if value in "NnFf ": + return False + if value in "YyTt": + return True + raise ValueError("[%s] Invalid logical value %r" % (self.name, value)) + + def encodeValue(self, value): + """Return a character from the "TF?" set. + + Return: + Return value is "T" if ``value`` is True + "?" if value is -1 or False otherwise. + + """ + if value is True: + return "T" + if value == -1: + return "?" + return "F" + + +class DbfMemoFieldDef(DbfFieldDef): + """Definition of the memo field. + + Note: memos aren't currenly completely supported. + + """ + + typeCode = "M" + defaultValue = " " * 10 + length = 10 + + def decodeValue(self, value): + """Return int .dbt block number decoded from the string object.""" + #return int(value) + raise NotImplementedError + + def encodeValue(self, value): + """Return raw data string encoded from a ``value``. + + Note: this is an internal method. + + """ + #return str(value)[:self.length].ljust(self.length) + raise NotImplementedError + + +class DbfDateFieldDef(DbfFieldDef): + """Definition of the date field.""" + + typeCode = "D" + defaultValue = utils.classproperty(lambda cls: datetime.date.today()) + # "yyyymmdd" gives us 8 characters + length = 8 + + def decodeValue(self, value): + """Return a ``datetime.date`` instance decoded from ``value``.""" + if value.strip(): + return utils.getDate(value) + else: + return None + + def encodeValue(self, value): + """Return a string-encoded value. + + ``value`` argument should be a value suitable for the + `utils.getDate` call. + + Return: + Return value is a string in format "yyyymmdd". + + """ + if value: + return utils.getDate(value).strftime("%Y%m%d") + else: + return " " * self.length + + +class DbfDateTimeFieldDef(DbfFieldDef): + """Definition of the timestamp field.""" + + # a difference between JDN (Julian Day Number) + # and GDN (Gregorian Day Number). note, that GDN < JDN + JDN_GDN_DIFF = 1721425 + typeCode = "T" + defaultValue = utils.classproperty(lambda cls: datetime.datetime.now()) + # two 32-bits integers representing JDN and amount of + # milliseconds respectively gives us 8 bytes. + # note, that values must be encoded in LE byteorder. + length = 8 + + def decodeValue(self, value): + """Return a `datetime.datetime` instance.""" + assert len(value) == self.length + # LE byteorder + _jdn, _msecs = struct.unpack("<2I", value) + if _jdn >= 1: + _rv = datetime.datetime.fromordinal(_jdn - self.JDN_GDN_DIFF) + _rv += datetime.timedelta(0, _msecs / 1000.0) + else: + # empty date + _rv = None + return _rv + + def encodeValue(self, value): + """Return a string-encoded ``value``.""" + if value: + value = utils.getDateTime(value) + # LE byteorder + _rv = struct.pack("<2I", value.toordinal() + self.JDN_GDN_DIFF, + (value.hour * 3600 + value.minute * 60 + value.second) * 1000) + else: + _rv = "\0" * self.length + assert len(_rv) == self.length + return _rv + + +_fieldsRegistry = {} + +def registerField(fieldCls): + """Register field definition class. + + ``fieldCls`` should be subclass of the `DbfFieldDef`. + + Use `lookupFor` to retrieve field definition class + by the type code. + + """ + assert fieldCls.typeCode is not None, "Type code isn't defined" + # XXX: use fieldCls.typeCode.upper()? in case of any decign + # don't forget to look to the same comment in ``lookupFor`` method + _fieldsRegistry[fieldCls.typeCode] = fieldCls + + +def lookupFor(typeCode): + """Return field definition class for the given type code. + + ``typeCode`` must be a single character. That type should be + previously registered. + + Use `registerField` to register new field class. + + Return: + Return value is a subclass of the `DbfFieldDef`. + + """ + # XXX: use typeCode.upper()? in case of any decign don't + # forget to look to the same comment in ``registerField`` + return _fieldsRegistry[typeCode] + +## register generic types + +for (_name, _val) in globals().items(): + if isinstance(_val, type) and issubclass(_val, DbfFieldDef) \ + and (_name != "DbfFieldDef"): + __all__.append(_name) + registerField(_val) +del _name, _val + +# vim: et sts=4 sw=4 : diff --git a/src/tablib/packages/dbfpy/header.py b/src/tablib/packages/dbfpy/header.py new file mode 100644 index 0000000..03a877c --- /dev/null +++ b/src/tablib/packages/dbfpy/header.py @@ -0,0 +1,275 @@ +"""DBF header definition. + +TODO: + - handle encoding of the character fields + (encoding information stored in the DBF header) + +""" +"""History (most recent first): +16-sep-2010 [als] fromStream: fix century of the last update field +11-feb-2007 [als] added .ignoreErrors +10-feb-2007 [als] added __getitem__: return field definitions + by field name or field number (zero-based) +04-jul-2006 [als] added export declaration +15-dec-2005 [yc] created +""" + +__version__ = "$Revision: 1.6 $"[11:-2] +__date__ = "$Date: 2010/09/16 05:06:39 $"[7:-2] + +__all__ = ["DbfHeader"] + +try: + import cStringIO +except ImportError: + # when we're in python3, we cStringIO has been replaced by io.StringIO + import io as cStringIO +import datetime +import struct +import time + +from . import fields +from . import utils + + +class DbfHeader(object): + """Dbf header definition. + + For more information about dbf header format visit + `http://www.clicketyclick.dk/databases/xbase/format/dbf.html#DBF_STRUCT` + + Examples: + Create an empty dbf header and add some field definitions: + dbfh = DbfHeader() + dbfh.addField(("name", "C", 10)) + dbfh.addField(("date", "D")) + dbfh.addField(DbfNumericFieldDef("price", 5, 2)) + Create a dbf header with field definitions: + dbfh = DbfHeader([ + ("name", "C", 10), + ("date", "D"), + DbfNumericFieldDef("price", 5, 2), + ]) + + """ + + __slots__ = ("signature", "fields", "lastUpdate", "recordLength", + "recordCount", "headerLength", "changed", "_ignore_errors") + + ## instance construction and initialization methods + + def __init__(self, fields=None, headerLength=0, recordLength=0, + recordCount=0, signature=0x03, lastUpdate=None, ignoreErrors=False, + ): + """Initialize instance. + + Arguments: + fields: + a list of field definitions; + recordLength: + size of the records; + headerLength: + size of the header; + recordCount: + number of records stored in DBF; + signature: + version number (aka signature). using 0x03 as a default meaning + "File without DBT". for more information about this field visit + ``http://www.clicketyclick.dk/databases/xbase/format/dbf.html#DBF_NOTE_1_TARGET`` + lastUpdate: + date of the DBF's update. this could be a string ('yymmdd' or + 'yyyymmdd'), timestamp (int or float), datetime/date value, + a sequence (assuming (yyyy, mm, dd, ...)) or an object having + callable ``ticks`` field. + ignoreErrors: + error processing mode for DBF fields (boolean) + + """ + self.signature = signature + if fields is None: + self.fields = [] + else: + self.fields = list(fields) + self.lastUpdate = utils.getDate(lastUpdate) + self.recordLength = recordLength + self.headerLength = headerLength + self.recordCount = recordCount + self.ignoreErrors = ignoreErrors + # XXX: I'm not sure this is safe to + # initialize `self.changed` in this way + self.changed = bool(self.fields) + + # @classmethod + def fromString(cls, string): + """Return header instance from the string object.""" + return cls.fromStream(cStringIO.StringIO(str(string))) + fromString = classmethod(fromString) + + # @classmethod + def fromStream(cls, stream): + """Return header object from the stream.""" + stream.seek(0) + _data = stream.read(32) + (_cnt, _hdrLen, _recLen) = struct.unpack("<I2H", _data[4:12]) + #reserved = _data[12:32] + _year = ord(_data[1]) + if _year < 80: + # dBase II started at 1980. It is quite unlikely + # that actual last update date is before that year. + _year += 2000 + else: + _year += 1900 + ## create header object + _obj = cls(None, _hdrLen, _recLen, _cnt, ord(_data[0]), + (_year, ord(_data[2]), ord(_data[3]))) + ## append field definitions + # position 0 is for the deletion flag + _pos = 1 + _data = stream.read(1) + + # The field definitions are ended either by \x0D OR a newline + # character, so we need to handle both when reading from a stream. + # When writing, dbfpy appears to write newlines instead of \x0D. + while _data[0] not in ["\x0D", "\n"]: + _data += stream.read(31) + _fld = fields.lookupFor(_data[11]).fromString(_data, _pos) + _obj._addField(_fld) + _pos = _fld.end + _data = stream.read(1) + return _obj + fromStream = classmethod(fromStream) + + ## properties + + year = property(lambda self: self.lastUpdate.year) + month = property(lambda self: self.lastUpdate.month) + day = property(lambda self: self.lastUpdate.day) + + def ignoreErrors(self, value): + """Update `ignoreErrors` flag on self and all fields""" + self._ignore_errors = value = bool(value) + for _field in self.fields: + _field.ignoreErrors = value + ignoreErrors = property( + lambda self: self._ignore_errors, + ignoreErrors, + doc="""Error processing mode for DBF field value conversion + + if set, failing field value conversion will return + ``INVALID_VALUE`` instead of raising conversion error. + + """) + + ## object representation + + def __repr__(self): + _rv = """\ +Version (signature): 0x%02x + Last update: %s + Header length: %d + Record length: %d + Record count: %d + FieldName Type Len Dec +""" % (self.signature, self.lastUpdate, self.headerLength, + self.recordLength, self.recordCount) + _rv += "\n".join( + ["%10s %4s %3s %3s" % _fld.fieldInfo() for _fld in self.fields] + ) + return _rv + + ## internal methods + + def _addField(self, *defs): + """Internal variant of the `addField` method. + + This method doesn't set `self.changed` field to True. + + Return value is a length of the appended records. + Note: this method doesn't modify ``recordLength`` and + ``headerLength`` fields. Use `addField` instead of this + method if you don't exactly know what you're doing. + + """ + # insure we have dbf.DbfFieldDef instances first (instantiation + # from the tuple could raise an error, in such a case I don't + # wanna add any of the definitions -- all will be ignored) + _defs = [] + _recordLength = 0 + for _def in defs: + if isinstance(_def, fields.DbfFieldDef): + _obj = _def + else: + (_name, _type, _len, _dec) = (tuple(_def) + (None,) * 4)[:4] + _cls = fields.lookupFor(_type) + _obj = _cls(_name, _len, _dec, + ignoreErrors=self._ignore_errors) + _recordLength += _obj.length + _defs.append(_obj) + # and now extend field definitions and + # update record length + self.fields += _defs + return _recordLength + + ## interface methods + + def addField(self, *defs): + """Add field definition to the header. + + Examples: + dbfh.addField( + ("name", "C", 20), + dbf.DbfCharacterFieldDef("surname", 20), + dbf.DbfDateFieldDef("birthdate"), + ("member", "L"), + ) + dbfh.addField(("price", "N", 5, 2)) + dbfh.addField(dbf.DbfNumericFieldDef("origprice", 5, 2)) + + """ + _oldLen = self.recordLength + self.recordLength += self._addField(*defs) + if not _oldLen: + self.recordLength += 1 + # XXX: may be just use: + # self.recordeLength += self._addField(*defs) + bool(not _oldLen) + # recalculate headerLength + self.headerLength = 32 + (32 * len(self.fields)) + 1 + self.changed = True + + def write(self, stream): + """Encode and write header to the stream.""" + stream.seek(0) + stream.write(self.toString()) + stream.write("".join([_fld.toString() for _fld in self.fields])) + stream.write(chr(0x0D)) # cr at end of all hdr data + self.changed = False + + def toString(self): + """Returned 32 chars length string with encoded header.""" + return struct.pack("<4BI2H", + self.signature, + self.year - 1900, + self.month, + self.day, + self.recordCount, + self.headerLength, + self.recordLength) + "\0" * 20 + + def setCurrentDate(self): + """Update ``self.lastUpdate`` field with current date value.""" + self.lastUpdate = datetime.date.today() + + def __getitem__(self, item): + """Return a field definition by numeric index or name string""" + if isinstance(item, basestring): + _name = item.upper() + for _field in self.fields: + if _field.name == _name: + return _field + else: + raise KeyError(item) + else: + # item must be field index + return self.fields[item] + +# vim: et sts=4 sw=4 : diff --git a/src/tablib/packages/dbfpy/record.py b/src/tablib/packages/dbfpy/record.py new file mode 100644 index 0000000..97bbfb3 --- /dev/null +++ b/src/tablib/packages/dbfpy/record.py @@ -0,0 +1,262 @@ +"""DBF record definition. + +""" +"""History (most recent first): +11-feb-2007 [als] __repr__: added special case for invalid field values +10-feb-2007 [als] added .rawFromStream() +30-oct-2006 [als] fix record length in .fromStream() +04-jul-2006 [als] added export declaration +20-dec-2005 [yc] DbfRecord.write() -> DbfRecord._write(); + added delete() method. +16-dec-2005 [yc] record definition moved from `dbf`. +""" + +__version__ = "$Revision: 1.7 $"[11:-2] +__date__ = "$Date: 2007/02/11 09:05:49 $"[7:-2] + +__all__ = ["DbfRecord"] + +from itertools import izip + +import utils + +class DbfRecord(object): + """DBF record. + + Instances of this class shouldn't be created manualy, + use `dbf.Dbf.newRecord` instead. + + Class implements mapping/sequence interface, so + fields could be accessed via their names or indexes + (names is a preffered way to access fields). + + Hint: + Use `store` method to save modified record. + + Examples: + Add new record to the database: + db = Dbf(filename) + rec = db.newRecord() + rec["FIELD1"] = value1 + rec["FIELD2"] = value2 + rec.store() + Or the same, but modify existed + (second in this case) record: + db = Dbf(filename) + rec = db[2] + rec["FIELD1"] = value1 + rec["FIELD2"] = value2 + rec.store() + + """ + + __slots__ = "dbf", "index", "deleted", "fieldData" + + ## creation and initialization + + def __init__(self, dbf, index=None, deleted=False, data=None): + """Instance initialiation. + + Arguments: + dbf: + A `Dbf.Dbf` instance this record belonogs to. + index: + An integer record index or None. If this value is + None, record will be appended to the DBF. + deleted: + Boolean flag indicating whether this record + is a deleted record. + data: + A sequence or None. This is a data of the fields. + If this argument is None, default values will be used. + + """ + self.dbf = dbf + # XXX: I'm not sure ``index`` is necessary + self.index = index + self.deleted = deleted + if data is None: + self.fieldData = [_fd.defaultValue for _fd in dbf.header.fields] + else: + self.fieldData = list(data) + + # XXX: validate self.index before calculating position? + position = property(lambda self: self.dbf.header.headerLength + \ + self.index * self.dbf.header.recordLength) + + def rawFromStream(cls, dbf, index): + """Return raw record contents read from the stream. + + Arguments: + dbf: + A `Dbf.Dbf` instance containing the record. + index: + Index of the record in the records' container. + This argument can't be None in this call. + + Return value is a string containing record data in DBF format. + + """ + # XXX: may be write smth assuming, that current stream + # position is the required one? it could save some + # time required to calculate where to seek in the file + dbf.stream.seek(dbf.header.headerLength + + index * dbf.header.recordLength) + return dbf.stream.read(dbf.header.recordLength) + rawFromStream = classmethod(rawFromStream) + + def fromStream(cls, dbf, index): + """Return a record read from the stream. + + Arguments: + dbf: + A `Dbf.Dbf` instance new record should belong to. + index: + Index of the record in the records' container. + This argument can't be None in this call. + + Return value is an instance of the current class. + + """ + return cls.fromString(dbf, cls.rawFromStream(dbf, index), index) + fromStream = classmethod(fromStream) + + def fromString(cls, dbf, string, index=None): + """Return record read from the string object. + + Arguments: + dbf: + A `Dbf.Dbf` instance new record should belong to. + string: + A string new record should be created from. + index: + Index of the record in the container. If this + argument is None, record will be appended. + + Return value is an instance of the current class. + + """ + return cls(dbf, index, string[0]=="*", + [_fd.decodeFromRecord(string) for _fd in dbf.header.fields]) + fromString = classmethod(fromString) + + ## object representation + + def __repr__(self): + _template = "%%%ds: %%s (%%s)" % max([len(_fld) + for _fld in self.dbf.fieldNames]) + _rv = [] + for _fld in self.dbf.fieldNames: + _val = self[_fld] + if _val is utils.INVALID_VALUE: + _rv.append(_template % + (_fld, "None", "value cannot be decoded")) + else: + _rv.append(_template % (_fld, _val, type(_val))) + return "\n".join(_rv) + + ## protected methods + + def _write(self): + """Write data to the dbf stream. + + Note: + This isn't a public method, it's better to + use 'store' instead publically. + Be design ``_write`` method should be called + only from the `Dbf` instance. + + + """ + self._validateIndex(False) + self.dbf.stream.seek(self.position) + self.dbf.stream.write(self.toString()) + # FIXME: may be move this write somewhere else? + # why we should check this condition for each record? + if self.index == len(self.dbf): + # this is the last record, + # we should write SUB (ASCII 26) + self.dbf.stream.write("\x1A") + + ## utility methods + + def _validateIndex(self, allowUndefined=True, checkRange=False): + """Valid ``self.index`` value. + + If ``allowUndefined`` argument is True functions does nothing + in case of ``self.index`` pointing to None object. + + """ + if self.index is None: + if not allowUndefined: + raise ValueError("Index is undefined") + elif self.index < 0: + raise ValueError("Index can't be negative (%s)" % self.index) + elif checkRange and self.index <= self.dbf.header.recordCount: + raise ValueError("There are only %d records in the DBF" % + self.dbf.header.recordCount) + + ## interface methods + + def store(self): + """Store current record in the DBF. + + If ``self.index`` is None, this record will be appended to the + records of the DBF this records belongs to; or replaced otherwise. + + """ + self._validateIndex() + if self.index is None: + self.index = len(self.dbf) + self.dbf.append(self) + else: + self.dbf[self.index] = self + + def delete(self): + """Mark method as deleted.""" + self.deleted = True + + def toString(self): + """Return string packed record values.""" + return "".join([" *"[self.deleted]] + [ + _def.encodeValue(_dat) + for (_def, _dat) in izip(self.dbf.header.fields, self.fieldData) + ]) + + def asList(self): + """Return a flat list of fields. + + Note: + Change of the list's values won't change + real values stored in this object. + + """ + return self.fieldData[:] + + def asDict(self): + """Return a dictionary of fields. + + Note: + Change of the dicts's values won't change + real values stored in this object. + + """ + return dict([_i for _i in izip(self.dbf.fieldNames, self.fieldData)]) + + def __getitem__(self, key): + """Return value by field name or field index.""" + if isinstance(key, (long, int)): + # integer index of the field + return self.fieldData[key] + # assuming string field name + return self.fieldData[self.dbf.indexOfFieldName(key)] + + def __setitem__(self, key, value): + """Set field value by integer index of the field or string name.""" + if isinstance(key, (int, long)): + # integer index of the field + return self.fieldData[key] + # assuming string field name + self.fieldData[self.dbf.indexOfFieldName(key)] = value + +# vim: et sts=4 sw=4 : diff --git a/src/tablib/packages/dbfpy/utils.py b/src/tablib/packages/dbfpy/utils.py new file mode 100644 index 0000000..cef8aa5 --- /dev/null +++ b/src/tablib/packages/dbfpy/utils.py @@ -0,0 +1,170 @@ +"""String utilities. + +TODO: + - allow strings in getDateTime routine; +""" +"""History (most recent first): +11-feb-2007 [als] added INVALID_VALUE +10-feb-2007 [als] allow date strings padded with spaces instead of zeroes +20-dec-2005 [yc] handle long objects in getDate/getDateTime +16-dec-2005 [yc] created from ``strutil`` module. +""" + +__version__ = "$Revision: 1.4 $"[11:-2] +__date__ = "$Date: 2007/02/11 08:57:17 $"[7:-2] + +import datetime +import time + + +def unzfill(str): + """Return a string without ASCII NULs. + + This function searchers for the first NUL (ASCII 0) occurance + and truncates string till that position. + + """ + try: + return str[:str.index('\0')] + except ValueError: + return str + + +def getDate(date=None): + """Return `datetime.date` instance. + + Type of the ``date`` argument could be one of the following: + None: + use current date value; + datetime.date: + this value will be returned; + datetime.datetime: + the result of the date.date() will be returned; + string: + assuming "%Y%m%d" or "%y%m%dd" format; + number: + assuming it's a timestamp (returned for example + by the time.time() call; + sequence: + assuming (year, month, day, ...) sequence; + + Additionaly, if ``date`` has callable ``ticks`` attribute, + it will be used and result of the called would be treated + as a timestamp value. + + """ + if date is None: + # use current value + return datetime.date.today() + if isinstance(date, datetime.date): + return date + if isinstance(date, datetime.datetime): + return date.date() + if isinstance(date, (int, long, float)): + # date is a timestamp + return datetime.date.fromtimestamp(date) + if isinstance(date, basestring): + date = date.replace(" ", "0") + if len(date) == 6: + # yymmdd + return datetime.date(*time.strptime(date, "%y%m%d")[:3]) + # yyyymmdd + return datetime.date(*time.strptime(date, "%Y%m%d")[:3]) + if hasattr(date, "__getitem__"): + # a sequence (assuming date/time tuple) + return datetime.date(*date[:3]) + return datetime.date.fromtimestamp(date.ticks()) + + +def getDateTime(value=None): + """Return `datetime.datetime` instance. + + Type of the ``value`` argument could be one of the following: + None: + use current date value; + datetime.date: + result will be converted to the `datetime.datetime` instance + using midnight; + datetime.datetime: + ``value`` will be returned as is; + string: + *** CURRENTLY NOT SUPPORTED ***; + number: + assuming it's a timestamp (returned for example + by the time.time() call; + sequence: + assuming (year, month, day, ...) sequence; + + Additionaly, if ``value`` has callable ``ticks`` attribute, + it will be used and result of the called would be treated + as a timestamp value. + + """ + if value is None: + # use current value + return datetime.datetime.today() + if isinstance(value, datetime.datetime): + return value + if isinstance(value, datetime.date): + return datetime.datetime.fromordinal(value.toordinal()) + if isinstance(value, (int, long, float)): + # value is a timestamp + return datetime.datetime.fromtimestamp(value) + if isinstance(value, basestring): + raise NotImplementedError("Strings aren't currently implemented") + if hasattr(value, "__getitem__"): + # a sequence (assuming date/time tuple) + return datetime.datetime(*tuple(value)[:6]) + return datetime.datetime.fromtimestamp(value.ticks()) + + +class classproperty(property): + """Works in the same way as a ``property``, but for the classes.""" + + def __get__(self, obj, cls): + return self.fget(cls) + + +class _InvalidValue(object): + + """Value returned from DBF records when field validation fails + + The value is not equal to anything except for itself + and equal to all empty values: None, 0, empty string etc. + In other words, invalid value is equal to None and not equal + to None at the same time. + + This value yields zero upon explicit conversion to a number type, + empty string for string types, and False for boolean. + + """ + + def __eq__(self, other): + return not other + + def __ne__(self, other): + return not (other is self) + + def __nonzero__(self): + return False + + def __int__(self): + return 0 + __long__ = __int__ + + def __float__(self): + return 0.0 + + def __str__(self): + return "" + + def __unicode__(self): + return u"" + + def __repr__(self): + return "<INVALID>" + +# invalid value is a constant singleton +INVALID_VALUE = _InvalidValue() + +# vim: set et sts=4 sw=4 : diff --git a/src/tablib/packages/dbfpy3/__init__.py b/src/tablib/packages/dbfpy3/__init__.py new file mode 100644 index 0000000..e69de29 --- /dev/null +++ b/src/tablib/packages/dbfpy3/__init__.py diff --git a/src/tablib/packages/dbfpy3/dbf.py b/src/tablib/packages/dbfpy3/dbf.py new file mode 100644 index 0000000..218353e --- /dev/null +++ b/src/tablib/packages/dbfpy3/dbf.py @@ -0,0 +1,297 @@ +#! /usr/bin/env python +"""DBF accessing helpers. + +FIXME: more documentation needed + +Examples: + + Create new table, setup structure, add records: + + dbf = Dbf(filename, new=True) + dbf.addField( + ("NAME", "C", 15), + ("SURNAME", "C", 25), + ("INITIALS", "C", 10), + ("BIRTHDATE", "D"), + ) + for (n, s, i, b) in ( + ("John", "Miller", "YC", (1980, 10, 11)), + ("Andy", "Larkin", "", (1980, 4, 11)), + ): + rec = dbf.newRecord() + rec["NAME"] = n + rec["SURNAME"] = s + rec["INITIALS"] = i + rec["BIRTHDATE"] = b + rec.store() + dbf.close() + + Open existed dbf, read some data: + + dbf = Dbf(filename, True) + for rec in dbf: + for fldName in dbf.fieldNames: + print('%s:\t %s (%s)' % (fldName, rec[fldName], + type(rec[fldName]))) + dbf.close() + +""" +"""History (most recent first): +11-feb-2007 [als] export INVALID_VALUE; + Dbf: added .ignoreErrors, .INVALID_VALUE +04-jul-2006 [als] added export declaration +20-dec-2005 [yc] removed fromStream and newDbf methods: + use argument of __init__ call must be used instead; + added class fields pointing to the header and + record classes. +17-dec-2005 [yc] split to several modules; reimplemented +13-dec-2005 [yc] adapted to the changes of the `strutil` module. +13-sep-2002 [als] support FoxPro Timestamp datatype +15-nov-1999 [jjk] documentation updates, add demo +24-aug-1998 [jjk] add some encodeValue methods (not tested), other tweaks +08-jun-1998 [jjk] fix problems, add more features +20-feb-1998 [jjk] fix problems, add more features +19-feb-1998 [jjk] add create/write capabilities +18-feb-1998 [jjk] from dbfload.py +""" + +__version__ = "$Revision: 1.7 $"[11:-2] +__date__ = "$Date: 2007/02/11 09:23:13 $"[7:-2] +__author__ = "Jeff Kunce <kuncej@mail.conservation.state.mo.us>" + +__all__ = ["Dbf"] + +from . import header +from . import record +from .utils import INVALID_VALUE + + +class Dbf(object): + """DBF accessor. + + FIXME: + docs and examples needed (dont' forget to tell + about problems adding new fields on the fly) + + Implementation notes: + ``_new`` field is used to indicate whether this is + a new data table. `addField` could be used only for + the new tables! If at least one record was appended + to the table it's structure couldn't be changed. + + """ + + __slots__ = ("name", "header", "stream", + "_changed", "_new", "_ignore_errors") + + HeaderClass = header.DbfHeader + RecordClass = record.DbfRecord + INVALID_VALUE = INVALID_VALUE + + # initialization and creation helpers + + def __init__(self, f, readOnly=False, new=False, ignoreErrors=False): + """Initialize instance. + + Arguments: + f: + Filename or file-like object. + new: + True if new data table must be created. Assume + data table exists if this argument is False. + readOnly: + if ``f`` argument is a string file will + be opend in read-only mode; in other cases + this argument is ignored. This argument is ignored + even if ``new`` argument is True. + headerObj: + `header.DbfHeader` instance or None. If this argument + is None, new empty header will be used with the + all fields set by default. + ignoreErrors: + if set, failing field value conversion will return + ``INVALID_VALUE`` instead of raising conversion error. + + """ + if isinstance(f, str): + # a filename + self.name = f + if new: + # new table (table file must be + # created or opened and truncated) + self.stream = open(f, "w+b") + else: + # tabe file must exist + self.stream = open(f, ("r+b", "rb")[bool(readOnly)]) + else: + # a stream + self.name = getattr(f, "name", "") + self.stream = f + if new: + # if this is a new table, header will be empty + self.header = self.HeaderClass() + else: + # or instantiated using stream + self.header = self.HeaderClass.fromStream(self.stream) + self.ignoreErrors = ignoreErrors + self._new = bool(new) + self._changed = False + + # properties + + closed = property(lambda self: self.stream.closed) + recordCount = property(lambda self: self.header.recordCount) + fieldNames = property( + lambda self: [_fld.name for _fld in self.header.fields]) + fieldDefs = property(lambda self: self.header.fields) + changed = property(lambda self: self._changed or self.header.changed) + + def ignoreErrors(self, value): + """Update `ignoreErrors` flag on the header object and self""" + self.header.ignoreErrors = self._ignore_errors = bool(value) + + ignoreErrors = property( + lambda self: self._ignore_errors, + ignoreErrors, + doc="""Error processing mode for DBF field value conversion + + if set, failing field value conversion will return + ``INVALID_VALUE`` instead of raising conversion error. + + """) + + # protected methods + + def _fixIndex(self, index): + """Return fixed index. + + This method fails if index isn't a numeric object + (long or int). Or index isn't in a valid range + (less or equal to the number of records in the db). + + If ``index`` is a negative number, it will be + treated as a negative indexes for list objects. + + Return: + Return value is numeric object maning valid index. + + """ + if not isinstance(index, int): + raise TypeError("Index must be a numeric object") + if index < 0: + # index from the right side + # fix it to the left-side index + index += len(self) + 1 + if index >= len(self): + raise IndexError("Record index out of range") + return index + + # iterface methods + + def close(self): + self.flush() + self.stream.close() + + def flush(self): + """Flush data to the associated stream.""" + if self.changed: + self.header.setCurrentDate() + self.header.write(self.stream) + self.stream.flush() + self._changed = False + + def indexOfFieldName(self, name): + """Index of field named ``name``.""" + # FIXME: move this to header class + names = [f.name for f in self.header.fields] + return names.index(name.upper()) + + def newRecord(self): + """Return new record, which belong to this table.""" + return self.RecordClass(self) + + def append(self, record): + """Append ``record`` to the database.""" + record.index = self.header.recordCount + record._write() + self.header.recordCount += 1 + self._changed = True + self._new = False + + def addField(self, *defs): + """Add field definitions. + + For more information see `header.DbfHeader.addField`. + + """ + if self._new: + self.header.addField(*defs) + else: + raise TypeError("At least one record was added, " + "structure can't be changed") + + # 'magic' methods (representation and sequence interface) + + def __repr__(self): + return "Dbf stream '%s'\n" % self.stream + repr(self.header) + + def __len__(self): + """Return number of records.""" + return self.recordCount + + def __getitem__(self, index): + """Return `DbfRecord` instance.""" + return self.RecordClass.fromStream(self, self._fixIndex(index)) + + def __setitem__(self, index, record): + """Write `DbfRecord` instance to the stream.""" + record.index = self._fixIndex(index) + record._write() + self._changed = True + self._new = False + + # def __del__(self): + # """Flush stream upon deletion of the object.""" + # self.flush() + + +def demo_read(filename): + _dbf = Dbf(filename, True) + for _rec in _dbf: + print() + print(repr(_rec)) + _dbf.close() + + +def demo_create(filename): + _dbf = Dbf(filename, new=True) + _dbf.addField( + ("NAME", "C", 15), + ("SURNAME", "C", 25), + ("INITIALS", "C", 10), + ("BIRTHDATE", "D"), + ) + for (_n, _s, _i, _b) in ( + ("John", "Miller", "YC", (1981, 1, 2)), + ("Andy", "Larkin", "AL", (1982, 3, 4)), + ("Bill", "Clinth", "", (1983, 5, 6)), + ("Bobb", "McNail", "", (1984, 7, 8)), + ): + _rec = _dbf.newRecord() + _rec["NAME"] = _n + _rec["SURNAME"] = _s + _rec["INITIALS"] = _i + _rec["BIRTHDATE"] = _b + _rec.store() + print(repr(_dbf)) + _dbf.close() + + +if __name__ == '__main__': + import sys + + _name = len(sys.argv) > 1 and sys.argv[1] or "county.dbf" + demo_create(_name) + demo_read(_name) + +# vim: set et sw=4 sts=4 : diff --git a/src/tablib/packages/dbfpy3/dbfnew.py b/src/tablib/packages/dbfpy3/dbfnew.py new file mode 100644 index 0000000..8fab275 --- /dev/null +++ b/src/tablib/packages/dbfpy3/dbfnew.py @@ -0,0 +1,183 @@ +#!/usr/bin/python +""".DBF creation helpers. + +Note: this is a legacy interface. New code should use Dbf class + for table creation (see examples in dbf.py) + +TODO: + - handle Memo fields. + - check length of the fields accoring to the + `http://www.clicketyclick.dk/databases/xbase/format/data_types.html` + +""" +"""History (most recent first) +04-jul-2006 [als] added export declaration; + updated for dbfpy 2.0 +15-dec-2005 [yc] define dbf_new.__slots__ +14-dec-2005 [yc] added vim modeline; retab'd; added doc-strings; + dbf_new now is a new class (inherited from object) +??-jun-2000 [--] added by Hans Fiby +""" + +__version__ = "$Revision: 1.4 $"[11:-2] +__date__ = "$Date: 2006/07/04 08:18:18 $"[7:-2] + +__all__ = ["dbf_new"] + +from .dbf import * +from .fields import * +from .header import * +from .record import * + + +class _FieldDefinition(object): + """Field definition. + + This is a simple structure, which contains ``name``, ``type``, + ``len``, ``dec`` and ``cls`` fields. + + Objects also implement get/setitem magic functions, so fields + could be accessed via sequence iterface, where 'name' has + index 0, 'type' index 1, 'len' index 2, 'dec' index 3 and + 'cls' could be located at index 4. + + """ + + __slots__ = "name", "type", "len", "dec", "cls" + + # WARNING: be attentive - dictionaries are mutable! + FLD_TYPES = { + # type: (cls, len) + "C": (DbfCharacterFieldDef, None), + "N": (DbfNumericFieldDef, None), + "L": (DbfLogicalFieldDef, 1), + # FIXME: support memos + # "M": (DbfMemoFieldDef), + "D": (DbfDateFieldDef, 8), + # FIXME: I'm not sure length should be 14 characters! + # but temporary I use it, cuz date is 8 characters + # and time 6 (hhmmss) + "T": (DbfDateTimeFieldDef, 14), + } + + def __init__(self, name, type, len=None, dec=0): + _cls, _len = self.FLD_TYPES[type] + if _len is None: + if len is None: + raise ValueError("Field length must be defined") + _len = len + self.name = name + self.type = type + self.len = _len + self.dec = dec + self.cls = _cls + + def getDbfField(self): + "Return `DbfFieldDef` instance from the current definition." + return self.cls(self.name, self.len, self.dec) + + def appendToHeader(self, dbfh): + """Create a `DbfFieldDef` instance and append it to the dbf header. + + Arguments: + dbfh: `DbfHeader` instance. + + """ + _dbff = self.getDbfField() + dbfh.addField(_dbff) + + +class dbf_new(object): + """New .DBF creation helper. + + Example Usage: + + dbfn = dbf_new() + dbfn.add_field("name",'C',80) + dbfn.add_field("price",'N',10,2) + dbfn.add_field("date",'D',8) + dbfn.write("tst.dbf") + + Note: + This module cannot handle Memo-fields, + they are special. + + """ + + __slots__ = ("fields",) + + FieldDefinitionClass = _FieldDefinition + + def __init__(self): + self.fields = [] + + def add_field(self, name, typ, len, dec=0): + """Add field definition. + + Arguments: + name: + field name (str object). field name must not + contain ASCII NULs and it's length shouldn't + exceed 10 characters. + typ: + type of the field. this must be a single character + from the "CNLMDT" set meaning character, numeric, + logical, memo, date and date/time respectively. + len: + length of the field. this argument is used only for + the character and numeric fields. all other fields + have fixed length. + FIXME: use None as a default for this argument? + dec: + decimal precision. used only for the numric fields. + + """ + self.fields.append(self.FieldDefinitionClass(name, typ, len, dec)) + + def write(self, filename): + """Create empty .DBF file using current structure.""" + _dbfh = DbfHeader() + _dbfh.setCurrentDate() + for _fldDef in self.fields: + _fldDef.appendToHeader(_dbfh) + + _dbfStream = open(filename, "wb") + _dbfh.write(_dbfStream) + _dbfStream.close() + + +if __name__ == '__main__': + # create a new DBF-File + dbfn = dbf_new() + dbfn.add_field("name", 'C', 80) + dbfn.add_field("price", 'N', 10, 2) + dbfn.add_field("date", 'D', 8) + dbfn.write("tst.dbf") + # test new dbf + print("*** created tst.dbf: ***") + dbft = Dbf('tst.dbf', readOnly=0) + print(repr(dbft)) + # add a record + rec = DbfRecord(dbft) + rec['name'] = 'something' + rec['price'] = 10.5 + rec['date'] = (2000, 1, 12) + rec.store() + # add another record + rec = DbfRecord(dbft) + rec['name'] = 'foo and bar' + rec['price'] = 12234 + rec['date'] = (1992, 7, 15) + rec.store() + + # show the records + print("*** inserted 2 records into tst.dbf: ***") + print(repr(dbft)) + for i1 in range(len(dbft)): + rec = dbft[i1] + for fldName in dbft.fieldNames: + print('%s:\t %s' % (fldName, rec[fldName])) + print() + dbft.close() + +# vim: set et sts=4 sw=4 : diff --git a/src/tablib/packages/dbfpy3/fields.py b/src/tablib/packages/dbfpy3/fields.py new file mode 100644 index 0000000..87916fe --- /dev/null +++ b/src/tablib/packages/dbfpy3/fields.py @@ -0,0 +1,466 @@ +"""DBF fields definitions. + +TODO: + - make memos work +""" +"""History (most recent first): +26-may-2009 [als] DbfNumericFieldDef.decodeValue: strip zero bytes +05-feb-2009 [als] DbfDateFieldDef.encodeValue: empty arg produces empty date +16-sep-2008 [als] DbfNumericFieldDef decoding looks for decimal point + in the value to select float or integer return type +13-mar-2008 [als] check field name length in constructor +11-feb-2007 [als] handle value conversion errors +10-feb-2007 [als] DbfFieldDef: added .rawFromRecord() +01-dec-2006 [als] Timestamp columns use None for empty values +31-oct-2006 [als] support field types 'F' (float), 'I' (integer) + and 'Y' (currency); + automate export and registration of field classes +04-jul-2006 [als] added export declaration +10-mar-2006 [als] decode empty values for Date and Logical fields; + show field name in errors +10-mar-2006 [als] fix Numeric value decoding: according to spec, + value always is string representation of the number; + ensure that encoded Numeric value fits into the field +20-dec-2005 [yc] use field names in upper case +15-dec-2005 [yc] field definitions moved from `dbf`. +""" + +__version__ = "$Revision: 1.14 $"[11:-2] +__date__ = "$Date: 2009/05/26 05:16:51 $"[7:-2] + +__all__ = ["lookupFor",] # field classes added at the end of the module + +import datetime +import struct +import sys + +from . import utils + +## abstract definitions + +class DbfFieldDef(object): + """Abstract field definition. + + Child classes must override ``type`` class attribute to provide datatype + infromation of the field definition. For more info about types visit + `http://www.clicketyclick.dk/databases/xbase/format/data_types.html` + + Also child classes must override ``defaultValue`` field to provide + default value for the field value. + + If child class has fixed length ``length`` class attribute must be + overriden and set to the valid value. None value means, that field + isn't of fixed length. + + Note: ``name`` field must not be changed after instantiation. + + """ + + __slots__ = ("name", "decimalCount", + "start", "end", "ignoreErrors") + + # length of the field, None in case of variable-length field, + # or a number if this field is a fixed-length field + length = None + + # field type. for more information about fields types visit + # `http://www.clicketyclick.dk/databases/xbase/format/data_types.html` + # must be overriden in child classes + typeCode = None + + # default value for the field. this field must be + # overriden in child classes + defaultValue = None + + def __init__(self, name, length=None, decimalCount=None, + start=None, stop=None, ignoreErrors=False, + ): + """Initialize instance.""" + assert self.typeCode is not None, "Type code must be overriden" + assert self.defaultValue is not None, "Default value must be overriden" + ## fix arguments + if len(name) >10: + raise ValueError("Field name \"%s\" is too long" % name) + name = str(name).upper() + if self.__class__.length is None: + if length is None: + raise ValueError("[%s] Length isn't specified" % name) + length = int(length) + if length <= 0: + raise ValueError("[%s] Length must be a positive integer" + % name) + else: + length = self.length + if decimalCount is None: + decimalCount = 0 + ## set fields + self.name = name + # FIXME: validate length according to the specification at + # http://www.clicketyclick.dk/databases/xbase/format/data_types.html + self.length = length + self.decimalCount = decimalCount + self.ignoreErrors = ignoreErrors + self.start = start + self.end = stop + + def __cmp__(self, other): + return cmp(self.name, str(other).upper()) + + def __hash__(self): + return hash(self.name) + + def fromString(cls, string, start, ignoreErrors=False): + """Decode dbf field definition from the string data. + + Arguments: + string: + a string, dbf definition is decoded from. length of + the string must be 32 bytes. + start: + position in the database file. + ignoreErrors: + initial error processing mode for the new field (boolean) + + """ + assert len(string) == 32 + _length = string[16] + return cls(utils.unzfill(string)[:11].decode('utf-8'), _length, + string[17], start, start + _length, ignoreErrors=ignoreErrors) + fromString = classmethod(fromString) + + def toString(self): + """Return encoded field definition. + + Return: + Return value is a string object containing encoded + definition of this field. + + """ + if sys.version_info < (2, 4): + # earlier versions did not support padding character + _name = self.name[:11] + "\0" * (11 - len(self.name)) + else: + _name = self.name.ljust(11, '\0') + return ( + _name + + self.typeCode + + #data address + chr(0) * 4 + + chr(self.length) + + chr(self.decimalCount) + + chr(0) * 14 + ) + + def __repr__(self): + return "%-10s %1s %3d %3d" % self.fieldInfo() + + def fieldInfo(self): + """Return field information. + + Return: + Return value is a (name, type, length, decimals) tuple. + + """ + return (self.name, self.typeCode, self.length, self.decimalCount) + + def rawFromRecord(self, record): + """Return a "raw" field value from the record string.""" + return record[self.start:self.end] + + def decodeFromRecord(self, record): + """Return decoded field value from the record string.""" + try: + return self.decodeValue(self.rawFromRecord(record)) + except: + if self.ignoreErrors: + return utils.INVALID_VALUE + else: + raise + + def decodeValue(self, value): + """Return decoded value from string value. + + This method shouldn't be used publicly. It's called from the + `decodeFromRecord` method. + + This is an abstract method and it must be overridden in child classes. + """ + raise NotImplementedError + + def encodeValue(self, value): + """Return str object containing encoded field value. + + This is an abstract method and it must be overriden in child classes. + """ + raise NotImplementedError + +## real classes + +class DbfCharacterFieldDef(DbfFieldDef): + """Definition of the character field.""" + + typeCode = "C" + defaultValue = b'' + + def decodeValue(self, value): + """Return string object. + + Return value is a ``value`` argument with stripped right spaces. + + """ + return value.rstrip(b' ').decode('utf-8') + + def encodeValue(self, value): + """Return raw data string encoded from a ``value``.""" + return str(value)[:self.length].ljust(self.length) + + +class DbfNumericFieldDef(DbfFieldDef): + """Definition of the numeric field.""" + + typeCode = "N" + # XXX: now I'm not sure it was a good idea to make a class field + # `defaultValue` instead of a generic method as it was implemented + # previously -- it's ok with all types except number, cuz + # if self.decimalCount is 0, we should return 0 and 0.0 otherwise. + defaultValue = 0 + + def decodeValue(self, value): + """Return a number decoded from ``value``. + + If decimals is zero, value will be decoded as an integer; + or as a float otherwise. + + Return: + Return value is a int (long) or float instance. + + """ + value = value.strip(b' \0') + if b'.' in value: + # a float (has decimal separator) + return float(value) + elif value: + # must be an integer + return int(value) + else: + return 0 + + def encodeValue(self, value): + """Return string containing encoded ``value``.""" + _rv = ("%*.*f" % (self.length, self.decimalCount, value)) + if len(_rv) > self.length: + _ppos = _rv.find(".") + if 0 <= _ppos <= self.length: + _rv = _rv[:self.length] + else: + raise ValueError("[%s] Numeric overflow: %s (field width: %i)" + % (self.name, _rv, self.length)) + return _rv + +class DbfFloatFieldDef(DbfNumericFieldDef): + """Definition of the float field - same as numeric.""" + + typeCode = "F" + +class DbfIntegerFieldDef(DbfFieldDef): + """Definition of the integer field.""" + + typeCode = "I" + length = 4 + defaultValue = 0 + + def decodeValue(self, value): + """Return an integer number decoded from ``value``.""" + return struct.unpack("<i", value)[0] + + def encodeValue(self, value): + """Return string containing encoded ``value``.""" + return struct.pack("<i", int(value)) + +class DbfCurrencyFieldDef(DbfFieldDef): + """Definition of the currency field.""" + + typeCode = "Y" + length = 8 + defaultValue = 0.0 + + def decodeValue(self, value): + """Return float number decoded from ``value``.""" + return struct.unpack("<q", value)[0] / 10000. + + def encodeValue(self, value): + """Return string containing encoded ``value``.""" + return struct.pack("<q", round(value * 10000)) + +class DbfLogicalFieldDef(DbfFieldDef): + """Definition of the logical field.""" + + typeCode = "L" + defaultValue = -1 + length = 1 + + def decodeValue(self, value): + """Return True, False or -1 decoded from ``value``.""" + # Note: value always is 1-char string + if value == "?": + return -1 + if value in "NnFf ": + return False + if value in "YyTt": + return True + raise ValueError("[%s] Invalid logical value %r" % (self.name, value)) + + def encodeValue(self, value): + """Return a character from the "TF?" set. + + Return: + Return value is "T" if ``value`` is True + "?" if value is -1 or False otherwise. + + """ + if value is True: + return "T" + if value == -1: + return "?" + return "F" + + +class DbfMemoFieldDef(DbfFieldDef): + """Definition of the memo field. + + Note: memos aren't currenly completely supported. + + """ + + typeCode = "M" + defaultValue = " " * 10 + length = 10 + + def decodeValue(self, value): + """Return int .dbt block number decoded from the string object.""" + #return int(value) + raise NotImplementedError + + def encodeValue(self, value): + """Return raw data string encoded from a ``value``. + + Note: this is an internal method. + + """ + #return str(value)[:self.length].ljust(self.length) + raise NotImplementedError + + +class DbfDateFieldDef(DbfFieldDef): + """Definition of the date field.""" + + typeCode = "D" + defaultValue = utils.classproperty(lambda cls: datetime.date.today()) + # "yyyymmdd" gives us 8 characters + length = 8 + + def decodeValue(self, value): + """Return a ``datetime.date`` instance decoded from ``value``.""" + if value.strip(): + return utils.getDate(value) + else: + return None + + def encodeValue(self, value): + """Return a string-encoded value. + + ``value`` argument should be a value suitable for the + `utils.getDate` call. + + Return: + Return value is a string in format "yyyymmdd". + + """ + if value: + return utils.getDate(value).strftime("%Y%m%d") + else: + return " " * self.length + + +class DbfDateTimeFieldDef(DbfFieldDef): + """Definition of the timestamp field.""" + + # a difference between JDN (Julian Day Number) + # and GDN (Gregorian Day Number). note, that GDN < JDN + JDN_GDN_DIFF = 1721425 + typeCode = "T" + defaultValue = utils.classproperty(lambda cls: datetime.datetime.now()) + # two 32-bits integers representing JDN and amount of + # milliseconds respectively gives us 8 bytes. + # note, that values must be encoded in LE byteorder. + length = 8 + + def decodeValue(self, value): + """Return a `datetime.datetime` instance.""" + assert len(value) == self.length + # LE byteorder + _jdn, _msecs = struct.unpack("<2I", value) + if _jdn >= 1: + _rv = datetime.datetime.fromordinal(_jdn - self.JDN_GDN_DIFF) + _rv += datetime.timedelta(0, _msecs / 1000.0) + else: + # empty date + _rv = None + return _rv + + def encodeValue(self, value): + """Return a string-encoded ``value``.""" + if value: + value = utils.getDateTime(value) + # LE byteorder + _rv = struct.pack("<2I", value.toordinal() + self.JDN_GDN_DIFF, + (value.hour * 3600 + value.minute * 60 + value.second) * 1000) + else: + _rv = "\0" * self.length + assert len(_rv) == self.length + return _rv + + +_fieldsRegistry = {} + +def registerField(fieldCls): + """Register field definition class. + + ``fieldCls`` should be subclass of the `DbfFieldDef`. + + Use `lookupFor` to retrieve field definition class + by the type code. + + """ + assert fieldCls.typeCode is not None, "Type code isn't defined" + # XXX: use fieldCls.typeCode.upper()? in case of any decign + # don't forget to look to the same comment in ``lookupFor`` method + _fieldsRegistry[fieldCls.typeCode] = fieldCls + + +def lookupFor(typeCode): + """Return field definition class for the given type code. + + ``typeCode`` must be a single character. That type should be + previously registered. + + Use `registerField` to register new field class. + + Return: + Return value is a subclass of the `DbfFieldDef`. + + """ + # XXX: use typeCode.upper()? in case of any decign don't + # forget to look to the same comment in ``registerField`` + return _fieldsRegistry[chr(typeCode)] + +## register generic types + +for (_name, _val) in list(globals().items()): + if isinstance(_val, type) and issubclass(_val, DbfFieldDef) \ + and (_name != "DbfFieldDef"): + __all__.append(_name) + registerField(_val) +del _name, _val + +# vim: et sts=4 sw=4 : diff --git a/src/tablib/packages/dbfpy3/header.py b/src/tablib/packages/dbfpy3/header.py new file mode 100644 index 0000000..6c0dc4f --- /dev/null +++ b/src/tablib/packages/dbfpy3/header.py @@ -0,0 +1,273 @@ +"""DBF header definition. + +TODO: + - handle encoding of the character fields + (encoding information stored in the DBF header) + +""" +"""History (most recent first): +16-sep-2010 [als] fromStream: fix century of the last update field +11-feb-2007 [als] added .ignoreErrors +10-feb-2007 [als] added __getitem__: return field definitions + by field name or field number (zero-based) +04-jul-2006 [als] added export declaration +15-dec-2005 [yc] created +""" + +__version__ = "$Revision: 1.6 $"[11:-2] +__date__ = "$Date: 2010/09/16 05:06:39 $"[7:-2] + +__all__ = ["DbfHeader"] + +import io +import datetime +import struct +import time +import sys + +from . import fields +from .utils import getDate + + +class DbfHeader(object): + """Dbf header definition. + + For more information about dbf header format visit + `http://www.clicketyclick.dk/databases/xbase/format/dbf.html#DBF_STRUCT` + + Examples: + Create an empty dbf header and add some field definitions: + dbfh = DbfHeader() + dbfh.addField(("name", "C", 10)) + dbfh.addField(("date", "D")) + dbfh.addField(DbfNumericFieldDef("price", 5, 2)) + Create a dbf header with field definitions: + dbfh = DbfHeader([ + ("name", "C", 10), + ("date", "D"), + DbfNumericFieldDef("price", 5, 2), + ]) + + """ + + __slots__ = ("signature", "fields", "lastUpdate", "recordLength", + "recordCount", "headerLength", "changed", "_ignore_errors") + + ## instance construction and initialization methods + + def __init__(self, fields=None, headerLength=0, recordLength=0, + recordCount=0, signature=0x03, lastUpdate=None, ignoreErrors=False, + ): + """Initialize instance. + + Arguments: + fields: + a list of field definitions; + recordLength: + size of the records; + headerLength: + size of the header; + recordCount: + number of records stored in DBF; + signature: + version number (aka signature). using 0x03 as a default meaning + "File without DBT". for more information about this field visit + ``http://www.clicketyclick.dk/databases/xbase/format/dbf.html#DBF_NOTE_1_TARGET`` + lastUpdate: + date of the DBF's update. this could be a string ('yymmdd' or + 'yyyymmdd'), timestamp (int or float), datetime/date value, + a sequence (assuming (yyyy, mm, dd, ...)) or an object having + callable ``ticks`` field. + ignoreErrors: + error processing mode for DBF fields (boolean) + + """ + self.signature = signature + if fields is None: + self.fields = [] + else: + self.fields = list(fields) + self.lastUpdate = getDate(lastUpdate) + self.recordLength = recordLength + self.headerLength = headerLength + self.recordCount = recordCount + self.ignoreErrors = ignoreErrors + # XXX: I'm not sure this is safe to + # initialize `self.changed` in this way + self.changed = bool(self.fields) + + # @classmethod + def fromString(cls, string): + """Return header instance from the string object.""" + return cls.fromStream(io.StringIO(str(string))) + fromString = classmethod(fromString) + + # @classmethod + def fromStream(cls, stream): + """Return header object from the stream.""" + stream.seek(0) + first_32 = stream.read(32) + if type(first_32) != bytes: + _data = bytes(first_32, sys.getfilesystemencoding()) + _data = first_32 + (_cnt, _hdrLen, _recLen) = struct.unpack("<I2H", _data[4:12]) + #reserved = _data[12:32] + _year = _data[1] + if _year < 80: + # dBase II started at 1980. It is quite unlikely + # that actual last update date is before that year. + _year += 2000 + else: + _year += 1900 + ## create header object + _obj = cls(None, _hdrLen, _recLen, _cnt, _data[0], + (_year, _data[2], _data[3])) + ## append field definitions + # position 0 is for the deletion flag + _pos = 1 + _data = stream.read(1) + while _data != b'\r': + _data += stream.read(31) + _fld = fields.lookupFor(_data[11]).fromString(_data, _pos) + _obj._addField(_fld) + _pos = _fld.end + _data = stream.read(1) + return _obj + fromStream = classmethod(fromStream) + + ## properties + + year = property(lambda self: self.lastUpdate.year) + month = property(lambda self: self.lastUpdate.month) + day = property(lambda self: self.lastUpdate.day) + + def ignoreErrors(self, value): + """Update `ignoreErrors` flag on self and all fields""" + self._ignore_errors = value = bool(value) + for _field in self.fields: + _field.ignoreErrors = value + ignoreErrors = property( + lambda self: self._ignore_errors, + ignoreErrors, + doc="""Error processing mode for DBF field value conversion + + if set, failing field value conversion will return + ``INVALID_VALUE`` instead of raising conversion error. + + """) + + ## object representation + + def __repr__(self): + _rv = """\ +Version (signature): 0x%02x + Last update: %s + Header length: %d + Record length: %d + Record count: %d + FieldName Type Len Dec +""" % (self.signature, self.lastUpdate, self.headerLength, + self.recordLength, self.recordCount) + _rv += "\n".join( + ["%10s %4s %3s %3s" % _fld.fieldInfo() for _fld in self.fields] + ) + return _rv + + ## internal methods + + def _addField(self, *defs): + """Internal variant of the `addField` method. + + This method doesn't set `self.changed` field to True. + + Return value is a length of the appended records. + Note: this method doesn't modify ``recordLength`` and + ``headerLength`` fields. Use `addField` instead of this + method if you don't exactly know what you're doing. + + """ + # insure we have dbf.DbfFieldDef instances first (instantiation + # from the tuple could raise an error, in such a case I don't + # wanna add any of the definitions -- all will be ignored) + _defs = [] + _recordLength = 0 + for _def in defs: + if isinstance(_def, fields.DbfFieldDef): + _obj = _def + else: + (_name, _type, _len, _dec) = (tuple(_def) + (None,) * 4)[:4] + _cls = fields.lookupFor(_type) + _obj = _cls(_name, _len, _dec, + ignoreErrors=self._ignore_errors) + _recordLength += _obj.length + _defs.append(_obj) + # and now extend field definitions and + # update record length + self.fields += _defs + return _recordLength + + ## interface methods + + def addField(self, *defs): + """Add field definition to the header. + + Examples: + dbfh.addField( + ("name", "C", 20), + dbf.DbfCharacterFieldDef("surname", 20), + dbf.DbfDateFieldDef("birthdate"), + ("member", "L"), + ) + dbfh.addField(("price", "N", 5, 2)) + dbfh.addField(dbf.DbfNumericFieldDef("origprice", 5, 2)) + + """ + _oldLen = self.recordLength + self.recordLength += self._addField(*defs) + if not _oldLen: + self.recordLength += 1 + # XXX: may be just use: + # self.recordeLength += self._addField(*defs) + bool(not _oldLen) + # recalculate headerLength + self.headerLength = 32 + (32 * len(self.fields)) + 1 + self.changed = True + + def write(self, stream): + """Encode and write header to the stream.""" + stream.seek(0) + stream.write(self.toString()) + fields = [_fld.toString() for _fld in self.fields] + stream.write(''.join(fields).encode(sys.getfilesystemencoding())) + stream.write(b'\x0D') # cr at end of all header data + self.changed = False + + def toString(self): + """Returned 32 chars length string with encoded header.""" + return struct.pack("<4BI2H", + self.signature, + self.year - 1900, + self.month, + self.day, + self.recordCount, + self.headerLength, + self.recordLength) + (b'\x00' * 20) + #TODO: figure out if bytes(utf-8) is correct here. + + def setCurrentDate(self): + """Update ``self.lastUpdate`` field with current date value.""" + self.lastUpdate = datetime.date.today() + + def __getitem__(self, item): + """Return a field definition by numeric index or name string""" + if isinstance(item, str): + _name = item.upper() + for _field in self.fields: + if _field.name == _name: + return _field + else: + raise KeyError(item) + else: + # item must be field index + return self.fields[item] + +# vim: et sts=4 sw=4 : diff --git a/src/tablib/packages/dbfpy3/record.py b/src/tablib/packages/dbfpy3/record.py new file mode 100644 index 0000000..d301476 --- /dev/null +++ b/src/tablib/packages/dbfpy3/record.py @@ -0,0 +1,266 @@ +"""DBF record definition. + +""" +"""History (most recent first): +11-feb-2007 [als] __repr__: added special case for invalid field values +10-feb-2007 [als] added .rawFromStream() +30-oct-2006 [als] fix record length in .fromStream() +04-jul-2006 [als] added export declaration +20-dec-2005 [yc] DbfRecord.write() -> DbfRecord._write(); + added delete() method. +16-dec-2005 [yc] record definition moved from `dbf`. +""" + +__version__ = "$Revision: 1.7 $"[11:-2] +__date__ = "$Date: 2007/02/11 09:05:49 $"[7:-2] + +__all__ = ["DbfRecord"] + +import sys + +from . import utils + +class DbfRecord(object): + """DBF record. + + Instances of this class shouldn't be created manualy, + use `dbf.Dbf.newRecord` instead. + + Class implements mapping/sequence interface, so + fields could be accessed via their names or indexes + (names is a preffered way to access fields). + + Hint: + Use `store` method to save modified record. + + Examples: + Add new record to the database: + db = Dbf(filename) + rec = db.newRecord() + rec["FIELD1"] = value1 + rec["FIELD2"] = value2 + rec.store() + Or the same, but modify existed + (second in this case) record: + db = Dbf(filename) + rec = db[2] + rec["FIELD1"] = value1 + rec["FIELD2"] = value2 + rec.store() + + """ + + __slots__ = "dbf", "index", "deleted", "fieldData" + + ## creation and initialization + + def __init__(self, dbf, index=None, deleted=False, data=None): + """Instance initialiation. + + Arguments: + dbf: + A `Dbf.Dbf` instance this record belonogs to. + index: + An integer record index or None. If this value is + None, record will be appended to the DBF. + deleted: + Boolean flag indicating whether this record + is a deleted record. + data: + A sequence or None. This is a data of the fields. + If this argument is None, default values will be used. + + """ + self.dbf = dbf + # XXX: I'm not sure ``index`` is necessary + self.index = index + self.deleted = deleted + if data is None: + self.fieldData = [_fd.defaultValue for _fd in dbf.header.fields] + else: + self.fieldData = list(data) + + # XXX: validate self.index before calculating position? + position = property(lambda self: self.dbf.header.headerLength + \ + self.index * self.dbf.header.recordLength) + + def rawFromStream(cls, dbf, index): + """Return raw record contents read from the stream. + + Arguments: + dbf: + A `Dbf.Dbf` instance containing the record. + index: + Index of the record in the records' container. + This argument can't be None in this call. + + Return value is a string containing record data in DBF format. + + """ + # XXX: may be write smth assuming, that current stream + # position is the required one? it could save some + # time required to calculate where to seek in the file + dbf.stream.seek(dbf.header.headerLength + + index * dbf.header.recordLength) + return dbf.stream.read(dbf.header.recordLength) + rawFromStream = classmethod(rawFromStream) + + def fromStream(cls, dbf, index): + """Return a record read from the stream. + + Arguments: + dbf: + A `Dbf.Dbf` instance new record should belong to. + index: + Index of the record in the records' container. + This argument can't be None in this call. + + Return value is an instance of the current class. + + """ + return cls.fromString(dbf, cls.rawFromStream(dbf, index), index) + fromStream = classmethod(fromStream) + + def fromString(cls, dbf, string, index=None): + """Return record read from the string object. + + Arguments: + dbf: + A `Dbf.Dbf` instance new record should belong to. + string: + A string new record should be created from. + index: + Index of the record in the container. If this + argument is None, record will be appended. + + Return value is an instance of the current class. + + """ + return cls(dbf, index, string[0]=="*", + [_fd.decodeFromRecord(string) for _fd in dbf.header.fields]) + fromString = classmethod(fromString) + + ## object representation + + def __repr__(self): + _template = "%%%ds: %%s (%%s)" % max([len(_fld) + for _fld in self.dbf.fieldNames]) + _rv = [] + for _fld in self.dbf.fieldNames: + _val = self[_fld] + if _val is utils.INVALID_VALUE: + _rv.append(_template % + (_fld, "None", "value cannot be decoded")) + else: + _rv.append(_template % (_fld, _val, type(_val))) + return "\n".join(_rv) + + ## protected methods + + def _write(self): + """Write data to the dbf stream. + + Note: + This isn't a public method, it's better to + use 'store' instead publically. + Be design ``_write`` method should be called + only from the `Dbf` instance. + + + """ + self._validateIndex(False) + self.dbf.stream.seek(self.position) + self.dbf.stream.write(bytes(self.toString(), + sys.getfilesystemencoding())) + # FIXME: may be move this write somewhere else? + # why we should check this condition for each record? + if self.index == len(self.dbf): + # this is the last record, + # we should write SUB (ASCII 26) + self.dbf.stream.write(b"\x1A") + + ## utility methods + + def _validateIndex(self, allowUndefined=True, checkRange=False): + """Valid ``self.index`` value. + + If ``allowUndefined`` argument is True functions does nothing + in case of ``self.index`` pointing to None object. + + """ + if self.index is None: + if not allowUndefined: + raise ValueError("Index is undefined") + elif self.index < 0: + raise ValueError("Index can't be negative (%s)" % self.index) + elif checkRange and self.index <= self.dbf.header.recordCount: + raise ValueError("There are only %d records in the DBF" % + self.dbf.header.recordCount) + + ## interface methods + + def store(self): + """Store current record in the DBF. + + If ``self.index`` is None, this record will be appended to the + records of the DBF this records belongs to; or replaced otherwise. + + """ + self._validateIndex() + if self.index is None: + self.index = len(self.dbf) + self.dbf.append(self) + else: + self.dbf[self.index] = self + + def delete(self): + """Mark method as deleted.""" + self.deleted = True + + def toString(self): + """Return string packed record values.""" +# for (_def, _dat) in zip(self.dbf.header.fields, self.fieldData): +# + + return "".join([" *"[self.deleted]] + [ + _def.encodeValue(_dat) + for (_def, _dat) in zip(self.dbf.header.fields, self.fieldData) + ]) + + def asList(self): + """Return a flat list of fields. + + Note: + Change of the list's values won't change + real values stored in this object. + + """ + return self.fieldData[:] + + def asDict(self): + """Return a dictionary of fields. + + Note: + Change of the dicts's values won't change + real values stored in this object. + + """ + return dict([_i for _i in zip(self.dbf.fieldNames, self.fieldData)]) + + def __getitem__(self, key): + """Return value by field name or field index.""" + if isinstance(key, int): + # integer index of the field + return self.fieldData[key] + # assuming string field name + return self.fieldData[self.dbf.indexOfFieldName(key)] + + def __setitem__(self, key, value): + """Set field value by integer index of the field or string name.""" + if isinstance(key, int): + # integer index of the field + return self.fieldData[key] + # assuming string field name + self.fieldData[self.dbf.indexOfFieldName(key)] = value + +# vim: et sts=4 sw=4 : diff --git a/src/tablib/packages/dbfpy3/utils.py b/src/tablib/packages/dbfpy3/utils.py new file mode 100644 index 0000000..856ade8 --- /dev/null +++ b/src/tablib/packages/dbfpy3/utils.py @@ -0,0 +1,170 @@ +"""String utilities. + +TODO: + - allow strings in getDateTime routine; +""" +"""History (most recent first): +11-feb-2007 [als] added INVALID_VALUE +10-feb-2007 [als] allow date strings padded with spaces instead of zeroes +20-dec-2005 [yc] handle long objects in getDate/getDateTime +16-dec-2005 [yc] created from ``strutil`` module. +""" + +__version__ = "$Revision: 1.4 $"[11:-2] +__date__ = "$Date: 2007/02/11 08:57:17 $"[7:-2] + +import datetime +import time + + +def unzfill(str): + """Return a string without ASCII NULs. + + This function searchers for the first NUL (ASCII 0) occurance + and truncates string till that position. + + """ + try: + return str[:str.index(b'\0')] + except ValueError: + return str + + +def getDate(date=None): + """Return `datetime.date` instance. + + Type of the ``date`` argument could be one of the following: + None: + use current date value; + datetime.date: + this value will be returned; + datetime.datetime: + the result of the date.date() will be returned; + string: + assuming "%Y%m%d" or "%y%m%dd" format; + number: + assuming it's a timestamp (returned for example + by the time.time() call; + sequence: + assuming (year, month, day, ...) sequence; + + Additionaly, if ``date`` has callable ``ticks`` attribute, + it will be used and result of the called would be treated + as a timestamp value. + + """ + if date is None: + # use current value + return datetime.date.today() + if isinstance(date, datetime.date): + return date + if isinstance(date, datetime.datetime): + return date.date() + if isinstance(date, (int, float)): + # date is a timestamp + return datetime.date.fromtimestamp(date) + if isinstance(date, str): + date = date.replace(" ", "0") + if len(date) == 6: + # yymmdd + return datetime.date(*time.strptime(date, "%y%m%d")[:3]) + # yyyymmdd + return datetime.date(*time.strptime(date, "%Y%m%d")[:3]) + if hasattr(date, "__getitem__"): + # a sequence (assuming date/time tuple) + return datetime.date(*date[:3]) + return datetime.date.fromtimestamp(date.ticks()) + + +def getDateTime(value=None): + """Return `datetime.datetime` instance. + + Type of the ``value`` argument could be one of the following: + None: + use current date value; + datetime.date: + result will be converted to the `datetime.datetime` instance + using midnight; + datetime.datetime: + ``value`` will be returned as is; + string: + *** CURRENTLY NOT SUPPORTED ***; + number: + assuming it's a timestamp (returned for example + by the time.time() call; + sequence: + assuming (year, month, day, ...) sequence; + + Additionaly, if ``value`` has callable ``ticks`` attribute, + it will be used and result of the called would be treated + as a timestamp value. + + """ + if value is None: + # use current value + return datetime.datetime.today() + if isinstance(value, datetime.datetime): + return value + if isinstance(value, datetime.date): + return datetime.datetime.fromordinal(value.toordinal()) + if isinstance(value, (int, float)): + # value is a timestamp + return datetime.datetime.fromtimestamp(value) + if isinstance(value, str): + raise NotImplementedError("Strings aren't currently implemented") + if hasattr(value, "__getitem__"): + # a sequence (assuming date/time tuple) + return datetime.datetime(*tuple(value)[:6]) + return datetime.datetime.fromtimestamp(value.ticks()) + + +class classproperty(property): + """Works in the same way as a ``property``, but for the classes.""" + + def __get__(self, obj, cls): + return self.fget(cls) + + +class _InvalidValue(object): + + """Value returned from DBF records when field validation fails + + The value is not equal to anything except for itself + and equal to all empty values: None, 0, empty string etc. + In other words, invalid value is equal to None and not equal + to None at the same time. + + This value yields zero upon explicit conversion to a number type, + empty string for string types, and False for boolean. + + """ + + def __eq__(self, other): + return not other + + def __ne__(self, other): + return not (other is self) + + def __bool__(self): + return False + + def __int__(self): + return 0 + __long__ = __int__ + + def __float__(self): + return 0.0 + + def __str__(self): + return "" + + def __unicode__(self): + return "" + + def __repr__(self): + return "<INVALID>" + +# invalid value is a constant singleton +INVALID_VALUE = _InvalidValue() + +# vim: set et sts=4 sw=4 : diff --git a/src/tablib/packages/statistics.py b/src/tablib/packages/statistics.py new file mode 100644 index 0000000..e97a6c9 --- /dev/null +++ b/src/tablib/packages/statistics.py @@ -0,0 +1,24 @@ +from __future__ import division + + +def median(data): + """ + Return the median (middle value) of numeric data, using the common + "mean of middle two" method. If data is empty, ValueError is raised. + + Mimics the behaviour of Python3's statistics.median + + >>> median([1, 3, 5]) + 3 + >>> median([1, 3, 5, 7]) + 4.0 + + """ + data = sorted(data) + n = len(data) + if not n: + raise ValueError("No median for empty data") + i = n // 2 + if n % 2: + return data[i] + return (data[i - 1] + data[i]) / 2 |
