summaryrefslogtreecommitdiff
path: root/src
diff options
context:
space:
mode:
authorJannis Leidel <jannis@leidel.info>2019-10-18 15:57:13 +0200
committerGitHub <noreply@github.com>2019-10-18 15:57:13 +0200
commitf6bf14afd22d8e5b706670590cc95f29d4483434 (patch)
treea65fbc8b6fd1112222beb6f66c31082ffe231460 /src
parentf3d02aa3b088e2f9044fe5e4869e8c8cb91d2cdc (diff)
downloadtablib-f6bf14afd22d8e5b706670590cc95f29d4483434.tar.gz
Add project release config and cleanup project setup. (#398)
* Add project release config and use Travis build stages. Refs #378. * Restructure project to use src/ and tests/ directories. * Fix testing. * Remove eggs. * More fixes. - isort and flake8 config - manifest template update - tox ini extension - docs build fixes - docs content fixes * Docs and license cleanup.
Diffstat (limited to 'src')
-rw-r--r--src/tablib/__init__.py13
-rw-r--r--src/tablib/compat.py36
-rw-r--r--src/tablib/core.py1160
-rw-r--r--src/tablib/formats/__init__.py21
-rw-r--r--src/tablib/formats/_csv.py59
-rw-r--r--src/tablib/formats/_dbf.py87
-rw-r--r--src/tablib/formats/_df.py43
-rw-r--r--src/tablib/formats/_html.py65
-rw-r--r--src/tablib/formats/_jira.py39
-rw-r--r--src/tablib/formats/_json.py59
-rw-r--r--src/tablib/formats/_latex.py134
-rw-r--r--src/tablib/formats/_ods.py104
-rw-r--r--src/tablib/formats/_rst.py273
-rw-r--r--src/tablib/formats/_tsv.py30
-rw-r--r--src/tablib/formats/_xls.py137
-rw-r--r--src/tablib/formats/_xlsx.py144
-rw-r--r--src/tablib/formats/_yaml.py53
-rw-r--r--src/tablib/packages/__init__.py0
-rw-r--r--src/tablib/packages/dbfpy/__init__.py0
-rw-r--r--src/tablib/packages/dbfpy/dbf.py297
-rw-r--r--src/tablib/packages/dbfpy/dbfnew.py189
-rw-r--r--src/tablib/packages/dbfpy/fields.py466
-rw-r--r--src/tablib/packages/dbfpy/header.py275
-rw-r--r--src/tablib/packages/dbfpy/record.py262
-rw-r--r--src/tablib/packages/dbfpy/utils.py170
-rw-r--r--src/tablib/packages/dbfpy3/__init__.py0
-rw-r--r--src/tablib/packages/dbfpy3/dbf.py297
-rw-r--r--src/tablib/packages/dbfpy3/dbfnew.py183
-rw-r--r--src/tablib/packages/dbfpy3/fields.py466
-rw-r--r--src/tablib/packages/dbfpy3/header.py273
-rw-r--r--src/tablib/packages/dbfpy3/record.py266
-rw-r--r--src/tablib/packages/dbfpy3/utils.py170
-rw-r--r--src/tablib/packages/statistics.py24
33 files changed, 5795 insertions, 0 deletions
diff --git a/src/tablib/__init__.py b/src/tablib/__init__.py
new file mode 100644
index 0000000..fafb569
--- /dev/null
+++ b/src/tablib/__init__.py
@@ -0,0 +1,13 @@
+""" Tablib. """
+from pkg_resources import get_distribution, DistributionNotFound
+
+from tablib.core import (
+ Databook, Dataset, detect_format, import_set, import_book,
+ InvalidDatasetType, InvalidDimensions, UnsupportedFormat
+)
+
+try:
+ __version__ = get_distribution(__name__).version
+except DistributionNotFound:
+ # package is not installed
+ __version__ = None
diff --git a/src/tablib/compat.py b/src/tablib/compat.py
new file mode 100644
index 0000000..1582173
--- /dev/null
+++ b/src/tablib/compat.py
@@ -0,0 +1,36 @@
+# -*- coding: utf-8 -*-
+
+"""
+tablib.compat
+~~~~~~~~~~~~~
+
+Tablib compatiblity module.
+
+"""
+
+import sys
+
+is_py3 = (sys.version_info[0] > 2)
+
+
+if is_py3:
+ from io import StringIO
+ from statistics import median
+ from itertools import zip_longest as izip_longest
+ import csv
+ import tablib.packages.dbfpy3 as dbfpy
+
+ unicode = str
+ xrange = range
+
+else:
+ from StringIO import StringIO
+ from tablib.packages.statistics import median
+ from itertools import izip_longest
+ from backports import csv
+ import tablib.packages.dbfpy as dbfpy
+
+ unicode = unicode
+ xrange = xrange
+
+from MarkupPy import markup # Kept temporarily to avoid breaking existing imports
diff --git a/src/tablib/core.py b/src/tablib/core.py
new file mode 100644
index 0000000..65dd901
--- /dev/null
+++ b/src/tablib/core.py
@@ -0,0 +1,1160 @@
+# -*- coding: utf-8 -*-
+"""
+ tablib.core
+ ~~~~~~~~~~~
+
+ This module implements the central Tablib objects.
+
+ :copyright: (c) 2016 by Kenneth Reitz. 2019 Jazzband.
+ :license: MIT, see LICENSE for more details.
+"""
+
+from collections import OrderedDict
+from copy import copy
+from operator import itemgetter
+
+from tablib import formats
+
+from tablib.compat import unicode
+
+
+__title__ = 'tablib'
+__author__ = 'Kenneth Reitz'
+__license__ = 'MIT'
+__copyright__ = 'Copyright 2017 Kenneth Reitz. 2019 Jazzband.'
+__docformat__ = 'restructuredtext'
+
+
+class Row(object):
+ """Internal Row object. Mainly used for filtering."""
+
+ __slots__ = ['_row', 'tags']
+
+ def __init__(self, row=list(), tags=list()):
+ self._row = list(row)
+ self.tags = list(tags)
+
+ def __iter__(self):
+ return (col for col in self._row)
+
+ def __len__(self):
+ return len(self._row)
+
+ def __repr__(self):
+ return repr(self._row)
+
+ def __getslice__(self, i, j):
+ return self._row[i:j]
+
+ def __getitem__(self, i):
+ return self._row[i]
+
+ def __setitem__(self, i, value):
+ self._row[i] = value
+
+ def __delitem__(self, i):
+ del self._row[i]
+
+ def __getstate__(self):
+
+ slots = dict()
+
+ for slot in self.__slots__:
+ attribute = getattr(self, slot)
+ slots[slot] = attribute
+
+ return slots
+
+ def __setstate__(self, state):
+ for (k, v) in list(state.items()): setattr(self, k, v)
+
+ def rpush(self, value):
+ self.insert(0, value)
+
+ def lpush(self, value):
+ self.insert(len(value), value)
+
+ def append(self, value):
+ self.rpush(value)
+
+ def insert(self, index, value):
+ self._row.insert(index, value)
+
+ def __contains__(self, item):
+ return (item in self._row)
+
+ @property
+ def tuple(self):
+ """Tuple representation of :class:`Row`."""
+ return tuple(self._row)
+
+ @property
+ def list(self):
+ """List representation of :class:`Row`."""
+ return list(self._row)
+
+ def has_tag(self, tag):
+ """Returns true if current row contains tag."""
+
+ if tag == None:
+ return False
+ elif isinstance(tag, str):
+ return (tag in self.tags)
+ else:
+ return bool(len(set(tag) & set(self.tags)))
+
+
+class Dataset(object):
+ """The :class:`Dataset` object is the heart of Tablib. It provides all core
+ functionality.
+
+ Usually you create a :class:`Dataset` instance in your main module, and append
+ rows as you collect data. ::
+
+ data = tablib.Dataset()
+ data.headers = ('name', 'age')
+
+ for (name, age) in some_collector():
+ data.append((name, age))
+
+
+ Setting columns is similar. The column data length must equal the
+ current height of the data and headers must be set ::
+
+ data = tablib.Dataset()
+ data.headers = ('first_name', 'last_name')
+
+ data.append(('John', 'Adams'))
+ data.append(('George', 'Washington'))
+
+ data.append_col((90, 67), header='age')
+
+
+ You can also set rows and headers upon instantiation. This is useful if
+ dealing with dozens or hundreds of :class:`Dataset` objects. ::
+
+ headers = ('first_name', 'last_name')
+ data = [('John', 'Adams'), ('George', 'Washington')]
+
+ data = tablib.Dataset(*data, headers=headers)
+
+ :param \\*args: (optional) list of rows to populate Dataset
+ :param headers: (optional) list strings for Dataset header row
+ :param title: (optional) string to use as title of the Dataset
+
+
+ .. admonition:: Format Attributes Definition
+
+ If you look at the code, the various output/import formats are not
+ defined within the :class:`Dataset` object. To add support for a new format, see
+ :ref:`Adding New Formats <newformats>`.
+
+ """
+
+ _formats = {}
+
+ def __init__(self, *args, **kwargs):
+ self._data = list(Row(arg) for arg in args)
+ self.__headers = None
+
+ # ('title', index) tuples
+ self._separators = []
+
+ # (column, callback) tuples
+ self._formatters = []
+
+ self.headers = kwargs.get('headers')
+
+ self.title = kwargs.get('title')
+
+ self._register_formats()
+
+ def __len__(self):
+ return self.height
+
+ def __getitem__(self, key):
+ if isinstance(key, (str, unicode)):
+ if key in self.headers:
+ pos = self.headers.index(key) # get 'key' index from each data
+ return [row[pos] for row in self._data]
+ else:
+ raise KeyError
+ else:
+ _results = self._data[key]
+ if isinstance(_results, Row):
+ return _results.tuple
+ else:
+ return [result.tuple for result in _results]
+
+ def __setitem__(self, key, value):
+ self._validate(value)
+ self._data[key] = Row(value)
+
+ def __delitem__(self, key):
+ if isinstance(key, (str, unicode)):
+
+ if key in self.headers:
+
+ pos = self.headers.index(key)
+ del self.headers[pos]
+
+ for i, row in enumerate(self._data):
+
+ del row[pos]
+ self._data[i] = row
+ else:
+ raise KeyError
+ else:
+ del self._data[key]
+
+ def __repr__(self):
+ try:
+ return '<%s dataset>' % (self.title.lower())
+ except AttributeError:
+ return '<dataset object>'
+
+ def __unicode__(self):
+ result = []
+
+ # Add unicode representation of headers.
+ if self.__headers:
+ result.append([unicode(h) for h in self.__headers])
+
+ # Add unicode representation of rows.
+ result.extend(list(map(unicode, row)) for row in self._data)
+
+ lens = [list(map(len, row)) for row in result]
+ field_lens = list(map(max, zip(*lens)))
+
+ # delimiter between header and data
+ if self.__headers:
+ result.insert(1, ['-' * length for length in field_lens])
+
+ format_string = '|'.join('{%s:%s}' % item for item in enumerate(field_lens))
+
+ return '\n'.join(format_string.format(*row) for row in result)
+
+ def __str__(self):
+ return self.__unicode__()
+
+ # ---------
+ # Internals
+ # ---------
+
+ @classmethod
+ def _register_formats(cls):
+ """Adds format properties."""
+ for fmt in formats.available:
+ try:
+ try:
+ setattr(cls, fmt.title, property(fmt.export_set, fmt.import_set))
+ setattr(cls, 'get_%s' % fmt.title, fmt.export_set)
+ setattr(cls, 'set_%s' % fmt.title, fmt.import_set)
+ cls._formats[fmt.title] = (fmt.export_set, fmt.import_set)
+ except AttributeError:
+ setattr(cls, fmt.title, property(fmt.export_set))
+ setattr(cls, 'get_%s' % fmt.title, fmt.export_set)
+ cls._formats[fmt.title] = (fmt.export_set, None)
+
+ except AttributeError:
+ cls._formats[fmt.title] = (None, None)
+
+ def _validate(self, row=None, col=None, safety=False):
+ """Assures size of every row in dataset is of proper proportions."""
+ if row:
+ is_valid = (len(row) == self.width) if self.width else True
+ elif col:
+ if len(col) < 1:
+ is_valid = True
+ else:
+ is_valid = (len(col) == self.height) if self.height else True
+ else:
+ is_valid = all((len(x) == self.width for x in self._data))
+
+ if is_valid:
+ return True
+ else:
+ if not safety:
+ raise InvalidDimensions
+ return False
+
+ def _package(self, dicts=True, ordered=True):
+ """Packages Dataset into lists of dictionaries for transmission."""
+ # TODO: Dicts default to false?
+
+ _data = list(self._data)
+
+ if ordered:
+ dict_pack = OrderedDict
+ else:
+ dict_pack = dict
+
+ # Execute formatters
+ if self._formatters:
+ for row_i, row in enumerate(_data):
+ for col, callback in self._formatters:
+ try:
+ if col is None:
+ for j, c in enumerate(row):
+ _data[row_i][j] = callback(c)
+ else:
+ _data[row_i][col] = callback(row[col])
+ except IndexError:
+ raise InvalidDatasetIndex
+
+ if self.headers:
+ if dicts:
+ data = [dict_pack(list(zip(self.headers, data_row))) for data_row in _data]
+ else:
+ data = [list(self.headers)] + list(_data)
+ else:
+ data = [list(row) for row in _data]
+
+ return data
+
+ def _get_headers(self):
+ """An *optional* list of strings to be used for header rows and attribute names.
+
+ This must be set manually. The given list length must equal :class:`Dataset.width`.
+
+ """
+ return self.__headers
+
+ def _set_headers(self, collection):
+ """Validating headers setter."""
+ self._validate(collection)
+ if collection:
+ try:
+ self.__headers = list(collection)
+ except TypeError:
+ raise TypeError
+ else:
+ self.__headers = None
+
+ headers = property(_get_headers, _set_headers)
+
+ def _get_dict(self):
+ """A native Python representation of the :class:`Dataset` object. If headers have
+ been set, a list of Python dictionaries will be returned. If no headers have been set,
+ a list of tuples (rows) will be returned instead.
+
+ A dataset object can also be imported by setting the `Dataset.dict` attribute: ::
+
+ data = tablib.Dataset()
+ data.dict = [{'age': 90, 'first_name': 'Kenneth', 'last_name': 'Reitz'}]
+
+ """
+ return self._package()
+
+ def _set_dict(self, pickle):
+ """A native Python representation of the Dataset object. If headers have been
+ set, a list of Python dictionaries will be returned. If no headers have been
+ set, a list of tuples (rows) will be returned instead.
+
+ A dataset object can also be imported by setting the :class:`Dataset.dict` attribute. ::
+
+ data = tablib.Dataset()
+ data.dict = [{'age': 90, 'first_name': 'Kenneth', 'last_name': 'Reitz'}]
+
+ """
+
+ if not len(pickle):
+ return
+
+ # if list of rows
+ if isinstance(pickle[0], list):
+ self.wipe()
+ for row in pickle:
+ self.append(Row(row))
+
+ # if list of objects
+ elif isinstance(pickle[0], dict):
+ self.wipe()
+ self.headers = list(pickle[0].keys())
+ for row in pickle:
+ self.append(Row(list(row.values())))
+ else:
+ raise UnsupportedFormat
+
+ dict = property(_get_dict, _set_dict)
+
+ def _clean_col(self, col):
+ """Prepares the given column for insert/append."""
+
+ col = list(col)
+
+ if self.headers:
+ header = [col.pop(0)]
+ else:
+ header = []
+
+ if len(col) == 1 and hasattr(col[0], '__call__'):
+
+ col = list(map(col[0], self._data))
+ col = tuple(header + col)
+
+ return col
+
+ @property
+ def height(self):
+ """The number of rows currently in the :class:`Dataset`.
+ Cannot be directly modified.
+ """
+ return len(self._data)
+
+ @property
+ def width(self):
+ """The number of columns currently in the :class:`Dataset`.
+ Cannot be directly modified.
+ """
+
+ try:
+ return len(self._data[0])
+ except IndexError:
+ try:
+ return len(self.headers)
+ except TypeError:
+ return 0
+
+ def load(self, in_stream, format=None, **kwargs):
+ """
+ Import `in_stream` to the :class:`Dataset` object using the `format`.
+
+ :param \\*\\*kwargs: (optional) custom configuration to the format `import_set`.
+ """
+
+ if not format:
+ format = detect_format(in_stream)
+
+ export_set, import_set = self._formats.get(format, (None, None))
+ if not import_set:
+ raise UnsupportedFormat('Format {0} cannot be imported.'.format(format))
+
+ import_set(self, in_stream, **kwargs)
+ return self
+
+ def export(self, format, **kwargs):
+ """
+ Export :class:`Dataset` object to `format`.
+
+ :param \\*\\*kwargs: (optional) custom configuration to the format `export_set`.
+ """
+ export_set, import_set = self._formats.get(format, (None, None))
+ if not export_set:
+ raise UnsupportedFormat('Format {0} cannot be exported.'.format(format))
+
+ return export_set(self, **kwargs)
+
+ # -------
+ # Formats
+ # -------
+
+ @property
+ def xls():
+ """A Legacy Excel Spreadsheet representation of the :class:`Dataset` object, with :ref:`separators`. Cannot be set.
+
+ .. note::
+
+ XLS files are limited to a maximum of 65,000 rows. Use :class:`Dataset.xlsx` to avoid this limitation.
+
+ .. admonition:: Binary Warning
+
+ :class:`Dataset.xls` contains binary data, so make sure to write in binary mode::
+
+ with open('output.xls', 'wb') as f:
+ f.write(data.xls)
+ """
+ pass
+
+ @property
+ def xlsx():
+ """An Excel '07+ Spreadsheet representation of the :class:`Dataset` object, with :ref:`separators`. Cannot be set.
+
+ .. admonition:: Binary Warning
+
+ :class:`Dataset.xlsx` contains binary data, so make sure to write in binary mode::
+
+ with open('output.xlsx', 'wb') as f:
+ f.write(data.xlsx)
+ """
+ pass
+
+ @property
+ def ods():
+ """An OpenDocument Spreadsheet representation of the :class:`Dataset` object, with :ref:`separators`. Cannot be set.
+
+ .. admonition:: Binary Warning
+
+ :class:`Dataset.ods` contains binary data, so make sure to write in binary mode::
+
+ with open('output.ods', 'wb') as f:
+ f.write(data.ods)
+ """
+ pass
+
+ @property
+ def csv():
+ """A CSV representation of the :class:`Dataset` object. The top row will contain
+ headers, if they have been set. Otherwise, the top row will contain
+ the first row of the dataset.
+
+ A dataset object can also be imported by setting the :class:`Dataset.csv` attribute. ::
+
+ data = tablib.Dataset()
+ data.csv = 'age, first_name, last_name\\n90, John, Adams'
+
+ Import assumes (for now) that headers exist.
+
+ .. admonition:: Binary Warning for Python 2
+
+ :class:`Dataset.csv` uses \\r\\n line endings by default so, in Python 2, make
+ sure to write in binary mode::
+
+ with open('output.csv', 'wb') as f:
+ f.write(data.csv)
+
+ If you do not do this, and you export the file on Windows, your
+ CSV file will open in Excel with a blank line between each row.
+
+ .. admonition:: Line endings for Python 3
+
+ :class:`Dataset.csv` uses \\r\\n line endings by default so, in Python 3, make
+ sure to include newline='' otherwise you will get a blank line between each row
+ when you open the file in Excel::
+
+ with open('output.csv', 'w', newline='') as f:
+ f.write(data.csv)
+
+ If you do not do this, and you export the file on Windows, your
+ CSV file will open in Excel with a blank line between each row.
+ """
+ pass
+
+ @property
+ def tsv():
+ """A TSV representation of the :class:`Dataset` object. The top row will contain
+ headers, if they have been set. Otherwise, the top row will contain
+ the first row of the dataset.
+
+ A dataset object can also be imported by setting the :class:`Dataset.tsv` attribute. ::
+
+ data = tablib.Dataset()
+ data.tsv = 'age\tfirst_name\tlast_name\\n90\tJohn\tAdams'
+
+ Import assumes (for now) that headers exist.
+ """
+ pass
+
+ @property
+ def yaml():
+ """A YAML representation of the :class:`Dataset` object. If headers have been
+ set, a YAML list of objects will be returned. If no headers have
+ been set, a YAML list of lists (rows) will be returned instead.
+
+ A dataset object can also be imported by setting the :class:`Dataset.yaml` attribute: ::
+
+ data = tablib.Dataset()
+ data.yaml = '- {age: 90, first_name: John, last_name: Adams}'
+
+ Import assumes (for now) that headers exist.
+ """
+ pass
+
+ @property
+ def df():
+ """A DataFrame representation of the :class:`Dataset` object.
+
+ A dataset object can also be imported by setting the :class:`Dataset.df` attribute: ::
+
+ data = tablib.Dataset()
+ data.df = DataFrame(np.random.randn(6,4))
+
+ Import assumes (for now) that headers exist.
+ """
+ pass
+
+ @property
+ def json():
+ """A JSON representation of the :class:`Dataset` object. If headers have been
+ set, a JSON list of objects will be returned. If no headers have
+ been set, a JSON list of lists (rows) will be returned instead.
+
+ A dataset object can also be imported by setting the :class:`Dataset.json` attribute: ::
+
+ data = tablib.Dataset()
+ data.json = '[{"age": 90, "first_name": "John", "last_name": "Adams"}]'
+
+ Import assumes (for now) that headers exist.
+ """
+ pass
+
+ @property
+ def html():
+ """A HTML table representation of the :class:`Dataset` object. If
+ headers have been set, they will be used as table headers.
+
+ ..notice:: This method can be used for export only.
+ """
+ pass
+
+ @property
+ def dbf():
+ """A dBASE representation of the :class:`Dataset` object.
+
+ A dataset object can also be imported by setting the
+ :class:`Dataset.dbf` attribute. ::
+
+ # To import data from an existing DBF file:
+ data = tablib.Dataset()
+ data.dbf = open('existing_table.dbf', mode='rb').read()
+
+ # to import data from an ASCII-encoded bytestring:
+ data = tablib.Dataset()
+ data.dbf = '<bytestring of tabular data>'
+
+ .. admonition:: Binary Warning
+
+ :class:`Dataset.dbf` contains binary data, so make sure to write in binary mode::
+
+ with open('output.dbf', 'wb') as f:
+ f.write(data.dbf)
+ """
+ pass
+
+ @property
+ def latex():
+ """A LaTeX booktabs representation of the :class:`Dataset` object. If a
+ title has been set, it will be exported as the table caption.
+
+ .. note:: This method can be used for export only.
+ """
+ pass
+
+ @property
+ def jira():
+ """A Jira table representation of the :class:`Dataset` object.
+
+ .. note:: This method can be used for export only.
+ """
+ pass
+
+ # ----
+ # Rows
+ # ----
+
+ def insert(self, index, row, tags=list()):
+ """Inserts a row to the :class:`Dataset` at the given index.
+
+ Rows inserted must be the correct size (height or width).
+
+ The default behaviour is to insert the given row to the :class:`Dataset`
+ object at the given index.
+ """
+
+ self._validate(row)
+ self._data.insert(index, Row(row, tags=tags))
+
+ def rpush(self, row, tags=list()):
+ """Adds a row to the end of the :class:`Dataset`.
+ See :class:`Dataset.insert` for additional documentation.
+ """
+
+ self.insert(self.height, row=row, tags=tags)
+
+ def lpush(self, row, tags=list()):
+ """Adds a row to the top of the :class:`Dataset`.
+ See :class:`Dataset.insert` for additional documentation.
+ """
+
+ self.insert(0, row=row, tags=tags)
+
+ def append(self, row, tags=list()):
+ """Adds a row to the :class:`Dataset`.
+ See :class:`Dataset.insert` for additional documentation.
+ """
+
+ self.rpush(row, tags)
+
+ def extend(self, rows, tags=list()):
+ """Adds a list of rows to the :class:`Dataset` using
+ :class:`Dataset.append`
+ """
+
+ for row in rows:
+ self.append(row, tags)
+
+ def lpop(self):
+ """Removes and returns the first row of the :class:`Dataset`."""
+
+ cache = self[0]
+ del self[0]
+
+ return cache
+
+ def rpop(self):
+ """Removes and returns the last row of the :class:`Dataset`."""
+
+ cache = self[-1]
+ del self[-1]
+
+ return cache
+
+ def pop(self):
+ """Removes and returns the last row of the :class:`Dataset`."""
+
+ return self.rpop()
+
+ # -------
+ # Columns
+ # -------
+
+ def insert_col(self, index, col=None, header=None):
+ """Inserts a column to the :class:`Dataset` at the given index.
+
+ Columns inserted must be the correct height.
+
+ You can also insert a column of a single callable object, which will
+ add a new column with the return values of the callable each as an
+ item in the column. ::
+
+ data.append_col(col=random.randint)
+
+ If inserting a column, and :class:`Dataset.headers` is set, the
+ header attribute must be set, and will be considered the header for
+ that row.
+
+ See :ref:`dyncols` for an in-depth example.
+
+ .. versionchanged:: 0.9.0
+ If inserting a column, and :class:`Dataset.headers` is set, the
+ header attribute must be set, and will be considered the header for
+ that row.
+
+ .. versionadded:: 0.9.0
+ If inserting a row, you can add :ref:`tags <tags>` to the row you are inserting.
+ This gives you the ability to :class:`filter <Dataset.filter>` your
+ :class:`Dataset` later.
+
+ """
+
+ if col is None:
+ col = []
+
+ # Callable Columns...
+ if hasattr(col, '__call__'):
+ col = list(map(col, self._data))
+
+ col = self._clean_col(col)
+ self._validate(col=col)
+
+ if self.headers:
+ # pop the first item off, add to headers
+ if not header:
+ raise HeadersNeeded()
+
+ # corner case - if header is set without data
+ elif header and self.height == 0 and len(col):
+ raise InvalidDimensions
+
+ self.headers.insert(index, header)
+
+ if self.height and self.width:
+
+ for i, row in enumerate(self._data):
+
+ row.insert(index, col[i])
+ self._data[i] = row
+ else:
+ self._data = [Row([row]) for row in col]
+
+ def rpush_col(self, col, header=None):
+ """Adds a column to the end of the :class:`Dataset`.
+ See :class:`Dataset.insert` for additional documentation.
+ """
+
+ self.insert_col(self.width, col, header=header)
+
+ def lpush_col(self, col, header=None):
+ """Adds a column to the top of the :class:`Dataset`.
+ See :class:`Dataset.insert` for additional documentation.
+ """
+
+ self.insert_col(0, col, header=header)
+
+ def insert_separator(self, index, text='-'):
+ """Adds a separator to :class:`Dataset` at given index."""
+
+ sep = (index, text)
+ self._separators.append(sep)
+
+ def append_separator(self, text='-'):
+ """Adds a :ref:`separator <separators>` to the :class:`Dataset`."""
+
+ # change offsets if headers are or aren't defined
+ if not self.headers:
+ index = self.height if self.height else 0
+ else:
+ index = (self.height + 1) if self.height else 1
+
+ self.insert_separator(index, text)
+
+ def append_col(self, col, header=None):
+ """Adds a column to the :class:`Dataset`.
+ See :class:`Dataset.insert_col` for additional documentation.
+ """
+
+ self.rpush_col(col, header)
+
+ def get_col(self, index):
+ """Returns the column from the :class:`Dataset` at the given index."""
+
+ return [row[index] for row in self._data]
+
+ # ----
+ # Misc
+ # ----
+
+ def add_formatter(self, col, handler):
+ """Adds a formatter to the :class:`Dataset`.
+
+ .. versionadded:: 0.9.5
+
+ :param col: column to. Accepts index int or header str.
+ :param handler: reference to callback function to execute against
+ each cell value.
+ """
+
+ if isinstance(col, unicode):
+ if col in self.headers:
+ col = self.headers.index(col) # get 'key' index from each data
+ else:
+ raise KeyError
+
+ if not col > self.width:
+ self._formatters.append((col, handler))
+ else:
+ raise InvalidDatasetIndex
+
+ return True
+
+ def filter(self, tag):
+ """Returns a new instance of the :class:`Dataset`, excluding any rows
+ that do not contain the given :ref:`tags <tags>`.
+ """
+ _dset = copy(self)
+ _dset._data = [row for row in _dset._data if row.has_tag(tag)]
+
+ return _dset
+
+ def sort(self, col, reverse=False):
+ """Sort a :class:`Dataset` by a specific column, given string (for
+ header) or integer (for column index). The order can be reversed by
+ setting ``reverse`` to ``True``.
+
+ Returns a new :class:`Dataset` instance where columns have been
+ sorted.
+ """
+
+ if isinstance(col, (str, unicode)):
+
+ if not self.headers:
+ raise HeadersNeeded
+
+ _sorted = sorted(self.dict, key=itemgetter(col), reverse=reverse)
+ _dset = Dataset(headers=self.headers, title=self.title)
+
+ for item in _sorted:
+ row = [item[key] for key in self.headers]
+ _dset.append(row=row)
+
+ else:
+ if self.headers:
+ col = self.headers[col]
+
+ _sorted = sorted(self.dict, key=itemgetter(col), reverse=reverse)
+ _dset = Dataset(headers=self.headers, title=self.title)
+
+ for item in _sorted:
+ if self.headers:
+ row = [item[key] for key in self.headers]
+ else:
+ row = item
+ _dset.append(row=row)
+
+ return _dset
+
+ def transpose(self):
+ """Transpose a :class:`Dataset`, turning rows into columns and vice
+ versa, returning a new ``Dataset`` instance. The first row of the
+ original instance becomes the new header row."""
+
+ # Don't transpose if there is no data
+ if not self:
+ return
+
+ _dset = Dataset()
+ # The first element of the headers stays in the headers,
+ # it is our "hinge" on which we rotate the data
+ new_headers = [self.headers[0]] + self[self.headers[0]]
+
+ _dset.headers = new_headers
+ for index, column in enumerate(self.headers):
+
+ if column == self.headers[0]:
+ # It's in the headers, so skip it
+ continue
+
+ # Adding the column name as now they're a regular column
+ # Use `get_col(index)` in case there are repeated values
+ row_data = [column] + self.get_col(index)
+ row_data = Row(row_data)
+ _dset.append(row=row_data)
+ return _dset
+
+ def stack(self, other):
+ """Stack two :class:`Dataset` instances together by
+ joining at the row level, and return new combined
+ ``Dataset`` instance."""
+
+ if not isinstance(other, Dataset):
+ return
+
+ if self.width != other.width:
+ raise InvalidDimensions
+
+ # Copy the source data
+ _dset = copy(self)
+
+ rows_to_stack = [row for row in _dset._data]
+ other_rows = [row for row in other._data]
+
+ rows_to_stack.extend(other_rows)
+ _dset._data = rows_to_stack
+
+ return _dset
+
+ def stack_cols(self, other):
+ """Stack two :class:`Dataset` instances together by
+ joining at the column level, and return a new
+ combined ``Dataset`` instance. If either ``Dataset``
+ has headers set, than the other must as well."""
+
+ if not isinstance(other, Dataset):
+ return
+
+ if self.headers or other.headers:
+ if not self.headers or not other.headers:
+ raise HeadersNeeded
+
+ if self.height != other.height:
+ raise InvalidDimensions
+
+ try:
+ new_headers = self.headers + other.headers
+ except TypeError:
+ new_headers = None
+
+ _dset = Dataset()
+
+ for column in self.headers:
+ _dset.append_col(col=self[column])
+
+ for column in other.headers:
+ _dset.append_col(col=other[column])
+
+ _dset.headers = new_headers
+
+ return _dset
+
+ def remove_duplicates(self):
+ """Removes all duplicate rows from the :class:`Dataset` object
+ while maintaining the original order."""
+ seen = set()
+ self._data[:] = [row for row in self._data if not (tuple(row) in seen or seen.add(tuple(row)))]
+
+ def wipe(self):
+ """Removes all content and headers from the :class:`Dataset` object."""
+ self._data = list()
+ self.__headers = None
+
+ def subset(self, rows=None, cols=None):
+ """Returns a new instance of the :class:`Dataset`,
+ including only specified rows and columns.
+ """
+
+ # Don't return if no data
+ if not self:
+ return
+
+ if rows is None:
+ rows = list(range(self.height))
+
+ if cols is None:
+ cols = list(self.headers)
+
+ #filter out impossible rows and columns
+ rows = [row for row in rows if row in range(self.height)]
+ cols = [header for header in cols if header in self.headers]
+
+ _dset = Dataset()
+
+ #filtering rows and columns
+ _dset.headers = list(cols)
+
+ _dset._data = []
+ for row_no, row in enumerate(self._data):
+ data_row = []
+ for key in _dset.headers:
+ if key in self.headers:
+ pos = self.headers.index(key)
+ data_row.append(row[pos])
+ else:
+ raise KeyError
+
+ if row_no in rows:
+ _dset.append(row=Row(data_row))
+
+ return _dset
+
+
+class Databook(object):
+ """A book of :class:`Dataset` objects.
+ """
+
+ _formats = {}
+
+ def __init__(self, sets=None):
+
+ if sets is None:
+ self._datasets = list()
+ else:
+ self._datasets = sets
+
+ self._register_formats()
+
+ def __repr__(self):
+ try:
+ return '<%s databook>' % (self.title.lower())
+ except AttributeError:
+ return '<databook object>'
+
+ def wipe(self):
+ """Removes all :class:`Dataset` objects from the :class:`Databook`."""
+ self._datasets = []
+
+ @classmethod
+ def _register_formats(cls):
+ """Adds format properties."""
+ for fmt in formats.available:
+ try:
+ try:
+ setattr(cls, fmt.title, property(fmt.export_book, fmt.import_book))
+ cls._formats[fmt.title] = (fmt.export_book, fmt.import_book)
+ except AttributeError:
+ setattr(cls, fmt.title, property(fmt.export_book))
+ cls._formats[fmt.title] = (fmt.export_book, None)
+
+ except AttributeError:
+ cls._formats[fmt.title] = (None, None)
+
+ def sheets(self):
+ return self._datasets
+
+ def add_sheet(self, dataset):
+ """Adds given :class:`Dataset` to the :class:`Databook`."""
+ if isinstance(dataset, Dataset):
+ self._datasets.append(dataset)
+ else:
+ raise InvalidDatasetType
+
+ def _package(self, ordered=True):
+ """Packages :class:`Databook` for delivery."""
+ collector = []
+
+ if ordered:
+ dict_pack = OrderedDict
+ else:
+ dict_pack = dict
+
+ for dset in self._datasets:
+ collector.append(dict_pack(
+ title = dset.title,
+ data = dset._package(ordered=ordered)
+ ))
+ return collector
+
+ @property
+ def size(self):
+ """The number of the :class:`Dataset` objects within :class:`Databook`."""
+ return len(self._datasets)
+
+ def load(self, in_stream, format, **kwargs):
+ """
+ Import `in_stream` to the :class:`Databook` object using the `format`.
+
+ :param \\*\\*kwargs: (optional) custom configuration to the format `import_book`.
+ """
+
+ if not format:
+ format = detect_format(in_stream)
+
+ export_book, import_book = self._formats.get(format, (None, None))
+ if not import_book:
+ raise UnsupportedFormat('Format {0} cannot be loaded.'.format(format))
+
+ import_book(self, in_stream, **kwargs)
+ return self
+
+ def export(self, format, **kwargs):
+ """
+ Export :class:`Databook` object to `format`.
+
+ :param \\*\\*kwargs: (optional) custom configuration to the format `export_book`.
+ """
+ export_book, import_book = self._formats.get(format, (None, None))
+ if not export_book:
+ raise UnsupportedFormat('Format {0} cannot be exported.'.format(format))
+
+ return export_book(self, **kwargs)
+
+
+def detect_format(stream):
+ """Return format name of given stream."""
+ for fmt in formats.available:
+ try:
+ if fmt.detect(stream):
+ return fmt.title
+ except AttributeError:
+ pass
+
+
+def import_set(stream, format=None, **kwargs):
+ """Return dataset of given stream."""
+
+ return Dataset().load(stream, format, **kwargs)
+
+
+def import_book(stream, format=None, **kwargs):
+ """Return dataset of given stream."""
+
+ return Databook().load(stream, format, **kwargs)
+
+
+class InvalidDatasetType(Exception):
+ "Only Datasets can be added to a DataBook"
+
+
+class InvalidDimensions(Exception):
+ "Invalid size"
+
+
+class InvalidDatasetIndex(Exception):
+ "Outside of Dataset size"
+
+
+class HeadersNeeded(Exception):
+ "Header parameter must be given when appending a column in this Dataset."
+
+
+class UnsupportedFormat(NotImplementedError):
+ "Format is not supported"
diff --git a/src/tablib/formats/__init__.py b/src/tablib/formats/__init__.py
new file mode 100644
index 0000000..52cf472
--- /dev/null
+++ b/src/tablib/formats/__init__.py
@@ -0,0 +1,21 @@
+# -*- coding: utf-8 -*-
+
+""" Tablib - formats
+"""
+
+from . import _csv as csv
+from . import _json as json
+from . import _xls as xls
+from . import _yaml as yaml
+from . import _tsv as tsv
+from . import _html as html
+from . import _xlsx as xlsx
+from . import _ods as ods
+from . import _dbf as dbf
+from . import _latex as latex
+from . import _df as df
+from . import _rst as rst
+from . import _jira as jira
+
+# xlsx before as xls (xlrd) can also read xlsx
+available = (json, xlsx, xls, yaml, csv, dbf, tsv, html, jira, latex, ods, df, rst)
diff --git a/src/tablib/formats/_csv.py b/src/tablib/formats/_csv.py
new file mode 100644
index 0000000..5c03d6f
--- /dev/null
+++ b/src/tablib/formats/_csv.py
@@ -0,0 +1,59 @@
+# -*- coding: utf-8 -*-
+
+""" Tablib - *SV Support.
+"""
+
+from tablib.compat import csv, StringIO, unicode
+
+
+title = 'csv'
+extensions = ('csv',)
+
+
+DEFAULT_DELIMITER = unicode(',')
+
+
+def export_stream_set(dataset, **kwargs):
+ """Returns CSV representation of Dataset as file-like."""
+ stream = StringIO()
+
+ kwargs.setdefault('delimiter', DEFAULT_DELIMITER)
+
+ _csv = csv.writer(stream, **kwargs)
+
+ for row in dataset._package(dicts=False):
+ _csv.writerow(row)
+
+ stream.seek(0)
+ return stream
+
+
+def export_set(dataset, **kwargs):
+ """Returns CSV representation of Dataset."""
+ stream = export_stream_set(dataset, **kwargs)
+ return stream.getvalue()
+
+
+def import_set(dset, in_stream, headers=True, **kwargs):
+ """Returns dataset from CSV stream."""
+
+ dset.wipe()
+
+ kwargs.setdefault('delimiter', DEFAULT_DELIMITER)
+
+ rows = csv.reader(StringIO(in_stream), **kwargs)
+ for i, row in enumerate(rows):
+
+ if (i == 0) and (headers):
+ dset.headers = row
+ elif row:
+ dset.append(row)
+
+
+def detect(stream, delimiter=DEFAULT_DELIMITER):
+ """Returns True if given stream is valid CSV."""
+ try:
+ csv.Sniffer().sniff(stream, delimiters=delimiter)
+ return True
+ except Exception:
+ return False
diff --git a/src/tablib/formats/_dbf.py b/src/tablib/formats/_dbf.py
new file mode 100644
index 0000000..0d1c87b
--- /dev/null
+++ b/src/tablib/formats/_dbf.py
@@ -0,0 +1,87 @@
+# -*- coding: utf-8 -*-
+
+""" Tablib - DBF Support.
+"""
+import tempfile
+import struct
+import os
+
+from tablib.compat import StringIO
+from tablib.compat import dbfpy
+from tablib.compat import is_py3
+
+if is_py3:
+ from tablib.packages.dbfpy3 import dbf
+ from tablib.packages.dbfpy3 import dbfnew
+ from tablib.packages.dbfpy3 import record as dbfrecord
+ import io
+else:
+ from tablib.packages.dbfpy import dbf
+ from tablib.packages.dbfpy import dbfnew
+ from tablib.packages.dbfpy import record as dbfrecord
+
+
+title = 'dbf'
+extensions = ('csv',)
+
+DEFAULT_ENCODING = 'utf-8'
+
+def export_set(dataset):
+ """Returns DBF representation of a Dataset"""
+ new_dbf = dbfnew.dbf_new()
+ temp_file, temp_uri = tempfile.mkstemp()
+
+ # create the appropriate fields based on the contents of the first row
+ first_row = dataset[0]
+ for fieldname, field_value in zip(dataset.headers, first_row):
+ if type(field_value) in [int, float]:
+ new_dbf.add_field(fieldname, 'N', 10, 8)
+ else:
+ new_dbf.add_field(fieldname, 'C', 80)
+
+ new_dbf.write(temp_uri)
+
+ dbf_file = dbf.Dbf(temp_uri, readOnly=0)
+ for row in dataset:
+ record = dbfrecord.DbfRecord(dbf_file)
+ for fieldname, field_value in zip(dataset.headers, row):
+ record[fieldname] = field_value
+ record.store()
+
+ dbf_file.close()
+ dbf_stream = open(temp_uri, 'rb')
+ if is_py3:
+ stream = io.BytesIO(dbf_stream.read())
+ else:
+ stream = StringIO(dbf_stream.read())
+ dbf_stream.close()
+ os.close(temp_file)
+ os.remove(temp_uri)
+ return stream.getvalue()
+
+def import_set(dset, in_stream, headers=True):
+ """Returns a dataset from a DBF stream."""
+
+ dset.wipe()
+ if is_py3:
+ _dbf = dbf.Dbf(io.BytesIO(in_stream))
+ else:
+ _dbf = dbf.Dbf(StringIO(in_stream))
+ dset.headers = _dbf.fieldNames
+ for record in range(_dbf.recordCount):
+ row = [_dbf[record][f] for f in _dbf.fieldNames]
+ dset.append(row)
+
+def detect(stream):
+ """Returns True if the given stream is valid DBF"""
+ #_dbf = dbf.Table(StringIO(stream))
+ try:
+ if is_py3:
+ if type(stream) is not bytes:
+ stream = bytes(stream, 'utf-8')
+ _dbf = dbf.Dbf(io.BytesIO(stream), readOnly=True)
+ else:
+ _dbf = dbf.Dbf(StringIO(stream), readOnly=True)
+ return True
+ except Exception:
+ return False
diff --git a/src/tablib/formats/_df.py b/src/tablib/formats/_df.py
new file mode 100644
index 0000000..0c7ebec
--- /dev/null
+++ b/src/tablib/formats/_df.py
@@ -0,0 +1,43 @@
+""" Tablib - DataFrame Support.
+"""
+
+import sys
+from io import BytesIO
+
+try:
+ from pandas import DataFrame
+except ImportError:
+ DataFrame = None
+
+import tablib
+
+from tablib.compat import unicode
+
+title = 'df'
+extensions = ('df', )
+
+def detect(stream):
+ """Returns True if given stream is a DataFrame."""
+ if DataFrame is None:
+ return False
+ try:
+ DataFrame(stream)
+ return True
+ except ValueError:
+ return False
+
+
+def export_set(dset, index=None):
+ """Returns DataFrame representation of DataBook."""
+ if DataFrame is None:
+ raise NotImplementedError(
+ 'DataFrame Format requires `pandas` to be installed.'
+ ' Try `pip install tablib[pandas]`.')
+ dataframe = DataFrame(dset.dict, columns=dset.headers)
+ return dataframe
+
+
+def import_set(dset, in_stream):
+ """Returns dataset from DataFrame."""
+ dset.wipe()
+ dset.dict = in_stream.to_dict(orient='records')
diff --git a/src/tablib/formats/_html.py b/src/tablib/formats/_html.py
new file mode 100644
index 0000000..ac72bd6
--- /dev/null
+++ b/src/tablib/formats/_html.py
@@ -0,0 +1,65 @@
+# -*- coding: utf-8 -*-
+
+""" Tablib - HTML export support.
+"""
+
+import codecs
+import sys
+from io import BytesIO
+
+from MarkupPy import markup
+import tablib
+from tablib.compat import unicode
+
+BOOK_ENDINGS = 'h3'
+
+title = 'html'
+extensions = ('html', )
+
+
+def export_set(dataset):
+ """HTML representation of a Dataset."""
+
+ stream = BytesIO()
+
+ page = markup.page()
+ page.table.open()
+
+ if dataset.headers is not None:
+ new_header = [item if item is not None else '' for item in dataset.headers]
+
+ page.thead.open()
+ headers = markup.oneliner.th(new_header)
+ page.tr(headers)
+ page.thead.close()
+
+ for row in dataset:
+ new_row = [item if item is not None else '' for item in row]
+
+ html_row = markup.oneliner.td(new_row)
+ page.tr(html_row)
+
+ page.table.close()
+
+ # Allow unicode characters in output
+ wrapper = codecs.getwriter("utf8")(stream)
+ wrapper.writelines(unicode(page))
+
+ return stream.getvalue().decode('utf-8')
+
+
+def export_book(databook):
+ """HTML representation of a Databook."""
+
+ stream = BytesIO()
+
+ # Allow unicode characters in output
+ wrapper = codecs.getwriter("utf8")(stream)
+
+ for i, dset in enumerate(databook._datasets):
+ title = (dset.title if dset.title else 'Set %s' % (i))
+ wrapper.write('<%s>%s</%s>\n' % (BOOK_ENDINGS, title, BOOK_ENDINGS))
+ wrapper.write(dset.html)
+ wrapper.write('\n')
+
+ return stream.getvalue().decode('utf-8')
diff --git a/src/tablib/formats/_jira.py b/src/tablib/formats/_jira.py
new file mode 100644
index 0000000..55fce52
--- /dev/null
+++ b/src/tablib/formats/_jira.py
@@ -0,0 +1,39 @@
+# -*- coding: utf-8 -*-
+
+"""Tablib - Jira table export support.
+
+ Generates a Jira table from the dataset.
+"""
+from tablib.compat import unicode
+
+title = 'jira'
+
+
+def export_set(dataset):
+ """Formats the dataset according to the Jira table syntax:
+
+ ||heading 1||heading 2||heading 3||
+ |col A1|col A2|col A3|
+ |col B1|col B2|col B3|
+
+ :param dataset: dataset to serialize
+ :type dataset: tablib.core.Dataset
+ """
+
+ header = _get_header(dataset.headers) if dataset.headers else ''
+ body = _get_body(dataset)
+ return '%s\n%s' % (header, body) if header else body
+
+
+def _get_body(dataset):
+ return '\n'.join([_serialize_row(row) for row in dataset])
+
+
+def _get_header(headers):
+ return _serialize_row(headers, delimiter='||')
+
+
+def _serialize_row(row, delimiter='|'):
+ return '%s%s%s' % (delimiter,
+ delimiter.join([unicode(item) if item else ' ' for item in row]),
+ delimiter)
diff --git a/src/tablib/formats/_json.py b/src/tablib/formats/_json.py
new file mode 100644
index 0000000..009fb3c
--- /dev/null
+++ b/src/tablib/formats/_json.py
@@ -0,0 +1,59 @@
+# -*- coding: utf-8 -*-
+
+""" Tablib - JSON Support
+"""
+import decimal
+import json
+from uuid import UUID
+
+import tablib
+
+
+title = 'json'
+extensions = ('json', 'jsn')
+
+
+def serialize_objects_handler(obj):
+ if isinstance(obj, (decimal.Decimal, UUID)):
+ return str(obj)
+ elif hasattr(obj, 'isoformat'):
+ return obj.isoformat()
+ else:
+ return obj
+
+
+def export_set(dataset):
+ """Returns JSON representation of Dataset."""
+ return json.dumps(dataset.dict, default=serialize_objects_handler)
+
+
+def export_book(databook):
+ """Returns JSON representation of Databook."""
+ return json.dumps(databook._package(), default=serialize_objects_handler)
+
+
+def import_set(dset, in_stream):
+ """Returns dataset from JSON stream."""
+
+ dset.wipe()
+ dset.dict = json.loads(in_stream)
+
+
+def import_book(dbook, in_stream):
+ """Returns databook from JSON stream."""
+
+ dbook.wipe()
+ for sheet in json.loads(in_stream):
+ data = tablib.Dataset()
+ data.title = sheet['title']
+ data.dict = sheet['data']
+ dbook.add_sheet(data)
+
+
+def detect(stream):
+ """Returns True if given stream is valid JSON."""
+ try:
+ json.loads(stream)
+ return True
+ except (TypeError, ValueError):
+ return False
diff --git a/src/tablib/formats/_latex.py b/src/tablib/formats/_latex.py
new file mode 100644
index 0000000..44ee101
--- /dev/null
+++ b/src/tablib/formats/_latex.py
@@ -0,0 +1,134 @@
+# -*- coding: utf-8 -*-
+
+"""Tablib - LaTeX table export support.
+
+ Generates a LaTeX booktabs-style table from the dataset.
+"""
+import re
+
+from tablib.compat import unicode
+
+title = 'latex'
+extensions = ('tex',)
+
+TABLE_TEMPLATE = """\
+%% Note: add \\usepackage{booktabs} to your preamble
+%%
+\\begin{table}[!htbp]
+ \\centering
+ %(CAPTION)s
+ \\begin{tabular}{%(COLSPEC)s}
+ \\toprule
+%(HEADER)s
+ %(MIDRULE)s
+%(BODY)s
+ \\bottomrule
+ \\end{tabular}
+\\end{table}
+"""
+
+TEX_RESERVED_SYMBOLS_MAP = dict([
+ ('\\', '\\textbackslash{}'),
+ ('{', '\\{'),
+ ('}', '\\}'),
+ ('$', '\\$'),
+ ('&', '\\&'),
+ ('#', '\\#'),
+ ('^', '\\textasciicircum{}'),
+ ('_', '\\_'),
+ ('~', '\\textasciitilde{}'),
+ ('%', '\\%'),
+])
+
+TEX_RESERVED_SYMBOLS_RE = re.compile(
+ '(%s)' % '|'.join(map(re.escape, TEX_RESERVED_SYMBOLS_MAP.keys())))
+
+
+def export_set(dataset):
+ """Returns LaTeX representation of dataset
+
+ :param dataset: dataset to serialize
+ :type dataset: tablib.core.Dataset
+ """
+
+ caption = '\\caption{%s}' % dataset.title if dataset.title else '%'
+ colspec = _colspec(dataset.width)
+ header = _serialize_row(dataset.headers) if dataset.headers else ''
+ midrule = _midrule(dataset.width)
+ body = '\n'.join([_serialize_row(row) for row in dataset])
+ return TABLE_TEMPLATE % dict(CAPTION=caption, COLSPEC=colspec,
+ HEADER=header, MIDRULE=midrule, BODY=body)
+
+
+def _colspec(dataset_width):
+ """Generates the column specification for the LaTeX `tabular` environment
+ based on the dataset width.
+
+ The first column is justified to the left, all further columns are aligned
+ to the right.
+
+ .. note:: This is only a heuristic and most probably has to be fine-tuned
+ post export. Column alignment should depend on the data type, e.g., textual
+ content should usually be aligned to the left while numeric content almost
+ always should be aligned to the right.
+
+ :param dataset_width: width of the dataset
+ """
+
+ spec = 'l'
+ for _ in range(1, dataset_width):
+ spec += 'r'
+ return spec
+
+
+def _midrule(dataset_width):
+ """Generates the table `midrule`, which may be composed of several
+ `cmidrules`.
+
+ :param dataset_width: width of the dataset to serialize
+ """
+
+ if not dataset_width or dataset_width == 1:
+ return '\\midrule'
+ return ' '.join([_cmidrule(colindex, dataset_width) for colindex in
+ range(1, dataset_width + 1)])
+
+
+def _cmidrule(colindex, dataset_width):
+ """Generates the `cmidrule` for a single column with appropriate trimming
+ based on the column position.
+
+ :param colindex: Column index
+ :param dataset_width: width of the dataset
+ """
+
+ rule = '\\cmidrule(%s){%d-%d}'
+ if colindex == 1:
+ # Rule of first column is trimmed on the right
+ return rule % ('r', colindex, colindex)
+ if colindex == dataset_width:
+ # Rule of last column is trimmed on the left
+ return rule % ('l', colindex, colindex)
+ # Inner columns are trimmed on the left and right
+ return rule % ('lr', colindex, colindex)
+
+
+def _serialize_row(row):
+ """Returns string representation of a single row.
+
+ :param row: single dataset row
+ """
+
+ new_row = [_escape_tex_reserved_symbols(unicode(item)) if item else '' for
+ item in row]
+ return 6 * ' ' + ' & '.join(new_row) + ' \\\\'
+
+
+def _escape_tex_reserved_symbols(input):
+ """Escapes all TeX reserved symbols ('_', '~', etc.) in a string.
+
+ :param input: String to escape
+ """
+ def replace(match):
+ return TEX_RESERVED_SYMBOLS_MAP[match.group()]
+ return TEX_RESERVED_SYMBOLS_RE.sub(replace, input)
diff --git a/src/tablib/formats/_ods.py b/src/tablib/formats/_ods.py
new file mode 100644
index 0000000..dbf57c4
--- /dev/null
+++ b/src/tablib/formats/_ods.py
@@ -0,0 +1,104 @@
+# -*- coding: utf-8 -*-
+
+""" Tablib - ODF Support.
+"""
+
+from io import BytesIO
+from odf import opendocument, style, table, text
+from tablib.compat import unicode
+
+title = 'ods'
+extensions = ('ods',)
+
+bold = style.Style(name="bold", family="paragraph")
+bold.addElement(style.TextProperties(fontweight="bold", fontweightasian="bold", fontweightcomplex="bold"))
+
+def export_set(dataset):
+ """Returns ODF representation of Dataset."""
+
+ wb = opendocument.OpenDocumentSpreadsheet()
+ wb.automaticstyles.addElement(bold)
+
+ ws = table.Table(name=dataset.title if dataset.title else 'Tablib Dataset')
+ wb.spreadsheet.addElement(ws)
+ dset_sheet(dataset, ws)
+
+ stream = BytesIO()
+ wb.save(stream)
+ return stream.getvalue()
+
+
+def export_book(databook):
+ """Returns ODF representation of DataBook."""
+
+ wb = opendocument.OpenDocumentSpreadsheet()
+ wb.automaticstyles.addElement(bold)
+
+ for i, dset in enumerate(databook._datasets):
+ ws = table.Table(name=dset.title if dset.title else 'Sheet%s' % (i))
+ wb.spreadsheet.addElement(ws)
+ dset_sheet(dset, ws)
+
+ stream = BytesIO()
+ wb.save(stream)
+ return stream.getvalue()
+
+
+def dset_sheet(dataset, ws):
+ """Completes given worksheet from given Dataset."""
+ _package = dataset._package(dicts=False)
+
+ for i, sep in enumerate(dataset._separators):
+ _offset = i
+ _package.insert((sep[0] + _offset), (sep[1],))
+
+ for i, row in enumerate(_package):
+ row_number = i + 1
+ odf_row = table.TableRow(stylename=bold, defaultcellstylename='bold')
+ for j, col in enumerate(row):
+ try:
+ col = unicode(col, errors='ignore')
+ except TypeError:
+ ## col is already unicode
+ pass
+ ws.addElement(table.TableColumn())
+
+ # bold headers
+ if (row_number == 1) and dataset.headers:
+ odf_row.setAttribute('stylename', bold)
+ ws.addElement(odf_row)
+ cell = table.TableCell()
+ p = text.P()
+ p.addElement(text.Span(text=col, stylename=bold))
+ cell.addElement(p)
+ odf_row.addElement(cell)
+
+ # wrap the rest
+ else:
+ try:
+ if '\n' in col:
+ ws.addElement(odf_row)
+ cell = table.TableCell()
+ cell.addElement(text.P(text=col))
+ odf_row.addElement(cell)
+ else:
+ ws.addElement(odf_row)
+ cell = table.TableCell()
+ cell.addElement(text.P(text=col))
+ odf_row.addElement(cell)
+ except TypeError:
+ ws.addElement(odf_row)
+ cell = table.TableCell()
+ cell.addElement(text.P(text=col))
+ odf_row.addElement(cell)
+
+
+def detect(stream):
+ if isinstance(stream, bytes):
+ # load expects a file-like object.
+ stream = BytesIO(stream)
+ try:
+ opendocument.load(stream)
+ return True
+ except Exception:
+ return False
diff --git a/src/tablib/formats/_rst.py b/src/tablib/formats/_rst.py
new file mode 100644
index 0000000..4b53ad7
--- /dev/null
+++ b/src/tablib/formats/_rst.py
@@ -0,0 +1,273 @@
+# -*- coding: utf-8 -*-
+
+""" Tablib - reStructuredText Support
+"""
+from __future__ import division
+from __future__ import print_function
+from __future__ import unicode_literals
+
+from textwrap import TextWrapper
+
+from tablib.compat import (
+ median,
+ unicode,
+ izip_longest,
+)
+
+
+title = 'rst'
+extensions = ('rst',)
+
+
+MAX_TABLE_WIDTH = 80 # Roughly. It may be wider to avoid breaking words.
+
+
+JUSTIFY_LEFT = 'left'
+JUSTIFY_CENTER = 'center'
+JUSTIFY_RIGHT = 'right'
+JUSTIFY_VALUES = (JUSTIFY_LEFT, JUSTIFY_CENTER, JUSTIFY_RIGHT)
+
+
+def to_unicode(value):
+ if isinstance(value, bytes):
+ return value.decode('utf-8')
+ return unicode(value)
+
+
+def _max_word_len(text):
+ """
+ Return the length of the longest word in `text`.
+
+
+ >>> _max_word_len('Python Module for Tabular Datasets')
+ 8
+
+ """
+ return max((len(word) for word in text.split()))
+
+
+def _get_column_string_lengths(dataset):
+ """
+ Returns a list of string lengths of each column, and a list of
+ maximum word lengths.
+ """
+ if dataset.headers:
+ column_lengths = [[len(h)] for h in dataset.headers]
+ word_lens = [_max_word_len(h) for h in dataset.headers]
+ else:
+ column_lengths = [[] for _ in range(dataset.width)]
+ word_lens = [0 for _ in range(dataset.width)]
+ for row in dataset.dict:
+ values = iter(row.values() if hasattr(row, 'values') else row)
+ for i, val in enumerate(values):
+ text = to_unicode(val)
+ column_lengths[i].append(len(text))
+ word_lens[i] = max(word_lens[i], _max_word_len(text))
+ return column_lengths, word_lens
+
+
+def _row_to_lines(values, widths, wrapper, sep='|', justify=JUSTIFY_LEFT):
+ """
+ Returns a table row of wrapped values as a list of lines
+ """
+ if justify not in JUSTIFY_VALUES:
+ raise ValueError('Value of "justify" must be one of "{}"'.format(
+ '", "'.join(JUSTIFY_VALUES)
+ ))
+ if justify == JUSTIFY_LEFT:
+ just = lambda text, width: text.ljust(width)
+ elif justify == JUSTIFY_CENTER:
+ just = lambda text, width: text.center(width)
+ else:
+ just = lambda text, width: text.rjust(width)
+ lpad = sep + ' ' if sep else ''
+ rpad = ' ' + sep if sep else ''
+ pad = ' ' + sep + ' '
+ cells = []
+ for value, width in zip(values, widths):
+ wrapper.width = width
+ text = to_unicode(value)
+ cell = wrapper.wrap(text)
+ cells.append(cell)
+ lines = izip_longest(*cells, fillvalue='')
+ lines = (
+ (just(cell_line, widths[i]) for i, cell_line in enumerate(line))
+ for line in lines
+ )
+ lines = [''.join((lpad, pad.join(line), rpad)) for line in lines]
+ return lines
+
+
+def _get_column_widths(dataset, max_table_width=MAX_TABLE_WIDTH, pad_len=3):
+ """
+ Returns a list of column widths proportional to the median length
+ of the text in their cells.
+ """
+ str_lens, word_lens = _get_column_string_lengths(dataset)
+ median_lens = [int(median(lens)) for lens in str_lens]
+ total = sum(median_lens)
+ if total > max_table_width - (pad_len * len(median_lens)):
+ column_widths = (max_table_width * l // total for l in median_lens)
+ else:
+ column_widths = (l for l in median_lens)
+ # Allow for separator and padding:
+ column_widths = (w - pad_len if w > pad_len else w for w in column_widths)
+ # Rather widen table than break words:
+ column_widths = [max(w, l) for w, l in zip(column_widths, word_lens)]
+ return column_widths
+
+
+def export_set_as_simple_table(dataset, column_widths=None):
+ """
+ Returns reStructuredText grid table representation of dataset.
+ """
+ lines = []
+ wrapper = TextWrapper()
+ if column_widths is None:
+ column_widths = _get_column_widths(dataset, pad_len=2)
+ border = ' '.join(['=' * w for w in column_widths])
+
+ lines.append(border)
+ if dataset.headers:
+ lines.extend(_row_to_lines(
+ dataset.headers,
+ column_widths,
+ wrapper,
+ sep='',
+ justify=JUSTIFY_CENTER,
+ ))
+ lines.append(border)
+ for row in dataset.dict:
+ values = iter(row.values() if hasattr(row, 'values') else row)
+ lines.extend(_row_to_lines(values, column_widths, wrapper, ''))
+ lines.append(border)
+ return '\n'.join(lines)
+
+
+def export_set_as_grid_table(dataset, column_widths=None):
+ """
+ Returns reStructuredText grid table representation of dataset.
+
+
+ >>> from tablib import Dataset
+ >>> from tablib.formats import rst
+ >>> bits = ((0, 0), (1, 0), (0, 1), (1, 1))
+ >>> data = Dataset()
+ >>> data.headers = ['A', 'B', 'A and B']
+ >>> for a, b in bits:
+ ... data.append([bool(a), bool(b), bool(a * b)])
+ >>> print(rst.export_set(data, force_grid=True))
+ +-------+-------+-------+
+ | A | B | A and |
+ | | | B |
+ +=======+=======+=======+
+ | False | False | False |
+ +-------+-------+-------+
+ | True | False | False |
+ +-------+-------+-------+
+ | False | True | False |
+ +-------+-------+-------+
+ | True | True | True |
+ +-------+-------+-------+
+
+ """
+ lines = []
+ wrapper = TextWrapper()
+ if column_widths is None:
+ column_widths = _get_column_widths(dataset)
+ header_sep = '+=' + '=+='.join(['=' * w for w in column_widths]) + '=+'
+ row_sep = '+-' + '-+-'.join(['-' * w for w in column_widths]) + '-+'
+
+ lines.append(row_sep)
+ if dataset.headers:
+ lines.extend(_row_to_lines(
+ dataset.headers,
+ column_widths,
+ wrapper,
+ justify=JUSTIFY_CENTER,
+ ))
+ lines.append(header_sep)
+ for row in dataset.dict:
+ values = iter(row.values() if hasattr(row, 'values') else row)
+ lines.extend(_row_to_lines(values, column_widths, wrapper))
+ lines.append(row_sep)
+ return '\n'.join(lines)
+
+
+def _use_simple_table(head0, col0, width0):
+ """
+ Use a simple table if the text in the first column is never wrapped
+
+
+ >>> _use_simple_table('menu', ['egg', 'bacon'], 10)
+ True
+ >>> _use_simple_table(None, ['lobster thermidor', 'spam'], 10)
+ False
+
+ """
+ if head0 is not None:
+ head0 = to_unicode(head0)
+ if len(head0) > width0:
+ return False
+ for cell in col0:
+ cell = to_unicode(cell)
+ if len(cell) > width0:
+ return False
+ return True
+
+
+def export_set(dataset, **kwargs):
+ """
+ Returns reStructuredText table representation of dataset.
+
+ Returns a simple table if the text in the first column is never
+ wrapped, otherwise returns a grid table.
+
+
+ >>> from tablib import Dataset
+ >>> bits = ((0, 0), (1, 0), (0, 1), (1, 1))
+ >>> data = Dataset()
+ >>> data.headers = ['A', 'B', 'A and B']
+ >>> for a, b in bits:
+ ... data.append([bool(a), bool(b), bool(a * b)])
+ >>> table = data.rst
+ >>> table.split('\\n') == [
+ ... '===== ===== =====',
+ ... ' A B A and',
+ ... ' B ',
+ ... '===== ===== =====',
+ ... 'False False False',
+ ... 'True False False',
+ ... 'False True False',
+ ... 'True True True ',
+ ... '===== ===== =====',
+ ... ]
+ True
+
+ """
+ if not dataset.dict:
+ return ''
+ force_grid = kwargs.get('force_grid', False)
+ max_table_width = kwargs.get('max_table_width', MAX_TABLE_WIDTH)
+ column_widths = _get_column_widths(dataset, max_table_width)
+
+ use_simple_table = _use_simple_table(
+ dataset.headers[0] if dataset.headers else None,
+ dataset.get_col(0),
+ column_widths[0],
+ )
+ if use_simple_table and not force_grid:
+ return export_set_as_simple_table(dataset, column_widths)
+ else:
+ return export_set_as_grid_table(dataset, column_widths)
+
+
+def export_book(databook):
+ """
+ reStructuredText representation of a Databook.
+
+ Tables are separated by a blank line. All tables use the grid
+ format.
+ """
+ return '\n\n'.join(export_set(dataset, force_grid=True)
+ for dataset in databook._datasets)
diff --git a/src/tablib/formats/_tsv.py b/src/tablib/formats/_tsv.py
new file mode 100644
index 0000000..1c6d6a1
--- /dev/null
+++ b/src/tablib/formats/_tsv.py
@@ -0,0 +1,30 @@
+# -*- coding: utf-8 -*-
+
+""" Tablib - TSV (Tab Separated Values) Support.
+"""
+
+from tablib.compat import unicode
+from tablib.formats._csv import (
+ export_set as export_set_wrapper,
+ import_set as import_set_wrapper,
+ detect as detect_wrapper,
+)
+
+title = 'tsv'
+extensions = ('tsv',)
+
+DELIMITER = unicode('\t')
+
+def export_set(dataset):
+ """Returns TSV representation of Dataset."""
+ return export_set_wrapper(dataset, delimiter=DELIMITER)
+
+
+def import_set(dset, in_stream, headers=True):
+ """Returns dataset from TSV stream."""
+ return import_set_wrapper(dset, in_stream, headers=headers, delimiter=DELIMITER)
+
+
+def detect(stream):
+ """Returns True if given stream is valid TSV."""
+ return detect_wrapper(stream, delimiter=DELIMITER)
diff --git a/src/tablib/formats/_xls.py b/src/tablib/formats/_xls.py
new file mode 100644
index 0000000..88e8636
--- /dev/null
+++ b/src/tablib/formats/_xls.py
@@ -0,0 +1,137 @@
+# -*- coding: utf-8 -*-
+
+""" Tablib - XLS Support.
+"""
+
+import sys
+from io import BytesIO
+
+from tablib.compat import xrange
+import tablib
+import xlrd
+import xlwt
+from xlrd.biffh import XLRDError
+
+title = 'xls'
+extensions = ('xls',)
+
+# special styles
+wrap = xlwt.easyxf("alignment: wrap on")
+bold = xlwt.easyxf("font: bold on")
+
+
+def detect(stream):
+ """Returns True if given stream is a readable excel file."""
+ try:
+ xlrd.open_workbook(file_contents=stream)
+ return True
+ except Exception:
+ pass
+ try:
+ xlrd.open_workbook(file_contents=stream.read())
+ return True
+ except Exception:
+ pass
+ try:
+ xlrd.open_workbook(filename=stream)
+ return True
+ except Exception:
+ return False
+
+
+def export_set(dataset):
+ """Returns XLS representation of Dataset."""
+
+ wb = xlwt.Workbook(encoding='utf8')
+ ws = wb.add_sheet(dataset.title if dataset.title else 'Tablib Dataset')
+
+ dset_sheet(dataset, ws)
+
+ stream = BytesIO()
+ wb.save(stream)
+ return stream.getvalue()
+
+
+def export_book(databook):
+ """Returns XLS representation of DataBook."""
+
+ wb = xlwt.Workbook(encoding='utf8')
+
+ for i, dset in enumerate(databook._datasets):
+ ws = wb.add_sheet(dset.title if dset.title else 'Sheet%s' % (i))
+
+ dset_sheet(dset, ws)
+
+ stream = BytesIO()
+ wb.save(stream)
+ return stream.getvalue()
+
+
+def import_set(dset, in_stream, headers=True):
+ """Returns databook from XLS stream."""
+
+ dset.wipe()
+
+ xls_book = xlrd.open_workbook(file_contents=in_stream)
+ sheet = xls_book.sheet_by_index(0)
+
+ dset.title = sheet.name
+
+ for i in xrange(sheet.nrows):
+ if (i == 0) and (headers):
+ dset.headers = sheet.row_values(0)
+ else:
+ dset.append(sheet.row_values(i))
+
+def import_book(dbook, in_stream, headers=True):
+ """Returns databook from XLS stream."""
+
+ dbook.wipe()
+
+ xls_book = xlrd.open_workbook(file_contents=in_stream)
+
+ for sheet in xls_book.sheets():
+ data = tablib.Dataset()
+ data.title = sheet.name
+
+ for i in xrange(sheet.nrows):
+ if (i == 0) and (headers):
+ data.headers = sheet.row_values(0)
+ else:
+ data.append(sheet.row_values(i))
+
+ dbook.add_sheet(data)
+
+
+def dset_sheet(dataset, ws):
+ """Completes given worksheet from given Dataset."""
+ _package = dataset._package(dicts=False)
+
+ for i, sep in enumerate(dataset._separators):
+ _offset = i
+ _package.insert((sep[0] + _offset), (sep[1],))
+
+ for i, row in enumerate(_package):
+ for j, col in enumerate(row):
+
+ # bold headers
+ if (i == 0) and dataset.headers:
+ ws.write(i, j, col, bold)
+
+ # frozen header row
+ ws.panes_frozen = True
+ ws.horz_split_pos = 1
+
+ # bold separators
+ elif len(row) < dataset.width:
+ ws.write(i, j, col, bold)
+
+ # wrap the rest
+ else:
+ try:
+ if '\n' in col:
+ ws.write(i, j, col, wrap)
+ else:
+ ws.write(i, j, col)
+ except TypeError:
+ ws.write(i, j, col)
diff --git a/src/tablib/formats/_xlsx.py b/src/tablib/formats/_xlsx.py
new file mode 100644
index 0000000..f8f21c2
--- /dev/null
+++ b/src/tablib/formats/_xlsx.py
@@ -0,0 +1,144 @@
+# -*- coding: utf-8 -*-
+
+""" Tablib - XLSX Support.
+"""
+
+import sys
+from io import BytesIO
+
+import openpyxl
+import tablib
+
+Workbook = openpyxl.workbook.Workbook
+ExcelWriter = openpyxl.writer.excel.ExcelWriter
+get_column_letter = openpyxl.utils.get_column_letter
+
+from tablib.compat import unicode
+
+
+title = 'xlsx'
+extensions = ('xlsx',)
+
+
+def detect(stream):
+ """Returns True if given stream is a readable excel file."""
+ if isinstance(stream, bytes):
+ # load_workbook expects a file-like object.
+ stream = BytesIO(stream)
+ try:
+ openpyxl.reader.excel.load_workbook(stream, read_only=True)
+ return True
+ except Exception:
+ return False
+
+def export_set(dataset, freeze_panes=True):
+ """Returns XLSX representation of Dataset."""
+
+ wb = Workbook()
+ ws = wb.worksheets[0]
+ ws.title = dataset.title if dataset.title else 'Tablib Dataset'
+
+ dset_sheet(dataset, ws, freeze_panes=freeze_panes)
+
+ stream = BytesIO()
+ wb.save(stream)
+ return stream.getvalue()
+
+
+def export_book(databook, freeze_panes=True):
+ """Returns XLSX representation of DataBook."""
+
+ wb = Workbook()
+ for sheet in wb.worksheets:
+ wb.remove(sheet)
+ for i, dset in enumerate(databook._datasets):
+ ws = wb.create_sheet()
+ ws.title = dset.title if dset.title else 'Sheet%s' % (i)
+
+ dset_sheet(dset, ws, freeze_panes=freeze_panes)
+
+ stream = BytesIO()
+ wb.save(stream)
+ return stream.getvalue()
+
+
+def import_set(dset, in_stream, headers=True):
+ """Returns databook from XLS stream."""
+
+ dset.wipe()
+
+ xls_book = openpyxl.reader.excel.load_workbook(BytesIO(in_stream), read_only=True)
+ sheet = xls_book.active
+
+ dset.title = sheet.title
+
+ for i, row in enumerate(sheet.rows):
+ row_vals = [c.value for c in row]
+ if (i == 0) and (headers):
+ dset.headers = row_vals
+ else:
+ dset.append(row_vals)
+
+
+def import_book(dbook, in_stream, headers=True):
+ """Returns databook from XLS stream."""
+
+ dbook.wipe()
+
+ xls_book = openpyxl.reader.excel.load_workbook(BytesIO(in_stream), read_only=True)
+
+ for sheet in xls_book.worksheets:
+ data = tablib.Dataset()
+ data.title = sheet.title
+
+ for i, row in enumerate(sheet.rows):
+ row_vals = [c.value for c in row]
+ if (i == 0) and (headers):
+ data.headers = row_vals
+ else:
+ data.append(row_vals)
+
+ dbook.add_sheet(data)
+
+
+def dset_sheet(dataset, ws, freeze_panes=True):
+ """Completes given worksheet from given Dataset."""
+ _package = dataset._package(dicts=False)
+
+ for i, sep in enumerate(dataset._separators):
+ _offset = i
+ _package.insert((sep[0] + _offset), (sep[1],))
+
+ bold = openpyxl.styles.Font(bold=True)
+ wrap_text = openpyxl.styles.Alignment(wrap_text=True)
+
+ for i, row in enumerate(_package):
+ row_number = i + 1
+ for j, col in enumerate(row):
+ col_idx = get_column_letter(j + 1)
+ cell = ws['%s%s' % (col_idx, row_number)]
+
+ # bold headers
+ if (row_number == 1) and dataset.headers:
+ cell.font = bold
+ if freeze_panes:
+ # Export Freeze only after first Line
+ ws.freeze_panes = 'A2'
+
+ # bold separators
+ elif len(row) < dataset.width:
+ cell.font = bold
+
+ # wrap the rest
+ else:
+ try:
+ str_col_value = unicode(col)
+ except TypeError:
+ str_col_value = ''
+ if '\n' in str_col_value:
+ cell.alignment = wrap_text
+
+ try:
+ cell.value = col
+ except (ValueError, TypeError):
+ cell.value = unicode(col)
diff --git a/src/tablib/formats/_yaml.py b/src/tablib/formats/_yaml.py
new file mode 100644
index 0000000..3d17baf
--- /dev/null
+++ b/src/tablib/formats/_yaml.py
@@ -0,0 +1,53 @@
+# -*- coding: utf-8 -*-
+
+""" Tablib - YAML Support.
+"""
+
+import tablib
+import yaml
+
+title = 'yaml'
+extensions = ('yaml', 'yml')
+
+
+def export_set(dataset):
+ """Returns YAML representation of Dataset."""
+
+ return yaml.safe_dump(dataset._package(ordered=False))
+
+
+def export_book(databook):
+ """Returns YAML representation of Databook."""
+ return yaml.safe_dump(databook._package(ordered=False))
+
+
+def import_set(dset, in_stream):
+ """Returns dataset from YAML stream."""
+
+ dset.wipe()
+ dset.dict = yaml.safe_load(in_stream)
+
+
+def import_book(dbook, in_stream):
+ """Returns databook from YAML stream."""
+
+ dbook.wipe()
+
+ for sheet in yaml.safe_load(in_stream):
+ data = tablib.Dataset()
+ data.title = sheet['title']
+ data.dict = sheet['data']
+ dbook.add_sheet(data)
+
+
+def detect(stream):
+ """Returns True if given stream is valid YAML."""
+ try:
+ _yaml = yaml.safe_load(stream)
+ if isinstance(_yaml, (list, tuple, dict)):
+ return True
+ else:
+ return False
+ except (yaml.parser.ParserError, yaml.reader.ReaderError,
+ yaml.scanner.ScannerError):
+ return False
diff --git a/src/tablib/packages/__init__.py b/src/tablib/packages/__init__.py
new file mode 100644
index 0000000..e69de29
--- /dev/null
+++ b/src/tablib/packages/__init__.py
diff --git a/src/tablib/packages/dbfpy/__init__.py b/src/tablib/packages/dbfpy/__init__.py
new file mode 100644
index 0000000..e69de29
--- /dev/null
+++ b/src/tablib/packages/dbfpy/__init__.py
diff --git a/src/tablib/packages/dbfpy/dbf.py b/src/tablib/packages/dbfpy/dbf.py
new file mode 100644
index 0000000..8147d0e
--- /dev/null
+++ b/src/tablib/packages/dbfpy/dbf.py
@@ -0,0 +1,297 @@
+#! /usr/bin/env python
+"""DBF accessing helpers.
+
+FIXME: more documentation needed
+
+Examples:
+
+ Create new table, setup structure, add records:
+
+ dbf = Dbf(filename, new=True)
+ dbf.addField(
+ ("NAME", "C", 15),
+ ("SURNAME", "C", 25),
+ ("INITIALS", "C", 10),
+ ("BIRTHDATE", "D"),
+ )
+ for (n, s, i, b) in (
+ ("John", "Miller", "YC", (1980, 10, 11)),
+ ("Andy", "Larkin", "", (1980, 4, 11)),
+ ):
+ rec = dbf.newRecord()
+ rec["NAME"] = n
+ rec["SURNAME"] = s
+ rec["INITIALS"] = i
+ rec["BIRTHDATE"] = b
+ rec.store()
+ dbf.close()
+
+ Open existed dbf, read some data:
+
+ dbf = Dbf(filename, True)
+ for rec in dbf:
+ for fldName in dbf.fieldNames:
+ print('%s:\t %s (%s)' % (fldName, rec[fldName],
+ type(rec[fldName])))
+ print()
+ dbf.close()
+
+"""
+"""History (most recent first):
+11-feb-2007 [als] export INVALID_VALUE;
+ Dbf: added .ignoreErrors, .INVALID_VALUE
+04-jul-2006 [als] added export declaration
+20-dec-2005 [yc] removed fromStream and newDbf methods:
+ use argument of __init__ call must be used instead;
+ added class fields pointing to the header and
+ record classes.
+17-dec-2005 [yc] split to several modules; reimplemented
+13-dec-2005 [yc] adapted to the changes of the `strutil` module.
+13-sep-2002 [als] support FoxPro Timestamp datatype
+15-nov-1999 [jjk] documentation updates, add demo
+24-aug-1998 [jjk] add some encodeValue methods (not tested), other tweaks
+08-jun-1998 [jjk] fix problems, add more features
+20-feb-1998 [jjk] fix problems, add more features
+19-feb-1998 [jjk] add create/write capabilities
+18-feb-1998 [jjk] from dbfload.py
+"""
+
+__version__ = "$Revision: 1.7 $"[11:-2]
+__date__ = "$Date: 2007/02/11 09:23:13 $"[7:-2]
+__author__ = "Jeff Kunce <kuncej@mail.conservation.state.mo.us>"
+
+__all__ = ["Dbf"]
+
+from . import header
+from . import record
+from utils import INVALID_VALUE
+
+
+class Dbf(object):
+ """DBF accessor.
+
+ FIXME:
+ docs and examples needed (dont' forget to tell
+ about problems adding new fields on the fly)
+
+ Implementation notes:
+ ``_new`` field is used to indicate whether this is
+ a new data table. `addField` could be used only for
+ the new tables! If at least one record was appended
+ to the table it's structure couldn't be changed.
+
+ """
+
+ __slots__ = ("name", "header", "stream",
+ "_changed", "_new", "_ignore_errors")
+
+ HeaderClass = header.DbfHeader
+ RecordClass = record.DbfRecord
+ INVALID_VALUE = INVALID_VALUE
+
+ # initialization and creation helpers
+
+ def __init__(self, f, readOnly=False, new=False, ignoreErrors=False):
+ """Initialize instance.
+
+ Arguments:
+ f:
+ Filename or file-like object.
+ new:
+ True if new data table must be created. Assume
+ data table exists if this argument is False.
+ readOnly:
+ if ``f`` argument is a string file will
+ be opend in read-only mode; in other cases
+ this argument is ignored. This argument is ignored
+ even if ``new`` argument is True.
+ headerObj:
+ `header.DbfHeader` instance or None. If this argument
+ is None, new empty header will be used with the
+ all fields set by default.
+ ignoreErrors:
+ if set, failing field value conversion will return
+ ``INVALID_VALUE`` instead of raising conversion error.
+
+ """
+ if isinstance(f, basestring):
+ # a filename
+ self.name = f
+ if new:
+ # new table (table file must be
+ # created or opened and truncated)
+ self.stream = file(f, "w+b")
+ else:
+ # tabe file must exist
+ self.stream = file(f, ("r+b", "rb")[bool(readOnly)])
+ else:
+ # a stream
+ self.name = getattr(f, "name", "")
+ self.stream = f
+ if new:
+ # if this is a new table, header will be empty
+ self.header = self.HeaderClass()
+ else:
+ # or instantiated using stream
+ self.header = self.HeaderClass.fromStream(self.stream)
+ self.ignoreErrors = ignoreErrors
+ self._new = bool(new)
+ self._changed = False
+
+ # properties
+
+ closed = property(lambda self: self.stream.closed)
+ recordCount = property(lambda self: self.header.recordCount)
+ fieldNames = property(
+ lambda self: [_fld.name for _fld in self.header.fields])
+ fieldDefs = property(lambda self: self.header.fields)
+ changed = property(lambda self: self._changed or self.header.changed)
+
+ def ignoreErrors(self, value):
+ """Update `ignoreErrors` flag on the header object and self"""
+ self.header.ignoreErrors = self._ignore_errors = bool(value)
+
+ ignoreErrors = property(
+ lambda self: self._ignore_errors,
+ ignoreErrors,
+ doc="""Error processing mode for DBF field value conversion
+
+ if set, failing field value conversion will return
+ ``INVALID_VALUE`` instead of raising conversion error.
+
+ """)
+
+ # protected methods
+
+ def _fixIndex(self, index):
+ """Return fixed index.
+
+ This method fails if index isn't a numeric object
+ (long or int). Or index isn't in a valid range
+ (less or equal to the number of records in the db).
+
+ If ``index`` is a negative number, it will be
+ treated as a negative indexes for list objects.
+
+ Return:
+ Return value is numeric object maning valid index.
+
+ """
+ if not isinstance(index, (int, long)):
+ raise TypeError("Index must be a numeric object")
+ if index < 0:
+ # index from the right side
+ # fix it to the left-side index
+ index += len(self) + 1
+ if index >= len(self):
+ raise IndexError("Record index out of range")
+ return index
+
+ # iterface methods
+
+ def close(self):
+ self.flush()
+ self.stream.close()
+
+ def flush(self):
+ """Flush data to the associated stream."""
+ if self.changed:
+ self.header.setCurrentDate()
+ self.header.write(self.stream)
+ self.stream.flush()
+ self._changed = False
+
+ def indexOfFieldName(self, name):
+ """Index of field named ``name``."""
+ # FIXME: move this to header class
+ return self.header.fields.index(name)
+
+ def newRecord(self):
+ """Return new record, which belong to this table."""
+ return self.RecordClass(self)
+
+ def append(self, record):
+ """Append ``record`` to the database."""
+ record.index = self.header.recordCount
+ record._write()
+ self.header.recordCount += 1
+ self._changed = True
+ self._new = False
+
+ def addField(self, *defs):
+ """Add field definitions.
+
+ For more information see `header.DbfHeader.addField`.
+
+ """
+ if self._new:
+ self.header.addField(*defs)
+ else:
+ raise TypeError("At least one record was added, "
+ "structure can't be changed")
+
+ # 'magic' methods (representation and sequence interface)
+
+ def __repr__(self):
+ return "Dbf stream '%s'\n" % self.stream + repr(self.header)
+
+ def __len__(self):
+ """Return number of records."""
+ return self.recordCount
+
+ def __getitem__(self, index):
+ """Return `DbfRecord` instance."""
+ return self.RecordClass.fromStream(self, self._fixIndex(index))
+
+ def __setitem__(self, index, record):
+ """Write `DbfRecord` instance to the stream."""
+ record.index = self._fixIndex(index)
+ record._write()
+ self._changed = True
+ self._new = False
+
+ # def __del__(self):
+ # """Flush stream upon deletion of the object."""
+ # self.flush()
+
+
+def demo_read(filename):
+ _dbf = Dbf(filename, True)
+ for _rec in _dbf:
+ print()
+ print(repr(_rec))
+ _dbf.close()
+
+
+def demo_create(filename):
+ _dbf = Dbf(filename, new=True)
+ _dbf.addField(
+ ("NAME", "C", 15),
+ ("SURNAME", "C", 25),
+ ("INITIALS", "C", 10),
+ ("BIRTHDATE", "D"),
+ )
+ for (_n, _s, _i, _b) in (
+ ("John", "Miller", "YC", (1981, 1, 2)),
+ ("Andy", "Larkin", "AL", (1982, 3, 4)),
+ ("Bill", "Clinth", "", (1983, 5, 6)),
+ ("Bobb", "McNail", "", (1984, 7, 8)),
+ ):
+ _rec = _dbf.newRecord()
+ _rec["NAME"] = _n
+ _rec["SURNAME"] = _s
+ _rec["INITIALS"] = _i
+ _rec["BIRTHDATE"] = _b
+ _rec.store()
+ print(repr(_dbf))
+ _dbf.close()
+
+
+if __name__ == '__main__':
+ import sys
+
+ _name = len(sys.argv) > 1 and sys.argv[1] or "county.dbf"
+ demo_create(_name)
+ demo_read(_name)
+
+ # vim: set et sw=4 sts=4 :
diff --git a/src/tablib/packages/dbfpy/dbfnew.py b/src/tablib/packages/dbfpy/dbfnew.py
new file mode 100644
index 0000000..008d307
--- /dev/null
+++ b/src/tablib/packages/dbfpy/dbfnew.py
@@ -0,0 +1,189 @@
+#!/usr/bin/python
+""".DBF creation helpers.
+
+Note: this is a legacy interface. New code should use Dbf class
+ for table creation (see examples in dbf.py)
+
+TODO:
+ - handle Memo fields.
+ - check length of the fields accoring to the
+ `http://www.clicketyclick.dk/databases/xbase/format/data_types.html`
+
+"""
+"""History (most recent first)
+04-jul-2006 [als] added export declaration;
+ updated for dbfpy 2.0
+15-dec-2005 [yc] define dbf_new.__slots__
+14-dec-2005 [yc] added vim modeline; retab'd; added doc-strings;
+ dbf_new now is a new class (inherited from object)
+??-jun-2000 [--] added by Hans Fiby
+"""
+
+__version__ = "$Revision: 1.4 $"[11:-2]
+__date__ = "$Date: 2006/07/04 08:18:18 $"[7:-2]
+
+__all__ = ["dbf_new"]
+
+from dbf import *
+from fields import *
+from header import *
+from record import *
+
+
+class _FieldDefinition(object):
+ """Field definition.
+
+ This is a simple structure, which contains ``name``, ``type``,
+ ``len``, ``dec`` and ``cls`` fields.
+
+ Objects also implement get/setitem magic functions, so fields
+ could be accessed via sequence iterface, where 'name' has
+ index 0, 'type' index 1, 'len' index 2, 'dec' index 3 and
+ 'cls' could be located at index 4.
+
+ """
+
+ __slots__ = "name", "type", "len", "dec", "cls"
+
+ # WARNING: be attentive - dictionaries are mutable!
+ FLD_TYPES = {
+ # type: (cls, len)
+ "C": (DbfCharacterFieldDef, None),
+ "N": (DbfNumericFieldDef, None),
+ "L": (DbfLogicalFieldDef, 1),
+ # FIXME: support memos
+ # "M": (DbfMemoFieldDef),
+ "D": (DbfDateFieldDef, 8),
+ # FIXME: I'm not sure length should be 14 characters!
+ # but temporary I use it, cuz date is 8 characters
+ # and time 6 (hhmmss)
+ "T": (DbfDateTimeFieldDef, 14),
+ }
+
+ def __init__(self, name, type, len=None, dec=0):
+ _cls, _len = self.FLD_TYPES[type]
+ if _len is None:
+ if len is None:
+ raise ValueError("Field length must be defined")
+ _len = len
+ self.name = name
+ self.type = type
+ self.len = _len
+ self.dec = dec
+ self.cls = _cls
+
+ def getDbfField(self):
+ "Return `DbfFieldDef` instance from the current definition."
+ return self.cls(self.name, self.len, self.dec)
+
+ def appendToHeader(self, dbfh):
+ """Create a `DbfFieldDef` instance and append it to the dbf header.
+
+ Arguments:
+ dbfh: `DbfHeader` instance.
+
+ """
+ _dbff = self.getDbfField()
+ dbfh.addField(_dbff)
+
+
+class dbf_new(object):
+ """New .DBF creation helper.
+
+ Example Usage:
+
+ dbfn = dbf_new()
+ dbfn.add_field("name",'C',80)
+ dbfn.add_field("price",'N',10,2)
+ dbfn.add_field("date",'D',8)
+ dbfn.write("tst.dbf")
+
+ Note:
+ This module cannot handle Memo-fields,
+ they are special.
+
+ """
+
+ __slots__ = ("fields",)
+
+ FieldDefinitionClass = _FieldDefinition
+
+ def __init__(self):
+ self.fields = []
+
+ def add_field(self, name, typ, len, dec=0):
+ """Add field definition.
+
+ Arguments:
+ name:
+ field name (str object). field name must not
+ contain ASCII NULs and it's length shouldn't
+ exceed 10 characters.
+ typ:
+ type of the field. this must be a single character
+ from the "CNLMDT" set meaning character, numeric,
+ logical, memo, date and date/time respectively.
+ len:
+ length of the field. this argument is used only for
+ the character and numeric fields. all other fields
+ have fixed length.
+ FIXME: use None as a default for this argument?
+ dec:
+ decimal precision. used only for the numric fields.
+
+ """
+ self.fields.append(self.FieldDefinitionClass(name, typ, len, dec))
+
+ def write(self, filename):
+ """Create empty .DBF file using current structure."""
+ _dbfh = DbfHeader()
+ _dbfh.setCurrentDate()
+ for _fldDef in self.fields:
+ _fldDef.appendToHeader(_dbfh)
+ _dbfStream = file(filename, "wb")
+ _dbfh.write(_dbfStream)
+ _dbfStream.close()
+
+ def write_stream(self, stream):
+ _dbfh = DbfHeader()
+ _dbfh.setCurrentDate()
+ for _fldDef in self.fields:
+ _fldDef.appendToHeader(_dbfh)
+ _dbfh.write(stream)
+
+
+if __name__ == '__main__':
+ # create a new DBF-File
+ dbfn = dbf_new()
+ dbfn.add_field("name", 'C', 80)
+ dbfn.add_field("price", 'N', 10, 2)
+ dbfn.add_field("date", 'D', 8)
+ dbfn.write("tst.dbf")
+ # test new dbf
+ print("*** created tst.dbf: ***")
+ dbft = Dbf('tst.dbf', readOnly=0)
+ print(repr(dbft))
+ # add a record
+ rec = DbfRecord(dbft)
+ rec['name'] = 'something'
+ rec['price'] = 10.5
+ rec['date'] = (2000, 1, 12)
+ rec.store()
+ # add another record
+ rec = DbfRecord(dbft)
+ rec['name'] = 'foo and bar'
+ rec['price'] = 12234
+ rec['date'] = (1992, 7, 15)
+ rec.store()
+
+ # show the records
+ print("*** inserted 2 records into tst.dbf: ***")
+ print(repr(dbft))
+ for i1 in range(len(dbft)):
+ rec = dbft[i1]
+ for fldName in dbft.fieldNames:
+ print('%s:\t %s' % (fldName, rec[fldName]))
+ print()
+ dbft.close()
+
+ # vim: set et sts=4 sw=4 :
diff --git a/src/tablib/packages/dbfpy/fields.py b/src/tablib/packages/dbfpy/fields.py
new file mode 100644
index 0000000..69cd436
--- /dev/null
+++ b/src/tablib/packages/dbfpy/fields.py
@@ -0,0 +1,466 @@
+"""DBF fields definitions.
+
+TODO:
+ - make memos work
+"""
+"""History (most recent first):
+26-may-2009 [als] DbfNumericFieldDef.decodeValue: strip zero bytes
+05-feb-2009 [als] DbfDateFieldDef.encodeValue: empty arg produces empty date
+16-sep-2008 [als] DbfNumericFieldDef decoding looks for decimal point
+ in the value to select float or integer return type
+13-mar-2008 [als] check field name length in constructor
+11-feb-2007 [als] handle value conversion errors
+10-feb-2007 [als] DbfFieldDef: added .rawFromRecord()
+01-dec-2006 [als] Timestamp columns use None for empty values
+31-oct-2006 [als] support field types 'F' (float), 'I' (integer)
+ and 'Y' (currency);
+ automate export and registration of field classes
+04-jul-2006 [als] added export declaration
+10-mar-2006 [als] decode empty values for Date and Logical fields;
+ show field name in errors
+10-mar-2006 [als] fix Numeric value decoding: according to spec,
+ value always is string representation of the number;
+ ensure that encoded Numeric value fits into the field
+20-dec-2005 [yc] use field names in upper case
+15-dec-2005 [yc] field definitions moved from `dbf`.
+"""
+
+__version__ = "$Revision: 1.14 $"[11:-2]
+__date__ = "$Date: 2009/05/26 05:16:51 $"[7:-2]
+
+__all__ = ["lookupFor",] # field classes added at the end of the module
+
+import datetime
+import struct
+import sys
+
+from . import utils
+
+## abstract definitions
+
+class DbfFieldDef(object):
+ """Abstract field definition.
+
+ Child classes must override ``type`` class attribute to provide datatype
+ infromation of the field definition. For more info about types visit
+ `http://www.clicketyclick.dk/databases/xbase/format/data_types.html`
+
+ Also child classes must override ``defaultValue`` field to provide
+ default value for the field value.
+
+ If child class has fixed length ``length`` class attribute must be
+ overriden and set to the valid value. None value means, that field
+ isn't of fixed length.
+
+ Note: ``name`` field must not be changed after instantiation.
+
+ """
+
+ __slots__ = ("name", "length", "decimalCount",
+ "start", "end", "ignoreErrors")
+
+ # length of the field, None in case of variable-length field,
+ # or a number if this field is a fixed-length field
+ length = None
+
+ # field type. for more information about fields types visit
+ # `http://www.clicketyclick.dk/databases/xbase/format/data_types.html`
+ # must be overriden in child classes
+ typeCode = None
+
+ # default value for the field. this field must be
+ # overriden in child classes
+ defaultValue = None
+
+ def __init__(self, name, length=None, decimalCount=None,
+ start=None, stop=None, ignoreErrors=False,
+ ):
+ """Initialize instance."""
+ assert self.typeCode is not None, "Type code must be overriden"
+ assert self.defaultValue is not None, "Default value must be overriden"
+ ## fix arguments
+ if len(name) >10:
+ raise ValueError("Field name \"%s\" is too long" % name)
+ name = str(name).upper()
+ if self.__class__.length is None:
+ if length is None:
+ raise ValueError("[%s] Length isn't specified" % name)
+ length = int(length)
+ if length <= 0:
+ raise ValueError("[%s] Length must be a positive integer"
+ % name)
+ else:
+ length = self.length
+ if decimalCount is None:
+ decimalCount = 0
+ ## set fields
+ self.name = name
+ # FIXME: validate length according to the specification at
+ # http://www.clicketyclick.dk/databases/xbase/format/data_types.html
+ self.length = length
+ self.decimalCount = decimalCount
+ self.ignoreErrors = ignoreErrors
+ self.start = start
+ self.end = stop
+
+ def __cmp__(self, other):
+ return cmp(self.name, str(other).upper())
+
+ def __hash__(self):
+ return hash(self.name)
+
+ def fromString(cls, string, start, ignoreErrors=False):
+ """Decode dbf field definition from the string data.
+
+ Arguments:
+ string:
+ a string, dbf definition is decoded from. length of
+ the string must be 32 bytes.
+ start:
+ position in the database file.
+ ignoreErrors:
+ initial error processing mode for the new field (boolean)
+
+ """
+ assert len(string) == 32
+ _length = ord(string[16])
+ return cls(utils.unzfill(string)[:11], _length, ord(string[17]),
+ start, start + _length, ignoreErrors=ignoreErrors)
+ fromString = classmethod(fromString)
+
+ def toString(self):
+ """Return encoded field definition.
+
+ Return:
+ Return value is a string object containing encoded
+ definition of this field.
+
+ """
+ if sys.version_info < (2, 4):
+ # earlier versions did not support padding character
+ _name = self.name[:11] + "\0" * (11 - len(self.name))
+ else:
+ _name = self.name.ljust(11, '\0')
+ return (
+ _name +
+ self.typeCode +
+ #data address
+ chr(0) * 4 +
+ chr(self.length) +
+ chr(self.decimalCount) +
+ chr(0) * 14
+ )
+
+ def __repr__(self):
+ return "%-10s %1s %3d %3d" % self.fieldInfo()
+
+ def fieldInfo(self):
+ """Return field information.
+
+ Return:
+ Return value is a (name, type, length, decimals) tuple.
+
+ """
+ return (self.name, self.typeCode, self.length, self.decimalCount)
+
+ def rawFromRecord(self, record):
+ """Return a "raw" field value from the record string."""
+ return record[self.start:self.end]
+
+ def decodeFromRecord(self, record):
+ """Return decoded field value from the record string."""
+ try:
+ return self.decodeValue(self.rawFromRecord(record))
+ except:
+ if self.ignoreErrors:
+ return utils.INVALID_VALUE
+ else:
+ raise
+
+ def decodeValue(self, value):
+ """Return decoded value from string value.
+
+ This method shouldn't be used publicly. It's called from the
+ `decodeFromRecord` method.
+
+ This is an abstract method and it must be overridden in child classes.
+ """
+ raise NotImplementedError
+
+ def encodeValue(self, value):
+ """Return str object containing encoded field value.
+
+ This is an abstract method and it must be overriden in child classes.
+ """
+ raise NotImplementedError
+
+## real classes
+
+class DbfCharacterFieldDef(DbfFieldDef):
+ """Definition of the character field."""
+
+ typeCode = "C"
+ defaultValue = ""
+
+ def decodeValue(self, value):
+ """Return string object.
+
+ Return value is a ``value`` argument with stripped right spaces.
+
+ """
+ return value.rstrip(" ")
+
+ def encodeValue(self, value):
+ """Return raw data string encoded from a ``value``."""
+ return str(value)[:self.length].ljust(self.length)
+
+
+class DbfNumericFieldDef(DbfFieldDef):
+ """Definition of the numeric field."""
+
+ typeCode = "N"
+ # XXX: now I'm not sure it was a good idea to make a class field
+ # `defaultValue` instead of a generic method as it was implemented
+ # previously -- it's ok with all types except number, cuz
+ # if self.decimalCount is 0, we should return 0 and 0.0 otherwise.
+ defaultValue = 0
+
+ def decodeValue(self, value):
+ """Return a number decoded from ``value``.
+
+ If decimals is zero, value will be decoded as an integer;
+ or as a float otherwise.
+
+ Return:
+ Return value is a int (long) or float instance.
+
+ """
+ value = value.strip(" \0")
+ if "." in value:
+ # a float (has decimal separator)
+ return float(value)
+ elif value:
+ # must be an integer
+ return int(value)
+ else:
+ return 0
+
+ def encodeValue(self, value):
+ """Return string containing encoded ``value``."""
+ _rv = ("%*.*f" % (self.length, self.decimalCount, value))
+ if len(_rv) > self.length:
+ _ppos = _rv.find(".")
+ if 0 <= _ppos <= self.length:
+ _rv = _rv[:self.length]
+ else:
+ raise ValueError("[%s] Numeric overflow: %s (field width: %i)"
+ % (self.name, _rv, self.length))
+ return _rv
+
+class DbfFloatFieldDef(DbfNumericFieldDef):
+ """Definition of the float field - same as numeric."""
+
+ typeCode = "F"
+
+class DbfIntegerFieldDef(DbfFieldDef):
+ """Definition of the integer field."""
+
+ typeCode = "I"
+ length = 4
+ defaultValue = 0
+
+ def decodeValue(self, value):
+ """Return an integer number decoded from ``value``."""
+ return struct.unpack("<i", value)[0]
+
+ def encodeValue(self, value):
+ """Return string containing encoded ``value``."""
+ return struct.pack("<i", int(value))
+
+class DbfCurrencyFieldDef(DbfFieldDef):
+ """Definition of the currency field."""
+
+ typeCode = "Y"
+ length = 8
+ defaultValue = 0.0
+
+ def decodeValue(self, value):
+ """Return float number decoded from ``value``."""
+ return struct.unpack("<q", value)[0] / 10000.
+
+ def encodeValue(self, value):
+ """Return string containing encoded ``value``."""
+ return struct.pack("<q", round(value * 10000))
+
+class DbfLogicalFieldDef(DbfFieldDef):
+ """Definition of the logical field."""
+
+ typeCode = "L"
+ defaultValue = -1
+ length = 1
+
+ def decodeValue(self, value):
+ """Return True, False or -1 decoded from ``value``."""
+ # Note: value always is 1-char string
+ if value == "?":
+ return -1
+ if value in "NnFf ":
+ return False
+ if value in "YyTt":
+ return True
+ raise ValueError("[%s] Invalid logical value %r" % (self.name, value))
+
+ def encodeValue(self, value):
+ """Return a character from the "TF?" set.
+
+ Return:
+ Return value is "T" if ``value`` is True
+ "?" if value is -1 or False otherwise.
+
+ """
+ if value is True:
+ return "T"
+ if value == -1:
+ return "?"
+ return "F"
+
+
+class DbfMemoFieldDef(DbfFieldDef):
+ """Definition of the memo field.
+
+ Note: memos aren't currenly completely supported.
+
+ """
+
+ typeCode = "M"
+ defaultValue = " " * 10
+ length = 10
+
+ def decodeValue(self, value):
+ """Return int .dbt block number decoded from the string object."""
+ #return int(value)
+ raise NotImplementedError
+
+ def encodeValue(self, value):
+ """Return raw data string encoded from a ``value``.
+
+ Note: this is an internal method.
+
+ """
+ #return str(value)[:self.length].ljust(self.length)
+ raise NotImplementedError
+
+
+class DbfDateFieldDef(DbfFieldDef):
+ """Definition of the date field."""
+
+ typeCode = "D"
+ defaultValue = utils.classproperty(lambda cls: datetime.date.today())
+ # "yyyymmdd" gives us 8 characters
+ length = 8
+
+ def decodeValue(self, value):
+ """Return a ``datetime.date`` instance decoded from ``value``."""
+ if value.strip():
+ return utils.getDate(value)
+ else:
+ return None
+
+ def encodeValue(self, value):
+ """Return a string-encoded value.
+
+ ``value`` argument should be a value suitable for the
+ `utils.getDate` call.
+
+ Return:
+ Return value is a string in format "yyyymmdd".
+
+ """
+ if value:
+ return utils.getDate(value).strftime("%Y%m%d")
+ else:
+ return " " * self.length
+
+
+class DbfDateTimeFieldDef(DbfFieldDef):
+ """Definition of the timestamp field."""
+
+ # a difference between JDN (Julian Day Number)
+ # and GDN (Gregorian Day Number). note, that GDN < JDN
+ JDN_GDN_DIFF = 1721425
+ typeCode = "T"
+ defaultValue = utils.classproperty(lambda cls: datetime.datetime.now())
+ # two 32-bits integers representing JDN and amount of
+ # milliseconds respectively gives us 8 bytes.
+ # note, that values must be encoded in LE byteorder.
+ length = 8
+
+ def decodeValue(self, value):
+ """Return a `datetime.datetime` instance."""
+ assert len(value) == self.length
+ # LE byteorder
+ _jdn, _msecs = struct.unpack("<2I", value)
+ if _jdn >= 1:
+ _rv = datetime.datetime.fromordinal(_jdn - self.JDN_GDN_DIFF)
+ _rv += datetime.timedelta(0, _msecs / 1000.0)
+ else:
+ # empty date
+ _rv = None
+ return _rv
+
+ def encodeValue(self, value):
+ """Return a string-encoded ``value``."""
+ if value:
+ value = utils.getDateTime(value)
+ # LE byteorder
+ _rv = struct.pack("<2I", value.toordinal() + self.JDN_GDN_DIFF,
+ (value.hour * 3600 + value.minute * 60 + value.second) * 1000)
+ else:
+ _rv = "\0" * self.length
+ assert len(_rv) == self.length
+ return _rv
+
+
+_fieldsRegistry = {}
+
+def registerField(fieldCls):
+ """Register field definition class.
+
+ ``fieldCls`` should be subclass of the `DbfFieldDef`.
+
+ Use `lookupFor` to retrieve field definition class
+ by the type code.
+
+ """
+ assert fieldCls.typeCode is not None, "Type code isn't defined"
+ # XXX: use fieldCls.typeCode.upper()? in case of any decign
+ # don't forget to look to the same comment in ``lookupFor`` method
+ _fieldsRegistry[fieldCls.typeCode] = fieldCls
+
+
+def lookupFor(typeCode):
+ """Return field definition class for the given type code.
+
+ ``typeCode`` must be a single character. That type should be
+ previously registered.
+
+ Use `registerField` to register new field class.
+
+ Return:
+ Return value is a subclass of the `DbfFieldDef`.
+
+ """
+ # XXX: use typeCode.upper()? in case of any decign don't
+ # forget to look to the same comment in ``registerField``
+ return _fieldsRegistry[typeCode]
+
+## register generic types
+
+for (_name, _val) in globals().items():
+ if isinstance(_val, type) and issubclass(_val, DbfFieldDef) \
+ and (_name != "DbfFieldDef"):
+ __all__.append(_name)
+ registerField(_val)
+del _name, _val
+
+# vim: et sts=4 sw=4 :
diff --git a/src/tablib/packages/dbfpy/header.py b/src/tablib/packages/dbfpy/header.py
new file mode 100644
index 0000000..03a877c
--- /dev/null
+++ b/src/tablib/packages/dbfpy/header.py
@@ -0,0 +1,275 @@
+"""DBF header definition.
+
+TODO:
+ - handle encoding of the character fields
+ (encoding information stored in the DBF header)
+
+"""
+"""History (most recent first):
+16-sep-2010 [als] fromStream: fix century of the last update field
+11-feb-2007 [als] added .ignoreErrors
+10-feb-2007 [als] added __getitem__: return field definitions
+ by field name or field number (zero-based)
+04-jul-2006 [als] added export declaration
+15-dec-2005 [yc] created
+"""
+
+__version__ = "$Revision: 1.6 $"[11:-2]
+__date__ = "$Date: 2010/09/16 05:06:39 $"[7:-2]
+
+__all__ = ["DbfHeader"]
+
+try:
+ import cStringIO
+except ImportError:
+ # when we're in python3, we cStringIO has been replaced by io.StringIO
+ import io as cStringIO
+import datetime
+import struct
+import time
+
+from . import fields
+from . import utils
+
+
+class DbfHeader(object):
+ """Dbf header definition.
+
+ For more information about dbf header format visit
+ `http://www.clicketyclick.dk/databases/xbase/format/dbf.html#DBF_STRUCT`
+
+ Examples:
+ Create an empty dbf header and add some field definitions:
+ dbfh = DbfHeader()
+ dbfh.addField(("name", "C", 10))
+ dbfh.addField(("date", "D"))
+ dbfh.addField(DbfNumericFieldDef("price", 5, 2))
+ Create a dbf header with field definitions:
+ dbfh = DbfHeader([
+ ("name", "C", 10),
+ ("date", "D"),
+ DbfNumericFieldDef("price", 5, 2),
+ ])
+
+ """
+
+ __slots__ = ("signature", "fields", "lastUpdate", "recordLength",
+ "recordCount", "headerLength", "changed", "_ignore_errors")
+
+ ## instance construction and initialization methods
+
+ def __init__(self, fields=None, headerLength=0, recordLength=0,
+ recordCount=0, signature=0x03, lastUpdate=None, ignoreErrors=False,
+ ):
+ """Initialize instance.
+
+ Arguments:
+ fields:
+ a list of field definitions;
+ recordLength:
+ size of the records;
+ headerLength:
+ size of the header;
+ recordCount:
+ number of records stored in DBF;
+ signature:
+ version number (aka signature). using 0x03 as a default meaning
+ "File without DBT". for more information about this field visit
+ ``http://www.clicketyclick.dk/databases/xbase/format/dbf.html#DBF_NOTE_1_TARGET``
+ lastUpdate:
+ date of the DBF's update. this could be a string ('yymmdd' or
+ 'yyyymmdd'), timestamp (int or float), datetime/date value,
+ a sequence (assuming (yyyy, mm, dd, ...)) or an object having
+ callable ``ticks`` field.
+ ignoreErrors:
+ error processing mode for DBF fields (boolean)
+
+ """
+ self.signature = signature
+ if fields is None:
+ self.fields = []
+ else:
+ self.fields = list(fields)
+ self.lastUpdate = utils.getDate(lastUpdate)
+ self.recordLength = recordLength
+ self.headerLength = headerLength
+ self.recordCount = recordCount
+ self.ignoreErrors = ignoreErrors
+ # XXX: I'm not sure this is safe to
+ # initialize `self.changed` in this way
+ self.changed = bool(self.fields)
+
+ # @classmethod
+ def fromString(cls, string):
+ """Return header instance from the string object."""
+ return cls.fromStream(cStringIO.StringIO(str(string)))
+ fromString = classmethod(fromString)
+
+ # @classmethod
+ def fromStream(cls, stream):
+ """Return header object from the stream."""
+ stream.seek(0)
+ _data = stream.read(32)
+ (_cnt, _hdrLen, _recLen) = struct.unpack("<I2H", _data[4:12])
+ #reserved = _data[12:32]
+ _year = ord(_data[1])
+ if _year < 80:
+ # dBase II started at 1980. It is quite unlikely
+ # that actual last update date is before that year.
+ _year += 2000
+ else:
+ _year += 1900
+ ## create header object
+ _obj = cls(None, _hdrLen, _recLen, _cnt, ord(_data[0]),
+ (_year, ord(_data[2]), ord(_data[3])))
+ ## append field definitions
+ # position 0 is for the deletion flag
+ _pos = 1
+ _data = stream.read(1)
+
+ # The field definitions are ended either by \x0D OR a newline
+ # character, so we need to handle both when reading from a stream.
+ # When writing, dbfpy appears to write newlines instead of \x0D.
+ while _data[0] not in ["\x0D", "\n"]:
+ _data += stream.read(31)
+ _fld = fields.lookupFor(_data[11]).fromString(_data, _pos)
+ _obj._addField(_fld)
+ _pos = _fld.end
+ _data = stream.read(1)
+ return _obj
+ fromStream = classmethod(fromStream)
+
+ ## properties
+
+ year = property(lambda self: self.lastUpdate.year)
+ month = property(lambda self: self.lastUpdate.month)
+ day = property(lambda self: self.lastUpdate.day)
+
+ def ignoreErrors(self, value):
+ """Update `ignoreErrors` flag on self and all fields"""
+ self._ignore_errors = value = bool(value)
+ for _field in self.fields:
+ _field.ignoreErrors = value
+ ignoreErrors = property(
+ lambda self: self._ignore_errors,
+ ignoreErrors,
+ doc="""Error processing mode for DBF field value conversion
+
+ if set, failing field value conversion will return
+ ``INVALID_VALUE`` instead of raising conversion error.
+
+ """)
+
+ ## object representation
+
+ def __repr__(self):
+ _rv = """\
+Version (signature): 0x%02x
+ Last update: %s
+ Header length: %d
+ Record length: %d
+ Record count: %d
+ FieldName Type Len Dec
+""" % (self.signature, self.lastUpdate, self.headerLength,
+ self.recordLength, self.recordCount)
+ _rv += "\n".join(
+ ["%10s %4s %3s %3s" % _fld.fieldInfo() for _fld in self.fields]
+ )
+ return _rv
+
+ ## internal methods
+
+ def _addField(self, *defs):
+ """Internal variant of the `addField` method.
+
+ This method doesn't set `self.changed` field to True.
+
+ Return value is a length of the appended records.
+ Note: this method doesn't modify ``recordLength`` and
+ ``headerLength`` fields. Use `addField` instead of this
+ method if you don't exactly know what you're doing.
+
+ """
+ # insure we have dbf.DbfFieldDef instances first (instantiation
+ # from the tuple could raise an error, in such a case I don't
+ # wanna add any of the definitions -- all will be ignored)
+ _defs = []
+ _recordLength = 0
+ for _def in defs:
+ if isinstance(_def, fields.DbfFieldDef):
+ _obj = _def
+ else:
+ (_name, _type, _len, _dec) = (tuple(_def) + (None,) * 4)[:4]
+ _cls = fields.lookupFor(_type)
+ _obj = _cls(_name, _len, _dec,
+ ignoreErrors=self._ignore_errors)
+ _recordLength += _obj.length
+ _defs.append(_obj)
+ # and now extend field definitions and
+ # update record length
+ self.fields += _defs
+ return _recordLength
+
+ ## interface methods
+
+ def addField(self, *defs):
+ """Add field definition to the header.
+
+ Examples:
+ dbfh.addField(
+ ("name", "C", 20),
+ dbf.DbfCharacterFieldDef("surname", 20),
+ dbf.DbfDateFieldDef("birthdate"),
+ ("member", "L"),
+ )
+ dbfh.addField(("price", "N", 5, 2))
+ dbfh.addField(dbf.DbfNumericFieldDef("origprice", 5, 2))
+
+ """
+ _oldLen = self.recordLength
+ self.recordLength += self._addField(*defs)
+ if not _oldLen:
+ self.recordLength += 1
+ # XXX: may be just use:
+ # self.recordeLength += self._addField(*defs) + bool(not _oldLen)
+ # recalculate headerLength
+ self.headerLength = 32 + (32 * len(self.fields)) + 1
+ self.changed = True
+
+ def write(self, stream):
+ """Encode and write header to the stream."""
+ stream.seek(0)
+ stream.write(self.toString())
+ stream.write("".join([_fld.toString() for _fld in self.fields]))
+ stream.write(chr(0x0D)) # cr at end of all hdr data
+ self.changed = False
+
+ def toString(self):
+ """Returned 32 chars length string with encoded header."""
+ return struct.pack("<4BI2H",
+ self.signature,
+ self.year - 1900,
+ self.month,
+ self.day,
+ self.recordCount,
+ self.headerLength,
+ self.recordLength) + "\0" * 20
+
+ def setCurrentDate(self):
+ """Update ``self.lastUpdate`` field with current date value."""
+ self.lastUpdate = datetime.date.today()
+
+ def __getitem__(self, item):
+ """Return a field definition by numeric index or name string"""
+ if isinstance(item, basestring):
+ _name = item.upper()
+ for _field in self.fields:
+ if _field.name == _name:
+ return _field
+ else:
+ raise KeyError(item)
+ else:
+ # item must be field index
+ return self.fields[item]
+
+# vim: et sts=4 sw=4 :
diff --git a/src/tablib/packages/dbfpy/record.py b/src/tablib/packages/dbfpy/record.py
new file mode 100644
index 0000000..97bbfb3
--- /dev/null
+++ b/src/tablib/packages/dbfpy/record.py
@@ -0,0 +1,262 @@
+"""DBF record definition.
+
+"""
+"""History (most recent first):
+11-feb-2007 [als] __repr__: added special case for invalid field values
+10-feb-2007 [als] added .rawFromStream()
+30-oct-2006 [als] fix record length in .fromStream()
+04-jul-2006 [als] added export declaration
+20-dec-2005 [yc] DbfRecord.write() -> DbfRecord._write();
+ added delete() method.
+16-dec-2005 [yc] record definition moved from `dbf`.
+"""
+
+__version__ = "$Revision: 1.7 $"[11:-2]
+__date__ = "$Date: 2007/02/11 09:05:49 $"[7:-2]
+
+__all__ = ["DbfRecord"]
+
+from itertools import izip
+
+import utils
+
+class DbfRecord(object):
+ """DBF record.
+
+ Instances of this class shouldn't be created manualy,
+ use `dbf.Dbf.newRecord` instead.
+
+ Class implements mapping/sequence interface, so
+ fields could be accessed via their names or indexes
+ (names is a preffered way to access fields).
+
+ Hint:
+ Use `store` method to save modified record.
+
+ Examples:
+ Add new record to the database:
+ db = Dbf(filename)
+ rec = db.newRecord()
+ rec["FIELD1"] = value1
+ rec["FIELD2"] = value2
+ rec.store()
+ Or the same, but modify existed
+ (second in this case) record:
+ db = Dbf(filename)
+ rec = db[2]
+ rec["FIELD1"] = value1
+ rec["FIELD2"] = value2
+ rec.store()
+
+ """
+
+ __slots__ = "dbf", "index", "deleted", "fieldData"
+
+ ## creation and initialization
+
+ def __init__(self, dbf, index=None, deleted=False, data=None):
+ """Instance initialiation.
+
+ Arguments:
+ dbf:
+ A `Dbf.Dbf` instance this record belonogs to.
+ index:
+ An integer record index or None. If this value is
+ None, record will be appended to the DBF.
+ deleted:
+ Boolean flag indicating whether this record
+ is a deleted record.
+ data:
+ A sequence or None. This is a data of the fields.
+ If this argument is None, default values will be used.
+
+ """
+ self.dbf = dbf
+ # XXX: I'm not sure ``index`` is necessary
+ self.index = index
+ self.deleted = deleted
+ if data is None:
+ self.fieldData = [_fd.defaultValue for _fd in dbf.header.fields]
+ else:
+ self.fieldData = list(data)
+
+ # XXX: validate self.index before calculating position?
+ position = property(lambda self: self.dbf.header.headerLength + \
+ self.index * self.dbf.header.recordLength)
+
+ def rawFromStream(cls, dbf, index):
+ """Return raw record contents read from the stream.
+
+ Arguments:
+ dbf:
+ A `Dbf.Dbf` instance containing the record.
+ index:
+ Index of the record in the records' container.
+ This argument can't be None in this call.
+
+ Return value is a string containing record data in DBF format.
+
+ """
+ # XXX: may be write smth assuming, that current stream
+ # position is the required one? it could save some
+ # time required to calculate where to seek in the file
+ dbf.stream.seek(dbf.header.headerLength +
+ index * dbf.header.recordLength)
+ return dbf.stream.read(dbf.header.recordLength)
+ rawFromStream = classmethod(rawFromStream)
+
+ def fromStream(cls, dbf, index):
+ """Return a record read from the stream.
+
+ Arguments:
+ dbf:
+ A `Dbf.Dbf` instance new record should belong to.
+ index:
+ Index of the record in the records' container.
+ This argument can't be None in this call.
+
+ Return value is an instance of the current class.
+
+ """
+ return cls.fromString(dbf, cls.rawFromStream(dbf, index), index)
+ fromStream = classmethod(fromStream)
+
+ def fromString(cls, dbf, string, index=None):
+ """Return record read from the string object.
+
+ Arguments:
+ dbf:
+ A `Dbf.Dbf` instance new record should belong to.
+ string:
+ A string new record should be created from.
+ index:
+ Index of the record in the container. If this
+ argument is None, record will be appended.
+
+ Return value is an instance of the current class.
+
+ """
+ return cls(dbf, index, string[0]=="*",
+ [_fd.decodeFromRecord(string) for _fd in dbf.header.fields])
+ fromString = classmethod(fromString)
+
+ ## object representation
+
+ def __repr__(self):
+ _template = "%%%ds: %%s (%%s)" % max([len(_fld)
+ for _fld in self.dbf.fieldNames])
+ _rv = []
+ for _fld in self.dbf.fieldNames:
+ _val = self[_fld]
+ if _val is utils.INVALID_VALUE:
+ _rv.append(_template %
+ (_fld, "None", "value cannot be decoded"))
+ else:
+ _rv.append(_template % (_fld, _val, type(_val)))
+ return "\n".join(_rv)
+
+ ## protected methods
+
+ def _write(self):
+ """Write data to the dbf stream.
+
+ Note:
+ This isn't a public method, it's better to
+ use 'store' instead publically.
+ Be design ``_write`` method should be called
+ only from the `Dbf` instance.
+
+
+ """
+ self._validateIndex(False)
+ self.dbf.stream.seek(self.position)
+ self.dbf.stream.write(self.toString())
+ # FIXME: may be move this write somewhere else?
+ # why we should check this condition for each record?
+ if self.index == len(self.dbf):
+ # this is the last record,
+ # we should write SUB (ASCII 26)
+ self.dbf.stream.write("\x1A")
+
+ ## utility methods
+
+ def _validateIndex(self, allowUndefined=True, checkRange=False):
+ """Valid ``self.index`` value.
+
+ If ``allowUndefined`` argument is True functions does nothing
+ in case of ``self.index`` pointing to None object.
+
+ """
+ if self.index is None:
+ if not allowUndefined:
+ raise ValueError("Index is undefined")
+ elif self.index < 0:
+ raise ValueError("Index can't be negative (%s)" % self.index)
+ elif checkRange and self.index <= self.dbf.header.recordCount:
+ raise ValueError("There are only %d records in the DBF" %
+ self.dbf.header.recordCount)
+
+ ## interface methods
+
+ def store(self):
+ """Store current record in the DBF.
+
+ If ``self.index`` is None, this record will be appended to the
+ records of the DBF this records belongs to; or replaced otherwise.
+
+ """
+ self._validateIndex()
+ if self.index is None:
+ self.index = len(self.dbf)
+ self.dbf.append(self)
+ else:
+ self.dbf[self.index] = self
+
+ def delete(self):
+ """Mark method as deleted."""
+ self.deleted = True
+
+ def toString(self):
+ """Return string packed record values."""
+ return "".join([" *"[self.deleted]] + [
+ _def.encodeValue(_dat)
+ for (_def, _dat) in izip(self.dbf.header.fields, self.fieldData)
+ ])
+
+ def asList(self):
+ """Return a flat list of fields.
+
+ Note:
+ Change of the list's values won't change
+ real values stored in this object.
+
+ """
+ return self.fieldData[:]
+
+ def asDict(self):
+ """Return a dictionary of fields.
+
+ Note:
+ Change of the dicts's values won't change
+ real values stored in this object.
+
+ """
+ return dict([_i for _i in izip(self.dbf.fieldNames, self.fieldData)])
+
+ def __getitem__(self, key):
+ """Return value by field name or field index."""
+ if isinstance(key, (long, int)):
+ # integer index of the field
+ return self.fieldData[key]
+ # assuming string field name
+ return self.fieldData[self.dbf.indexOfFieldName(key)]
+
+ def __setitem__(self, key, value):
+ """Set field value by integer index of the field or string name."""
+ if isinstance(key, (int, long)):
+ # integer index of the field
+ return self.fieldData[key]
+ # assuming string field name
+ self.fieldData[self.dbf.indexOfFieldName(key)] = value
+
+# vim: et sts=4 sw=4 :
diff --git a/src/tablib/packages/dbfpy/utils.py b/src/tablib/packages/dbfpy/utils.py
new file mode 100644
index 0000000..cef8aa5
--- /dev/null
+++ b/src/tablib/packages/dbfpy/utils.py
@@ -0,0 +1,170 @@
+"""String utilities.
+
+TODO:
+ - allow strings in getDateTime routine;
+"""
+"""History (most recent first):
+11-feb-2007 [als] added INVALID_VALUE
+10-feb-2007 [als] allow date strings padded with spaces instead of zeroes
+20-dec-2005 [yc] handle long objects in getDate/getDateTime
+16-dec-2005 [yc] created from ``strutil`` module.
+"""
+
+__version__ = "$Revision: 1.4 $"[11:-2]
+__date__ = "$Date: 2007/02/11 08:57:17 $"[7:-2]
+
+import datetime
+import time
+
+
+def unzfill(str):
+ """Return a string without ASCII NULs.
+
+ This function searchers for the first NUL (ASCII 0) occurance
+ and truncates string till that position.
+
+ """
+ try:
+ return str[:str.index('\0')]
+ except ValueError:
+ return str
+
+
+def getDate(date=None):
+ """Return `datetime.date` instance.
+
+ Type of the ``date`` argument could be one of the following:
+ None:
+ use current date value;
+ datetime.date:
+ this value will be returned;
+ datetime.datetime:
+ the result of the date.date() will be returned;
+ string:
+ assuming "%Y%m%d" or "%y%m%dd" format;
+ number:
+ assuming it's a timestamp (returned for example
+ by the time.time() call;
+ sequence:
+ assuming (year, month, day, ...) sequence;
+
+ Additionaly, if ``date`` has callable ``ticks`` attribute,
+ it will be used and result of the called would be treated
+ as a timestamp value.
+
+ """
+ if date is None:
+ # use current value
+ return datetime.date.today()
+ if isinstance(date, datetime.date):
+ return date
+ if isinstance(date, datetime.datetime):
+ return date.date()
+ if isinstance(date, (int, long, float)):
+ # date is a timestamp
+ return datetime.date.fromtimestamp(date)
+ if isinstance(date, basestring):
+ date = date.replace(" ", "0")
+ if len(date) == 6:
+ # yymmdd
+ return datetime.date(*time.strptime(date, "%y%m%d")[:3])
+ # yyyymmdd
+ return datetime.date(*time.strptime(date, "%Y%m%d")[:3])
+ if hasattr(date, "__getitem__"):
+ # a sequence (assuming date/time tuple)
+ return datetime.date(*date[:3])
+ return datetime.date.fromtimestamp(date.ticks())
+
+
+def getDateTime(value=None):
+ """Return `datetime.datetime` instance.
+
+ Type of the ``value`` argument could be one of the following:
+ None:
+ use current date value;
+ datetime.date:
+ result will be converted to the `datetime.datetime` instance
+ using midnight;
+ datetime.datetime:
+ ``value`` will be returned as is;
+ string:
+ *** CURRENTLY NOT SUPPORTED ***;
+ number:
+ assuming it's a timestamp (returned for example
+ by the time.time() call;
+ sequence:
+ assuming (year, month, day, ...) sequence;
+
+ Additionaly, if ``value`` has callable ``ticks`` attribute,
+ it will be used and result of the called would be treated
+ as a timestamp value.
+
+ """
+ if value is None:
+ # use current value
+ return datetime.datetime.today()
+ if isinstance(value, datetime.datetime):
+ return value
+ if isinstance(value, datetime.date):
+ return datetime.datetime.fromordinal(value.toordinal())
+ if isinstance(value, (int, long, float)):
+ # value is a timestamp
+ return datetime.datetime.fromtimestamp(value)
+ if isinstance(value, basestring):
+ raise NotImplementedError("Strings aren't currently implemented")
+ if hasattr(value, "__getitem__"):
+ # a sequence (assuming date/time tuple)
+ return datetime.datetime(*tuple(value)[:6])
+ return datetime.datetime.fromtimestamp(value.ticks())
+
+
+class classproperty(property):
+ """Works in the same way as a ``property``, but for the classes."""
+
+ def __get__(self, obj, cls):
+ return self.fget(cls)
+
+
+class _InvalidValue(object):
+
+ """Value returned from DBF records when field validation fails
+
+ The value is not equal to anything except for itself
+ and equal to all empty values: None, 0, empty string etc.
+ In other words, invalid value is equal to None and not equal
+ to None at the same time.
+
+ This value yields zero upon explicit conversion to a number type,
+ empty string for string types, and False for boolean.
+
+ """
+
+ def __eq__(self, other):
+ return not other
+
+ def __ne__(self, other):
+ return not (other is self)
+
+ def __nonzero__(self):
+ return False
+
+ def __int__(self):
+ return 0
+ __long__ = __int__
+
+ def __float__(self):
+ return 0.0
+
+ def __str__(self):
+ return ""
+
+ def __unicode__(self):
+ return u""
+
+ def __repr__(self):
+ return "<INVALID>"
+
+# invalid value is a constant singleton
+INVALID_VALUE = _InvalidValue()
+
+# vim: set et sts=4 sw=4 :
diff --git a/src/tablib/packages/dbfpy3/__init__.py b/src/tablib/packages/dbfpy3/__init__.py
new file mode 100644
index 0000000..e69de29
--- /dev/null
+++ b/src/tablib/packages/dbfpy3/__init__.py
diff --git a/src/tablib/packages/dbfpy3/dbf.py b/src/tablib/packages/dbfpy3/dbf.py
new file mode 100644
index 0000000..218353e
--- /dev/null
+++ b/src/tablib/packages/dbfpy3/dbf.py
@@ -0,0 +1,297 @@
+#! /usr/bin/env python
+"""DBF accessing helpers.
+
+FIXME: more documentation needed
+
+Examples:
+
+ Create new table, setup structure, add records:
+
+ dbf = Dbf(filename, new=True)
+ dbf.addField(
+ ("NAME", "C", 15),
+ ("SURNAME", "C", 25),
+ ("INITIALS", "C", 10),
+ ("BIRTHDATE", "D"),
+ )
+ for (n, s, i, b) in (
+ ("John", "Miller", "YC", (1980, 10, 11)),
+ ("Andy", "Larkin", "", (1980, 4, 11)),
+ ):
+ rec = dbf.newRecord()
+ rec["NAME"] = n
+ rec["SURNAME"] = s
+ rec["INITIALS"] = i
+ rec["BIRTHDATE"] = b
+ rec.store()
+ dbf.close()
+
+ Open existed dbf, read some data:
+
+ dbf = Dbf(filename, True)
+ for rec in dbf:
+ for fldName in dbf.fieldNames:
+ print('%s:\t %s (%s)' % (fldName, rec[fldName],
+ type(rec[fldName])))
+ dbf.close()
+
+"""
+"""History (most recent first):
+11-feb-2007 [als] export INVALID_VALUE;
+ Dbf: added .ignoreErrors, .INVALID_VALUE
+04-jul-2006 [als] added export declaration
+20-dec-2005 [yc] removed fromStream and newDbf methods:
+ use argument of __init__ call must be used instead;
+ added class fields pointing to the header and
+ record classes.
+17-dec-2005 [yc] split to several modules; reimplemented
+13-dec-2005 [yc] adapted to the changes of the `strutil` module.
+13-sep-2002 [als] support FoxPro Timestamp datatype
+15-nov-1999 [jjk] documentation updates, add demo
+24-aug-1998 [jjk] add some encodeValue methods (not tested), other tweaks
+08-jun-1998 [jjk] fix problems, add more features
+20-feb-1998 [jjk] fix problems, add more features
+19-feb-1998 [jjk] add create/write capabilities
+18-feb-1998 [jjk] from dbfload.py
+"""
+
+__version__ = "$Revision: 1.7 $"[11:-2]
+__date__ = "$Date: 2007/02/11 09:23:13 $"[7:-2]
+__author__ = "Jeff Kunce <kuncej@mail.conservation.state.mo.us>"
+
+__all__ = ["Dbf"]
+
+from . import header
+from . import record
+from .utils import INVALID_VALUE
+
+
+class Dbf(object):
+ """DBF accessor.
+
+ FIXME:
+ docs and examples needed (dont' forget to tell
+ about problems adding new fields on the fly)
+
+ Implementation notes:
+ ``_new`` field is used to indicate whether this is
+ a new data table. `addField` could be used only for
+ the new tables! If at least one record was appended
+ to the table it's structure couldn't be changed.
+
+ """
+
+ __slots__ = ("name", "header", "stream",
+ "_changed", "_new", "_ignore_errors")
+
+ HeaderClass = header.DbfHeader
+ RecordClass = record.DbfRecord
+ INVALID_VALUE = INVALID_VALUE
+
+ # initialization and creation helpers
+
+ def __init__(self, f, readOnly=False, new=False, ignoreErrors=False):
+ """Initialize instance.
+
+ Arguments:
+ f:
+ Filename or file-like object.
+ new:
+ True if new data table must be created. Assume
+ data table exists if this argument is False.
+ readOnly:
+ if ``f`` argument is a string file will
+ be opend in read-only mode; in other cases
+ this argument is ignored. This argument is ignored
+ even if ``new`` argument is True.
+ headerObj:
+ `header.DbfHeader` instance or None. If this argument
+ is None, new empty header will be used with the
+ all fields set by default.
+ ignoreErrors:
+ if set, failing field value conversion will return
+ ``INVALID_VALUE`` instead of raising conversion error.
+
+ """
+ if isinstance(f, str):
+ # a filename
+ self.name = f
+ if new:
+ # new table (table file must be
+ # created or opened and truncated)
+ self.stream = open(f, "w+b")
+ else:
+ # tabe file must exist
+ self.stream = open(f, ("r+b", "rb")[bool(readOnly)])
+ else:
+ # a stream
+ self.name = getattr(f, "name", "")
+ self.stream = f
+ if new:
+ # if this is a new table, header will be empty
+ self.header = self.HeaderClass()
+ else:
+ # or instantiated using stream
+ self.header = self.HeaderClass.fromStream(self.stream)
+ self.ignoreErrors = ignoreErrors
+ self._new = bool(new)
+ self._changed = False
+
+ # properties
+
+ closed = property(lambda self: self.stream.closed)
+ recordCount = property(lambda self: self.header.recordCount)
+ fieldNames = property(
+ lambda self: [_fld.name for _fld in self.header.fields])
+ fieldDefs = property(lambda self: self.header.fields)
+ changed = property(lambda self: self._changed or self.header.changed)
+
+ def ignoreErrors(self, value):
+ """Update `ignoreErrors` flag on the header object and self"""
+ self.header.ignoreErrors = self._ignore_errors = bool(value)
+
+ ignoreErrors = property(
+ lambda self: self._ignore_errors,
+ ignoreErrors,
+ doc="""Error processing mode for DBF field value conversion
+
+ if set, failing field value conversion will return
+ ``INVALID_VALUE`` instead of raising conversion error.
+
+ """)
+
+ # protected methods
+
+ def _fixIndex(self, index):
+ """Return fixed index.
+
+ This method fails if index isn't a numeric object
+ (long or int). Or index isn't in a valid range
+ (less or equal to the number of records in the db).
+
+ If ``index`` is a negative number, it will be
+ treated as a negative indexes for list objects.
+
+ Return:
+ Return value is numeric object maning valid index.
+
+ """
+ if not isinstance(index, int):
+ raise TypeError("Index must be a numeric object")
+ if index < 0:
+ # index from the right side
+ # fix it to the left-side index
+ index += len(self) + 1
+ if index >= len(self):
+ raise IndexError("Record index out of range")
+ return index
+
+ # iterface methods
+
+ def close(self):
+ self.flush()
+ self.stream.close()
+
+ def flush(self):
+ """Flush data to the associated stream."""
+ if self.changed:
+ self.header.setCurrentDate()
+ self.header.write(self.stream)
+ self.stream.flush()
+ self._changed = False
+
+ def indexOfFieldName(self, name):
+ """Index of field named ``name``."""
+ # FIXME: move this to header class
+ names = [f.name for f in self.header.fields]
+ return names.index(name.upper())
+
+ def newRecord(self):
+ """Return new record, which belong to this table."""
+ return self.RecordClass(self)
+
+ def append(self, record):
+ """Append ``record`` to the database."""
+ record.index = self.header.recordCount
+ record._write()
+ self.header.recordCount += 1
+ self._changed = True
+ self._new = False
+
+ def addField(self, *defs):
+ """Add field definitions.
+
+ For more information see `header.DbfHeader.addField`.
+
+ """
+ if self._new:
+ self.header.addField(*defs)
+ else:
+ raise TypeError("At least one record was added, "
+ "structure can't be changed")
+
+ # 'magic' methods (representation and sequence interface)
+
+ def __repr__(self):
+ return "Dbf stream '%s'\n" % self.stream + repr(self.header)
+
+ def __len__(self):
+ """Return number of records."""
+ return self.recordCount
+
+ def __getitem__(self, index):
+ """Return `DbfRecord` instance."""
+ return self.RecordClass.fromStream(self, self._fixIndex(index))
+
+ def __setitem__(self, index, record):
+ """Write `DbfRecord` instance to the stream."""
+ record.index = self._fixIndex(index)
+ record._write()
+ self._changed = True
+ self._new = False
+
+ # def __del__(self):
+ # """Flush stream upon deletion of the object."""
+ # self.flush()
+
+
+def demo_read(filename):
+ _dbf = Dbf(filename, True)
+ for _rec in _dbf:
+ print()
+ print(repr(_rec))
+ _dbf.close()
+
+
+def demo_create(filename):
+ _dbf = Dbf(filename, new=True)
+ _dbf.addField(
+ ("NAME", "C", 15),
+ ("SURNAME", "C", 25),
+ ("INITIALS", "C", 10),
+ ("BIRTHDATE", "D"),
+ )
+ for (_n, _s, _i, _b) in (
+ ("John", "Miller", "YC", (1981, 1, 2)),
+ ("Andy", "Larkin", "AL", (1982, 3, 4)),
+ ("Bill", "Clinth", "", (1983, 5, 6)),
+ ("Bobb", "McNail", "", (1984, 7, 8)),
+ ):
+ _rec = _dbf.newRecord()
+ _rec["NAME"] = _n
+ _rec["SURNAME"] = _s
+ _rec["INITIALS"] = _i
+ _rec["BIRTHDATE"] = _b
+ _rec.store()
+ print(repr(_dbf))
+ _dbf.close()
+
+
+if __name__ == '__main__':
+ import sys
+
+ _name = len(sys.argv) > 1 and sys.argv[1] or "county.dbf"
+ demo_create(_name)
+ demo_read(_name)
+
+# vim: set et sw=4 sts=4 :
diff --git a/src/tablib/packages/dbfpy3/dbfnew.py b/src/tablib/packages/dbfpy3/dbfnew.py
new file mode 100644
index 0000000..8fab275
--- /dev/null
+++ b/src/tablib/packages/dbfpy3/dbfnew.py
@@ -0,0 +1,183 @@
+#!/usr/bin/python
+""".DBF creation helpers.
+
+Note: this is a legacy interface. New code should use Dbf class
+ for table creation (see examples in dbf.py)
+
+TODO:
+ - handle Memo fields.
+ - check length of the fields accoring to the
+ `http://www.clicketyclick.dk/databases/xbase/format/data_types.html`
+
+"""
+"""History (most recent first)
+04-jul-2006 [als] added export declaration;
+ updated for dbfpy 2.0
+15-dec-2005 [yc] define dbf_new.__slots__
+14-dec-2005 [yc] added vim modeline; retab'd; added doc-strings;
+ dbf_new now is a new class (inherited from object)
+??-jun-2000 [--] added by Hans Fiby
+"""
+
+__version__ = "$Revision: 1.4 $"[11:-2]
+__date__ = "$Date: 2006/07/04 08:18:18 $"[7:-2]
+
+__all__ = ["dbf_new"]
+
+from .dbf import *
+from .fields import *
+from .header import *
+from .record import *
+
+
+class _FieldDefinition(object):
+ """Field definition.
+
+ This is a simple structure, which contains ``name``, ``type``,
+ ``len``, ``dec`` and ``cls`` fields.
+
+ Objects also implement get/setitem magic functions, so fields
+ could be accessed via sequence iterface, where 'name' has
+ index 0, 'type' index 1, 'len' index 2, 'dec' index 3 and
+ 'cls' could be located at index 4.
+
+ """
+
+ __slots__ = "name", "type", "len", "dec", "cls"
+
+ # WARNING: be attentive - dictionaries are mutable!
+ FLD_TYPES = {
+ # type: (cls, len)
+ "C": (DbfCharacterFieldDef, None),
+ "N": (DbfNumericFieldDef, None),
+ "L": (DbfLogicalFieldDef, 1),
+ # FIXME: support memos
+ # "M": (DbfMemoFieldDef),
+ "D": (DbfDateFieldDef, 8),
+ # FIXME: I'm not sure length should be 14 characters!
+ # but temporary I use it, cuz date is 8 characters
+ # and time 6 (hhmmss)
+ "T": (DbfDateTimeFieldDef, 14),
+ }
+
+ def __init__(self, name, type, len=None, dec=0):
+ _cls, _len = self.FLD_TYPES[type]
+ if _len is None:
+ if len is None:
+ raise ValueError("Field length must be defined")
+ _len = len
+ self.name = name
+ self.type = type
+ self.len = _len
+ self.dec = dec
+ self.cls = _cls
+
+ def getDbfField(self):
+ "Return `DbfFieldDef` instance from the current definition."
+ return self.cls(self.name, self.len, self.dec)
+
+ def appendToHeader(self, dbfh):
+ """Create a `DbfFieldDef` instance and append it to the dbf header.
+
+ Arguments:
+ dbfh: `DbfHeader` instance.
+
+ """
+ _dbff = self.getDbfField()
+ dbfh.addField(_dbff)
+
+
+class dbf_new(object):
+ """New .DBF creation helper.
+
+ Example Usage:
+
+ dbfn = dbf_new()
+ dbfn.add_field("name",'C',80)
+ dbfn.add_field("price",'N',10,2)
+ dbfn.add_field("date",'D',8)
+ dbfn.write("tst.dbf")
+
+ Note:
+ This module cannot handle Memo-fields,
+ they are special.
+
+ """
+
+ __slots__ = ("fields",)
+
+ FieldDefinitionClass = _FieldDefinition
+
+ def __init__(self):
+ self.fields = []
+
+ def add_field(self, name, typ, len, dec=0):
+ """Add field definition.
+
+ Arguments:
+ name:
+ field name (str object). field name must not
+ contain ASCII NULs and it's length shouldn't
+ exceed 10 characters.
+ typ:
+ type of the field. this must be a single character
+ from the "CNLMDT" set meaning character, numeric,
+ logical, memo, date and date/time respectively.
+ len:
+ length of the field. this argument is used only for
+ the character and numeric fields. all other fields
+ have fixed length.
+ FIXME: use None as a default for this argument?
+ dec:
+ decimal precision. used only for the numric fields.
+
+ """
+ self.fields.append(self.FieldDefinitionClass(name, typ, len, dec))
+
+ def write(self, filename):
+ """Create empty .DBF file using current structure."""
+ _dbfh = DbfHeader()
+ _dbfh.setCurrentDate()
+ for _fldDef in self.fields:
+ _fldDef.appendToHeader(_dbfh)
+
+ _dbfStream = open(filename, "wb")
+ _dbfh.write(_dbfStream)
+ _dbfStream.close()
+
+
+if __name__ == '__main__':
+ # create a new DBF-File
+ dbfn = dbf_new()
+ dbfn.add_field("name", 'C', 80)
+ dbfn.add_field("price", 'N', 10, 2)
+ dbfn.add_field("date", 'D', 8)
+ dbfn.write("tst.dbf")
+ # test new dbf
+ print("*** created tst.dbf: ***")
+ dbft = Dbf('tst.dbf', readOnly=0)
+ print(repr(dbft))
+ # add a record
+ rec = DbfRecord(dbft)
+ rec['name'] = 'something'
+ rec['price'] = 10.5
+ rec['date'] = (2000, 1, 12)
+ rec.store()
+ # add another record
+ rec = DbfRecord(dbft)
+ rec['name'] = 'foo and bar'
+ rec['price'] = 12234
+ rec['date'] = (1992, 7, 15)
+ rec.store()
+
+ # show the records
+ print("*** inserted 2 records into tst.dbf: ***")
+ print(repr(dbft))
+ for i1 in range(len(dbft)):
+ rec = dbft[i1]
+ for fldName in dbft.fieldNames:
+ print('%s:\t %s' % (fldName, rec[fldName]))
+ print()
+ dbft.close()
+
+# vim: set et sts=4 sw=4 :
diff --git a/src/tablib/packages/dbfpy3/fields.py b/src/tablib/packages/dbfpy3/fields.py
new file mode 100644
index 0000000..87916fe
--- /dev/null
+++ b/src/tablib/packages/dbfpy3/fields.py
@@ -0,0 +1,466 @@
+"""DBF fields definitions.
+
+TODO:
+ - make memos work
+"""
+"""History (most recent first):
+26-may-2009 [als] DbfNumericFieldDef.decodeValue: strip zero bytes
+05-feb-2009 [als] DbfDateFieldDef.encodeValue: empty arg produces empty date
+16-sep-2008 [als] DbfNumericFieldDef decoding looks for decimal point
+ in the value to select float or integer return type
+13-mar-2008 [als] check field name length in constructor
+11-feb-2007 [als] handle value conversion errors
+10-feb-2007 [als] DbfFieldDef: added .rawFromRecord()
+01-dec-2006 [als] Timestamp columns use None for empty values
+31-oct-2006 [als] support field types 'F' (float), 'I' (integer)
+ and 'Y' (currency);
+ automate export and registration of field classes
+04-jul-2006 [als] added export declaration
+10-mar-2006 [als] decode empty values for Date and Logical fields;
+ show field name in errors
+10-mar-2006 [als] fix Numeric value decoding: according to spec,
+ value always is string representation of the number;
+ ensure that encoded Numeric value fits into the field
+20-dec-2005 [yc] use field names in upper case
+15-dec-2005 [yc] field definitions moved from `dbf`.
+"""
+
+__version__ = "$Revision: 1.14 $"[11:-2]
+__date__ = "$Date: 2009/05/26 05:16:51 $"[7:-2]
+
+__all__ = ["lookupFor",] # field classes added at the end of the module
+
+import datetime
+import struct
+import sys
+
+from . import utils
+
+## abstract definitions
+
+class DbfFieldDef(object):
+ """Abstract field definition.
+
+ Child classes must override ``type`` class attribute to provide datatype
+ infromation of the field definition. For more info about types visit
+ `http://www.clicketyclick.dk/databases/xbase/format/data_types.html`
+
+ Also child classes must override ``defaultValue`` field to provide
+ default value for the field value.
+
+ If child class has fixed length ``length`` class attribute must be
+ overriden and set to the valid value. None value means, that field
+ isn't of fixed length.
+
+ Note: ``name`` field must not be changed after instantiation.
+
+ """
+
+ __slots__ = ("name", "decimalCount",
+ "start", "end", "ignoreErrors")
+
+ # length of the field, None in case of variable-length field,
+ # or a number if this field is a fixed-length field
+ length = None
+
+ # field type. for more information about fields types visit
+ # `http://www.clicketyclick.dk/databases/xbase/format/data_types.html`
+ # must be overriden in child classes
+ typeCode = None
+
+ # default value for the field. this field must be
+ # overriden in child classes
+ defaultValue = None
+
+ def __init__(self, name, length=None, decimalCount=None,
+ start=None, stop=None, ignoreErrors=False,
+ ):
+ """Initialize instance."""
+ assert self.typeCode is not None, "Type code must be overriden"
+ assert self.defaultValue is not None, "Default value must be overriden"
+ ## fix arguments
+ if len(name) >10:
+ raise ValueError("Field name \"%s\" is too long" % name)
+ name = str(name).upper()
+ if self.__class__.length is None:
+ if length is None:
+ raise ValueError("[%s] Length isn't specified" % name)
+ length = int(length)
+ if length <= 0:
+ raise ValueError("[%s] Length must be a positive integer"
+ % name)
+ else:
+ length = self.length
+ if decimalCount is None:
+ decimalCount = 0
+ ## set fields
+ self.name = name
+ # FIXME: validate length according to the specification at
+ # http://www.clicketyclick.dk/databases/xbase/format/data_types.html
+ self.length = length
+ self.decimalCount = decimalCount
+ self.ignoreErrors = ignoreErrors
+ self.start = start
+ self.end = stop
+
+ def __cmp__(self, other):
+ return cmp(self.name, str(other).upper())
+
+ def __hash__(self):
+ return hash(self.name)
+
+ def fromString(cls, string, start, ignoreErrors=False):
+ """Decode dbf field definition from the string data.
+
+ Arguments:
+ string:
+ a string, dbf definition is decoded from. length of
+ the string must be 32 bytes.
+ start:
+ position in the database file.
+ ignoreErrors:
+ initial error processing mode for the new field (boolean)
+
+ """
+ assert len(string) == 32
+ _length = string[16]
+ return cls(utils.unzfill(string)[:11].decode('utf-8'), _length,
+ string[17], start, start + _length, ignoreErrors=ignoreErrors)
+ fromString = classmethod(fromString)
+
+ def toString(self):
+ """Return encoded field definition.
+
+ Return:
+ Return value is a string object containing encoded
+ definition of this field.
+
+ """
+ if sys.version_info < (2, 4):
+ # earlier versions did not support padding character
+ _name = self.name[:11] + "\0" * (11 - len(self.name))
+ else:
+ _name = self.name.ljust(11, '\0')
+ return (
+ _name +
+ self.typeCode +
+ #data address
+ chr(0) * 4 +
+ chr(self.length) +
+ chr(self.decimalCount) +
+ chr(0) * 14
+ )
+
+ def __repr__(self):
+ return "%-10s %1s %3d %3d" % self.fieldInfo()
+
+ def fieldInfo(self):
+ """Return field information.
+
+ Return:
+ Return value is a (name, type, length, decimals) tuple.
+
+ """
+ return (self.name, self.typeCode, self.length, self.decimalCount)
+
+ def rawFromRecord(self, record):
+ """Return a "raw" field value from the record string."""
+ return record[self.start:self.end]
+
+ def decodeFromRecord(self, record):
+ """Return decoded field value from the record string."""
+ try:
+ return self.decodeValue(self.rawFromRecord(record))
+ except:
+ if self.ignoreErrors:
+ return utils.INVALID_VALUE
+ else:
+ raise
+
+ def decodeValue(self, value):
+ """Return decoded value from string value.
+
+ This method shouldn't be used publicly. It's called from the
+ `decodeFromRecord` method.
+
+ This is an abstract method and it must be overridden in child classes.
+ """
+ raise NotImplementedError
+
+ def encodeValue(self, value):
+ """Return str object containing encoded field value.
+
+ This is an abstract method and it must be overriden in child classes.
+ """
+ raise NotImplementedError
+
+## real classes
+
+class DbfCharacterFieldDef(DbfFieldDef):
+ """Definition of the character field."""
+
+ typeCode = "C"
+ defaultValue = b''
+
+ def decodeValue(self, value):
+ """Return string object.
+
+ Return value is a ``value`` argument with stripped right spaces.
+
+ """
+ return value.rstrip(b' ').decode('utf-8')
+
+ def encodeValue(self, value):
+ """Return raw data string encoded from a ``value``."""
+ return str(value)[:self.length].ljust(self.length)
+
+
+class DbfNumericFieldDef(DbfFieldDef):
+ """Definition of the numeric field."""
+
+ typeCode = "N"
+ # XXX: now I'm not sure it was a good idea to make a class field
+ # `defaultValue` instead of a generic method as it was implemented
+ # previously -- it's ok with all types except number, cuz
+ # if self.decimalCount is 0, we should return 0 and 0.0 otherwise.
+ defaultValue = 0
+
+ def decodeValue(self, value):
+ """Return a number decoded from ``value``.
+
+ If decimals is zero, value will be decoded as an integer;
+ or as a float otherwise.
+
+ Return:
+ Return value is a int (long) or float instance.
+
+ """
+ value = value.strip(b' \0')
+ if b'.' in value:
+ # a float (has decimal separator)
+ return float(value)
+ elif value:
+ # must be an integer
+ return int(value)
+ else:
+ return 0
+
+ def encodeValue(self, value):
+ """Return string containing encoded ``value``."""
+ _rv = ("%*.*f" % (self.length, self.decimalCount, value))
+ if len(_rv) > self.length:
+ _ppos = _rv.find(".")
+ if 0 <= _ppos <= self.length:
+ _rv = _rv[:self.length]
+ else:
+ raise ValueError("[%s] Numeric overflow: %s (field width: %i)"
+ % (self.name, _rv, self.length))
+ return _rv
+
+class DbfFloatFieldDef(DbfNumericFieldDef):
+ """Definition of the float field - same as numeric."""
+
+ typeCode = "F"
+
+class DbfIntegerFieldDef(DbfFieldDef):
+ """Definition of the integer field."""
+
+ typeCode = "I"
+ length = 4
+ defaultValue = 0
+
+ def decodeValue(self, value):
+ """Return an integer number decoded from ``value``."""
+ return struct.unpack("<i", value)[0]
+
+ def encodeValue(self, value):
+ """Return string containing encoded ``value``."""
+ return struct.pack("<i", int(value))
+
+class DbfCurrencyFieldDef(DbfFieldDef):
+ """Definition of the currency field."""
+
+ typeCode = "Y"
+ length = 8
+ defaultValue = 0.0
+
+ def decodeValue(self, value):
+ """Return float number decoded from ``value``."""
+ return struct.unpack("<q", value)[0] / 10000.
+
+ def encodeValue(self, value):
+ """Return string containing encoded ``value``."""
+ return struct.pack("<q", round(value * 10000))
+
+class DbfLogicalFieldDef(DbfFieldDef):
+ """Definition of the logical field."""
+
+ typeCode = "L"
+ defaultValue = -1
+ length = 1
+
+ def decodeValue(self, value):
+ """Return True, False or -1 decoded from ``value``."""
+ # Note: value always is 1-char string
+ if value == "?":
+ return -1
+ if value in "NnFf ":
+ return False
+ if value in "YyTt":
+ return True
+ raise ValueError("[%s] Invalid logical value %r" % (self.name, value))
+
+ def encodeValue(self, value):
+ """Return a character from the "TF?" set.
+
+ Return:
+ Return value is "T" if ``value`` is True
+ "?" if value is -1 or False otherwise.
+
+ """
+ if value is True:
+ return "T"
+ if value == -1:
+ return "?"
+ return "F"
+
+
+class DbfMemoFieldDef(DbfFieldDef):
+ """Definition of the memo field.
+
+ Note: memos aren't currenly completely supported.
+
+ """
+
+ typeCode = "M"
+ defaultValue = " " * 10
+ length = 10
+
+ def decodeValue(self, value):
+ """Return int .dbt block number decoded from the string object."""
+ #return int(value)
+ raise NotImplementedError
+
+ def encodeValue(self, value):
+ """Return raw data string encoded from a ``value``.
+
+ Note: this is an internal method.
+
+ """
+ #return str(value)[:self.length].ljust(self.length)
+ raise NotImplementedError
+
+
+class DbfDateFieldDef(DbfFieldDef):
+ """Definition of the date field."""
+
+ typeCode = "D"
+ defaultValue = utils.classproperty(lambda cls: datetime.date.today())
+ # "yyyymmdd" gives us 8 characters
+ length = 8
+
+ def decodeValue(self, value):
+ """Return a ``datetime.date`` instance decoded from ``value``."""
+ if value.strip():
+ return utils.getDate(value)
+ else:
+ return None
+
+ def encodeValue(self, value):
+ """Return a string-encoded value.
+
+ ``value`` argument should be a value suitable for the
+ `utils.getDate` call.
+
+ Return:
+ Return value is a string in format "yyyymmdd".
+
+ """
+ if value:
+ return utils.getDate(value).strftime("%Y%m%d")
+ else:
+ return " " * self.length
+
+
+class DbfDateTimeFieldDef(DbfFieldDef):
+ """Definition of the timestamp field."""
+
+ # a difference between JDN (Julian Day Number)
+ # and GDN (Gregorian Day Number). note, that GDN < JDN
+ JDN_GDN_DIFF = 1721425
+ typeCode = "T"
+ defaultValue = utils.classproperty(lambda cls: datetime.datetime.now())
+ # two 32-bits integers representing JDN and amount of
+ # milliseconds respectively gives us 8 bytes.
+ # note, that values must be encoded in LE byteorder.
+ length = 8
+
+ def decodeValue(self, value):
+ """Return a `datetime.datetime` instance."""
+ assert len(value) == self.length
+ # LE byteorder
+ _jdn, _msecs = struct.unpack("<2I", value)
+ if _jdn >= 1:
+ _rv = datetime.datetime.fromordinal(_jdn - self.JDN_GDN_DIFF)
+ _rv += datetime.timedelta(0, _msecs / 1000.0)
+ else:
+ # empty date
+ _rv = None
+ return _rv
+
+ def encodeValue(self, value):
+ """Return a string-encoded ``value``."""
+ if value:
+ value = utils.getDateTime(value)
+ # LE byteorder
+ _rv = struct.pack("<2I", value.toordinal() + self.JDN_GDN_DIFF,
+ (value.hour * 3600 + value.minute * 60 + value.second) * 1000)
+ else:
+ _rv = "\0" * self.length
+ assert len(_rv) == self.length
+ return _rv
+
+
+_fieldsRegistry = {}
+
+def registerField(fieldCls):
+ """Register field definition class.
+
+ ``fieldCls`` should be subclass of the `DbfFieldDef`.
+
+ Use `lookupFor` to retrieve field definition class
+ by the type code.
+
+ """
+ assert fieldCls.typeCode is not None, "Type code isn't defined"
+ # XXX: use fieldCls.typeCode.upper()? in case of any decign
+ # don't forget to look to the same comment in ``lookupFor`` method
+ _fieldsRegistry[fieldCls.typeCode] = fieldCls
+
+
+def lookupFor(typeCode):
+ """Return field definition class for the given type code.
+
+ ``typeCode`` must be a single character. That type should be
+ previously registered.
+
+ Use `registerField` to register new field class.
+
+ Return:
+ Return value is a subclass of the `DbfFieldDef`.
+
+ """
+ # XXX: use typeCode.upper()? in case of any decign don't
+ # forget to look to the same comment in ``registerField``
+ return _fieldsRegistry[chr(typeCode)]
+
+## register generic types
+
+for (_name, _val) in list(globals().items()):
+ if isinstance(_val, type) and issubclass(_val, DbfFieldDef) \
+ and (_name != "DbfFieldDef"):
+ __all__.append(_name)
+ registerField(_val)
+del _name, _val
+
+# vim: et sts=4 sw=4 :
diff --git a/src/tablib/packages/dbfpy3/header.py b/src/tablib/packages/dbfpy3/header.py
new file mode 100644
index 0000000..6c0dc4f
--- /dev/null
+++ b/src/tablib/packages/dbfpy3/header.py
@@ -0,0 +1,273 @@
+"""DBF header definition.
+
+TODO:
+ - handle encoding of the character fields
+ (encoding information stored in the DBF header)
+
+"""
+"""History (most recent first):
+16-sep-2010 [als] fromStream: fix century of the last update field
+11-feb-2007 [als] added .ignoreErrors
+10-feb-2007 [als] added __getitem__: return field definitions
+ by field name or field number (zero-based)
+04-jul-2006 [als] added export declaration
+15-dec-2005 [yc] created
+"""
+
+__version__ = "$Revision: 1.6 $"[11:-2]
+__date__ = "$Date: 2010/09/16 05:06:39 $"[7:-2]
+
+__all__ = ["DbfHeader"]
+
+import io
+import datetime
+import struct
+import time
+import sys
+
+from . import fields
+from .utils import getDate
+
+
+class DbfHeader(object):
+ """Dbf header definition.
+
+ For more information about dbf header format visit
+ `http://www.clicketyclick.dk/databases/xbase/format/dbf.html#DBF_STRUCT`
+
+ Examples:
+ Create an empty dbf header and add some field definitions:
+ dbfh = DbfHeader()
+ dbfh.addField(("name", "C", 10))
+ dbfh.addField(("date", "D"))
+ dbfh.addField(DbfNumericFieldDef("price", 5, 2))
+ Create a dbf header with field definitions:
+ dbfh = DbfHeader([
+ ("name", "C", 10),
+ ("date", "D"),
+ DbfNumericFieldDef("price", 5, 2),
+ ])
+
+ """
+
+ __slots__ = ("signature", "fields", "lastUpdate", "recordLength",
+ "recordCount", "headerLength", "changed", "_ignore_errors")
+
+ ## instance construction and initialization methods
+
+ def __init__(self, fields=None, headerLength=0, recordLength=0,
+ recordCount=0, signature=0x03, lastUpdate=None, ignoreErrors=False,
+ ):
+ """Initialize instance.
+
+ Arguments:
+ fields:
+ a list of field definitions;
+ recordLength:
+ size of the records;
+ headerLength:
+ size of the header;
+ recordCount:
+ number of records stored in DBF;
+ signature:
+ version number (aka signature). using 0x03 as a default meaning
+ "File without DBT". for more information about this field visit
+ ``http://www.clicketyclick.dk/databases/xbase/format/dbf.html#DBF_NOTE_1_TARGET``
+ lastUpdate:
+ date of the DBF's update. this could be a string ('yymmdd' or
+ 'yyyymmdd'), timestamp (int or float), datetime/date value,
+ a sequence (assuming (yyyy, mm, dd, ...)) or an object having
+ callable ``ticks`` field.
+ ignoreErrors:
+ error processing mode for DBF fields (boolean)
+
+ """
+ self.signature = signature
+ if fields is None:
+ self.fields = []
+ else:
+ self.fields = list(fields)
+ self.lastUpdate = getDate(lastUpdate)
+ self.recordLength = recordLength
+ self.headerLength = headerLength
+ self.recordCount = recordCount
+ self.ignoreErrors = ignoreErrors
+ # XXX: I'm not sure this is safe to
+ # initialize `self.changed` in this way
+ self.changed = bool(self.fields)
+
+ # @classmethod
+ def fromString(cls, string):
+ """Return header instance from the string object."""
+ return cls.fromStream(io.StringIO(str(string)))
+ fromString = classmethod(fromString)
+
+ # @classmethod
+ def fromStream(cls, stream):
+ """Return header object from the stream."""
+ stream.seek(0)
+ first_32 = stream.read(32)
+ if type(first_32) != bytes:
+ _data = bytes(first_32, sys.getfilesystemencoding())
+ _data = first_32
+ (_cnt, _hdrLen, _recLen) = struct.unpack("<I2H", _data[4:12])
+ #reserved = _data[12:32]
+ _year = _data[1]
+ if _year < 80:
+ # dBase II started at 1980. It is quite unlikely
+ # that actual last update date is before that year.
+ _year += 2000
+ else:
+ _year += 1900
+ ## create header object
+ _obj = cls(None, _hdrLen, _recLen, _cnt, _data[0],
+ (_year, _data[2], _data[3]))
+ ## append field definitions
+ # position 0 is for the deletion flag
+ _pos = 1
+ _data = stream.read(1)
+ while _data != b'\r':
+ _data += stream.read(31)
+ _fld = fields.lookupFor(_data[11]).fromString(_data, _pos)
+ _obj._addField(_fld)
+ _pos = _fld.end
+ _data = stream.read(1)
+ return _obj
+ fromStream = classmethod(fromStream)
+
+ ## properties
+
+ year = property(lambda self: self.lastUpdate.year)
+ month = property(lambda self: self.lastUpdate.month)
+ day = property(lambda self: self.lastUpdate.day)
+
+ def ignoreErrors(self, value):
+ """Update `ignoreErrors` flag on self and all fields"""
+ self._ignore_errors = value = bool(value)
+ for _field in self.fields:
+ _field.ignoreErrors = value
+ ignoreErrors = property(
+ lambda self: self._ignore_errors,
+ ignoreErrors,
+ doc="""Error processing mode for DBF field value conversion
+
+ if set, failing field value conversion will return
+ ``INVALID_VALUE`` instead of raising conversion error.
+
+ """)
+
+ ## object representation
+
+ def __repr__(self):
+ _rv = """\
+Version (signature): 0x%02x
+ Last update: %s
+ Header length: %d
+ Record length: %d
+ Record count: %d
+ FieldName Type Len Dec
+""" % (self.signature, self.lastUpdate, self.headerLength,
+ self.recordLength, self.recordCount)
+ _rv += "\n".join(
+ ["%10s %4s %3s %3s" % _fld.fieldInfo() for _fld in self.fields]
+ )
+ return _rv
+
+ ## internal methods
+
+ def _addField(self, *defs):
+ """Internal variant of the `addField` method.
+
+ This method doesn't set `self.changed` field to True.
+
+ Return value is a length of the appended records.
+ Note: this method doesn't modify ``recordLength`` and
+ ``headerLength`` fields. Use `addField` instead of this
+ method if you don't exactly know what you're doing.
+
+ """
+ # insure we have dbf.DbfFieldDef instances first (instantiation
+ # from the tuple could raise an error, in such a case I don't
+ # wanna add any of the definitions -- all will be ignored)
+ _defs = []
+ _recordLength = 0
+ for _def in defs:
+ if isinstance(_def, fields.DbfFieldDef):
+ _obj = _def
+ else:
+ (_name, _type, _len, _dec) = (tuple(_def) + (None,) * 4)[:4]
+ _cls = fields.lookupFor(_type)
+ _obj = _cls(_name, _len, _dec,
+ ignoreErrors=self._ignore_errors)
+ _recordLength += _obj.length
+ _defs.append(_obj)
+ # and now extend field definitions and
+ # update record length
+ self.fields += _defs
+ return _recordLength
+
+ ## interface methods
+
+ def addField(self, *defs):
+ """Add field definition to the header.
+
+ Examples:
+ dbfh.addField(
+ ("name", "C", 20),
+ dbf.DbfCharacterFieldDef("surname", 20),
+ dbf.DbfDateFieldDef("birthdate"),
+ ("member", "L"),
+ )
+ dbfh.addField(("price", "N", 5, 2))
+ dbfh.addField(dbf.DbfNumericFieldDef("origprice", 5, 2))
+
+ """
+ _oldLen = self.recordLength
+ self.recordLength += self._addField(*defs)
+ if not _oldLen:
+ self.recordLength += 1
+ # XXX: may be just use:
+ # self.recordeLength += self._addField(*defs) + bool(not _oldLen)
+ # recalculate headerLength
+ self.headerLength = 32 + (32 * len(self.fields)) + 1
+ self.changed = True
+
+ def write(self, stream):
+ """Encode and write header to the stream."""
+ stream.seek(0)
+ stream.write(self.toString())
+ fields = [_fld.toString() for _fld in self.fields]
+ stream.write(''.join(fields).encode(sys.getfilesystemencoding()))
+ stream.write(b'\x0D') # cr at end of all header data
+ self.changed = False
+
+ def toString(self):
+ """Returned 32 chars length string with encoded header."""
+ return struct.pack("<4BI2H",
+ self.signature,
+ self.year - 1900,
+ self.month,
+ self.day,
+ self.recordCount,
+ self.headerLength,
+ self.recordLength) + (b'\x00' * 20)
+ #TODO: figure out if bytes(utf-8) is correct here.
+
+ def setCurrentDate(self):
+ """Update ``self.lastUpdate`` field with current date value."""
+ self.lastUpdate = datetime.date.today()
+
+ def __getitem__(self, item):
+ """Return a field definition by numeric index or name string"""
+ if isinstance(item, str):
+ _name = item.upper()
+ for _field in self.fields:
+ if _field.name == _name:
+ return _field
+ else:
+ raise KeyError(item)
+ else:
+ # item must be field index
+ return self.fields[item]
+
+# vim: et sts=4 sw=4 :
diff --git a/src/tablib/packages/dbfpy3/record.py b/src/tablib/packages/dbfpy3/record.py
new file mode 100644
index 0000000..d301476
--- /dev/null
+++ b/src/tablib/packages/dbfpy3/record.py
@@ -0,0 +1,266 @@
+"""DBF record definition.
+
+"""
+"""History (most recent first):
+11-feb-2007 [als] __repr__: added special case for invalid field values
+10-feb-2007 [als] added .rawFromStream()
+30-oct-2006 [als] fix record length in .fromStream()
+04-jul-2006 [als] added export declaration
+20-dec-2005 [yc] DbfRecord.write() -> DbfRecord._write();
+ added delete() method.
+16-dec-2005 [yc] record definition moved from `dbf`.
+"""
+
+__version__ = "$Revision: 1.7 $"[11:-2]
+__date__ = "$Date: 2007/02/11 09:05:49 $"[7:-2]
+
+__all__ = ["DbfRecord"]
+
+import sys
+
+from . import utils
+
+class DbfRecord(object):
+ """DBF record.
+
+ Instances of this class shouldn't be created manualy,
+ use `dbf.Dbf.newRecord` instead.
+
+ Class implements mapping/sequence interface, so
+ fields could be accessed via their names or indexes
+ (names is a preffered way to access fields).
+
+ Hint:
+ Use `store` method to save modified record.
+
+ Examples:
+ Add new record to the database:
+ db = Dbf(filename)
+ rec = db.newRecord()
+ rec["FIELD1"] = value1
+ rec["FIELD2"] = value2
+ rec.store()
+ Or the same, but modify existed
+ (second in this case) record:
+ db = Dbf(filename)
+ rec = db[2]
+ rec["FIELD1"] = value1
+ rec["FIELD2"] = value2
+ rec.store()
+
+ """
+
+ __slots__ = "dbf", "index", "deleted", "fieldData"
+
+ ## creation and initialization
+
+ def __init__(self, dbf, index=None, deleted=False, data=None):
+ """Instance initialiation.
+
+ Arguments:
+ dbf:
+ A `Dbf.Dbf` instance this record belonogs to.
+ index:
+ An integer record index or None. If this value is
+ None, record will be appended to the DBF.
+ deleted:
+ Boolean flag indicating whether this record
+ is a deleted record.
+ data:
+ A sequence or None. This is a data of the fields.
+ If this argument is None, default values will be used.
+
+ """
+ self.dbf = dbf
+ # XXX: I'm not sure ``index`` is necessary
+ self.index = index
+ self.deleted = deleted
+ if data is None:
+ self.fieldData = [_fd.defaultValue for _fd in dbf.header.fields]
+ else:
+ self.fieldData = list(data)
+
+ # XXX: validate self.index before calculating position?
+ position = property(lambda self: self.dbf.header.headerLength + \
+ self.index * self.dbf.header.recordLength)
+
+ def rawFromStream(cls, dbf, index):
+ """Return raw record contents read from the stream.
+
+ Arguments:
+ dbf:
+ A `Dbf.Dbf` instance containing the record.
+ index:
+ Index of the record in the records' container.
+ This argument can't be None in this call.
+
+ Return value is a string containing record data in DBF format.
+
+ """
+ # XXX: may be write smth assuming, that current stream
+ # position is the required one? it could save some
+ # time required to calculate where to seek in the file
+ dbf.stream.seek(dbf.header.headerLength +
+ index * dbf.header.recordLength)
+ return dbf.stream.read(dbf.header.recordLength)
+ rawFromStream = classmethod(rawFromStream)
+
+ def fromStream(cls, dbf, index):
+ """Return a record read from the stream.
+
+ Arguments:
+ dbf:
+ A `Dbf.Dbf` instance new record should belong to.
+ index:
+ Index of the record in the records' container.
+ This argument can't be None in this call.
+
+ Return value is an instance of the current class.
+
+ """
+ return cls.fromString(dbf, cls.rawFromStream(dbf, index), index)
+ fromStream = classmethod(fromStream)
+
+ def fromString(cls, dbf, string, index=None):
+ """Return record read from the string object.
+
+ Arguments:
+ dbf:
+ A `Dbf.Dbf` instance new record should belong to.
+ string:
+ A string new record should be created from.
+ index:
+ Index of the record in the container. If this
+ argument is None, record will be appended.
+
+ Return value is an instance of the current class.
+
+ """
+ return cls(dbf, index, string[0]=="*",
+ [_fd.decodeFromRecord(string) for _fd in dbf.header.fields])
+ fromString = classmethod(fromString)
+
+ ## object representation
+
+ def __repr__(self):
+ _template = "%%%ds: %%s (%%s)" % max([len(_fld)
+ for _fld in self.dbf.fieldNames])
+ _rv = []
+ for _fld in self.dbf.fieldNames:
+ _val = self[_fld]
+ if _val is utils.INVALID_VALUE:
+ _rv.append(_template %
+ (_fld, "None", "value cannot be decoded"))
+ else:
+ _rv.append(_template % (_fld, _val, type(_val)))
+ return "\n".join(_rv)
+
+ ## protected methods
+
+ def _write(self):
+ """Write data to the dbf stream.
+
+ Note:
+ This isn't a public method, it's better to
+ use 'store' instead publically.
+ Be design ``_write`` method should be called
+ only from the `Dbf` instance.
+
+
+ """
+ self._validateIndex(False)
+ self.dbf.stream.seek(self.position)
+ self.dbf.stream.write(bytes(self.toString(),
+ sys.getfilesystemencoding()))
+ # FIXME: may be move this write somewhere else?
+ # why we should check this condition for each record?
+ if self.index == len(self.dbf):
+ # this is the last record,
+ # we should write SUB (ASCII 26)
+ self.dbf.stream.write(b"\x1A")
+
+ ## utility methods
+
+ def _validateIndex(self, allowUndefined=True, checkRange=False):
+ """Valid ``self.index`` value.
+
+ If ``allowUndefined`` argument is True functions does nothing
+ in case of ``self.index`` pointing to None object.
+
+ """
+ if self.index is None:
+ if not allowUndefined:
+ raise ValueError("Index is undefined")
+ elif self.index < 0:
+ raise ValueError("Index can't be negative (%s)" % self.index)
+ elif checkRange and self.index <= self.dbf.header.recordCount:
+ raise ValueError("There are only %d records in the DBF" %
+ self.dbf.header.recordCount)
+
+ ## interface methods
+
+ def store(self):
+ """Store current record in the DBF.
+
+ If ``self.index`` is None, this record will be appended to the
+ records of the DBF this records belongs to; or replaced otherwise.
+
+ """
+ self._validateIndex()
+ if self.index is None:
+ self.index = len(self.dbf)
+ self.dbf.append(self)
+ else:
+ self.dbf[self.index] = self
+
+ def delete(self):
+ """Mark method as deleted."""
+ self.deleted = True
+
+ def toString(self):
+ """Return string packed record values."""
+# for (_def, _dat) in zip(self.dbf.header.fields, self.fieldData):
+#
+
+ return "".join([" *"[self.deleted]] + [
+ _def.encodeValue(_dat)
+ for (_def, _dat) in zip(self.dbf.header.fields, self.fieldData)
+ ])
+
+ def asList(self):
+ """Return a flat list of fields.
+
+ Note:
+ Change of the list's values won't change
+ real values stored in this object.
+
+ """
+ return self.fieldData[:]
+
+ def asDict(self):
+ """Return a dictionary of fields.
+
+ Note:
+ Change of the dicts's values won't change
+ real values stored in this object.
+
+ """
+ return dict([_i for _i in zip(self.dbf.fieldNames, self.fieldData)])
+
+ def __getitem__(self, key):
+ """Return value by field name or field index."""
+ if isinstance(key, int):
+ # integer index of the field
+ return self.fieldData[key]
+ # assuming string field name
+ return self.fieldData[self.dbf.indexOfFieldName(key)]
+
+ def __setitem__(self, key, value):
+ """Set field value by integer index of the field or string name."""
+ if isinstance(key, int):
+ # integer index of the field
+ return self.fieldData[key]
+ # assuming string field name
+ self.fieldData[self.dbf.indexOfFieldName(key)] = value
+
+# vim: et sts=4 sw=4 :
diff --git a/src/tablib/packages/dbfpy3/utils.py b/src/tablib/packages/dbfpy3/utils.py
new file mode 100644
index 0000000..856ade8
--- /dev/null
+++ b/src/tablib/packages/dbfpy3/utils.py
@@ -0,0 +1,170 @@
+"""String utilities.
+
+TODO:
+ - allow strings in getDateTime routine;
+"""
+"""History (most recent first):
+11-feb-2007 [als] added INVALID_VALUE
+10-feb-2007 [als] allow date strings padded with spaces instead of zeroes
+20-dec-2005 [yc] handle long objects in getDate/getDateTime
+16-dec-2005 [yc] created from ``strutil`` module.
+"""
+
+__version__ = "$Revision: 1.4 $"[11:-2]
+__date__ = "$Date: 2007/02/11 08:57:17 $"[7:-2]
+
+import datetime
+import time
+
+
+def unzfill(str):
+ """Return a string without ASCII NULs.
+
+ This function searchers for the first NUL (ASCII 0) occurance
+ and truncates string till that position.
+
+ """
+ try:
+ return str[:str.index(b'\0')]
+ except ValueError:
+ return str
+
+
+def getDate(date=None):
+ """Return `datetime.date` instance.
+
+ Type of the ``date`` argument could be one of the following:
+ None:
+ use current date value;
+ datetime.date:
+ this value will be returned;
+ datetime.datetime:
+ the result of the date.date() will be returned;
+ string:
+ assuming "%Y%m%d" or "%y%m%dd" format;
+ number:
+ assuming it's a timestamp (returned for example
+ by the time.time() call;
+ sequence:
+ assuming (year, month, day, ...) sequence;
+
+ Additionaly, if ``date`` has callable ``ticks`` attribute,
+ it will be used and result of the called would be treated
+ as a timestamp value.
+
+ """
+ if date is None:
+ # use current value
+ return datetime.date.today()
+ if isinstance(date, datetime.date):
+ return date
+ if isinstance(date, datetime.datetime):
+ return date.date()
+ if isinstance(date, (int, float)):
+ # date is a timestamp
+ return datetime.date.fromtimestamp(date)
+ if isinstance(date, str):
+ date = date.replace(" ", "0")
+ if len(date) == 6:
+ # yymmdd
+ return datetime.date(*time.strptime(date, "%y%m%d")[:3])
+ # yyyymmdd
+ return datetime.date(*time.strptime(date, "%Y%m%d")[:3])
+ if hasattr(date, "__getitem__"):
+ # a sequence (assuming date/time tuple)
+ return datetime.date(*date[:3])
+ return datetime.date.fromtimestamp(date.ticks())
+
+
+def getDateTime(value=None):
+ """Return `datetime.datetime` instance.
+
+ Type of the ``value`` argument could be one of the following:
+ None:
+ use current date value;
+ datetime.date:
+ result will be converted to the `datetime.datetime` instance
+ using midnight;
+ datetime.datetime:
+ ``value`` will be returned as is;
+ string:
+ *** CURRENTLY NOT SUPPORTED ***;
+ number:
+ assuming it's a timestamp (returned for example
+ by the time.time() call;
+ sequence:
+ assuming (year, month, day, ...) sequence;
+
+ Additionaly, if ``value`` has callable ``ticks`` attribute,
+ it will be used and result of the called would be treated
+ as a timestamp value.
+
+ """
+ if value is None:
+ # use current value
+ return datetime.datetime.today()
+ if isinstance(value, datetime.datetime):
+ return value
+ if isinstance(value, datetime.date):
+ return datetime.datetime.fromordinal(value.toordinal())
+ if isinstance(value, (int, float)):
+ # value is a timestamp
+ return datetime.datetime.fromtimestamp(value)
+ if isinstance(value, str):
+ raise NotImplementedError("Strings aren't currently implemented")
+ if hasattr(value, "__getitem__"):
+ # a sequence (assuming date/time tuple)
+ return datetime.datetime(*tuple(value)[:6])
+ return datetime.datetime.fromtimestamp(value.ticks())
+
+
+class classproperty(property):
+ """Works in the same way as a ``property``, but for the classes."""
+
+ def __get__(self, obj, cls):
+ return self.fget(cls)
+
+
+class _InvalidValue(object):
+
+ """Value returned from DBF records when field validation fails
+
+ The value is not equal to anything except for itself
+ and equal to all empty values: None, 0, empty string etc.
+ In other words, invalid value is equal to None and not equal
+ to None at the same time.
+
+ This value yields zero upon explicit conversion to a number type,
+ empty string for string types, and False for boolean.
+
+ """
+
+ def __eq__(self, other):
+ return not other
+
+ def __ne__(self, other):
+ return not (other is self)
+
+ def __bool__(self):
+ return False
+
+ def __int__(self):
+ return 0
+ __long__ = __int__
+
+ def __float__(self):
+ return 0.0
+
+ def __str__(self):
+ return ""
+
+ def __unicode__(self):
+ return ""
+
+ def __repr__(self):
+ return "<INVALID>"
+
+# invalid value is a constant singleton
+INVALID_VALUE = _InvalidValue()
+
+# vim: set et sts=4 sw=4 :
diff --git a/src/tablib/packages/statistics.py b/src/tablib/packages/statistics.py
new file mode 100644
index 0000000..e97a6c9
--- /dev/null
+++ b/src/tablib/packages/statistics.py
@@ -0,0 +1,24 @@
+from __future__ import division
+
+
+def median(data):
+ """
+ Return the median (middle value) of numeric data, using the common
+ "mean of middle two" method. If data is empty, ValueError is raised.
+
+ Mimics the behaviour of Python3's statistics.median
+
+ >>> median([1, 3, 5])
+ 3
+ >>> median([1, 3, 5, 7])
+ 4.0
+
+ """
+ data = sorted(data)
+ n = len(data)
+ if not n:
+ raise ValueError("No median for empty data")
+ i = n // 2
+ if n % 2:
+ return data[i]
+ return (data[i - 1] + data[i]) / 2