summaryrefslogtreecommitdiff
path: root/tablib/core.py
blob: 5727a69bea53c15a476739c6004d4db3d8310646 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
# -*- coding: utf-8 -*-
"""
    tablib.core
    ~~~~~~~~~~~

    This module implements the central tablib objects.

    :copyright: (c) 2011 by Kenneth Reitz.
    :license: MIT, see LICENSE for more details.
"""

from copy import copy
from operator import itemgetter

from tablib import formats


__title__ = 'tablib'
__version__ = '0.9.2'
__build__ = 0x000902
__author__ = 'Kenneth Reitz'
__license__ = 'MIT'
__copyright__ = 'Copyright 2011 Kenneth Reitz'


class Row(object):
	"""Internal Row object. Mainly used for filtering."""

	__slots__ = ['tuple', '_row', 'tags']

	def __init__(self, row=list(), tags=list()):
		self._row = list(row)
		self.tags = list(tags)

	def __iter__(self):
		return (col for col in self._row)

	def __len__(self):
		return len(self._row)

	def __repr__(self):
		return repr(self._row)

	def __getslice__(self, i, j):
		return self._row[i,j]

	def __getitem__(self, i):
		return self._row[i]

	def __setitem__(self, i, value):
		self._row[i] = value

	def __delitem__(self, i):
		del self._row[i]

	def __getstate__(self):
		result = dict()
		result['_row'] = self._row
		result['tags'] = self.tags

		return result

	def __setstate__(self, state):
		self._row = state['_row']
		self.tags = state['tags']

	def append(self, value):
		self._row.append(value)

	def insert(self, index, value):
		self._row.insert(index, value)

	def __contains__(self, item):
		return (item in self._row)

	@property
	def tuple(self):
		'''Tuple representation of :class:`Row`.'''
		return tuple(self._row)

	@property
	def list(self):
		'''List representation of :class:`Row`.'''
		return list(self._row)

	def has_tag(self, tag):
		"""Returns true if current row contains tag."""

		if tag == None:
			return False
		elif isinstance(tag, basestring):
			return (tag in self.tags)
		else:
			return True if len(set(tag) & set(self.tags)) else False


class Dataset(object):
	"""The :class:`Dataset` object is the heart of Tablib. It provides all core
    functionality.

    Usually you create a :class:`Dataset` instance in your main module, and append
    rows and columns as you collect data. ::

        data = tablib.Dataset()
        data.headers = ('name', 'age')

        for (name, age) in some_collector():
            data.append((name, age))

    You can also set rows and headers upon instantiation. This is useful if dealing
    with dozens or hundres of :class:`Dataset` objects. ::

        headers = ('first_name', 'last_name')
        data = [('John', 'Adams'), ('George', 'Washington')]

        data = tablib.Dataset(*data, headers=headers)


    :param \*args: (optional) list of rows to populate Dataset
    :param headers: (optional) list strings for Dataset header row


    .. admonition:: Format Attributes Definition

     If you look at the code, the various output/import formats are not
     defined within the :class:`Dataset` object. To add support for a new format, see
     :ref:`Adding New Formats <newformats>`.

	"""

	def __init__(self, *args, **kwargs):
		self._data = list(Row(arg) for arg in args)
		self.__headers = None

		# ('title', index) tuples
		self._separators = []

		try:
			self.headers = kwargs['headers']
		except KeyError:
			self.headers = None

		try:
			self.title = kwargs['title']
		except KeyError:
			self.title = None

		self._register_formats()


	def __len__(self):
		return self.height


	def __getitem__(self, key):
		if isinstance(key, basestring):
			if key in self.headers:
				pos = self.headers.index(key) # get 'key' index from each data
				return [row[pos] for row in self._data]
			else:
				raise KeyError
		else:
			_results = self._data[key]
			if isinstance(_results, Row):
				return _results.tuple
			else:
				return [result.tuple for result in _results]


	def __setitem__(self, key, value):
		self._validate(value)
		self._data[key] = Row(value)


	def __delitem__(self, key):
		if isinstance(key, basestring):

			if key in self.headers:

				pos = self.headers.index(key)
				del self.headers[pos]

				for i, row in enumerate(self._data):

					del row[pos]
					self._data[i] = row
			else:
				raise KeyError
		else:
			del self._data[key]


	def __repr__(self):
		try:
			return '<%s dataset>' % (self.title.lower())
		except AttributeError:
			return '<dataset object>'


	@classmethod
	def _register_formats(cls):
		"""Adds format properties."""
		for fmt in formats.available:
			try:
				try:
					setattr(cls, fmt.title, property(fmt.export_set, fmt.import_set))
				except AttributeError:
					setattr(cls, fmt.title, property(fmt.export_set))

			except AttributeError:
				pass


	def _validate(self, row=None, col=None, safety=False):
		"""Assures size of every row in dataset is of proper proportions."""
		if row:
			is_valid = (len(row) == self.width) if self.width else True
		elif col:
			if len(col) < 1:
				is_valid = True
			else:
				is_valid = (len(col) == self.height) if self.height else True
		else:
			is_valid = all((len(x) == self.width for x in self._data))

		if is_valid:
			return True
		else:
			if not safety:
				raise InvalidDimensions
			return False


	def _package(self, dicts=True):
		"""Packages Dataset into lists of dictionaries for transmission."""

		if self.headers:
			if dicts:
				data = [dict(zip(self.headers, data_row)) for data_row in self ._data]
			else:
				data = [list(self.headers)] + list(self._data)
		else:
			data = [list(row) for row in self._data]

		return data


	def _clean_col(self, col):
		"""Prepares the given column for insert/append."""

		col = list(col)

		if self.headers:
			header = [col.pop(0)]
		else:
			header = []

		if len(col) == 1 and callable(col[0]):
			col = map(col[0], self._data)
		col = tuple(header + col)

		return col


	@property
	def height(self):
		"""The number of rows currently in the :class:`Dataset`.
		   Cannot be directly modified.
		"""
		return len(self._data)


	@property
	def width(self):
		"""The number of columns currently in the :class:`Dataset`.
		   Cannot be directly modified.
		"""

		try:
			return len(self._data[0])
		except IndexError:
			try:
				return len(self.headers)
			except TypeError:
				return 0


	@property
	def headers(self):
		"""An *optional* list of strings to be used for header rows and attribute names.

        This must be set manually. The given list length must equal :class:`Dataset.width`.

		"""
		return self.__headers


	@headers.setter
	def headers(self, collection):
		"""Validating headers setter."""
		self._validate(collection)
		if collection:
			try:
				self.__headers = list(collection)
			except TypeError:
				raise TypeError
		else:
			self.__headers = None


	@property
	def dict(self):
		"""A JSON representation of the :class:`Dataset` object. If headers have been
        set, a JSON list of objects will be returned. If no headers have
        been set, a JSON list of lists (rows) will be returned instead.

        A dataset object can also be imported by setting the `Dataset.json` attribute: ::

            data = tablib.Dataset()
            data.json = '[{"last_name": "Adams","age": 90,"first_name": "John"}]'

		"""
		return self._package()


	@dict.setter
	def dict(self, pickle):
		"""A native Python representation of the Dataset object. If headers have been
	    set, a list of Python dictionaries will be returned. If no headers have been
	    set, a list of tuples (rows) will be returned instead.

	    A dataset object can also be imported by setting the :class:`Dataset.dict` attribute. ::

	        data = tablib.Dataset()
	        data.dict = [{'age': 90, 'first_name': 'Kenneth', 'last_name': 'Reitz'}]

		"""
		if not len(pickle):
			return

		# if list of rows
		if isinstance(pickle[0], list):
			self.wipe()
			for row in pickle:
				self.append(Row(row))

		# if list of objects
		elif isinstance(pickle[0], dict):
			self.wipe()
			self.headers = pickle[0].keys()
			for row in pickle:
				self.append(Row(row.values()))
		else:
			raise UnsupportedFormat

	@property
	def xls():
		"""An Excel Spreadsheet representation of the :class:`Dataset` object, with :ref:`seperators`. Cannot be set.

	     .. admonition:: Binary Warning

	         :class:`Dataset.xls` contains binary data, so make sure to write in binary mode::

	            with open('output.xls', 'wb') as f:
	                f.write(data.xls)'
		"""
		pass


	@property
	def csv():
		"""A CSV representation of the :class:`Dataset` object. The top row will contain
	    headers, if they have been set. Otherwise, the top row will contain
	    the first row of the dataset.

	    A dataset object can also be imported by setting the :class:`Dataset.csv` attribute. ::

	        data = tablib.Dataset()
	        data.csv = 'age, first_name, last_name\\n90, John, Adams'

	    Import assumes (for now) that headers exist.
		"""
		pass

	@property
	def tsv():
		"""A TSV representation of the :class:`Dataset` object. The top row will contain
	    headers, if they have been set. Otherwise, the top row will contain
	    the first row of the dataset.

	    A dataset object can also be imported by setting the :class:`Dataset.tsv` attribute. ::

	        data = tablib.Dataset()
	        data.tsv = 'age\tfirst_name\tlast_name\\n90\tJohn\tAdams'

	    Import assumes (for now) that headers exist.
		"""

	@property
	def yaml():
		"""A YAML representation of the :class:`Dataset` object. If headers have been
	    set, a YAML list of objects will be returned. If no headers have
	    been set, a YAML list of lists (rows) will be returned instead.

	    A dataset object can also be imported by setting the :class:`Dataset.json` attribute: ::

	        data = tablib.Dataset()
	        data.yaml = '- {age: 90, first_name: John, last_name: Adams}'

	    Import assumes (for now) that headers exist.
		"""
		pass


	@property
	def json():
		"""A JSON representation of the :class:`Dataset` object. If headers have been
	    set, a JSON list of objects will be returned. If no headers have
	    been set, a JSON list of lists (rows) will be returned instead.

	    A dataset object can also be imported by setting the :class:`Dataset.json` attribute: ::

	        data = tablib.Dataset()
	        data.json = '[{age: 90, first_name: "John", liast_name: "Adams"}]'

	    Import assumes (for now) that headers exist.
		"""

	@property
	def html():
		"""A HTML table representation of the :class:`Dataset` object. If
		headers have been set, they will be used as table headers.

		..notice:: This method can be used for export only.
		"""
		pass

	def append(self, row=None, col=None, header=None, tags=list()):
		"""Adds a row or column to the :class:`Dataset`.
		Usage is  :class:`Dataset.insert` for documentation.
        """

		if row is not None:
			self.insert(self.height, row=row, tags=tags)
		elif col is not None:
			self.insert(self.width, col=col, header=header)

	def insert_separator(self, index, text='-'):
		"""Adds a separator to :class:`Dataset` at given index."""

		sep = (index, text)
		self._separators.append(sep)


	def append_separator(self, text='-'):
		"""Adds a :ref:`seperator <seperators>` to the :class:`Dataset`."""

		# change offsets if headers are or aren't defined
		if not self.headers:
			index = self.height if self.height else 0
		else:
			index = (self.height + 1) if self.height else 1

		self.insert_separator(index, text)


	def insert(self, index, row=None, col=None, header=None, tags=list()):
		"""Inserts a row or column to the :class:`Dataset` at the given index.

        Rows and columns inserted must be the correct size (height or width).

        The default behaviour is to insert the given row to the :class:`Dataset`
        object at the given index. If the ``col`` parameter is given, however,
        a new column will be insert to the :class:`Dataset` object instead.

        You can also insert a column of a single callable object, which will
        add a new column with the return values of the callable each as an
        item in the column. ::

            data.append(col=random.randint)

        See :ref:`dyncols` for an in-depth example.

        .. versionchanged:: 0.9.0
           If inserting a column, and :class:`Dataset.headers` is set, the
           header attribute must be set, and will be considered the header for
           that row.

        .. versionadded:: 0.9.0
           If inserting a row, you can add :ref:`tags <tags>` to the row you are inserting.
           This gives you the ability to :class:`filter <Dataset.filter>` your
           :class:`Dataset` later.

		"""
		if row:
			self._validate(row)
			self._data.insert(index, Row(row, tags=tags))
		elif col:
			col = list(col)

			# Callable Columns...
			if len(col) == 1 and callable(col[0]):
				col = map(col[0], self._data)

			col = self._clean_col(col)
			self._validate(col=col)

			if self.headers:
				# pop the first item off, add to headers
				if not header:
					raise HeadersNeeded()
				self.headers.insert(index, header)

			if self.height and self.width:

				for i, row in enumerate(self._data):

					row.insert(index, col[i])
					self._data[i] = row
			else:
				self._data = [Row([row]) for row in col]

	def filter(self, tag):
		"""Returns a new instance of the :class:`Dataset`, excluding any rows
		that do not contain the given :ref:`tags <tags>`.
		"""
		_dset = copy(self)
		_dset._data = [row for row in _dset._data if row.has_tag(tag)]

		return _dset

	def sort(self, col, reverse=False):

		"""Sort a :class:`Dataset` by a specific column. The order can be
		reversed by setting ``reverse`` to ``True``. Requires headers to be
		set. Returns a new :class:`Dataset` instance where columns have been
		sorted."""

		if not self.headers:
			raise HeadersNeeded

		_sorted = sorted(self.dict, key=itemgetter(col), reverse=reverse)
		_dset = Dataset(headers=self.headers)

		for item in _sorted:
			row = [item[key] for key in self.headers]
			_dset.append(row=row)

		return _dset

	def transpose(self):
		"""Transpose a :class:`Dataset`, turning rows into columns and vice
		versa, returning a new ``Dataset`` instance. The first row of the
		original instance becomes the new header row."""

		# Don't transpose if there is no data
		if not self:
			return

		_dset = Dataset()
		# The first element of the headers stays in the headers,
		# it is our "hinge" on which we rotate the data
		new_headers = [self.headers[0]] + self[self.headers[0]]

		_dset.headers = new_headers
		for column in self.headers:

			if column == self.headers[0]:
				# It's in the headers, so skip it
				continue

			# Adding the column name as now they're a regular column
			row_data = [column] + self[column]
			row_data = Row(row_data)
			_dset.append(row=row_data)

		return _dset


	def stack_rows(self, other):
		"""Stack two :class:`Dataset` instances together by
		joining at the row level, and return new combined
		``Dataset`` instance."""

		if not isinstance(other, Dataset):
			return

		if self.width != other.width:
			raise InvalidDimensions

		# Copy the source data
		_dset = copy(self)

		rows_to_stack = [row for row in _dset._data]
		other_rows = [row for row in other._data]

		rows_to_stack.extend(other_rows)
		_dset._data = rows_to_stack

		return _dset


	def stack_columns(self, other):
		"""Stack two :class:`Dataset` instances together by
		joining at the column level, and return a new
		combined ``Dataset`` instance. If either ``Dataset``
		has headers set, than the other must as well."""

		if not isinstance(other, Dataset):
			return

		if self.headers or other.headers:
			if not self.headers or not other.headers:
				raise HeadersNeeded

		if self.height != other.height:
			raise InvalidDimensions

		try:
			new_headers = self.headers + other.headers
		except TypeError:
			new_headers = None

		_dset = Dataset()

		for column in self.headers:
			_dset.append(col=self[column])

		for column in other.headers:
			_dset.append(col=other[column])

		_dset.headers = new_headers

		return _dset

	def wipe(self):
		"""Removes all content and headers from the :class:`Dataset` object."""
		self._data = list()
		self.__headers = None


class Databook(object):
	"""A book of :class:`Dataset` objects.
	"""

	def __init__(self, sets=None):

		if sets is None:
			self._datasets = list()
		else:
			self._datasets = sets

		self._register_formats()

	def __repr__(self):
		try:
			return '<%s databook>' % (self.title.lower())
		except AttributeError:
			return '<databook object>'


	def wipe(self):
		"""Removes all :class:`Dataset` objects from the :class:`Databook`."""
		self._datasets = []


	@classmethod
	def _register_formats(cls):
		"""Adds format properties."""
		for fmt in formats.available:
			try:
				try:
					setattr(cls, fmt.title, property(fmt.export_book, fmt.import_book))
				except AttributeError:
					setattr(cls, fmt.title, property(fmt.export_book))

			except AttributeError:
				pass


	def add_sheet(self, dataset):
		"""Adds given :class:`Dataset` to the :class:`Databook`."""
		if type(dataset) is Dataset:
			self._datasets.append(dataset)
		else:
			raise InvalidDatasetType


	def _package(self):
		"""Packages :class:`Databook` for delivery."""
		collector = []
		for dset in self._datasets:
			collector.append(dict(
				title = dset.title,
				data = dset.dict
			))
		return collector


	@property
	def size(self):
		"""The number of the :class:`Dataset` objects within :class:`Databook`."""
		return len(self._datasets)


def detect(stream):
	"""Return (format, stream) of given stream."""
	for fmt in formats.available:
		try:
			if fmt.detect(stream):
				return (fmt, stream)
		except AttributeError:
			pass
	return (None, stream)


def import_set(stream):
	"""Return dataset of given stream."""
	(format, stream) = detect(stream)

	try:
		data = Dataset()
		format.import_set(data, stream)
		return data

	except AttributeError, e:
		return None


class InvalidDatasetType(Exception):
	"Only Datasets can be added to a DataBook"


class InvalidDimensions(Exception):
	"Invalid size"

class HeadersNeeded(Exception):
	"Header parameter must be given when appending a column in this Dataset."

class UnsupportedFormat(NotImplementedError):
	"Format is not supported"