summaryrefslogtreecommitdiff
path: root/gitdb
diff options
context:
space:
mode:
authorSebastian Thiel <byronimo@gmail.com>2011-03-31 23:42:18 +0200
committerSebastian Thiel <byronimo@gmail.com>2011-03-31 23:42:18 +0200
commit17d9d1395fb6d18d553e085150138463b5827a2f (patch)
treec5c4c455e3dd684f1bfe7beaa1bc065800e11fca /gitdb
parentdf570f00f611073a20796128ca167474aa7826fc (diff)
parent0c7a3ec9829caa6632afd3e46901be67c63ae7fa (diff)
downloadgitdb-17d9d1395fb6d18d553e085150138463b5827a2f.tar.gz
Merge branch 'pack_writing'
Diffstat (limited to 'gitdb')
-rw-r--r--gitdb/_delta_apply.c19
-rw-r--r--gitdb/_delta_apply.h6
-rw-r--r--gitdb/exc.py3
-rw-r--r--gitdb/fun.py30
-rw-r--r--gitdb/pack.py265
-rw-r--r--gitdb/stream.py18
-rw-r--r--gitdb/test/lib.py7
-rw-r--r--gitdb/test/performance/test_pack_streaming.py43
-rw-r--r--gitdb/test/test_pack.py73
9 files changed, 422 insertions, 42 deletions
diff --git a/gitdb/_delta_apply.c b/gitdb/_delta_apply.c
index 96ab30a..f03e7ea 100644
--- a/gitdb/_delta_apply.c
+++ b/gitdb/_delta_apply.c
@@ -1,4 +1,4 @@
-#include "_delta_apply.h"
+#include <_delta_apply.h>
#include <stdint.h>
#include <assert.h>
#include <stdio.h>
@@ -463,7 +463,7 @@ void DIV_reset(DeltaInfoVector* vec)
// Append one chunk to the end of the list, and return a pointer to it
// It will not have been initialized !
-static inline
+inline
DeltaInfo* DIV_append(DeltaInfoVector* vec)
{
if (vec->size + 1 > vec->reserved_size){
@@ -703,7 +703,7 @@ typedef struct {
} DeltaChunkList;
-static
+
int DCL_init(DeltaChunkList*self, PyObject *args, PyObject *kwds)
{
if(args && PySequence_Size(args) > 0){
@@ -715,20 +715,20 @@ int DCL_init(DeltaChunkList*self, PyObject *args, PyObject *kwds)
return 0;
}
-static
+
void DCL_dealloc(DeltaChunkList* self)
{
TSI_destroy(&(self->istream));
}
-static
+
PyObject* DCL_py_rbound(DeltaChunkList* self)
{
return PyLong_FromUnsignedLongLong(self->istream.target_size);
}
// Write using a write function, taking remaining bytes from a base buffer
-static
+
PyObject* DCL_apply(DeltaChunkList* self, PyObject* args)
{
PyObject* pybuf = 0;
@@ -769,13 +769,13 @@ PyObject* DCL_apply(DeltaChunkList* self, PyObject* args)
Py_RETURN_NONE;
}
-static PyMethodDef DCL_methods[] = {
+PyMethodDef DCL_methods[] = {
{"apply", (PyCFunction)DCL_apply, METH_VARARGS, "Apply the given iterable of delta streams" },
{"rbound", (PyCFunction)DCL_py_rbound, METH_NOARGS, NULL},
{NULL} /* Sentinel */
};
-static PyTypeObject DeltaChunkListType = {
+PyTypeObject DeltaChunkListType = {
PyObject_HEAD_INIT(NULL)
0, /*ob_size*/
"DeltaChunkList", /*tp_name*/
@@ -897,7 +897,7 @@ uint compute_chunk_count(const uchar* data, const uchar* dend, bool read_header)
return num_chunks;
}
-static PyObject* connect_deltas(PyObject *self, PyObject *dstreams)
+PyObject* connect_deltas(PyObject *self, PyObject *dstreams)
{
// obtain iterator
PyObject* stream_iter = 0;
@@ -1088,7 +1088,6 @@ _error:
// Write using a write function, taking remaining bytes from a base buffer
// replaces the corresponding method in python
-static
PyObject* apply_delta(PyObject* self, PyObject* args)
{
PyObject* pybbuf = 0;
diff --git a/gitdb/_delta_apply.h b/gitdb/_delta_apply.h
index 3e7e5f9..1fcd538 100644
--- a/gitdb/_delta_apply.h
+++ b/gitdb/_delta_apply.h
@@ -1,6 +1,6 @@
#include <Python.h>
-static PyObject* connect_deltas(PyObject *self, PyObject *dstreams);
-static PyObject* apply_delta(PyObject* self, PyObject* args);
+extern PyObject* connect_deltas(PyObject *self, PyObject *dstreams);
+extern PyObject* apply_delta(PyObject* self, PyObject* args);
-static PyTypeObject DeltaChunkListType;
+extern PyTypeObject DeltaChunkListType;
diff --git a/gitdb/exc.py b/gitdb/exc.py
index 96fa874..e087047 100644
--- a/gitdb/exc.py
+++ b/gitdb/exc.py
@@ -17,6 +17,9 @@ class BadObject(ODBError):
def __str__(self):
return "BadObject: %s" % to_hex_sha(self.args[0])
+
+class ParseError(ODBError):
+ """Thrown if the parsing of a file failed due to an invalid format"""
class AmbiguousObjectName(ODBError):
"""Thrown if a possibly shortened name does not uniquely represent a single object
diff --git a/gitdb/fun.py b/gitdb/fun.py
index 3e035db..5bbe8ef 100644
--- a/gitdb/fun.py
+++ b/gitdb/fun.py
@@ -48,7 +48,7 @@ chunk_size = 1000*mmap.PAGESIZE
__all__ = ('is_loose_object', 'loose_object_header_info', 'msb_size', 'pack_object_header_info',
'write_object', 'loose_object_header', 'stream_copy', 'apply_delta_data',
- 'is_equal_canonical_sha', 'connect_deltas', 'DeltaChunkList')
+ 'is_equal_canonical_sha', 'connect_deltas', 'DeltaChunkList', 'create_pack_object_header')
#{ Structures
@@ -411,15 +411,25 @@ def pack_object_header_info(data):
size += (c & 0x7f) << s
s += 7
# END character loop
-
- try:
- return (type_id, size, i)
- except KeyError:
- # invalid object type - we could try to be smart now and decode part
- # of the stream to get the info, problem is that we had trouble finding
- # the exact start of the content stream
- raise BadObjectType(type_id)
- # END handle exceptions
+ return (type_id, size, i)
+
+def create_pack_object_header(obj_type, obj_size):
+ """:return: string defining the pack header comprised of the object type
+ and its incompressed size in bytes
+ :parmam obj_type: pack type_id of the object
+ :param obj_size: uncompressed size in bytes of the following object stream"""
+ c = 0 # 1 byte
+ hdr = str() # output string
+
+ c = (obj_type << 4) | (obj_size & 0xf)
+ obj_size >>= 4
+ while obj_size:
+ hdr += chr(c | 0x80)
+ c = obj_size & 0x7f
+ obj_size >>= 7
+ #END until size is consumed
+ hdr += chr(c)
+ return hdr
def msb_size(data, offset=0):
"""
diff --git a/gitdb/pack.py b/gitdb/pack.py
index 09e7def..7ae9786 100644
--- a/gitdb/pack.py
+++ b/gitdb/pack.py
@@ -5,16 +5,19 @@
"""Contains PackIndexFile and PackFile implementations"""
from gitdb.exc import (
BadObject,
- UnsupportedOperation
+ UnsupportedOperation,
+ ParseError
)
from util import (
zlib,
LazyMixin,
unpack_from,
+ bin_to_hex,
file_contents_ro_filepath,
)
from fun import (
+ create_pack_object_header,
pack_object_header_info,
is_equal_canonical_sha,
type_id_to_type_map,
@@ -47,6 +50,7 @@ from stream import (
DeltaApplyReader,
Sha1Writer,
NullStream,
+ FlexibleSha1Writer
)
from struct import (
@@ -54,7 +58,10 @@ from struct import (
unpack,
)
+from binascii import crc32
+
from itertools import izip
+import tempfile
import array
import os
import sys
@@ -98,8 +105,7 @@ def pack_object_at(data, offset, as_stream):
# REF DELTA
elif type_id == REF_DELTA:
total_rela_offset = data_rela_offset+20
- ref_sha = data[data_rela_offset:total_rela_offset]
- delta_info = ref_sha
+ delta_info = data[data_rela_offset:total_rela_offset]
# BASE OBJECT
else:
# assume its a base object
@@ -120,10 +126,120 @@ def pack_object_at(data, offset, as_stream):
return abs_data_offset, ODeltaPackInfo(offset, type_id, uncomp_size, delta_info)
# END handle info
# END handle stream
-
+
+def write_stream_to_pack(read, write, zstream, base_crc=None):
+ """Copy a stream as read from read function, zip it, and write the result.
+ Count the number of written bytes and return it
+ :param base_crc: if not None, the crc will be the base for all compressed data
+ we consecutively write and generate a crc32 from. If None, no crc will be generated
+ :return: tuple(no bytes read, no bytes written, crc32) crc might be 0 if base_crc
+ was false"""
+ br = 0 # bytes read
+ bw = 0 # bytes written
+ want_crc = base_crc is not None
+ crc = 0
+ if want_crc:
+ crc = base_crc
+ #END initialize crc
+
+ while True:
+ chunk = read(chunk_size)
+ br += len(chunk)
+ compressed = zstream.compress(chunk)
+ bw += len(compressed)
+ write(compressed) # cannot assume return value
+
+ if want_crc:
+ crc = crc32(compressed, crc)
+ #END handle crc
+
+ if len(chunk) != chunk_size:
+ break
+ #END copy loop
+
+ compressed = zstream.flush()
+ bw += len(compressed)
+ write(compressed)
+ if want_crc:
+ crc = crc32(compressed, crc)
+ #END handle crc
+
+ return (br, bw, crc)
+
+
#} END utilities
+class IndexWriter(object):
+ """Utility to cache index information, allowing to write all information later
+ in one go to the given stream
+ :note: currently only writes v2 indices"""
+ __slots__ = '_objs'
+
+ def __init__(self):
+ self._objs = list()
+
+ def append(self, binsha, crc, offset):
+ """Append one piece of object information"""
+ self._objs.append((binsha, crc, offset))
+
+ def write(self, pack_sha, write):
+ """Write the index file using the given write method
+ :param pack_sha: binary sha over the whole pack that we index
+ :return: sha1 binary sha over all index file contents"""
+ # sort for sha1 hash
+ self._objs.sort(key=lambda o: o[0])
+
+ sha_writer = FlexibleSha1Writer(write)
+ sha_write = sha_writer.write
+ sha_write(PackIndexFile.index_v2_signature)
+ sha_write(pack(">L", PackIndexFile.index_version_default))
+
+ # fanout
+ tmplist = list((0,)*256) # fanout or list with 64 bit offsets
+ for t in self._objs:
+ tmplist[ord(t[0][0])] += 1
+ #END prepare fanout
+ for i in xrange(255):
+ v = tmplist[i]
+ sha_write(pack('>L', v))
+ tmplist[i+1] += v
+ #END write each fanout entry
+ sha_write(pack('>L', tmplist[255]))
+
+ # sha1 ordered
+ # save calls, that is push them into c
+ sha_write(''.join(t[0] for t in self._objs))
+
+ # crc32
+ for t in self._objs:
+ sha_write(pack('>L', t[1]&0xffffffff))
+ #END for each crc
+
+ tmplist = list()
+ # offset 32
+ for t in self._objs:
+ ofs = t[2]
+ if ofs > 0x7fffffff:
+ tmplist.append(ofs)
+ ofs = 0x80000000 + len(tmplist)-1
+ #END hande 64 bit offsets
+ sha_write(pack('>L', ofs&0xffffffff))
+ #END for each offset
+
+ # offset 64
+ for ofs in tmplist:
+ sha_write(pack(">Q", ofs))
+ #END for each offset
+
+ # trailer
+ assert(len(pack_sha) == 20)
+ sha_write(pack_sha)
+ sha = sha_writer.sha(as_hex=False)
+ write(sha)
+ return sha
+
+
class PackIndexFile(LazyMixin):
"""A pack index provides offsets into the corresponding pack, allowing to find
@@ -136,6 +252,8 @@ class PackIndexFile(LazyMixin):
# used in v2 indices
_sha_list_offset = 8 + 1024
+ index_v2_signature = '\377tOc'
+ index_version_default = 2
def __init__(self, indexpath):
super(PackIndexFile, self).__init__()
@@ -156,7 +274,7 @@ class PackIndexFile(LazyMixin):
# to access the fanout table or related properties
# CHECK VERSION
- self._version = (self._data[:4] == '\377tOc' and 2) or 1
+ self._version = (self._data[:4] == self.index_v2_signature and 2) or 1
if self._version == 2:
version_id = unpack_from(">L", self._data, 4)[0]
assert version_id == self._version, "Unsupported index version: %i" % version_id
@@ -384,6 +502,8 @@ class PackFile(LazyMixin):
case"""
__slots__ = ('_packpath', '_data', '_size', '_version')
+ pack_signature = 0x5041434b # 'PACK'
+ pack_version_default = 2
# offset into our data at which the first object starts
first_object_offset = 3*4 # header bytes
@@ -397,15 +517,19 @@ class PackFile(LazyMixin):
self._data = file_contents_ro_filepath(self._packpath)
# read the header information
- type_id, self._version, self._size = unpack_from(">4sLL", self._data, 0)
+ type_id, self._version, self._size = unpack_from(">LLL", self._data, 0)
# TODO: figure out whether we should better keep the lock, or maybe
# add a .keep file instead ?
else: # must be '_size' or '_version'
# read header info - we do that just with a file stream
- type_id, self._version, self._size = unpack(">4sLL", open(self._packpath).read(12))
+ type_id, self._version, self._size = unpack(">LLL", open(self._packpath).read(12))
# END handle header
+ if type_id != self.pack_signature:
+ raise ParseError("Invalid pack signature: %i" % type_id)
+ #END assert type id
+
def _iter_objects(self, start_offset, as_stream=True):
"""Handle the actual iteration of objects within this pack"""
data = self._data
@@ -532,6 +656,9 @@ class PackEntity(LazyMixin):
def _set_cache_(self, attr):
# currently this can only be _offset_map
+ # TODO: make this a simple sorted offset array which can be bisected
+ # to find the respective entry, from which we can take a +1 easily
+ # This might be slower, but should also be much lighter in memory !
offsets_sorted = sorted(self._index.offsets())
last_offset = len(self._pack.data()) - self._pack.footer_size
assert offsets_sorted, "Cannot handle empty indices"
@@ -561,11 +688,10 @@ class PackEntity(LazyMixin):
def _iter_objects(self, as_stream):
"""Iterate over all objects in our index and yield their OInfo or OStream instences"""
- indexfile = self._index
+ _sha = self._index.sha
_object = self._object
- for index in xrange(indexfile.size()):
- sha = indexfile.sha(index)
- yield _object(sha, as_stream, index)
+ for index in xrange(self._index.size()):
+ yield _object(_sha(index), as_stream, index)
# END for each index
def _object(self, sha, as_stream, index=-1):
@@ -653,7 +779,9 @@ class PackEntity(LazyMixin):
"""
Verify that the stream at the given sha is valid.
- :param use_crc: if True, the index' crc for the sha is used to determine
+ :param use_crc: if True, the index' crc is run over the compressed stream of
+ the object, which is much faster than checking the sha1. It is also
+ more prone to unnoticed corruption or manipulation.
:param sha: 20 byte sha1 of the object whose stream to verify
whether the compressed stream of the object is valid. If it is
a delta, this only verifies that the delta's data is valid, not the
@@ -760,5 +888,118 @@ class PackEntity(LazyMixin):
return self.collect_streams_at_offset(self._index.offset(self._sha_to_index(sha)))
+ @classmethod
+ def write_pack(cls, object_iter, pack_write, index_write=None,
+ object_count = None, zlib_compression = zlib.Z_BEST_SPEED):
+ """
+ Create a new pack by putting all objects obtained by the object_iterator
+ into a pack which is written using the pack_write method.
+ The respective index is produced as well if index_write is not Non.
+
+ :param object_iter: iterator yielding odb output objects
+ :param pack_write: function to receive strings to write into the pack stream
+ :param indx_write: if not None, the function writes the index file corresponding
+ to the pack.
+ :param object_count: if you can provide the amount of objects in your iteration,
+ this would be the place to put it. Otherwise we have to pre-iterate and store
+ all items into a list to get the number, which uses more memory than necessary.
+ :param zlib_compression: the zlib compression level to use
+ :return: tuple(pack_sha, index_binsha) binary sha over all the contents of the pack
+ and over all contents of the index. If index_write was None, index_binsha will be None
+ :note: The destination of the write functions is up to the user. It could
+ be a socket, or a file for instance
+ :note: writes only undeltified objects"""
+ objs = object_iter
+ if not object_count:
+ if not isinstance(object_iter, (tuple, list)):
+ objs = list(object_iter)
+ #END handle list type
+ object_count = len(objs)
+ #END handle object
+
+ pack_writer = FlexibleSha1Writer(pack_write)
+ pwrite = pack_writer.write
+ ofs = 0 # current offset into the pack file
+ index = None
+ wants_index = index_write is not None
+
+ # write header
+ pwrite(pack('>LLL', PackFile.pack_signature, PackFile.pack_version_default, object_count))
+ ofs += 12
+
+ if wants_index:
+ index = IndexWriter()
+ #END handle index header
+
+ actual_count = 0
+ for obj in objs:
+ actual_count += 1
+ crc = 0
+
+ # object header
+ hdr = create_pack_object_header(obj.type_id, obj.size)
+ if index_write:
+ crc = crc32(hdr)
+ else:
+ crc = None
+ #END handle crc
+ pwrite(hdr)
+
+ # data stream
+ zstream = zlib.compressobj(zlib_compression)
+ ostream = obj.stream
+ br, bw, crc = write_stream_to_pack(ostream.read, pwrite, zstream, base_crc = crc)
+ assert(br == obj.size)
+ if wants_index:
+ index.append(obj.binsha, crc, ofs)
+ #END handle index
+
+ ofs += len(hdr) + bw
+ if actual_count == object_count:
+ break
+ #END abort once we are done
+ #END for each object
+
+ if actual_count != object_count:
+ raise ValueError("Expected to write %i objects into pack, but received only %i from iterators" % (object_count, actual_count))
+ #END count assertion
+
+ # write footer
+ pack_sha = pack_writer.sha(as_hex = False)
+ assert len(pack_sha) == 20
+ pack_write(pack_sha)
+ ofs += len(pack_sha) # just for completeness ;)
+
+ index_sha = None
+ if wants_index:
+ index_sha = index.write(pack_sha, index_write)
+ #END handle index
+
+ return pack_sha, index_sha
+
+ @classmethod
+ def create(cls, object_iter, base_dir, object_count = None, zlib_compression = zlib.Z_BEST_SPEED):
+ """Create a new on-disk entity comprised of a properly named pack file and a properly named
+ and corresponding index file. The pack contains all OStream objects contained in object iter.
+ :param base_dir: directory which is to contain the files
+ :return: PackEntity instance initialized with the new pack
+ :note: for more information on the other parameters see the write_pack method"""
+ pack_fd, pack_path = tempfile.mkstemp('', 'pack', base_dir)
+ index_fd, index_path = tempfile.mkstemp('', 'index', base_dir)
+ pack_write = lambda d: os.write(pack_fd, d)
+ index_write = lambda d: os.write(index_fd, d)
+
+ pack_binsha, index_binsha = cls.write_pack(object_iter, pack_write, index_write, object_count, zlib_compression)
+ os.close(pack_fd)
+ os.close(index_fd)
+
+ fmt = "pack-%s.%s"
+ new_pack_path = os.path.join(base_dir, fmt % (bin_to_hex(pack_binsha), 'pack'))
+ new_index_path = os.path.join(base_dir, fmt % (bin_to_hex(pack_binsha), 'idx'))
+ os.rename(pack_path, new_pack_path)
+ os.rename(index_path, new_index_path)
+
+ return cls(new_pack_path)
+
#} END interface
diff --git a/gitdb/stream.py b/gitdb/stream.py
index 6c3b8d3..8010a05 100644
--- a/gitdb/stream.py
+++ b/gitdb/stream.py
@@ -33,7 +33,9 @@ try:
except ImportError:
pass
-__all__ = ('DecompressMemMapReader', 'FDCompressedSha1Writer', 'DeltaApplyReader')
+__all__ = ( 'DecompressMemMapReader', 'FDCompressedSha1Writer', 'DeltaApplyReader',
+ 'Sha1Writer', 'FlexibleSha1Writer', 'ZippedStoreShaWriter', 'FDCompressedSha1Writer',
+ 'FDStream', 'NullStream')
#{ RO Streams
@@ -557,6 +559,20 @@ class Sha1Writer(object):
#} END interface
+class FlexibleSha1Writer(Sha1Writer):
+ """Writer producing a sha1 while passing on the written bytes to the given
+ write function"""
+ __slots__ = 'writer'
+
+ def __init__(self, writer):
+ Sha1Writer.__init__(self)
+ self.writer = writer
+
+ def write(self, data):
+ Sha1Writer.write(self, data)
+ self.writer(data)
+
+
class ZippedStoreShaWriter(Sha1Writer):
"""Remembers everything someone writes to it and generates a sha"""
__slots__ = ('buf', 'zip')
diff --git a/gitdb/test/lib.py b/gitdb/test/lib.py
index 342234a..50645be 100644
--- a/gitdb/test/lib.py
+++ b/gitdb/test/lib.py
@@ -42,19 +42,22 @@ def with_rw_directory(func):
def wrapper(self):
path = tempfile.mktemp(prefix=func.__name__)
os.mkdir(path)
+ keep = False
try:
try:
return func(self, path)
except Exception:
print >> sys.stderr, "Test %s.%s failed, output is at %r" % (type(self).__name__, func.__name__, path)
+ keep = True
raise
finally:
# Need to collect here to be sure all handles have been closed. It appears
# a windows-only issue. In fact things should be deleted, as well as
# memory maps closed, once objects go out of scope. For some reason
# though this is not the case here unless we collect explicitly.
- gc.collect()
- shutil.rmtree(path)
+ if not keep:
+ gc.collect()
+ shutil.rmtree(path)
# END handle exception
# END wrapper
diff --git a/gitdb/test/performance/test_pack_streaming.py b/gitdb/test/performance/test_pack_streaming.py
index 22a62a3..795ed1e 100644
--- a/gitdb/test/performance/test_pack_streaming.py
+++ b/gitdb/test/performance/test_pack_streaming.py
@@ -8,14 +8,57 @@ from lib import (
)
from gitdb.db.pack import PackedDB
+from gitdb.stream import NullStream
+from gitdb.pack import PackEntity
import os
import sys
from time import time
+from nose import SkipTest
+
+class CountedNullStream(NullStream):
+ __slots__ = '_bw'
+ def __init__(self):
+ self._bw = 0
+
+ def bytes_written(self):
+ return self._bw
+
+ def write(self, d):
+ self._bw += NullStream.write(self, d)
+
class TestPackStreamingPerformance(TestBigRepoR):
+ def test_pack_writing(self):
+ # see how fast we can write a pack from object streams.
+ # This will not be fast, as we take time for decompressing the streams as well
+ ostream = CountedNullStream()
+ pdb = PackedDB(os.path.join(self.gitrepopath, "objects/pack"))
+
+ ni = 5000
+ count = 0
+ total_size = 0
+ st = time()
+ objs = list()
+ for sha in pdb.sha_iter():
+ count += 1
+ objs.append(pdb.stream(sha))
+ if count == ni:
+ break
+ #END gather objects for pack-writing
+ elapsed = time() - st
+ print >> sys.stderr, "PDB Streaming: Got %i streams by sha in in %f s ( %f streams/s )" % (ni, elapsed, ni / elapsed)
+
+ st = time()
+ PackEntity.write_pack(objs, ostream.write)
+ elapsed = time() - st
+ total_kb = ostream.bytes_written() / 1000
+ print >> sys.stderr, "PDB Streaming: Wrote pack of size %i kb in %f s (%f kb/s)" % (total_kb, elapsed, total_kb/elapsed)
+
+
def test_stream_reading(self):
+ raise SkipTest()
pdb = PackedDB(os.path.join(self.gitrepopath, "objects/pack"))
# streaming only, meant for --with-profile runs
diff --git a/gitdb/test/test_pack.py b/gitdb/test/test_pack.py
index 928f0cd..4a7f1ca 100644
--- a/gitdb/test/test_pack.py
+++ b/gitdb/test/test_pack.py
@@ -25,8 +25,12 @@ from gitdb.base import (
from gitdb.fun import delta_types
from gitdb.exc import UnsupportedOperation
from gitdb.util import to_bin_sha
-from itertools import izip
+from itertools import izip, chain
+from nose import SkipTest
+
import os
+import sys
+import tempfile
#{ Utilities
@@ -134,7 +138,9 @@ class TestPack(TestBase):
self._assert_pack_file(pack, version, size)
# END for each pack to test
- def test_pack_entity(self):
+ @with_rw_directory
+ def test_pack_entity(self, rw_dir):
+ pack_objs = list()
for packinfo, indexinfo in ( (self.packfile_v2_1, self.packindexfile_v1),
(self.packfile_v2_2, self.packindexfile_v2),
(self.packfile_v2_3_ascii, self.packindexfile_v2_3_ascii)):
@@ -143,6 +149,7 @@ class TestPack(TestBase):
entity = PackEntity(packfile)
assert entity.pack().path() == packfile
assert entity.index().path() == indexfile
+ pack_objs.extend(entity.stream_iter())
count = 0
for info, stream in izip(entity.info_iter(), entity.stream_iter()):
@@ -174,9 +181,67 @@ class TestPack(TestBase):
# END for each info, stream tuple
assert count == size
- # END for each entity
+ # END for each entity
+
+ # pack writing - write all packs into one
+ # index path can be None
+ pack_path = tempfile.mktemp('', "pack", rw_dir)
+ index_path = tempfile.mktemp('', 'index', rw_dir)
+ iteration = 0
+ def rewind_streams():
+ for obj in pack_objs:
+ obj.stream.seek(0)
+ #END utility
+ for ppath, ipath, num_obj in zip((pack_path, )*2, (index_path, None), (len(pack_objs), None)):
+ pfile = open(ppath, 'wb')
+ iwrite = None
+ if ipath:
+ ifile = open(ipath, 'wb')
+ iwrite = ifile.write
+ #END handle ip
+
+ # make sure we rewind the streams ... we work on the same objects over and over again
+ if iteration > 0:
+ rewind_streams()
+ #END rewind streams
+ iteration += 1
+
+ pack_sha, index_sha = PackEntity.write_pack(pack_objs, pfile.write, iwrite, object_count=num_obj)
+ pfile.close()
+ assert os.path.getsize(ppath) > 100
+
+ # verify pack
+ pf = PackFile(ppath)
+ assert pf.size() == len(pack_objs)
+ assert pf.version() == PackFile.pack_version_default
+ assert pf.checksum() == pack_sha
+
+ # verify index
+ if ipath is not None:
+ ifile.close()
+ assert os.path.getsize(ipath) > 100
+ idx = PackIndexFile(ipath)
+ assert idx.version() == PackIndexFile.index_version_default
+ assert idx.packfile_checksum() == pack_sha
+ assert idx.indexfile_checksum() == index_sha
+ assert idx.size() == len(pack_objs)
+ #END verify files exist
+ #END for each packpath, indexpath pair
+
+ # verify the packs throughly
+ rewind_streams()
+ entity = PackEntity.create(pack_objs, rw_dir)
+ count = 0
+ for info in entity.info_iter():
+ count += 1
+ for use_crc in range(2):
+ assert entity.is_valid_stream(info.binsha, use_crc)
+ # END for each crc mode
+ #END for each info
+ assert count == len(pack_objs)
+
def test_pack_64(self):
# TODO: hex-edit a pack helping us to verify that we can handle 64 byte offsets
# of course without really needing such a huge pack
- pass
+ raise SkipTest()