summaryrefslogtreecommitdiff
path: root/gitdb
diff options
context:
space:
mode:
authorSebastian Thiel <byronimo@gmail.com>2011-03-31 23:40:04 +0200
committerSebastian Thiel <byronimo@gmail.com>2011-03-31 23:40:10 +0200
commit0c7a3ec9829caa6632afd3e46901be67c63ae7fa (patch)
treec5c4c455e3dd684f1bfe7beaa1bc065800e11fca /gitdb
parent184a776960efdc2a83eac571c9c046ffcee3e7c8 (diff)
downloadgitdb-0c7a3ec9829caa6632afd3e46901be67c63ae7fa.tar.gz
Fixed _perf module, which built, but didn't link dynamically. All the time, I think it never successfully imported, but its hard to believe this slipped by.
Added performance test for pack-writing, which isn't really showing what I want as it currently read data from a densly compressed pack which takes most of the time in the nearly pure python implementation. Compared to c++, all the measured performance is just below anything I'd want to use. But we shouldn't forget this is just a test implementation, writing packs is quite simple actually, if you leave out the delta compression part and the delta logic
Diffstat (limited to 'gitdb')
-rw-r--r--gitdb/_delta_apply.c19
-rw-r--r--gitdb/_delta_apply.h6
-rw-r--r--gitdb/test/performance/test_pack_streaming.py43
3 files changed, 55 insertions, 13 deletions
diff --git a/gitdb/_delta_apply.c b/gitdb/_delta_apply.c
index 96ab30a..f03e7ea 100644
--- a/gitdb/_delta_apply.c
+++ b/gitdb/_delta_apply.c
@@ -1,4 +1,4 @@
-#include "_delta_apply.h"
+#include <_delta_apply.h>
#include <stdint.h>
#include <assert.h>
#include <stdio.h>
@@ -463,7 +463,7 @@ void DIV_reset(DeltaInfoVector* vec)
// Append one chunk to the end of the list, and return a pointer to it
// It will not have been initialized !
-static inline
+inline
DeltaInfo* DIV_append(DeltaInfoVector* vec)
{
if (vec->size + 1 > vec->reserved_size){
@@ -703,7 +703,7 @@ typedef struct {
} DeltaChunkList;
-static
+
int DCL_init(DeltaChunkList*self, PyObject *args, PyObject *kwds)
{
if(args && PySequence_Size(args) > 0){
@@ -715,20 +715,20 @@ int DCL_init(DeltaChunkList*self, PyObject *args, PyObject *kwds)
return 0;
}
-static
+
void DCL_dealloc(DeltaChunkList* self)
{
TSI_destroy(&(self->istream));
}
-static
+
PyObject* DCL_py_rbound(DeltaChunkList* self)
{
return PyLong_FromUnsignedLongLong(self->istream.target_size);
}
// Write using a write function, taking remaining bytes from a base buffer
-static
+
PyObject* DCL_apply(DeltaChunkList* self, PyObject* args)
{
PyObject* pybuf = 0;
@@ -769,13 +769,13 @@ PyObject* DCL_apply(DeltaChunkList* self, PyObject* args)
Py_RETURN_NONE;
}
-static PyMethodDef DCL_methods[] = {
+PyMethodDef DCL_methods[] = {
{"apply", (PyCFunction)DCL_apply, METH_VARARGS, "Apply the given iterable of delta streams" },
{"rbound", (PyCFunction)DCL_py_rbound, METH_NOARGS, NULL},
{NULL} /* Sentinel */
};
-static PyTypeObject DeltaChunkListType = {
+PyTypeObject DeltaChunkListType = {
PyObject_HEAD_INIT(NULL)
0, /*ob_size*/
"DeltaChunkList", /*tp_name*/
@@ -897,7 +897,7 @@ uint compute_chunk_count(const uchar* data, const uchar* dend, bool read_header)
return num_chunks;
}
-static PyObject* connect_deltas(PyObject *self, PyObject *dstreams)
+PyObject* connect_deltas(PyObject *self, PyObject *dstreams)
{
// obtain iterator
PyObject* stream_iter = 0;
@@ -1088,7 +1088,6 @@ _error:
// Write using a write function, taking remaining bytes from a base buffer
// replaces the corresponding method in python
-static
PyObject* apply_delta(PyObject* self, PyObject* args)
{
PyObject* pybbuf = 0;
diff --git a/gitdb/_delta_apply.h b/gitdb/_delta_apply.h
index 3e7e5f9..1fcd538 100644
--- a/gitdb/_delta_apply.h
+++ b/gitdb/_delta_apply.h
@@ -1,6 +1,6 @@
#include <Python.h>
-static PyObject* connect_deltas(PyObject *self, PyObject *dstreams);
-static PyObject* apply_delta(PyObject* self, PyObject* args);
+extern PyObject* connect_deltas(PyObject *self, PyObject *dstreams);
+extern PyObject* apply_delta(PyObject* self, PyObject* args);
-static PyTypeObject DeltaChunkListType;
+extern PyTypeObject DeltaChunkListType;
diff --git a/gitdb/test/performance/test_pack_streaming.py b/gitdb/test/performance/test_pack_streaming.py
index 22a62a3..795ed1e 100644
--- a/gitdb/test/performance/test_pack_streaming.py
+++ b/gitdb/test/performance/test_pack_streaming.py
@@ -8,14 +8,57 @@ from lib import (
)
from gitdb.db.pack import PackedDB
+from gitdb.stream import NullStream
+from gitdb.pack import PackEntity
import os
import sys
from time import time
+from nose import SkipTest
+
+class CountedNullStream(NullStream):
+ __slots__ = '_bw'
+ def __init__(self):
+ self._bw = 0
+
+ def bytes_written(self):
+ return self._bw
+
+ def write(self, d):
+ self._bw += NullStream.write(self, d)
+
class TestPackStreamingPerformance(TestBigRepoR):
+ def test_pack_writing(self):
+ # see how fast we can write a pack from object streams.
+ # This will not be fast, as we take time for decompressing the streams as well
+ ostream = CountedNullStream()
+ pdb = PackedDB(os.path.join(self.gitrepopath, "objects/pack"))
+
+ ni = 5000
+ count = 0
+ total_size = 0
+ st = time()
+ objs = list()
+ for sha in pdb.sha_iter():
+ count += 1
+ objs.append(pdb.stream(sha))
+ if count == ni:
+ break
+ #END gather objects for pack-writing
+ elapsed = time() - st
+ print >> sys.stderr, "PDB Streaming: Got %i streams by sha in in %f s ( %f streams/s )" % (ni, elapsed, ni / elapsed)
+
+ st = time()
+ PackEntity.write_pack(objs, ostream.write)
+ elapsed = time() - st
+ total_kb = ostream.bytes_written() / 1000
+ print >> sys.stderr, "PDB Streaming: Wrote pack of size %i kb in %f s (%f kb/s)" % (total_kb, elapsed, total_kb/elapsed)
+
+
def test_stream_reading(self):
+ raise SkipTest()
pdb = PackedDB(os.path.join(self.gitrepopath, "objects/pack"))
# streaming only, meant for --with-profile runs