7 Commits
Author SHA1 Message Date
Patrick McCarty 6127bd6095 Release v1.0.3
This release fixes a bug that results in significantly improved diff
creation times in certain cases.

Signed-off-by: Patrick McCarty <patrick.mccarty@intel.com>
2018-02-22 18:04:24 -08:00
Patrick McCarty 85b1c5345b Find the longest match in the suffix-sorted array
Fixes #2

The current implementation performs a binary search through the
suffix-sorted array to find byte sequence matches between old and new
files. However, the algorithm is not optimal when it repeatedly matches
very short strings, leading to performance issues as reported in issue
 #2.

This commit changes the algorithm to consider *all* matches encountered
during the binary search and choose the longest of these matches.

The overall performance impact is yet to be determined, but it appears
to yield a small percentage increase in diff creation time (expected)
and a large percentage *decrease* for the case of diffing the files from
issue #2. In my testing, the creation time there decreases from approx
64 minutes to 4.5 seconds.

Signed-off-by: Patrick McCarty <patrick.mccarty@intel.com>
2018-02-22 18:01:17 -08:00
Patrick McCarty d039492824 Fix code style
Signed-off-by: Patrick McCarty <patrick.mccarty@intel.com>
2017-06-12 11:33:09 -07:00
Patrick McCarty 43a817a2e5 build: add 'compliant' makefile target for fixing code style
Signed-off-by: Patrick McCarty <patrick.mccarty@intel.com>
2017-06-12 11:32:33 -07:00
Patrick McCarty 4e868b4772 Split suffix sort code into a separate source file
To prepare for the possibility of testing other suffix sort algorithms
in the future, split this code into a separate source file for clarity.

Signed-off-by: Patrick McCarty <patrick.mccarty@intel.com>
2017-06-12 11:24:09 -07:00
Patrick McCarty 71b9d7e78a Allow for skipping valgrind use in tests
In case the system valgrind is not functioning properly, allow the test
suite to run without using the tool by setting SKIP_VALGRIND=1 for `make
check`.

Signed-off-by: Patrick McCarty <patrick.mccarty@intel.com>
2016-12-13 10:07:22 -08:00
Patrick McCarty b46a4e2c0f Run clang-format to fix code style
Signed-off-by: Patrick McCarty <patrick.mccarty@intel.com>
2016-12-13 09:49:33 -08:00
8 changed files with 293 additions and 207 deletions
+23 -1
View File
@@ -67,7 +67,8 @@ lib_LTLIBRARIES = \
libbsdiff_la_SOURCES = \
src/diff.c \
src/patch.c
src/patch.c \
src/sufsort.c
libbsdiff_la_LIBADD = \
$(zlib_LIBS)
@@ -146,6 +147,27 @@ dist_check_SCRIPTS = \
test/run.sh
endif
compliant:
@git diff --quiet --exit-code include src; ret=$$?; \
if [ $$ret -eq 1 ]; then \
echo "Error: can only check code style when include/ and src/ are clean."; \
echo "Stash or commit your changes and try again."; \
exit $$ret; \
elif [ $$ret -gt 1 ]; then \
exit $$ret; \
fi; \
clang-format -i -style=file include/*.h src/*.c; ret=$$?; \
if [ $$ret -ne 0 ]; then \
exit $$ret; \
fi; \
git diff --quiet --exit-code include src; ret=$$?; \
if [ $$ret -eq 1 ]; then \
echo "Code style issues found. Run 'git diff' to view issues."; \
elif [ $$ret -eq 0 ]; then \
echo "No code style issues found."; \
fi; \
exit $$ret
release:
@git rev-parse v$(PACKAGE_VERSION) &> /dev/null; \
if [ "$$?" -eq 0 ]; then \
+1 -1
View File
@@ -1,5 +1,5 @@
AC_PREREQ([2.66])
AC_INIT([bsdiff], [1.0.2], [patrick.mccarty@intel.com])
AC_INIT([bsdiff], [1.0.3], [patrick.mccarty@intel.com])
AC_CONFIG_MACRO_DIR([m4])
AC_PROG_CC
AC_PROG_CC_STDC
+4 -1
View File
@@ -2,6 +2,7 @@
#define __INCLUDE_GUARD_BSHEADER_H
#include <stdint.h>
#include <sys/types.h> // for u_char
#include "bsdiff.h"
@@ -58,7 +59,7 @@ struct header_v20 {
uint64_t extra_length;
uint64_t old_file_length;
uint64_t new_file_length;
uint64_t mtime; /* unused */
uint64_t mtime; /* unused */
uint32_t file_mode;
uint32_t file_owner;
uint32_t file_group;
@@ -177,4 +178,6 @@ static inline int eblock_get_enc(enc_flags_t enc)
}
}
int qsufsort(int64_t *, int64_t *, u_char *, int64_t);
#endif
+48 -185
View File
@@ -32,8 +32,6 @@ __FBSDID
#define _GNU_SOURCE
#include "config.h"
#include <sys/types.h>
#ifdef BSDIFF_WITH_BZIP2
#include <bzlib.h>
#endif
@@ -45,20 +43,20 @@ __FBSDID
#include <lzma.h>
#endif
#include <assert.h>
#include <endian.h>
#include <grp.h>
#include <pthread.h>
#include <pwd.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/mman.h>
#include <sys/stat.h>
#include <sys/types.h>
#include <unistd.h>
#include <zlib.h>
#include <endian.h>
#include <stdint.h>
#include <sys/types.h>
#include <sys/stat.h>
#include <pwd.h>
#include <grp.h>
#include <pthread.h>
#include <assert.h>
#include <sys/mman.h>
#include "bsheader.h"
@@ -76,172 +74,6 @@ static int bsdiff_fulldl;
#undef MIN
#define MIN(x, y) (((x) < (y)) ? (x) : (y))
/* NOTES:
* I and V are chunks of memory (arrays) with length = (oldfile size +1) * sizeof(int64_t).
* Additionally, we pass in arraylen now. The parent function qsufsort receives it, so it
* should be available here as well for error checking.
* start: is actually the point in the array sent in during the suffix sort, which sorts by
* small blocks/chunks.
* len: refers to the length of the current chunk being processed - NOT the array length(s).
* h: will never be more than 8, and increases by *2 during suffix sort (h += h) */
static void split(int64_t *I, int64_t *V, int64_t arraylen, int64_t start, int64_t len,
int64_t h)
{
int64_t i, j, k, x, tmp, jj, kk;
if (len < 16) {
for (k = start; k < start + len; k += j) {
j = 1;
x = V[I[k] + h];
for (i = 1; k + i < start + len; i++) {
if (V[I[k + i] + h] < x) {
x = V[I[k + i] + h];
j = 0;
}
if (V[I[k + i] + h] == x) {
tmp = I[k + j];
I[k + j] = I[k + i];
I[k + i] = tmp;
j++;
}
}
for (i = 0; i < j; i++) {
V[I[k + i]] = k + j - 1;
}
if (j == 1) {
I[k] = -1;
}
}
return;
}
x = V[I[start + len / 2] + h];
jj = 0;
kk = 0;
for (i = start; i < start + len; i++) {
if (V[I[i] + h] < x) {
jj++;
}
if (V[I[i] + h] == x) {
kk++;
}
}
jj += start;
kk += jj;
i = start;
j = 0;
k = 0;
while (i < jj) {
if (V[I[i] + h] < x) {
i++;
} else if (V[I[i] + h] == x) {
tmp = I[i];
I[i] = I[jj + j];
I[jj + j] = tmp;
j++;
} else {
tmp = I[i];
I[i] = I[kk + k];
I[kk + k] = tmp;
k++;
}
}
while (jj + j < kk) {
if (V[I[jj + j] + h] == x) {
j++;
} else {
tmp = I[jj + j];
I[jj + j] = I[kk + k];
I[kk + k] = tmp;
k++;
}
}
if (jj > start) {
split(I, V, arraylen, start, jj - start, h);
}
for (i = 0; i < kk - jj; i++) {
V[I[jj + i]] = kk - 1;
}
if (jj == kk - 1) {
I[jj] = -1;
}
if (start + len > kk) {
split(I, V, arraylen, kk, start + len - kk, h);
}
}
/* The old_data (previous file data) is passed into this suffix sort and sorted
* accordingly using the I and V arrays, which are both of length oldsize +1. */
static int qsufsort(int64_t *I, int64_t *V, u_char *old, int64_t oldsize)
{
int64_t buckets[QSUF_BUCKET_SIZE];
int64_t i, h, len;
for (i = 0; i < QSUF_BUCKET_SIZE; i++) {
buckets[i] = 0;
}
for (i = 0; i < oldsize; i++) {
buckets[old[i]]++;
}
for (i = 1; i < QSUF_BUCKET_SIZE; i++) {
buckets[i] += buckets[i - 1];
}
for (i = QSUF_BUCKET_SIZE - 1; i > 0; i--) {
buckets[i] = buckets[i - 1];
}
buckets[0] = 0;
for (i = 0; i < oldsize; i++) {
if (buckets[old[i]] > oldsize + 1) {
return -1;
}
I[++buckets[old[i]]] = i;
}
for (i = 0; i < oldsize; i++) {
V[i] = buckets[old[i]];
}
V[oldsize] = 0;
for (i = 1; i < QSUF_BUCKET_SIZE; i++) {
if (buckets[i] == buckets[i - 1] + 1) {
I[buckets[i]] = -1;
}
}
I[0] = -1;
for (h = 1; I[0] != -(oldsize + 1); h += h) {
len = 0;
for (i = 0; i < oldsize + 1;) {
if (I[i] < 0) {
len -= I[i];
i -= I[i];
} else {
if (len) {
I[i - len] = -len;
}
len = V[I[i]] + 1 - i;
split(I, V, oldsize, i, len, h);
i += len;
len = 0;
}
}
if (len) {
I[i - len] = -len;
}
}
for (i = 0; i < oldsize + 1; i++) {
I[V[i]] = i;
}
return 0;
}
static int64_t matchlen(u_char *old, int64_t oldsize, u_char *new,
int64_t newsize)
{
@@ -256,27 +88,58 @@ static int64_t matchlen(u_char *old, int64_t oldsize, u_char *new,
return i;
}
int64_t max_len = 0;
/**
* Finds the longest matching array of bytes between the OLD and NEW file. The
* old file is suffix-sorted; the suffix-sorted array is stored at I, and
* indices to search between are indicated by ST (start) and EN (end). Returns
* the length of the match, and POS is updated to the position of the match
* within OLD.
*/
static int64_t search(int64_t *I, u_char *old, int64_t oldsize,
u_char *new, int64_t newsize, int64_t st, int64_t en,
int64_t *pos)
{
int64_t x, y;
/* Initialize max_len for the binary search */
if (st == 0 && en == oldsize) {
max_len = matchlen(old, oldsize, new, newsize);
*pos = I[st];
}
/* The binary search terminates here when "en" and "st" are adjacent
* indices in the suffix-sorted array. */
if (en - st < 2) {
x = matchlen(old + I[st], oldsize - I[st], new, newsize);
y = matchlen(old + I[en], oldsize - I[en], new, newsize);
if (x > y) {
if (x > max_len) {
max_len = x;
*pos = I[st];
return x;
} else {
*pos = I[en];
return y;
}
y = matchlen(old + I[en], oldsize - I[en], new, newsize);
if (y > max_len) {
max_len = y;
*pos = I[en];
}
return max_len;
}
x = st + (en - st) / 2;
if (memcmp(old + I[x], new, MIN(oldsize - I[x], newsize)) < 0) {
int64_t length = MIN(oldsize - I[x], newsize);
u_char *oldoffset = old + I[x];
/* This match *could* be the longest one, so check for that here */
int64_t tmp = matchlen(oldoffset, length, new, length);
if (tmp > max_len) {
max_len = tmp;
*pos = I[x];
}
/* Determine how to continue the binary search */
if (memcmp(oldoffset, new, length) < 0) {
return search(I, old, oldsize, new, newsize, x, en, pos);
} else {
return search(I, old, oldsize, new, newsize, st, x, pos);
+3 -3
View File
@@ -30,12 +30,12 @@
*/
#define _GNU_SOURCE
#include <assert.h>
#include <stdio.h>
#include <stdlib.h>
#include <unistd.h>
#include <string.h>
#include <assert.h>
#include <time.h>
#include <unistd.h>
#include "bsdiff.h"
#include "bsheader.h"
@@ -110,7 +110,7 @@ static void print_v20_header(struct header_v20 *h, FILE *f)
if (h->mtime == 0) {
printf("Mtime:\t(not set, as expected)\n");
} else {
printf("Mtime:\t%s (probably means there is a bug)\n", ctime((const time_t*)&h->mtime));
printf("Mtime:\t%s (probably means there is a bug)\n", ctime((const time_t *)&h->mtime));
}
printf("Mode:\t%4o\n", h->file_mode);
printf("Uid:\t%d\n", h->file_owner);
+14 -14
View File
@@ -44,23 +44,23 @@ __FBSDID
#include <lzma.h>
#endif
#include <stdlib.h>
#include <stdio.h>
#include <string.h>
#include <unistd.h>
#include <zlib.h>
#include <stdint.h>
#include <sys/types.h>
#include <sys/stat.h>
#include <pwd.h>
#include <grp.h>
#include <fcntl.h>
#include <limits.h>
#include <linux/fs.h>
#include <assert.h>
#include <endian.h>
#include <errno.h>
#include <fcntl.h>
#include <grp.h>
#include <limits.h>
#include <linux/fs.h>
#include <pwd.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/mman.h>
#include <sys/stat.h>
#include <sys/types.h>
#include <unistd.h>
#include <zlib.h>
#include "bsheader.h"
@@ -259,7 +259,7 @@ typedef struct {
#ifdef BSDIFF_WITH_BZIP2
BZFILE *bz2; /* method = BZIP2 */
#endif
gzFile gz; /* method = GZIP */
gzFile gz; /* method = GZIP */
#ifdef BSDIFF_WITH_LZMA
xzfile *xz; /* method = XZ */
#endif
+193
View File
@@ -0,0 +1,193 @@
/*-
* Copyright 2003-2005 Colin Percival
* All rights reserved
*
* Redistribution and use in source and binary forms, with or without
* modification, are permitted providing that the following conditions
* are met:
* 1. Redistributions of source code must retain the above copyright
* notice, this list of conditions and the following disclaimer.
* 2. Redistributions in binary form must reproduce the above copyright
* notice, this list of conditions and the following disclaimer in the
* documentation and/or other materials provided with the distribution.
*
* THIS SOFTWARE IS PROVIDED BY THE AUTHOR ``AS IS'' AND ANY EXPRESS OR
* IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE IMPLIED
* WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR PURPOSE
* ARE DISCLAIMED. IN NO EVENT SHALL THE AUTHOR BE LIABLE FOR ANY
* DIRECT, INDIRECT, INCIDENTAL, SPECIAL, EXEMPLARY, OR CONSEQUENTIAL
* DAMAGES (INCLUDING, BUT NOT LIMITED TO, PROCUREMENT OF SUBSTITUTE GOODS
* OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR BUSINESS INTERRUPTION)
* HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER IN CONTRACT,
* STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING
* IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
* POSSIBILITY OF SUCH DAMAGE.
*/
#include "bsheader.h"
/* NOTES:
* I and V are chunks of memory (arrays) with length = (oldfile size +1) * sizeof(int64_t).
* Additionally, we pass in arraylen now. The parent function qsufsort receives it, so it
* should be available here as well for error checking.
* start: is actually the point in the array sent in during the suffix sort, which sorts by
* small blocks/chunks.
* len: refers to the length of the current chunk being processed - NOT the array length(s).
* h: will never be more than 8, and increases by *2 during suffix sort (h += h) */
static void split(int64_t *I, int64_t *V, int64_t arraylen, int64_t start, int64_t len,
int64_t h)
{
int64_t i, j, k, x, tmp, jj, kk;
if (len < 16) {
for (k = start; k < start + len; k += j) {
j = 1;
x = V[I[k] + h];
for (i = 1; k + i < start + len; i++) {
if (V[I[k + i] + h] < x) {
x = V[I[k + i] + h];
j = 0;
}
if (V[I[k + i] + h] == x) {
tmp = I[k + j];
I[k + j] = I[k + i];
I[k + i] = tmp;
j++;
}
}
for (i = 0; i < j; i++) {
V[I[k + i]] = k + j - 1;
}
if (j == 1) {
I[k] = -1;
}
}
return;
}
x = V[I[start + len / 2] + h];
jj = 0;
kk = 0;
for (i = start; i < start + len; i++) {
if (V[I[i] + h] < x) {
jj++;
}
if (V[I[i] + h] == x) {
kk++;
}
}
jj += start;
kk += jj;
i = start;
j = 0;
k = 0;
while (i < jj) {
if (V[I[i] + h] < x) {
i++;
} else if (V[I[i] + h] == x) {
tmp = I[i];
I[i] = I[jj + j];
I[jj + j] = tmp;
j++;
} else {
tmp = I[i];
I[i] = I[kk + k];
I[kk + k] = tmp;
k++;
}
}
while (jj + j < kk) {
if (V[I[jj + j] + h] == x) {
j++;
} else {
tmp = I[jj + j];
I[jj + j] = I[kk + k];
I[kk + k] = tmp;
k++;
}
}
if (jj > start) {
split(I, V, arraylen, start, jj - start, h);
}
for (i = 0; i < kk - jj; i++) {
V[I[jj + i]] = kk - 1;
}
if (jj == kk - 1) {
I[jj] = -1;
}
if (start + len > kk) {
split(I, V, arraylen, kk, start + len - kk, h);
}
}
/* The old_data (previous file data) is passed into this suffix sort and sorted
* accordingly using the I and V arrays, which are both of length oldsize +1. */
int qsufsort(int64_t *I, int64_t *V, u_char *old, int64_t oldsize)
{
int64_t buckets[QSUF_BUCKET_SIZE];
int64_t i, h, len;
for (i = 0; i < QSUF_BUCKET_SIZE; i++) {
buckets[i] = 0;
}
for (i = 0; i < oldsize; i++) {
buckets[old[i]]++;
}
for (i = 1; i < QSUF_BUCKET_SIZE; i++) {
buckets[i] += buckets[i - 1];
}
for (i = QSUF_BUCKET_SIZE - 1; i > 0; i--) {
buckets[i] = buckets[i - 1];
}
buckets[0] = 0;
for (i = 0; i < oldsize; i++) {
if (buckets[old[i]] > oldsize + 1) {
return -1;
}
I[++buckets[old[i]]] = i;
}
for (i = 0; i < oldsize; i++) {
V[i] = buckets[old[i]];
}
V[oldsize] = 0;
for (i = 1; i < QSUF_BUCKET_SIZE; i++) {
if (buckets[i] == buckets[i - 1] + 1) {
I[buckets[i]] = -1;
}
}
I[0] = -1;
for (h = 1; I[0] != -(oldsize + 1); h += h) {
len = 0;
for (i = 0; i < oldsize + 1;) {
if (I[i] < 0) {
len -= I[i];
i -= I[i];
} else {
if (len) {
I[i - len] = -len;
}
len = V[I[i]] + 1 - i;
split(I, V, oldsize, i, len, h);
i += len;
len = 0;
}
}
if (len) {
I[i - len] = -len;
}
}
for (i = 0; i < oldsize + 1; i++) {
I[V[i]] = i;
}
return 0;
}
+7 -2
View File
@@ -8,10 +8,15 @@ testnum=0
sudo rm -f *.diff *.out
VALGRIND="valgrind -q"
if [ -n "$SKIP_VALGRIND" ]; then
VALGRIND=""
fi
libdir="$abs_builddir/.libs"
ldpath="LD_LIBRARY_PATH=$libdir"
BSDIFF="sudo $ldpath valgrind -q $libdir/bsdiff"
BSPATCH="sudo $ldpath valgrind -q $libdir/bspatch"
BSDIFF="sudo $ldpath $VALGRIND $libdir/bsdiff"
BSPATCH="sudo $ldpath $VALGRIND $libdir/bspatch"
# If exit status is 0, the test succeeded. Else it failed.
check_success() {