Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 18 additions & 1 deletion Makefile
Original file line number Diff line number Diff line change
Expand Up @@ -6,10 +6,12 @@ ifeq ($(UNAME),Darwin)
CLANG = /opt/homebrew/opt/llvm/bin/clang
SED_INPLACE = sed -i ''
BASE64 = base64 -i uncompress.wasm -o uncompress.wasm.base64
BASE64_COMP = base64 -i compress.wasm -o compress.wasm.base64
else
CLANG = clang
SED_INPLACE = sed -i
BASE64 = base64 -w 0 uncompress.wasm > uncompress.wasm.base64
BASE64_COMP = base64 -w 0 compress.wasm > compress.wasm.base64
endif

uncompress.wasm.base64: uncompress.wasm
Expand All @@ -24,8 +26,23 @@ uncompress.wasm: c/uncompress.c
-Wl,--no-entry \
-o uncompress.wasm c/uncompress.c

compress.wasm.base64: compress.wasm
$(BASE64_COMP)
$(SED_INPLACE) 's|const wasm64 = .*|const wasm64 = '"'`cat compress.wasm.base64`'"'|' js/compress.js

compress.wasm: c/compress.c
$(CLANG) --target=wasm32 \
-O3 \
-nostdlib \
-Wl,--export-all \
-Wl,--no-entry \
-o compress.wasm c/compress.c

main: c/main.c c/uncompress.c
$(CLANG) -g -o main c/main.c c/uncompress.c

bench: compress.wasm.base64 uncompress.wasm.base64
node benchmark.js

clean:
rm -f main uncompress.wasm uncompress.wasm.base64
rm -f main uncompress.wasm uncompress.wasm.base64 compress.wasm compress.wasm.base64
97 changes: 53 additions & 44 deletions benchmark.js
Original file line number Diff line number Diff line change
@@ -1,52 +1,61 @@

import { compress } from 'snappyjs'
import { snappyCompressor } from './js/compress.js'
import { snappyUncompressor } from './js/uncompress.js'

const fileSize = 200_000_000
// Generate synthetic LLM completion JSON
const completions = []
const sentences = [
'The function processes the input data and returns a transformed result that can be used by downstream components in the pipeline for further analysis and visualization.',
'First, we need to validate the parameters before proceeding with the operation to ensure that all required fields are present and conform to the expected schema definitions.',
'This approach improves performance by caching intermediate computations in a hash table, allowing subsequent requests with similar parameters to bypass expensive recalculations entirely.',
'The algorithm iterates through each element and applies the transformation using a map-reduce pattern that enables efficient parallel processing across multiple CPU cores when available.',
'Error handling is implemented to gracefully manage unexpected conditions, including network timeouts, malformed responses, and resource exhaustion scenarios that may occur during execution.',
'The response includes metadata about the processing time and token usage, which can be used for billing purposes, performance monitoring, and capacity planning in production environments.',
'Consider using async operations for better throughput in production, especially when dealing with I/O-bound workloads such as database queries, API calls, and file system operations.',
'The configuration supports multiple output formats including JSON and XML, with additional options for CSV export, Protocol Buffers serialization, and custom format handlers defined by plugins.',
]

const compressed = time(`generate and compress ${fileSize.toLocaleString()}`, () => {
// Generate input array with random data
const input = new Uint8Array(fileSize)
for (let i = 0; i < fileSize; i++) {
input[i] = Math.floor(Math.random() * 16)
// Create 50 completions with some repeated patterns
for (let i = 0; i < 50; i++) {
const content = []
// conversation length
for (let j = 0; j < 20; j++) {
content.push(sentences[(i + j) % sentences.length])
}
return compress(input)
})
console.log(`compressed ${fileSize.toLocaleString()} bytes to ${compressed.length.toLocaleString()} bytes`)
completions.push({
id: `chatcmpl-${i.toString(36).padStart(6, '0')}`,
object: 'chat.completion',
created: 1700000000 + i,
model: 'gpt-4-turbo',
choices: [{
index: 0,
message: { role: 'assistant', content: content.join(' ') },
finish_reason: 'stop',
}],
usage: { prompt_tokens: 150 + i, completion_tokens: 200 + i, total_tokens: 350 + i * 2 },
})
}

const snappyUncompress = snappyUncompressor()
timeWithStdDev('uncompress wasm', () => snappyUncompress(compressed, fileSize))
const input = new TextEncoder().encode(JSON.stringify(completions))
console.log('Input:', (input.length / 1024).toFixed(1), 'KB')

// time('uncompress snappyjs', () => uncompress(compressed, fileSize))
// Create compressor and decompressor instances
const compress = snappyCompressor()
const uncompress = snappyUncompressor()

/**
* @param {string} name
* @param {() => any} fn
* @returns {any}
*/
function time(name, fn) {
const start = performance.now()
const output = fn()
const ms = performance.now() - start
console.log(`${name} took ${ms} ms`)
return output
}
// Get compressed size
const compressed = compress(input)
console.log('Compressed:', (compressed.length / 1024).toFixed(1), 'KB (' + (compressed.length / input.length * 100).toFixed(1) + '%)')

/**
* @param {string} name
* @param {() => void} fn
* @param {number} iterations
*/
function timeWithStdDev(name, fn, iterations = 20) {
const times = []
for (let i = 0; i < iterations; i++) {
const start = performance.now()
fn()
const ms = performance.now() - start
times.push(ms)
}
const mean = times.reduce((a, b) => a + b, 0) / times.length
const variance = times.reduce((a, b) => a + (b - mean) ** 2, 0) / times.length
const stdDev = Math.sqrt(variance)
console.log(`${name} took ${mean} ms ± ${stdDev} ms`)
}
// Benchmark compression
const compIters = 1000
const compStart = performance.now()
for (let i = 0; i < compIters; i++) compress(input)
const compTime = performance.now() - compStart
console.log('Compress:', (input.length * compIters / 1024 / 1024 / (compTime / 1000)).toFixed(0), 'MB/s')

// Benchmark decompression
const decIters = 1000
const decStart = performance.now()
for (let i = 0; i < decIters; i++) uncompress(compressed, input.length)
const decTime = performance.now() - decStart
console.log('Decompress:', (input.length * decIters / 1024 / 1024 / (decTime / 1000)).toFixed(0), 'MB/s')
187 changes: 187 additions & 0 deletions c/compress.c
Original file line number Diff line number Diff line change
@@ -0,0 +1,187 @@
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>

void *memcpy(void *dest, const void *src, size_t n) {
char *d = dest;
const char *s = src;
while (n--) {
*d++ = *s++;
}
return dest;
}

/* Unaligned load macros for fast comparison */
#define LOAD32(p) ({ uint32_t _v; memcpy(&_v, (p), 4); _v; })
#define LOAD64(p) ({ uint64_t _v; memcpy(&_v, (p), 8); _v; })

/* Larger hash table for fewer collisions = faster lookups */
#define HASH_BITS 14
#define HASH_SIZE (1 << HASH_BITS)

/* Fast hash using multiply-shift - better distribution */
static inline uint32_t hash(const unsigned char *ptr) {
return (LOAD32(ptr) * 0x1e35a7bd) >> (32 - HASH_BITS);
}

/* Write varint to output, return bytes written */
static inline int write_varint(unsigned char *op, uint32_t val) {
unsigned char *start = op;
while (val >= 0x80) {
*op++ = (val & 0x7f) | 0x80;
val >>= 7;
}
*op++ = val;
return op - start;
}

/* Emit literal bytes, return bytes written */
static int emit_literal(unsigned char *op, const unsigned char *literal, uint32_t len) {
unsigned char *start = op;
if (len <= 60) {
*op++ = ((len - 1) << 2);
} else if (len < 256) {
*op++ = (60 << 2);
*op++ = len - 1;
} else {
*op++ = (61 << 2);
*op++ = (len - 1) & 0xff;
*op++ = ((len - 1) >> 8) & 0xff;
}
memcpy(op, literal, len);
return (op - start) + len;
}

/* Emit copy operation, return bytes written */
static inline int emit_copy(unsigned char *op, uint32_t offset, uint32_t len) {
unsigned char *start = op;

/* Emit copies in chunks of max 64 bytes for COPY_2, 11 for COPY_1 */
while (len > 0) {
if (offset < 2048 && len >= 4 && len < 12) {
/* COPY_1: 2 bytes, offset 0-2047, len 4-11 */
*op++ = ((len - 4) << 2) | ((offset >> 8) << 5) | 0x01;
*op++ = offset & 0xff;
return op - start;
} else {
/* COPY_2: 3 bytes, offset 0-65535, len 1-64 */
uint32_t copy_len = len > 64 ? 64 : len;
*op++ = ((copy_len - 1) << 2) | 0x02;
*op++ = offset & 0xff;
*op++ = (offset >> 8) & 0xff;
len -= copy_len;
}
}

return op - start;
}

/* Find match length using 8-byte chunks */
static inline uint32_t find_match_length(const unsigned char *s1, const unsigned char *s2,
const unsigned char *s2_end) {
const unsigned char *s2_start = s2;

/* Compare 8 bytes at a time */
while (s2 + 8 <= s2_end) {
uint64_t a = LOAD64(s1);
uint64_t b = LOAD64(s2);
if (a != b) {
/* Find first differing byte using XOR and count trailing zeros */
uint64_t diff = a ^ b;
/* Count trailing zero bytes (little-endian) */
uint32_t matching = __builtin_ctzll(diff) >> 3;
return (s2 - s2_start) + matching;
}
s1 += 8;
s2 += 8;
}

/* Handle remaining bytes */
while (s2 < s2_end && *s1 == *s2) {
s1++;
s2++;
}

return s2 - s2_start;
}

/**
* Compress data using Snappy algorithm.
* Returns the compressed size, or negative on error.
*
* Note: output buffer must be at least input_length + input_length/6 + 32 bytes.
*/
int compress(const char *input, size_t input_length, char *output) {
const unsigned char *ip = (const unsigned char *)input;
const unsigned char *ip_end = ip + input_length;
unsigned char *op = (unsigned char *)output;

/* Hash table: stores offset of last occurrence */
static uint32_t table[HASH_SIZE];

/* Write uncompressed length as varint */
op += write_varint(op, input_length);

/* Handle empty input */
if (input_length == 0) {
return op - (unsigned char *)output;
}

/* Handle very short input (< 4 bytes can't have matches) */
if (input_length < 4) {
op += emit_literal(op, ip, input_length);
return op - (unsigned char *)output;
}

const unsigned char *lit_start = ip; /* Start of pending literal */
const unsigned char *ip_limit = ip_end - 4; /* Last position we can hash */
const unsigned char *base = ip;

while (ip <= ip_limit) {
uint32_t h = hash(ip);
uint32_t candidate_offset = table[h];
const unsigned char *candidate = base + candidate_offset;

/* Update hash table with current position */
table[h] = ip - base;

/* Check for match using 32-bit comparison */
uint32_t offset = ip - candidate;
if (offset > 0 && offset < 65536 && candidate >= base &&
LOAD32(candidate) == LOAD32(ip)) {

/* Found a match - emit pending literals first */
if (ip > lit_start) {
op += emit_literal(op, lit_start, ip - lit_start);
}

/* Find match length using fast 8-byte comparison */
uint32_t match_len = 4 + find_match_length(candidate + 4, ip + 4, ip_end);

/* Emit copy */
op += emit_copy(op, offset, match_len);

/* Skip matched bytes, updating hash table periodically for better compression */
ip += match_len;

/* Update hash for positions we skipped (improves compression) */
if (match_len > 4 && ip <= ip_limit) {
/* Hash a few positions within the match for future reference */
table[hash(ip - 2)] = (ip - 2) - base;
}

/* Start new literal */
lit_start = ip;
} else {
ip++;
}
}

/* Emit any remaining literals */
uint32_t remaining = ip_end - lit_start;
if (remaining > 0) {
op += emit_literal(op, lit_start, remaining);
}

return op - (unsigned char *)output;
}
4 changes: 0 additions & 4 deletions global.d.ts

This file was deleted.

11 changes: 11 additions & 0 deletions js/compress.d.ts
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
/**
* Compress data using snappy compression.
*
* @param {Uint8Array} input
*/
export function snappyCompress(input: Uint8Array): Uint8Array

/**
* Load wasm and return compressor function.
*/
export function snappyCompressor(): (input: Uint8Array) => Uint8Array
Loading