Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
40 changes: 36 additions & 4 deletions .github/workflows/test.yml
Original file line number Diff line number Diff line change
Expand Up @@ -7,11 +7,15 @@ env:

jobs:
ubuntu:
name: Test ${{matrix.os}}
name: Test ${{matrix.os}} ${{matrix.build_type}}
runs-on: ${{matrix.os}}
strategy:
matrix:
os: [ubuntu-22.04, ubuntu-24.04, ubuntu-26.04]
# Release defines NDEBUG, which makes every ASSERT() a no-op. Testing
# only Debug cannot catch a defect that an assertion was masking, and
# Release is what actually ships.
build_type: [Debug, Release]

steps:
- uses: actions/checkout@v7
Expand All @@ -26,17 +30,45 @@ jobs:
- name: Configure CMake
shell: bash
working-directory: ${{github.workspace}}/build
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=$BUILD_TYPE
run: cmake $GITHUB_WORKSPACE -DCMAKE_BUILD_TYPE=${{matrix.build_type}}

- name: Build
working-directory: ${{github.workspace}}/build
shell: bash
run: cmake --build . --config $BUILD_TYPE
run: cmake --build . --config ${{matrix.build_type}}

- name: Test
working-directory: ${{github.workspace}}/build
shell: bash
run: ctest -C $BUILD_TYPE
run: ctest -C ${{matrix.build_type}} --output-on-failure

sanitizers:
name: Test sanitizers
runs-on: ubuntu-24.04

steps:
- uses: actions/checkout@v7

- name: Install dependencies
shell: bash
run: sudo apt update && sudo apt install libcunit1-dev --yes

- name: Build and run the test suite under ASan, UBSan and LeakSanitizer
shell: bash
run: |
gcc -std=c11 -g -O1 -fsanitize=address,undefined \
-fno-omit-frame-pointer -fno-sanitize-recover=all \
-Iinclude -Isrc src/*.c unit_tests.c -o unit_tests_asan -lcunit
ASAN_OPTIONS=detect_leaks=1 ./unit_tests_asan

- name: Build the library with warnings as errors
shell: bash
run: |
for f in src/*.c; do \
gcc -c -std=c11 -O2 -Werror -Wall -Wextra -Wshadow -Wcast-qual \
-Wstrict-prototypes -Wmissing-prototypes -Wsign-compare \
-Wpointer-arith -Iinclude -Isrc "$f" -o /dev/null; \
done

debian:
name: Test debian ${{matrix.os}}
Expand Down
10 changes: 6 additions & 4 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -13,13 +13,15 @@ libdict is a C library that provides the following data structures with efficien
* [weight-balanced tree](https://en.wikipedia.org/wiki/Weight-balanced_tree)
* [path-reduction tree](https://cs.uwaterloo.ca/research/tr/1982/CS-82-07.pdf)
* [treap](http://en.wikipedia.org/wiki/Treap)
* [hashtable using separate chaining](http://en.wikipedia.org/wiki/Hashtable#Separate_chaining)
* [hashtable using open addressing with linear probing](http://en.wikipedia.org/wiki/Hashtable#Open_addressing)
* [hashtable using separate chaining](http://en.wikipedia.org/wiki/Hash_table#Separate_chaining)
* [hashtable using open addressing with linear probing](http://en.wikipedia.org/wiki/Hash_table#Open_addressing)
* [skip list](https://en.wikipedia.org/wiki/Skip_list)

All data structures in this library support insert, search, and remove, and have bidirectional iterators. The sorted data structures (everything but hash tables) support near-search operations: searching for the key greater or equal to, strictly greater than, lesser or equal to, or strictly less than, a given key. The tree data structures also support the selecting the nth element; this takes linear time, except in path-reduction and weight-balanced trees, where it only takes logarithmic time.
All data structures in this library support insert, search, and remove, and have bidirectional iterators. The sorted data structures (everything but hash tables) support near-search operations: searching for the key greater or equal to, strictly greater than, lesser or equal to, or strictly less than, a given key. The tree data structures also support selecting the nth element; this takes linear time, except in path-reduction and weight-balanced trees, where it only takes logarithmic time.

The API and code are written with efficiency as a primary concern. For example, an insert call returns a boolean indicating whether or not the key was already present in the dictionary (i.e. whether there was an insertion or a collision), and a pointer to the location of the associated data. Thus, an insert-or-update operation can be supported with a single traversal of the data structure. In addition, almost all recursive algorithms have been rewritten to use iteration instead.
Iterator `remove` and `compare` are supported on all containers, including the hash tables.

The API and code are written with efficiency as a primary concern. For example, an insert call returns a boolean (`dict_insert_result.inserted`) indicating whether or not the key was inserted, i.e. `true` when the key was not already present and `false` when an existing entry was found, and a pointer to the location of the associated data. Thus, an insert-or-update operation can be supported with a single traversal of the data structure. In addition, almost all recursive algorithms have been rewritten to use iteration instead.

Documentation is generated by Doxygen on every commit to master and is available [here](https://rtbrick.github.io/libdict/html).

Expand Down
4 changes: 2 additions & 2 deletions TODO
Original file line number Diff line number Diff line change
@@ -1,7 +1,7 @@
[ ] Use restrict keyword wherever appropriate.
[X] Fix skiplist prev pointers.
[ ] Implement incomplete functionality, e.g. iterator remove & compare.
[X] Reformat to 80 columns.
[X] Implement incomplete functionality, e.g. iterator remove & compare.
[ ] Reformat to 80 columns.
[ ] Optimize double-rotations.
[X] Fix bugs in weight balanced tree.
[ ] Optimize skiplist.
51 changes: 41 additions & 10 deletions anagram.c
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,15 @@ struct WordList {
WordList *next;
};

/* The tree owns the keys; the WordList data are freed while iterating, so
* only the key is released here. */
static void
key_free(void *key, void *datum)
{
(void)datum;
free(key);
}

int
main(int argc, char *argv[])
{
Expand All @@ -29,37 +38,59 @@ main(int argc, char *argv[])
exit(1);
}

dict_malloc_func = xmalloc;

rb_tree *tree = rb_tree_new(dict_str_cmp);

char buf[512];
while (fgets(buf, sizeof(buf), fp)) {
if (isupper(buf[0])) /* Disregard proper nouns. */
if (isupper((unsigned char) buf[0])) /* Disregard proper nouns. */
continue;

strtok(buf, "\r\n");
int freq[256] = { 0 };
memset(freq, 0, sizeof(freq));

ASSERT(buf[0] != '\0');

for (char *p = buf; *p; p++)
freq[tolower(*p)]++;
freq[tolower((unsigned char) *p)]++;

/* The signature encodes each letter count as a single digit, so a
* word with ten or more of the same letter cannot be represented
* unambiguously; skip it rather than mis-group it. */
int representable = 1;
for (int i = 1; i < 256; i++) {
if (freq[i] > 9) {
representable = 0;
break;
}
}
if (!representable) {
fprintf(stderr, "Skipping '%s': letter repeated 10 or more times.\n",
buf);
continue;
}

char name[1024];
char *p = name;
for (int i=1; i<256; i++) {
if (freq[i]) {
ASSERT(freq[i] < 10);

*p++ = (char) i;
*p++ = '0' + (char) freq[i];
}
}
*p = 0;

char *key = xstrdup(name);
dict_insert_result result = rb_tree_insert(tree, key);
if (!result.datum_ptr) {
free(key);
fprintf(stderr, "Insertion failed\n");
exit(1);
}
if (!result.inserted)
free(key); /* The tree kept the pre-existing key. */
WordList* word = xmalloc(sizeof(*word));
word->word = xstrdup(buf);
WordList** wordp = (WordList**) rb_tree_insert(tree, xstrdup(name)).datum_ptr;
WordList** wordp = (WordList**) result.datum_ptr;
word->next = *wordp;
*wordp = word;
}
Expand Down Expand Up @@ -87,9 +118,9 @@ main(int argc, char *argv[])
free(word);
word = next;
}
} while (rb_itor_next(itor));
}
rb_itor_free(itor);
rb_tree_free(tree, NULL);
rb_tree_free(tree, key_free);
fclose(fp);

return 0;
Expand Down
Loading
Loading