From 898fa1e5e8697222cf5c2cc0623e40f3f29d3679 Mon Sep 17 00:00:00 2001 From: Nikolai Nosov Date: Sat, 25 Jan 2020 17:26:16 +0400 Subject: [PATCH] add dictionary implementation --- framework/data_entry.h | 36 +++++ framework/dictionary.cpp | 71 ++++++++++ framework/dictionary.h | 275 +++++++++++++++++++++++++++++++++++++++ 3 files changed, 382 insertions(+) create mode 100644 framework/data_entry.h create mode 100644 framework/dictionary.cpp create mode 100644 framework/dictionary.h diff --git a/framework/data_entry.h b/framework/data_entry.h new file mode 100644 index 0000000..eb443d3 --- /dev/null +++ b/framework/data_entry.h @@ -0,0 +1,36 @@ +/* + * data_entry.h + * + * Created on: Oct 30, 2013 + * Author: nnosov + * + * Simple representation of data. Pointer to data and size of data. + */ + +#ifndef DATA_ENTRY_H_ +#define DATA_ENTRY_H_ + + +//system +#include + +typedef struct _SC_Data +{ + _SC_Data() + { + data = 0; + size =0; + } + + _SC_Data(const char* _data, uint32_t _size) + { + data = (char*)_data; + size = _size; + } + + char* data; + uint32_t size; +} SC_Data; + + +#endif /* DATA_ENTRY_H_ */ diff --git a/framework/dictionary.cpp b/framework/dictionary.cpp new file mode 100644 index 0000000..738efa5 --- /dev/null +++ b/framework/dictionary.cpp @@ -0,0 +1,71 @@ +/* + * dictionary.cpp + * + * Created on: Oct 28, 2014 + * Author: nnosov + */ +#include "dictionary.h" + +//framework +#include "logger/log.h" + +extern SC_LogBase* logger; + +typedef struct __key_value { + const char* key; + uint32_t keylen; + int value; +} key_value; + +#define KEY_VALUE(key, value) {key, sizeof(key), value} + +key_value utable[] = { + KEY_VALUE("key1", 1), + KEY_VALUE("key2", 2), + KEY_VALUE("key1", 3), +}; + + +bool SC_Dict::UnitTest() +{ + logger->Log(SEVERITY_DEBUG, "%s: starting ...", __PRETTY_FUNCTION__); + logger->Log(SEVERITY_DEBUG, "\t\tshift = %d, hash function = %p, number of buckets = %d, mask = 0x%X. count = %d", mShift, pHashFn, mSize, mMask, mCount); + logger->Log(SEVERITY_DEBUG, "... inserting values into the dictionary ..."); + for(uint32_t i = 0; i < sizeof(utable) / sizeof(key_value); i++) + { + logger->Log(SEVERITY_DEBUG, "\t\tKey = %s, Key size = %d, Key value = %d", utable[i].key, utable[i].keylen, utable[i].value); + Insert(utable[i].key, utable[i].keylen, &(utable[i].value)); + } + logger->Log(SEVERITY_DEBUG, "\t\tdictionary size = %d",mCount); + logger->Log(SEVERITY_DEBUG, "... went through the dictionary sequentially ..."); + for(void* data = First(); data; data = Next()) + { + logger->Log(SEVERITY_DEBUG, "\t\tvalue = %d", *((int*)data)); + } + + logger->Log(SEVERITY_DEBUG, "... went through the dictionary in key order ..."); + for(uint32_t i = 0; i < sizeof(utable) / sizeof(key_value); i++) + { + void* data = Lookup(utable[i].key, utable[i].keylen); + if(data) + logger->Log(SEVERITY_DEBUG, "\t\tKey = %s, Key size = %d, Key value = %d", utable[i].key, utable[i].keylen, *((int*)data)); + } + + void* data = Lookup(utable[0].key, utable[0].keylen); + if(data) + { + logger->Log(SEVERITY_DEBUG, "... removing first key[%s] ...", utable[0].key); + Remove(); + } + + logger->Log(SEVERITY_DEBUG, "... went through the dictionary in key order 2 ..."); + for(uint32_t i = 0; i < sizeof(utable) / sizeof(key_value); i++) + { + void* data = Lookup(utable[i].key, utable[i].keylen); + if(data) + logger->Log(SEVERITY_DEBUG, "\t\tKey = %s, Key size = %d, Key value = %d", utable[i].key, utable[i].keylen, *((int*)data)); + } + + logger->Log(SEVERITY_DEBUG, "%s: finished.", __PRETTY_FUNCTION__); + return true; +} diff --git a/framework/dictionary.h b/framework/dictionary.h new file mode 100644 index 0000000..bbda9be --- /dev/null +++ b/framework/dictionary.h @@ -0,0 +1,275 @@ +/* + * dictionary.h + * + * Created on: Aug 5, 2014 + * Author: nnosov + */ + +#ifndef DICTIONARY_H_ +#define DICTIONARY_H_ + +//system +#include +#include + +//framework +#include "data_entry.h" +#include "linkedlist.h" + +#define HASH_MIN_SHIFT (8) +#define HASH_MAX_SHIFT (15) + +typedef uint32_t (* SC_HashFn)(const void *key, uint32_t key_size); + +inline uint32_t hash_bernstein(const void* key, uint32_t key_size) +{ + uint32_t hash = 0; + register unsigned char* p = (unsigned char*)key; + + for (unsigned int i = 0; i < key_size; i++) + { + hash = 33 * hash + p[i]; + } + + return hash; +} + +inline uint32_t hash_sdbm(const void *key, uint32_t key_size) +{ + uint32_t hash = 0; + register unsigned char* p = (unsigned char*)key; + + for (uint32_t i = 0; i < key_size; i++) + { + hash = p[i] + (hash << 6) + (hash << 16) - hash; + } + + return hash; +} + +inline uint32_t hash_fnv(const void *key, uint32_t key_size ) +{ + register unsigned char *p = (unsigned char*)key; + uint32_t hash = 2166136261U; + + for (uint32_t i = 0; i < key_size; i++ ) + { + hash = ( hash * 16777619 ) ^ p[i]; + } + + return hash; +} + +typedef struct __SC_DictItem : public SC_ListNode +{ + __SC_DictItem() + { + data = 0; + hash = 0; + } + + ~__SC_DictItem() + { + if(key.data) { + delete[] key.data; + } + key.data = 0; + key.size = 0; + } + + void* data; + SC_Data key; + uint32_t hash; +} SC_DictItem; + +class SC_Dict +{ +private: + SC_Dict(SC_Dict& other) : pHashFn(0), mCurIdx(0), pCurItem(0), mCount(0), mShift(0), mMask(0), mBuckets(0), mSize(0){(void) other;} + +public: + SC_Dict(uint16_t shift = 8, SC_HashFn fn = hash_bernstein) : pHashFn(fn), mCurIdx(0), pCurItem(0), mCount(0) + { + if(shift <= HASH_MIN_SHIFT) { + mShift = HASH_MIN_SHIFT; + } else if(shift >= HASH_MAX_SHIFT) { + mShift = HASH_MAX_SHIFT; + } else { + mShift = shift; + } + + mSize = 1UL << shift; + mMask = mSize -1; + + mBuckets = new SC_ListHead[mSize]; + } + + virtual ~SC_Dict() + { + Clean(); + delete[] mBuckets; + } + + void Clean() + { + First(); + while(mCount) Remove(); + mCurIdx = 0; + pCurItem = 0; + } + + + inline void Insert(const void* key, uint32_t size, void* data) + { + SC_DictItem* item = new SC_DictItem; + item->hash = pHashFn(key, size); + item->data = data; + + // save key + item->key.data = new char[size]; + memcpy(item->key.data, key, size); + item->key.size = size; + + mCurIdx = item->hash & mMask; + + for(pCurItem = (SC_DictItem*)mBuckets[mCurIdx].First(); pCurItem ; pCurItem = (SC_DictItem*)mBuckets[mCurIdx].Next(pCurItem)) + { + if(pCurItem->key.size == item->key.size) + { + if(!memcmp(pCurItem->key.data, item->key.data, size)) + { + pCurItem->data = data; + delete item; + return; + } + } + } + mBuckets[mCurIdx].InsertLast(item); + pCurItem = item; + mCount++; + } + + void Remove() + { + SC_DictItem* item = pCurItem; + if(item) + { + MoveNext(); + mBuckets[item->hash & mMask].Remove(item); + if(pCurItem == item) + pCurItem = 0; + delete item; + mCount--; + } + } + + //returns first data in dictionary + inline void* First() + { + if(!mCount) + { + return 0; + } + + mCurIdx = 0; + pCurItem = 0; + MoveNext(); + + return pCurItem ? pCurItem->data : 0; + } + + //returns next data in dictionary + inline void* Next() + { + MoveNext(); + return pCurItem ? pCurItem->data : 0; + } + + //looking up for the next data + inline void* LookupNext(const void* key, uint32_t size) + { + uint32_t hash = pHashFn(key, size); + + for(SC_DictItem* curr = (SC_DictItem*)mBuckets[mCurIdx].Next(pCurItem); curr; curr = (SC_DictItem*)mBuckets[mCurIdx].Next(curr)) + { + if(curr->hash == hash && curr->key.size == size) + { + if(!memcmp(key, curr->key.data, size)) + { + pCurItem = curr; + return curr->data; + } + } + } + + return 0; + } + + //looking up for the first data corresponds to the key value + inline void* Lookup(const void* key, uint32_t size) + { + uint32_t hash = pHashFn(key, size); + uint32_t idx = hash & mMask; + for(SC_DictItem* curr = (SC_DictItem*)mBuckets[idx].First(); curr; curr = (SC_DictItem*)mBuckets[idx].Next(curr)) + { + if(curr->hash == hash && curr->key.size == size) + { + if(!memcmp(key, curr->key.data, size)) + { + mCurIdx = idx; + pCurItem = curr; + return curr->data; + } + } + } + + return 0; + } + + uint32_t Size() + { + return mCount; + } + +protected: + inline SC_DictItem* MoveNext() + { + if(mCurIdx >= mSize) + { + return 0; + } + + if(pCurItem) + { + pCurItem = (SC_DictItem*)mBuckets[mCurIdx].Next(pCurItem); + if(!pCurItem) + mCurIdx++; + } + + while(!pCurItem && mCurIdx < mSize) + { + pCurItem = (SC_DictItem*)mBuckets[mCurIdx].First(); + if(pCurItem) + break; + mCurIdx++; + } + + return pCurItem; + } +public: + virtual bool UnitTest(); + +protected: + + SC_HashFn pHashFn; //hash function + uint32_t mCurIdx; //index of current bucket + SC_DictItem* pCurItem; //current item + uint32_t mCount; //number of elements in dictionary + uint32_t mShift; //shift of the hash + uint32_t mMask; //mask to be applied to the hash + SC_ListHead* mBuckets; //Buckets of hash table + uint32_t mSize; //size of array in buckets +}; + + +#endif /* DICTIONARY_H_ */