mirror of
https://github.com/fanlumaster/googlepinyinime-rev.git
synced 2026-10-10 04:53:13 +08:00
format codes
This commit is contained in:
@@ -61,7 +61,7 @@ class AtomDictBase {
|
||||
* parameter.
|
||||
* @return True if succeed.
|
||||
*/
|
||||
virtual bool load_dict(const char *file_name, LemmaIdType start_id, LemmaIdType end_id) = 0;
|
||||
virtual bool load_dict(const char* file_name, LemmaIdType start_id, LemmaIdType end_id) = 0;
|
||||
|
||||
/**
|
||||
* Close this atom dictionary.
|
||||
@@ -119,7 +119,7 @@ class AtomDictBase {
|
||||
* @param lpi_num Used to return the newly added items.
|
||||
* @return The new mile stone for this extending. 0 if fail.
|
||||
*/
|
||||
virtual MileStoneHandle extend_dict(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num) = 0;
|
||||
virtual MileStoneHandle extend_dict(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num) = 0;
|
||||
|
||||
/**
|
||||
* Get lemma items with scores according to a spelling id stream.
|
||||
@@ -131,7 +131,7 @@ class AtomDictBase {
|
||||
* @param lpi_max The maximum size of the buffer to return result.
|
||||
* @return The number of matched items which have been filled in to lpi_items.
|
||||
*/
|
||||
virtual size_t get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lpi_items, size_t lpi_max) = 0;
|
||||
virtual size_t get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lpi_items, size_t lpi_max) = 0;
|
||||
|
||||
/**
|
||||
* Get a lemma string (The Chinese string) by the given lemma id.
|
||||
@@ -141,7 +141,7 @@ class AtomDictBase {
|
||||
* @param str_max The maximum size of the buffer.
|
||||
* @return The length of the string, 0 if fail.
|
||||
*/
|
||||
virtual uint16 get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max) = 0;
|
||||
virtual uint16 get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max) = 0;
|
||||
|
||||
/**
|
||||
* Get the full spelling ids for the given lemma id.
|
||||
@@ -155,7 +155,7 @@ class AtomDictBase {
|
||||
* case, splids_max is the number of valid ids in splids.
|
||||
* @return The number of ids in the buffer.
|
||||
*/
|
||||
virtual uint16 get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid) = 0;
|
||||
virtual uint16 get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid) = 0;
|
||||
|
||||
/**
|
||||
* Function used for prediction.
|
||||
@@ -170,7 +170,7 @@ class AtomDictBase {
|
||||
* from other atom dictionaries. A atom ditionary can just ignore it.
|
||||
* @return The number of prediction result from this atom dictionary.
|
||||
*/
|
||||
virtual size_t predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used) = 0;
|
||||
virtual size_t predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used) = 0;
|
||||
|
||||
/**
|
||||
* Add a lemma to the dictionary. If the dictionary allows to add new
|
||||
|
||||
+18
-18
@@ -36,38 +36,38 @@ class DictTrie;
|
||||
class DictBuilder {
|
||||
private:
|
||||
// The raw lemma array buffer.
|
||||
LemmaEntry *lemma_arr_;
|
||||
LemmaEntry* lemma_arr_;
|
||||
size_t lemma_num_;
|
||||
|
||||
// Used to store all possible single char items.
|
||||
// Two items may have the same Hanzi while their spelling ids are different.
|
||||
SingleCharItem *scis_;
|
||||
SingleCharItem* scis_;
|
||||
size_t scis_num_;
|
||||
|
||||
// In the tree, root's level is -1.
|
||||
// Lemma nodes for root, and level 0
|
||||
LmaNodeLE0 *lma_nodes_le0_;
|
||||
LmaNodeLE0* lma_nodes_le0_;
|
||||
|
||||
// Lemma nodes for layers whose levels are deeper than 0
|
||||
LmaNodeGE1 *lma_nodes_ge1_;
|
||||
LmaNodeGE1* lma_nodes_ge1_;
|
||||
|
||||
// Number of used lemma nodes
|
||||
size_t lma_nds_used_num_le0_;
|
||||
size_t lma_nds_used_num_ge1_;
|
||||
|
||||
// Used to store homophonies' ids.
|
||||
LemmaIdType *homo_idx_buf_;
|
||||
LemmaIdType* homo_idx_buf_;
|
||||
// Number of homophonies each of which only contains one Chinese character.
|
||||
size_t homo_idx_num_eq1_;
|
||||
// Number of homophonies each of which contains more than one character.
|
||||
size_t homo_idx_num_gt1_;
|
||||
|
||||
// The items with highest scores.
|
||||
LemmaEntry *top_lmas_;
|
||||
LemmaEntry* top_lmas_;
|
||||
size_t top_lmas_num_;
|
||||
|
||||
SpellingTable *spl_table_;
|
||||
SpellingParser *spl_parser_;
|
||||
SpellingTable* spl_table_;
|
||||
SpellingParser* spl_parser_;
|
||||
|
||||
#ifdef ___DO_STATISTICS___
|
||||
size_t max_sonbuf_len_[kMaxLemmaSize];
|
||||
@@ -96,21 +96,21 @@ class DictBuilder {
|
||||
// Build dictionary trie from the file fn_raw. File fn_validhzs provides
|
||||
// valid chars. If fn_validhzs is NULL, only chars in GB2312 will be
|
||||
// included.
|
||||
bool build_dict(const char *fn_raw, const char *fn_validhzs, DictTrie *dict_trie);
|
||||
bool build_dict(const char* fn_raw, const char* fn_validhzs, DictTrie* dict_trie);
|
||||
|
||||
private:
|
||||
// Fill in the buffer with id. The caller guarantees that the paramters are
|
||||
// vaild.
|
||||
void id_to_charbuf(unsigned char *buf, LemmaIdType id);
|
||||
void id_to_charbuf(unsigned char* buf, LemmaIdType id);
|
||||
|
||||
// Update the offset of sons for a node.
|
||||
void set_son_offset(LmaNodeGE1 *node, size_t offset);
|
||||
void set_son_offset(LmaNodeGE1* node, size_t offset);
|
||||
|
||||
// Update the offset of homophonies' ids for a node.
|
||||
void set_homo_id_buf_offset(LmaNodeGE1 *node, size_t offset);
|
||||
void set_homo_id_buf_offset(LmaNodeGE1* node, size_t offset);
|
||||
|
||||
// Format a speling string.
|
||||
void format_spelling_str(char *spl_str);
|
||||
void format_spelling_str(char* spl_str);
|
||||
|
||||
// Sort the lemma_arr by the hanzi string, and give each of unique items
|
||||
// a id. Why we need to sort the lemma list according to their Hanzi string
|
||||
@@ -130,23 +130,23 @@ class DictBuilder {
|
||||
// item_star to item_end)
|
||||
// parent is the parent node to update the necessary information
|
||||
// parent can be a member of LmaNodeLE0 or LmaNodeGE1
|
||||
bool construct_subset(void *parent, LemmaEntry *lemma_arr, size_t item_start, size_t item_end, size_t level);
|
||||
bool construct_subset(void* parent, LemmaEntry* lemma_arr, size_t item_start, size_t item_end, size_t level);
|
||||
|
||||
// Read valid Chinese Hanzis from the given file.
|
||||
// num is used to return number of chars.
|
||||
// The return buffer is sorted and caller needs to free the returned buffer.
|
||||
char16 *read_valid_hanzis(const char *fn_validhzs, size_t *num);
|
||||
char16* read_valid_hanzis(const char* fn_validhzs, size_t* num);
|
||||
|
||||
// Read a raw dictionary. max_item is the maximum number of items. If there
|
||||
// are more items in the ditionary, only the first max_item will be read.
|
||||
// Returned value is the number of items successfully read from the file.
|
||||
size_t read_raw_dict(const char *fn_raw, const char *fn_validhzs, size_t max_item);
|
||||
size_t read_raw_dict(const char* fn_raw, const char* fn_validhzs, size_t max_item);
|
||||
|
||||
// Try to find if a character is in hzs buffer.
|
||||
bool hz_in_hanzis_list(const char16 *hzs, size_t hzs_len, char16 hz);
|
||||
bool hz_in_hanzis_list(const char16* hzs, size_t hzs_len, char16 hz);
|
||||
|
||||
// Try to find if all characters in str are in hzs buffer.
|
||||
bool str_in_hanzis_list(const char16 *hzs, size_t hzs_len, const char16 *str, size_t str_len);
|
||||
bool str_in_hanzis_list(const char16* hzs, size_t hzs_len, const char16* str, size_t str_len);
|
||||
|
||||
// Get these lemmas with toppest scores.
|
||||
void get_top_lemmas();
|
||||
|
||||
+19
-19
@@ -30,15 +30,15 @@ class DictList {
|
||||
private:
|
||||
bool initialized_;
|
||||
|
||||
const SpellingTrie *spl_trie_;
|
||||
const SpellingTrie* spl_trie_;
|
||||
|
||||
// Number of SingCharItem. The first is blank, because id 0 is invalid.
|
||||
size_t scis_num_;
|
||||
char16 *scis_hz_;
|
||||
SpellingId *scis_splid_;
|
||||
char16* scis_hz_;
|
||||
SpellingId* scis_splid_;
|
||||
|
||||
// The large memory block to store the word list.
|
||||
char16 *buf_;
|
||||
char16* buf_;
|
||||
|
||||
// Starting position of those words whose lengths are i+1, counted in
|
||||
// char16
|
||||
@@ -46,7 +46,7 @@ class DictList {
|
||||
|
||||
uint32 start_id_[kMaxLemmaSize + 1];
|
||||
|
||||
int (*cmp_func_[kMaxLemmaSize])(const void *, const void *);
|
||||
int (*cmp_func_[kMaxLemmaSize])(const void*, const void*);
|
||||
|
||||
bool alloc_resource(size_t buf_size, size_t scim_num);
|
||||
|
||||
@@ -54,44 +54,44 @@ class DictList {
|
||||
|
||||
#ifdef ___BUILD_MODEL___
|
||||
// Calculate the requsted memory, including the start_pos[] buffer.
|
||||
size_t calculate_size(const LemmaEntry *lemma_arr, size_t lemma_num);
|
||||
size_t calculate_size(const LemmaEntry* lemma_arr, size_t lemma_num);
|
||||
|
||||
void fill_scis(const SingleCharItem *scis, size_t scis_num);
|
||||
void fill_scis(const SingleCharItem* scis, size_t scis_num);
|
||||
|
||||
// Copy the related content to the inner buffer
|
||||
// It should be called after calculate_size()
|
||||
void fill_list(const LemmaEntry *lemma_arr, size_t lemma_num);
|
||||
void fill_list(const LemmaEntry* lemma_arr, size_t lemma_num);
|
||||
|
||||
// Find the starting position for the buffer of those 2-character Chinese word
|
||||
// whose first character is the given Chinese character.
|
||||
char16 *find_pos2_startedbyhz(char16 hz_char);
|
||||
char16* find_pos2_startedbyhz(char16 hz_char);
|
||||
#endif
|
||||
|
||||
// Find the starting position for the buffer of those words whose lengths are
|
||||
// word_len. The given parameter cmp_func decides how many characters from
|
||||
// beginning will be used to compare.
|
||||
char16 *find_pos_startedbyhzs(const char16 last_hzs[], size_t word_Len, int (*cmp_func)(const void *, const void *));
|
||||
char16* find_pos_startedbyhzs(const char16 last_hzs[], size_t word_Len, int (*cmp_func)(const void*, const void*));
|
||||
|
||||
public:
|
||||
DictList();
|
||||
~DictList();
|
||||
|
||||
bool save_list(FILE *fp);
|
||||
bool load_list(FILE *fp);
|
||||
bool save_list(FILE* fp);
|
||||
bool load_list(FILE* fp);
|
||||
|
||||
#ifdef ___BUILD_MODEL___
|
||||
// Init the list from the LemmaEntry array.
|
||||
// lemma_arr should have been sorted by the hanzi_str, and have been given
|
||||
// ids from 1
|
||||
bool init_list(const SingleCharItem *scis, size_t scis_num, const LemmaEntry *lemma_arr, size_t lemma_num);
|
||||
bool init_list(const SingleCharItem* scis, size_t scis_num, const LemmaEntry* lemma_arr, size_t lemma_num);
|
||||
#endif
|
||||
|
||||
// Get the hanzi string for the given id
|
||||
uint16 get_lemma_str(LemmaIdType id_hz, char16 *str_buf, uint16 str_max);
|
||||
uint16 get_lemma_str(LemmaIdType id_hz, char16* str_buf, uint16 str_max);
|
||||
|
||||
void convert_to_hanzis(char16 *str, uint16 str_len);
|
||||
void convert_to_hanzis(char16* str, uint16 str_len);
|
||||
|
||||
void convert_to_scis_ids(char16 *str, uint16 str_len);
|
||||
void convert_to_scis_ids(char16* str, uint16 str_len);
|
||||
|
||||
// last_hzs stores the last n Chinese characters history, its length should be
|
||||
// less or equal than kMaxPredictSize.
|
||||
@@ -100,13 +100,13 @@ class DictList {
|
||||
// buf_len specifies the buffer length.
|
||||
// b4_used specifies how many items before predict_buf have been used.
|
||||
// Returned value is the number of newly added items.
|
||||
size_t predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used);
|
||||
size_t predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used);
|
||||
|
||||
// If half_splid is a valid half spelling id, return those full spelling
|
||||
// ids which share this half id.
|
||||
uint16 get_splids_for_hanzi(char16 hanzi, uint16 half_splid, uint16 *splids, uint16 max_splids);
|
||||
uint16 get_splids_for_hanzi(char16 hanzi, uint16 half_splid, uint16* splids, uint16 max_splids);
|
||||
|
||||
LemmaIdType get_lemma_id(const char16 *str, uint16 str_len);
|
||||
LemmaIdType get_lemma_id(const char16* str, uint16 str_len);
|
||||
};
|
||||
} // namespace ime_pinyin
|
||||
|
||||
|
||||
+29
-29
@@ -55,12 +55,12 @@ class DictTrie : AtomDictBase {
|
||||
uint16 mark_num;
|
||||
};
|
||||
|
||||
DictList *dict_list_;
|
||||
DictList* dict_list_;
|
||||
|
||||
const SpellingTrie *spl_trie_;
|
||||
const SpellingTrie* spl_trie_;
|
||||
|
||||
LmaNodeLE0 *root_; // Nodes for root and the first layer.
|
||||
LmaNodeGE1 *nodes_ge1_; // Nodes for other layers.
|
||||
LmaNodeLE0* root_; // Nodes for root and the first layer.
|
||||
LmaNodeGE1* nodes_ge1_; // Nodes for other layers.
|
||||
|
||||
// An quick index from spelling id to the LmaNodeLE0 node buffer, or
|
||||
// to the root_ buffer.
|
||||
@@ -71,69 +71,69 @@ class DictTrie : AtomDictBase {
|
||||
// corresponding full ids.
|
||||
// So, given an id splid, the son is:
|
||||
// root_[splid_le0_index_[splid - kFullSplIdStart]]
|
||||
uint16 *splid_le0_index_;
|
||||
uint16* splid_le0_index_;
|
||||
|
||||
size_t lma_node_num_le0_;
|
||||
size_t lma_node_num_ge1_;
|
||||
|
||||
// The first part is for homophnies, and the last top_lma_num_ items are
|
||||
// lemmas with highest scores.
|
||||
unsigned char *lma_idx_buf_;
|
||||
unsigned char* lma_idx_buf_;
|
||||
size_t lma_idx_buf_len_; // The total size of lma_idx_buf_ in byte.
|
||||
size_t total_lma_num_; // Total number of lemmas in this dictionary.
|
||||
size_t top_lmas_num_; // Number of lemma with highest scores.
|
||||
|
||||
// Parsing mark list used to mark the detailed extended statuses.
|
||||
ParsingMark *parsing_marks_;
|
||||
ParsingMark* parsing_marks_;
|
||||
// The position for next available mark.
|
||||
uint16 parsing_marks_pos_;
|
||||
|
||||
// Mile stone list used to mark the extended status.
|
||||
MileStone *mile_stones_;
|
||||
MileStone* mile_stones_;
|
||||
// The position for the next available mile stone. We use positions (except 0)
|
||||
// as handles.
|
||||
MileStoneHandle mile_stones_pos_;
|
||||
|
||||
// Get the offset of sons for a node.
|
||||
inline size_t get_son_offset(const LmaNodeGE1 *node);
|
||||
inline size_t get_son_offset(const LmaNodeGE1* node);
|
||||
|
||||
// Get the offset of homonious ids for a node.
|
||||
inline size_t get_homo_idx_buf_offset(const LmaNodeGE1 *node);
|
||||
inline size_t get_homo_idx_buf_offset(const LmaNodeGE1* node);
|
||||
|
||||
// Get the lemma id by the offset.
|
||||
inline LemmaIdType get_lemma_id(size_t id_offset);
|
||||
|
||||
void free_resource(bool free_dict_list);
|
||||
|
||||
bool load_dict(FILE *fp);
|
||||
bool load_dict(FILE* fp);
|
||||
|
||||
// Given a LmaNodeLE0 node, extract the lemmas specified by it, and fill
|
||||
// them into the lpi_items buffer.
|
||||
// This function is called by the search engine.
|
||||
size_t fill_lpi_buffer(LmaPsbItem lpi_items[], size_t max_size, LmaNodeLE0 *node);
|
||||
size_t fill_lpi_buffer(LmaPsbItem lpi_items[], size_t max_size, LmaNodeLE0* node);
|
||||
|
||||
// Given a LmaNodeGE1 node, extract the lemmas specified by it, and fill
|
||||
// them into the lpi_items buffer.
|
||||
// This function is called by inner functions extend_dict0(), extend_dict1()
|
||||
// and extend_dict2().
|
||||
size_t fill_lpi_buffer(LmaPsbItem lpi_items[], size_t max_size, size_t homo_buf_off, LmaNodeGE1 *node, uint16 lma_len);
|
||||
size_t fill_lpi_buffer(LmaPsbItem lpi_items[], size_t max_size, size_t homo_buf_off, LmaNodeGE1* node, uint16 lma_len);
|
||||
|
||||
// Extend in the trie from level 0.
|
||||
MileStoneHandle extend_dict0(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num);
|
||||
MileStoneHandle extend_dict0(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num);
|
||||
|
||||
// Extend in the trie from level 1.
|
||||
MileStoneHandle extend_dict1(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num);
|
||||
MileStoneHandle extend_dict1(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num);
|
||||
|
||||
// Extend in the trie from level 2.
|
||||
MileStoneHandle extend_dict2(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num);
|
||||
MileStoneHandle extend_dict2(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num);
|
||||
|
||||
// Try to extend the given spelling id buffer, and if the given id_lemma can
|
||||
// be successfully gotten, return true;
|
||||
// The given spelling ids are all valid full ids.
|
||||
bool try_extend(const uint16 *splids, uint16 splid_num, LemmaIdType id_lemma);
|
||||
bool try_extend(const uint16* splids, uint16 splid_num, LemmaIdType id_lemma);
|
||||
|
||||
#ifdef ___BUILD_MODEL___
|
||||
bool save_dict(FILE *fp);
|
||||
bool save_dict(FILE* fp);
|
||||
#endif // ___BUILD_MODEL___
|
||||
|
||||
// Increase limits to support longer Pinyin input.
|
||||
@@ -153,35 +153,35 @@ class DictTrie : AtomDictBase {
|
||||
// Construct the tree from the file fn_raw.
|
||||
// fn_validhzs provide the valid hanzi list. If fn_validhzs is
|
||||
// NULL, only chars in GB2312 will be included.
|
||||
bool build_dict(const char *fn_raw, const char *fn_validhzs);
|
||||
bool build_dict(const char* fn_raw, const char* fn_validhzs);
|
||||
|
||||
// Save the binary dictionary
|
||||
// Actually, the SpellingTrie/DictList instance will be also saved.
|
||||
bool save_dict(const char *filename);
|
||||
bool save_dict(const char* filename);
|
||||
#endif // ___BUILD_MODEL___
|
||||
|
||||
void convert_to_hanzis(char16 *str, uint16 str_len);
|
||||
void convert_to_hanzis(char16* str, uint16 str_len);
|
||||
|
||||
void convert_to_scis_ids(char16 *str, uint16 str_len);
|
||||
void convert_to_scis_ids(char16* str, uint16 str_len);
|
||||
|
||||
// Load a binary dictionary
|
||||
// The SpellingTrie instance/DictList will be also loaded
|
||||
bool load_dict(const char *filename, LemmaIdType start_id, LemmaIdType end_id);
|
||||
bool load_dict(const char* filename, LemmaIdType start_id, LemmaIdType end_id);
|
||||
bool load_dict_fd(int sys_fd, long start_offset, long length, LemmaIdType start_id, LemmaIdType end_id);
|
||||
bool close_dict() { return true; }
|
||||
size_t number_of_lemmas() { return 0; }
|
||||
|
||||
void reset_milestones(uint16 from_step, MileStoneHandle from_handle);
|
||||
|
||||
MileStoneHandle extend_dict(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num);
|
||||
MileStoneHandle extend_dict(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num);
|
||||
|
||||
size_t get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lpi_items, size_t lpi_max);
|
||||
size_t get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lpi_items, size_t lpi_max);
|
||||
|
||||
uint16 get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max);
|
||||
uint16 get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max);
|
||||
|
||||
uint16 get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid);
|
||||
uint16 get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid);
|
||||
|
||||
size_t predict(const char16 *last_hzs, uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used);
|
||||
size_t predict(const char16* last_hzs, uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used);
|
||||
|
||||
LemmaIdType put_lemma(char16 /*lemma_str*/[], uint16 /*splids*/[], uint16 /*lemma_len*/, uint16 /*count*/) { return 0; }
|
||||
|
||||
@@ -204,7 +204,7 @@ class DictTrie : AtomDictBase {
|
||||
|
||||
// Fill the lemmas with highest scores to the prediction buffer.
|
||||
// his_len is the history length to fill in the prediction buffer.
|
||||
size_t predict_top_lmas(size_t his_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used);
|
||||
size_t predict_top_lmas(size_t his_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used);
|
||||
};
|
||||
} // namespace ime_pinyin
|
||||
|
||||
|
||||
@@ -26,17 +26,17 @@ namespace ime_pinyin {
|
||||
// Used to cache LmaPsbItem list for half spelling ids.
|
||||
class LpiCache {
|
||||
private:
|
||||
static LpiCache *instance_;
|
||||
static LpiCache* instance_;
|
||||
static const int kMaxLpiCachePerId = 15;
|
||||
|
||||
LmaPsbItem *lpi_cache_;
|
||||
uint16 *lpi_cache_len_;
|
||||
LmaPsbItem* lpi_cache_;
|
||||
uint16* lpi_cache_len_;
|
||||
|
||||
public:
|
||||
LpiCache();
|
||||
~LpiCache();
|
||||
|
||||
static LpiCache &get_instance();
|
||||
static LpiCache& get_instance();
|
||||
|
||||
// Test if the LPI list of the given splid has been cached.
|
||||
// If splid is a full spelling id, it returns false, because we only cache
|
||||
|
||||
+26
-26
@@ -57,7 +57,7 @@ typedef struct {
|
||||
typedef struct MatrixNode {
|
||||
LemmaIdType id;
|
||||
float score;
|
||||
MatrixNode *from;
|
||||
MatrixNode* from;
|
||||
// From which DMI node. Used to trace the spelling segmentation.
|
||||
PoolPosType dmi_fr;
|
||||
uint16 step;
|
||||
@@ -98,7 +98,7 @@ typedef struct {
|
||||
uint16 dmi_has_full_id : 1;
|
||||
// Points to a MatrixNode of the current step to indicate which choice the
|
||||
// user selects.
|
||||
MatrixNode *mtrx_nd_fixed;
|
||||
MatrixNode* mtrx_nd_fixed;
|
||||
} MatrixRow, *PMatrixRow;
|
||||
|
||||
// When user inputs and selects candidates, the fixed lemma ids are stored in
|
||||
@@ -159,7 +159,7 @@ class MatrixSearch {
|
||||
bool inited_;
|
||||
|
||||
// Spelling trie.
|
||||
const SpellingTrie *spl_trie_;
|
||||
const SpellingTrie* spl_trie_;
|
||||
|
||||
// Used to indicate this switcher status: when "xian" is parseed, should
|
||||
// "xi an" also be extended. Default is false.
|
||||
@@ -170,13 +170,13 @@ class MatrixSearch {
|
||||
bool xi_an_enabled_;
|
||||
|
||||
// System dictionary.
|
||||
DictTrie *dict_trie_;
|
||||
DictTrie* dict_trie_;
|
||||
|
||||
// User dictionary.
|
||||
AtomDictBase *user_dict_;
|
||||
AtomDictBase* user_dict_;
|
||||
|
||||
// Spelling parser.
|
||||
SpellingParser *spl_parser_;
|
||||
SpellingParser* spl_parser_;
|
||||
|
||||
// The maximum allowed length of spelling string (such as a Pinyin string).
|
||||
size_t max_sps_len_;
|
||||
@@ -191,18 +191,18 @@ class MatrixSearch {
|
||||
size_t pys_decoded_len_;
|
||||
|
||||
// Shared buffer for multiple purposes.
|
||||
size_t *share_buf_;
|
||||
size_t* share_buf_;
|
||||
|
||||
MatrixNode *mtrx_nd_pool_;
|
||||
MatrixNode* mtrx_nd_pool_;
|
||||
PoolPosType mtrx_nd_pool_used_; // How many nodes used in the pool
|
||||
DictMatchInfo *dmi_pool_;
|
||||
DictMatchInfo* dmi_pool_;
|
||||
PoolPosType dmi_pool_used_; // How many items used in the pool
|
||||
|
||||
MatrixRow *matrix_; // The first row is for starting
|
||||
MatrixRow* matrix_; // The first row is for starting
|
||||
|
||||
DictExtPara *dep_; // Parameter used to extend DMI nodes.
|
||||
DictExtPara* dep_; // Parameter used to extend DMI nodes.
|
||||
|
||||
NPredictItem *npre_items_; // Used to do prediction
|
||||
NPredictItem* npre_items_; // Used to do prediction
|
||||
size_t npre_items_len_;
|
||||
|
||||
// The starting positions and lemma ids for the full sentence candidate.
|
||||
@@ -287,11 +287,11 @@ class MatrixSearch {
|
||||
// If pfullsent is not NULL, means the full sentence candidate may be the
|
||||
// same with the coming lemma string, if so, remove that lemma.
|
||||
// The result is sorted in descendant order by the frequency score.
|
||||
size_t get_lpis(const uint16 *splid_str, size_t splid_str_len, LmaPsbItem *lma_buf, size_t max_lma_buf, const char16 *pfullsent, bool sort_by_psb);
|
||||
size_t get_lpis(const uint16* splid_str, size_t splid_str_len, LmaPsbItem* lma_buf, size_t max_lma_buf, const char16* pfullsent, bool sort_by_psb);
|
||||
|
||||
uint16 get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max);
|
||||
uint16 get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max);
|
||||
|
||||
uint16 get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid);
|
||||
uint16 get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid);
|
||||
|
||||
// Extend a DMI node with a spelling id. ext_len is the length of the rows
|
||||
// to extend, actually, it is the size of the spelling string of splid.
|
||||
@@ -312,17 +312,17 @@ class MatrixSearch {
|
||||
// calling this function if necessary.
|
||||
//
|
||||
// The caller should guarantees that NULL != dep.
|
||||
size_t extend_dmi(DictExtPara *dep, DictMatchInfo *dmi_s);
|
||||
size_t extend_dmi(DictExtPara* dep, DictMatchInfo* dmi_s);
|
||||
|
||||
// Extend dmi for the composing phrase.
|
||||
size_t extend_dmi_c(DictExtPara *dep, DictMatchInfo *dmi_s);
|
||||
size_t extend_dmi_c(DictExtPara* dep, DictMatchInfo* dmi_s);
|
||||
|
||||
// Extend a MatrixNode with the give LmaPsbItem list.
|
||||
// res_row is the destination row number.
|
||||
// This function does not change mtrx_nd_pool_used_. Please change it after
|
||||
// calling this function if necessary.
|
||||
// return 0 always.
|
||||
size_t extend_mtrx_nd(MatrixNode *mtrx_nd, LmaPsbItem lpi_items[], size_t lpi_num, PoolPosType dmi_fr, size_t res_row);
|
||||
size_t extend_mtrx_nd(MatrixNode* mtrx_nd, LmaPsbItem lpi_items[], size_t lpi_num, PoolPosType dmi_fr, size_t res_row);
|
||||
|
||||
// Try to find a dmi node at step_to position, and the found dmi node should
|
||||
// match the given spelling id strings.
|
||||
@@ -341,7 +341,7 @@ class MatrixSearch {
|
||||
// The caller guarantees that the position is valid.
|
||||
bool is_split_at(uint16 pos);
|
||||
|
||||
void fill_dmi(DictMatchInfo *dmi, MileStoneHandle *handles, PoolPosType dmi_fr, uint16 spl_id, uint16 node_num, unsigned char dict_level, bool splid_end_split, unsigned char splstr_len, unsigned char all_full_id);
|
||||
void fill_dmi(DictMatchInfo* dmi, MileStoneHandle* handles, PoolPosType dmi_fr, uint16 spl_id, uint16 node_num, unsigned char dict_level, bool splid_end_split, unsigned char splstr_len, unsigned char all_full_id);
|
||||
|
||||
size_t inner_predict(const char16 fixed_scis_ids[], uint16 scis_num, char16 predict_buf[][kMaxPredictSize + 1], size_t buf_len);
|
||||
|
||||
@@ -364,9 +364,9 @@ class MatrixSearch {
|
||||
MatrixSearch();
|
||||
~MatrixSearch();
|
||||
|
||||
bool init(const char *fn_sys_dict, const char *fn_usr_dict);
|
||||
bool init(const char* fn_sys_dict, const char* fn_usr_dict);
|
||||
|
||||
bool init_fd(int sys_fd, long start_offset, long length, const char *fn_usr_dict);
|
||||
bool init_fd(int sys_fd, long start_offset, long length, const char* fn_usr_dict);
|
||||
|
||||
void set_max_lens(size_t max_sps_len, size_t max_hzs_len);
|
||||
|
||||
@@ -384,7 +384,7 @@ class MatrixSearch {
|
||||
|
||||
// Search a Pinyin string.
|
||||
// Return value is the position successfully parsed.
|
||||
size_t search(const char *py, size_t py_len);
|
||||
size_t search(const char* py, size_t py_len);
|
||||
|
||||
// Used to delete something in the Pinyin string kept by the engine, and do
|
||||
// a re-search.
|
||||
@@ -406,23 +406,23 @@ class MatrixSearch {
|
||||
|
||||
// Get the Pinyin string stored by the engine.
|
||||
// *decoded_len returns the length of the successfully decoded string.
|
||||
const char *get_pystr(size_t *decoded_len);
|
||||
const char* get_pystr(size_t* decoded_len);
|
||||
|
||||
// Get the spelling boundaries for the first sentence candidate.
|
||||
// Number of spellings will be returned. The number of valid elements in
|
||||
// spl_start is one more than the return value because the last one is used
|
||||
// to indicate the beginning of the next un-input speling.
|
||||
// For a Pinyin "women", the returned value is 2, spl_start is [0, 2, 5] .
|
||||
size_t get_spl_start(const uint16 *&spl_start);
|
||||
size_t get_spl_start(const uint16*& spl_start);
|
||||
|
||||
// Get one candiate string. If full sentence candidate is available, it will
|
||||
// be the first one.
|
||||
char16 *get_candidate(size_t cand_id, char16 *cand_str, size_t max_len);
|
||||
char16* get_candidate(size_t cand_id, char16* cand_str, size_t max_len);
|
||||
|
||||
// Get the first candiate, which is a "full sentence".
|
||||
// retstr_len is not NULL, it will be used to return the string length.
|
||||
// If only_unfixed is true, only unfixed part will be fetched.
|
||||
char16 *get_candidate0(char16 *cand_str, size_t max_len, uint16 *retstr_len, bool only_unfixed);
|
||||
char16* get_candidate0(char16* cand_str, size_t max_len, uint16* retstr_len, bool only_unfixed);
|
||||
|
||||
// Choose a candidate. The decoder will do a search after the fixed position.
|
||||
size_t choose(size_t cand_id);
|
||||
|
||||
@@ -21,9 +21,9 @@
|
||||
|
||||
namespace ime_pinyin {
|
||||
|
||||
void myqsort(void *p, size_t n, size_t es, int (*cmp)(const void *, const void *));
|
||||
void myqsort(void* p, size_t n, size_t es, int (*cmp)(const void*, const void*));
|
||||
|
||||
void *mybsearch(const void *key, const void *base, size_t nmemb, size_t size, int (*compar)(const void *, const void *));
|
||||
void* mybsearch(const void* key, const void* base, size_t nmemb, size_t size, int (*compar)(const void*, const void*));
|
||||
} // namespace ime_pinyin
|
||||
|
||||
#endif // PINYINIME_INCLUDE_MYSTDLIB_H__
|
||||
|
||||
+8
-8
@@ -45,7 +45,7 @@ class NGram {
|
||||
static const size_t kSysDictTotalFreq = 100000000;
|
||||
|
||||
private:
|
||||
static NGram *instance_;
|
||||
static NGram* instance_;
|
||||
|
||||
bool initialized_;
|
||||
size_t idx_num_;
|
||||
@@ -58,19 +58,19 @@ class NGram {
|
||||
float sys_score_compensation_;
|
||||
|
||||
#ifdef ___BUILD_MODEL___
|
||||
double *freq_codes_df_;
|
||||
double* freq_codes_df_;
|
||||
#endif
|
||||
LmaScoreType *freq_codes_;
|
||||
CODEBOOK_TYPE *lma_freq_idx_;
|
||||
LmaScoreType* freq_codes_;
|
||||
CODEBOOK_TYPE* lma_freq_idx_;
|
||||
|
||||
public:
|
||||
NGram();
|
||||
~NGram();
|
||||
|
||||
static NGram &get_instance();
|
||||
static NGram& get_instance();
|
||||
|
||||
bool save_ngram(FILE *fp);
|
||||
bool load_ngram(FILE *fp);
|
||||
bool save_ngram(FILE* fp);
|
||||
bool load_ngram(FILE* fp);
|
||||
|
||||
// Set the total frequency of all none system dictionaries.
|
||||
void set_total_freq_none_sys(size_t freq_none_sys);
|
||||
@@ -86,7 +86,7 @@ class NGram {
|
||||
|
||||
#ifdef ___BUILD_MODEL___
|
||||
// For constructing the unigram mode model.
|
||||
bool build_unigram(LemmaEntry *lemma_arr, size_t num, LemmaIdType next_idx_unused);
|
||||
bool build_unigram(LemmaEntry* lemma_arr, size_t num, LemmaIdType next_idx_unused);
|
||||
#endif
|
||||
};
|
||||
} // namespace ime_pinyin
|
||||
|
||||
@@ -33,7 +33,7 @@ namespace ime_pinyin {
|
||||
* @param fn_usr_dict The file name of the user dictionary.
|
||||
* @return true if open the decoder engine successfully.
|
||||
*/
|
||||
bool im_open_decoder(const char *fn_sys_dict, const char *fn_usr_dict);
|
||||
bool im_open_decoder(const char* fn_sys_dict, const char* fn_usr_dict);
|
||||
|
||||
/**
|
||||
* Open the decoder engine via the system dictionary FD and user dictionary
|
||||
@@ -47,7 +47,7 @@ bool im_open_decoder(const char *fn_sys_dict, const char *fn_usr_dict);
|
||||
* counted in byte.
|
||||
* @return true if succeed.
|
||||
*/
|
||||
bool im_open_decoder_fd(int sys_fd, long start_offset, long length, const char *fn_usr_dict);
|
||||
bool im_open_decoder_fd(int sys_fd, long start_offset, long length, const char* fn_usr_dict);
|
||||
|
||||
/**
|
||||
* Close the decoder engine.
|
||||
@@ -86,7 +86,7 @@ void im_flush_cache();
|
||||
* @param sps_len The length of the spelling string buffer.
|
||||
* @return The number of candidates.
|
||||
*/
|
||||
size_t im_search(const char *sps_buf, size_t sps_len);
|
||||
size_t im_search(const char* sps_buf, size_t sps_len);
|
||||
|
||||
/**
|
||||
* Make a delete operation in the current search result, and make research if
|
||||
@@ -122,7 +122,7 @@ size_t im_add_letter(char ch);
|
||||
* string is successfully parsed.
|
||||
* @return The spelling string kept by the decoder.
|
||||
*/
|
||||
const char *im_get_sps_str(size_t *decoded_len);
|
||||
const char* im_get_sps_str(size_t* decoded_len);
|
||||
|
||||
/**
|
||||
* Get a candidate(or choice) string.
|
||||
@@ -133,7 +133,7 @@ const char *im_get_sps_str(size_t *decoded_len);
|
||||
* @param max_len The maximum length of the buffer.
|
||||
* @return cand_str if succeeds, otherwise NULL.
|
||||
*/
|
||||
char16 *im_get_candidate(size_t cand_id, char16 *cand_str, size_t max_len);
|
||||
char16* im_get_candidate(size_t cand_id, char16* cand_str, size_t max_len);
|
||||
|
||||
/**
|
||||
* Get the segmentation information(the starting positions) of the spelling
|
||||
@@ -144,7 +144,7 @@ char16 *im_get_candidate(size_t cand_id, char16 *cand_str, size_t max_len);
|
||||
* elements in spl_start, and spl_start[L] is the posistion after the end of
|
||||
* the last spelling id.
|
||||
*/
|
||||
size_t im_get_spl_start_pos(const uint16 *&spl_start);
|
||||
size_t im_get_spl_start_pos(const uint16*& spl_start);
|
||||
|
||||
/**
|
||||
* Choose a candidate and make it fixed. If the candidate does not match
|
||||
@@ -187,7 +187,7 @@ bool im_cancel_input();
|
||||
* @param pre_buf Used to return prediction result list.
|
||||
* @return The number of predicted result string.
|
||||
*/
|
||||
size_t im_get_predicts(const char16 *his_buf, char16 (*&pre_buf)[kMaxPredictSize + 1]);
|
||||
size_t im_get_predicts(const char16* his_buf, char16 (*&pre_buf)[kMaxPredictSize + 1]);
|
||||
|
||||
/**
|
||||
* Enable Shengmus in ShouZiMu mode.
|
||||
|
||||
+17
-17
@@ -111,27 +111,27 @@ bool is_system_lemma(LemmaIdType lma_id);
|
||||
bool is_user_lemma(LemmaIdType lma_id);
|
||||
bool is_composing_lemma(LemmaIdType lma_id);
|
||||
|
||||
int cmp_lpi_with_psb(const void *p1, const void *p2);
|
||||
int cmp_lpi_with_unified_psb(const void *p1, const void *p2);
|
||||
int cmp_lpi_with_id(const void *p1, const void *p2);
|
||||
int cmp_lpi_with_hanzi(const void *p1, const void *p2);
|
||||
int cmp_lpi_with_psb(const void* p1, const void* p2);
|
||||
int cmp_lpi_with_unified_psb(const void* p1, const void* p2);
|
||||
int cmp_lpi_with_id(const void* p1, const void* p2);
|
||||
int cmp_lpi_with_hanzi(const void* p1, const void* p2);
|
||||
|
||||
int cmp_lpsi_with_str(const void *p1, const void *p2);
|
||||
int cmp_lpsi_with_str(const void* p1, const void* p2);
|
||||
|
||||
int cmp_hanzis_1(const void *p1, const void *p2);
|
||||
int cmp_hanzis_2(const void *p1, const void *p2);
|
||||
int cmp_hanzis_3(const void *p1, const void *p2);
|
||||
int cmp_hanzis_4(const void *p1, const void *p2);
|
||||
int cmp_hanzis_5(const void *p1, const void *p2);
|
||||
int cmp_hanzis_6(const void *p1, const void *p2);
|
||||
int cmp_hanzis_7(const void *p1, const void *p2);
|
||||
int cmp_hanzis_8(const void *p1, const void *p2);
|
||||
int cmp_hanzis_1(const void* p1, const void* p2);
|
||||
int cmp_hanzis_2(const void* p1, const void* p2);
|
||||
int cmp_hanzis_3(const void* p1, const void* p2);
|
||||
int cmp_hanzis_4(const void* p1, const void* p2);
|
||||
int cmp_hanzis_5(const void* p1, const void* p2);
|
||||
int cmp_hanzis_6(const void* p1, const void* p2);
|
||||
int cmp_hanzis_7(const void* p1, const void* p2);
|
||||
int cmp_hanzis_8(const void* p1, const void* p2);
|
||||
|
||||
int cmp_npre_by_score(const void *p1, const void *p2);
|
||||
int cmp_npre_by_hislen_score(const void *p1, const void *p2);
|
||||
int cmp_npre_by_hanzi_score(const void *p1, const void *p2);
|
||||
int cmp_npre_by_score(const void* p1, const void* p2);
|
||||
int cmp_npre_by_hislen_score(const void* p1, const void* p2);
|
||||
int cmp_npre_by_hanzi_score(const void* p1, const void* p2);
|
||||
|
||||
size_t remove_duplicate_npre(NPredictItem *npre_items, size_t npre_num);
|
||||
size_t remove_duplicate_npre(NPredictItem* npre_items, size_t npre_num);
|
||||
|
||||
size_t align_to_size_t(size_t size);
|
||||
|
||||
|
||||
@@ -24,7 +24,7 @@ namespace ime_pinyin {
|
||||
|
||||
class SpellingParser {
|
||||
protected:
|
||||
const SpellingTrie *spl_trie_;
|
||||
const SpellingTrie* spl_trie_;
|
||||
|
||||
public:
|
||||
SpellingParser();
|
||||
@@ -39,27 +39,27 @@ class SpellingParser {
|
||||
// If splstr starts with a character not in ['a'-z'] (it is a split char),
|
||||
// return 0.
|
||||
// Split char can only appear in the middle of the string or at the end.
|
||||
uint16 splstr_to_idxs(const char *splstr, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre);
|
||||
uint16 splstr_to_idxs(const char* splstr, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre);
|
||||
|
||||
// Similar to splstr_to_idxs(), the only difference is that splstr_to_idxs()
|
||||
// convert single-character Yunmus into half ids, while this function converts
|
||||
// them into full ids.
|
||||
uint16 splstr_to_idxs_f(const char *splstr, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre);
|
||||
uint16 splstr_to_idxs_f(const char* splstr, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre);
|
||||
|
||||
// Similar to splstr_to_idxs(), the only difference is that this function
|
||||
// uses char16 instead of char8.
|
||||
uint16 splstr16_to_idxs(const char16 *splstr, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre);
|
||||
uint16 splstr16_to_idxs(const char16* splstr, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre);
|
||||
|
||||
// Similar to splstr_to_idxs_f(), the only difference is that this function
|
||||
// uses char16 instead of char8.
|
||||
uint16 splstr16_to_idxs_f(const char16 *splstr16, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre);
|
||||
uint16 splstr16_to_idxs_f(const char16* splstr16, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre);
|
||||
|
||||
// If the given string is a spelling, return the id, others, return 0.
|
||||
// If the give string is a single char Yunmus like "A", and the char is
|
||||
// enabled in ShouZiMu mode, the returned spelling id will be a half id.
|
||||
// When the returned spelling id is a half id, *is_pre returns whether it
|
||||
// is a prefix of a full spelling string.
|
||||
uint16 get_splid_by_str(const char *splstr, uint16 str_len, bool *is_pre);
|
||||
uint16 get_splid_by_str(const char* splstr, uint16 str_len, bool* is_pre);
|
||||
|
||||
// If the given string is a spelling, return the id, others, return 0.
|
||||
// If the give string is a single char Yunmus like "a", no matter the char
|
||||
@@ -67,7 +67,7 @@ class SpellingParser {
|
||||
// a full id.
|
||||
// When the returned spelling id is a half id, *p_is_pre returns whether it
|
||||
// is a prefix of a full spelling string.
|
||||
uint16 get_splid_by_str_f(const char *splstr, uint16 str_len, bool *is_pre);
|
||||
uint16 get_splid_by_str_f(const char* splstr, uint16 str_len, bool* is_pre);
|
||||
|
||||
// Splitter chars are not included.
|
||||
bool is_valid_to_parse(char ch);
|
||||
@@ -82,7 +82,7 @@ class SpellingParser {
|
||||
// return 0.
|
||||
// Split char can only appear in the middle of the string or at the end.
|
||||
// The caller should guarantee NULL != splstr && str_len > 0 && NULL != splidx
|
||||
uint16 get_splids_parallel(const char *splstr, uint16 str_len, uint16 splidx[], uint16 max_size, uint16 &full_id_num, bool &is_pre);
|
||||
uint16 get_splids_parallel(const char* splstr, uint16 str_len, uint16 splidx[], uint16 max_size, uint16& full_id_num, bool& is_pre);
|
||||
};
|
||||
} // namespace ime_pinyin
|
||||
|
||||
|
||||
+44
-44
@@ -26,7 +26,7 @@
|
||||
|
||||
#ifdef _WIN32
|
||||
#include <time.h>
|
||||
#include <winsock.h> // timeval
|
||||
#include <winsock.h> // timeval
|
||||
#else
|
||||
#include <pthread.h>
|
||||
#include <sys/time.h>
|
||||
@@ -40,7 +40,7 @@ class UserDict : public AtomDictBase {
|
||||
UserDict();
|
||||
~UserDict();
|
||||
|
||||
bool load_dict(const char *file_name, LemmaIdType start_id, LemmaIdType end_id);
|
||||
bool load_dict(const char* file_name, LemmaIdType start_id, LemmaIdType end_id);
|
||||
|
||||
bool close_dict();
|
||||
|
||||
@@ -48,15 +48,15 @@ class UserDict : public AtomDictBase {
|
||||
|
||||
void reset_milestones(uint16 from_step, MileStoneHandle from_handle);
|
||||
|
||||
MileStoneHandle extend_dict(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num);
|
||||
MileStoneHandle extend_dict(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num);
|
||||
|
||||
size_t get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lpi_items, size_t lpi_max);
|
||||
size_t get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lpi_items, size_t lpi_max);
|
||||
|
||||
uint16 get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max);
|
||||
uint16 get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max);
|
||||
|
||||
uint16 get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid);
|
||||
uint16 get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid);
|
||||
|
||||
size_t predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used);
|
||||
size_t predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used);
|
||||
|
||||
// Full spelling ids are required
|
||||
LemmaIdType put_lemma(char16 lemma_str[], uint16 splids[], uint16 lemma_len, uint16 count);
|
||||
@@ -95,7 +95,7 @@ class UserDict : public AtomDictBase {
|
||||
* @param len length of lemmas string in UTF-16LE
|
||||
* @return newly added lemma count
|
||||
*/
|
||||
int put_lemmas_no_sync_from_utf16le_string(char16 *lemmas, int len);
|
||||
int put_lemmas_no_sync_from_utf16le_string(char16* lemmas, int len);
|
||||
|
||||
/**
|
||||
* Get lemmas need sync to a UTF-16LE string of above format.
|
||||
@@ -107,13 +107,13 @@ class UserDict : public AtomDictBase {
|
||||
* @param count output value of lemma returned
|
||||
* @return UTF-16LE string length
|
||||
*/
|
||||
int get_sync_lemmas_in_utf16le_string_from_beginning(char16 *str, int size, int *count);
|
||||
int get_sync_lemmas_in_utf16le_string_from_beginning(char16* str, int size, int* count);
|
||||
|
||||
#endif
|
||||
|
||||
struct UserDictStat {
|
||||
uint32 version;
|
||||
const char *file_name;
|
||||
const char* file_name;
|
||||
struct timeval load_time;
|
||||
struct timeval last_update;
|
||||
uint32 disk_size;
|
||||
@@ -129,19 +129,19 @@ class UserDict : public AtomDictBase {
|
||||
uint32 limit_lemma_size;
|
||||
};
|
||||
|
||||
bool state(UserDictStat *stat);
|
||||
bool state(UserDictStat* stat);
|
||||
|
||||
private:
|
||||
uint32 total_other_nfreq_;
|
||||
struct timeval load_time_;
|
||||
LemmaIdType start_id_;
|
||||
uint32 version_;
|
||||
uint8 *lemmas_;
|
||||
uint8* lemmas_;
|
||||
|
||||
// In-Memory-Only flag for each lemma
|
||||
static const uint8 kUserDictLemmaFlagRemove = 1;
|
||||
// Inuse lemmas' offset
|
||||
uint32 *offsets_;
|
||||
uint32* offsets_;
|
||||
// Highest bit in offset tells whether corresponding lemma is removed
|
||||
static const uint32 kUserDictOffsetFlagRemove = (1 << 31);
|
||||
// Maximum possible for the offset
|
||||
@@ -157,22 +157,22 @@ class UserDict : public AtomDictBase {
|
||||
static const uint64 kUserDictLMTSince = COARSE_UTC(2009, 1, 1, 0, 0, 0);
|
||||
|
||||
// Correspond to offsets_
|
||||
uint32 *scores_;
|
||||
uint32* scores_;
|
||||
// Following two fields are only valid in memory
|
||||
uint32 *ids_;
|
||||
uint32* ids_;
|
||||
#ifdef ___PREDICT_ENABLED___
|
||||
uint32 *predicts_;
|
||||
uint32* predicts_;
|
||||
#endif
|
||||
#ifdef ___SYNC_ENABLED___
|
||||
uint32 *syncs_;
|
||||
uint32* syncs_;
|
||||
size_t sync_count_size_;
|
||||
#endif
|
||||
uint32 *offsets_by_id_;
|
||||
uint32* offsets_by_id_;
|
||||
|
||||
size_t lemma_count_left_;
|
||||
size_t lemma_size_left_;
|
||||
|
||||
const char *dict_file_;
|
||||
const char* dict_file_;
|
||||
|
||||
// Be sure size is 4xN
|
||||
struct UserDictInfo {
|
||||
@@ -250,19 +250,19 @@ class UserDict : public AtomDictBase {
|
||||
|
||||
void cache_init();
|
||||
|
||||
void cache_push(UserDictCacheType type, UserDictSearchable *searchable, uint32 offset, uint32 length);
|
||||
void cache_push(UserDictCacheType type, UserDictSearchable* searchable, uint32 offset, uint32 length);
|
||||
|
||||
bool cache_hit(UserDictSearchable *searchable, uint32 *offset, uint32 *length);
|
||||
bool cache_hit(UserDictSearchable* searchable, uint32* offset, uint32* length);
|
||||
|
||||
bool load_cache(UserDictSearchable *searchable, uint32 *offset, uint32 *length);
|
||||
bool load_cache(UserDictSearchable* searchable, uint32* offset, uint32* length);
|
||||
|
||||
void save_cache(UserDictSearchable *searchable, uint32 offset, uint32 length);
|
||||
void save_cache(UserDictSearchable* searchable, uint32 offset, uint32 length);
|
||||
|
||||
void reset_cache();
|
||||
|
||||
bool load_miss_cache(UserDictSearchable *searchable);
|
||||
bool load_miss_cache(UserDictSearchable* searchable);
|
||||
|
||||
void save_miss_cache(UserDictSearchable *searchable);
|
||||
void save_miss_cache(UserDictSearchable* searchable);
|
||||
|
||||
void reset_miss_cache();
|
||||
#endif
|
||||
@@ -275,29 +275,29 @@ class UserDict : public AtomDictBase {
|
||||
|
||||
inline int build_score(uint64 lmt, int freq);
|
||||
|
||||
inline int64 utf16le_atoll(uint16 *s, int len);
|
||||
inline int64 utf16le_atoll(uint16* s, int len);
|
||||
|
||||
inline int utf16le_lltoa(int64 v, uint16 *s, int size);
|
||||
inline int utf16le_lltoa(int64 v, uint16* s, int size);
|
||||
|
||||
LemmaIdType _put_lemma(char16 lemma_str[], uint16 splids[], uint16 lemma_len, uint16 count, uint64 lmt);
|
||||
|
||||
size_t _get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lpi_items, size_t lpi_max, bool *need_extend);
|
||||
size_t _get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lpi_items, size_t lpi_max, bool* need_extend);
|
||||
|
||||
int _get_lemma_score(char16 lemma_str[], uint16 splids[], uint16 lemma_len);
|
||||
|
||||
int _get_lemma_score(LemmaIdType lemma_id);
|
||||
|
||||
int is_fuzzy_prefix_spell_id(const uint16 *id1, uint16 len1, const UserDictSearchable *searchable);
|
||||
int is_fuzzy_prefix_spell_id(const uint16* id1, uint16 len1, const UserDictSearchable* searchable);
|
||||
|
||||
bool is_prefix_spell_id(const uint16 *fullids, uint16 fulllen, const UserDictSearchable *searchable);
|
||||
bool is_prefix_spell_id(const uint16* fullids, uint16 fulllen, const UserDictSearchable* searchable);
|
||||
|
||||
uint32 get_dict_file_size(UserDictInfo *info);
|
||||
uint32 get_dict_file_size(UserDictInfo* info);
|
||||
|
||||
bool reset(const char *file);
|
||||
bool reset(const char* file);
|
||||
|
||||
bool validate(const char *file);
|
||||
bool validate(const char* file);
|
||||
|
||||
bool load(const char *file, LemmaIdType start_id);
|
||||
bool load(const char* file, LemmaIdType start_id);
|
||||
|
||||
bool is_valid_state();
|
||||
|
||||
@@ -311,22 +311,22 @@ class UserDict : public AtomDictBase {
|
||||
|
||||
char get_lemma_nchar(uint32 offset);
|
||||
|
||||
uint16 *get_lemma_spell_ids(uint32 offset);
|
||||
uint16* get_lemma_spell_ids(uint32 offset);
|
||||
|
||||
uint16 *get_lemma_word(uint32 offset);
|
||||
uint16* get_lemma_word(uint32 offset);
|
||||
|
||||
// Prepare searchable to fasten locate process
|
||||
void prepare_locate(UserDictSearchable *searchable, const uint16 *splids, uint16 len);
|
||||
void prepare_locate(UserDictSearchable* searchable, const uint16* splids, uint16 len);
|
||||
|
||||
// Compare initial letters only
|
||||
int32 fuzzy_compare_spell_id(const uint16 *id1, uint16 len1, const UserDictSearchable *searchable);
|
||||
int32 fuzzy_compare_spell_id(const uint16* id1, uint16 len1, const UserDictSearchable* searchable);
|
||||
|
||||
// Compare exactly two spell ids
|
||||
// First argument must be a full id spell id
|
||||
bool equal_spell_id(const uint16 *fullids, uint16 fulllen, const UserDictSearchable *searchable);
|
||||
bool equal_spell_id(const uint16* fullids, uint16 fulllen, const UserDictSearchable* searchable);
|
||||
|
||||
// Find first item by initial letters
|
||||
int32 locate_first_in_offsets(const UserDictSearchable *searchable);
|
||||
int32 locate_first_in_offsets(const UserDictSearchable* searchable);
|
||||
|
||||
LemmaIdType append_a_lemma(char16 lemma_str[], uint16 splids[], uint16 lemma_len, uint16 count, uint64 lmt);
|
||||
|
||||
@@ -335,9 +335,9 @@ class UserDict : public AtomDictBase {
|
||||
|
||||
bool remove_lemma_by_offset_index(int offset_index);
|
||||
#ifdef ___PREDICT_ENABLED___
|
||||
uint32 locate_where_to_insert_in_predicts(const uint16 *words, int lemma_len);
|
||||
uint32 locate_where_to_insert_in_predicts(const uint16* words, int lemma_len);
|
||||
|
||||
int32 locate_first_in_predicts(const uint16 *words, int lemma_len);
|
||||
int32 locate_first_in_predicts(const uint16* words, int lemma_len);
|
||||
|
||||
void remove_lemma_from_predict_list(uint32 offset);
|
||||
#endif
|
||||
@@ -359,9 +359,9 @@ class UserDict : public AtomDictBase {
|
||||
uint32 offset_index;
|
||||
};
|
||||
|
||||
inline void swap(UserDictScoreOffsetPair *sop, int i, int j);
|
||||
inline void swap(UserDictScoreOffsetPair* sop, int i, int j);
|
||||
|
||||
void shift_down(UserDictScoreOffsetPair *sop, int i, int n);
|
||||
void shift_down(UserDictScoreOffsetPair* sop, int i, int n);
|
||||
|
||||
// On-disk format for each lemma
|
||||
// +-------------+
|
||||
|
||||
@@ -30,21 +30,21 @@ typedef unsigned short char16;
|
||||
// Get a token from utf16_str,
|
||||
// Returned pointer is a '\0'-terminated utf16 string, or NULL
|
||||
// *utf16_str_next returns the next part of the string for further tokenizing
|
||||
char16 *utf16_strtok(char16 *utf16_str, size_t *token_size, char16 **utf16_str_next);
|
||||
char16* utf16_strtok(char16* utf16_str, size_t* token_size, char16** utf16_str_next);
|
||||
|
||||
int utf16_atoi(const char16 *utf16_str);
|
||||
int utf16_atoi(const char16* utf16_str);
|
||||
|
||||
float utf16_atof(const char16 *utf16_str);
|
||||
float utf16_atof(const char16* utf16_str);
|
||||
|
||||
size_t utf16_strlen(const char16 *utf16_str);
|
||||
size_t utf16_strlen(const char16* utf16_str);
|
||||
|
||||
int utf16_strcmp(const char16 *str1, const char16 *str2);
|
||||
int utf16_strncmp(const char16 *str1, const char16 *str2, size_t size);
|
||||
int utf16_strcmp(const char16* str1, const char16* str2);
|
||||
int utf16_strncmp(const char16* str1, const char16* str2, size_t size);
|
||||
|
||||
char16 *utf16_strcpy(char16 *dst, const char16 *src);
|
||||
char16 *utf16_strncpy(char16 *dst, const char16 *src, size_t size);
|
||||
char16* utf16_strcpy(char16* dst, const char16* src);
|
||||
char16* utf16_strncpy(char16* dst, const char16* src, size_t size);
|
||||
|
||||
char *utf16_strcpy_tochar(char *dst, const char16 *src);
|
||||
char* utf16_strcpy_tochar(char* dst, const char16* src);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
|
||||
+70
-70
@@ -38,10 +38,10 @@ static const size_t kSplTableHashLen = 2000;
|
||||
|
||||
// Compare a SingleCharItem, first by Hanzis, then by spelling ids, then by
|
||||
// frequencies.
|
||||
int cmp_scis_hz_splid_freq(const void *p1, const void *p2) {
|
||||
int cmp_scis_hz_splid_freq(const void* p1, const void* p2) {
|
||||
const SingleCharItem *s1, *s2;
|
||||
s1 = static_cast<const SingleCharItem *>(p1);
|
||||
s2 = static_cast<const SingleCharItem *>(p2);
|
||||
s1 = static_cast<const SingleCharItem*>(p1);
|
||||
s2 = static_cast<const SingleCharItem*>(p2);
|
||||
|
||||
if (s1->hz < s2->hz) return -1;
|
||||
if (s1->hz > s2->hz) return 1;
|
||||
@@ -57,10 +57,10 @@ int cmp_scis_hz_splid_freq(const void *p1, const void *p2) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
int cmp_scis_hz_splid(const void *p1, const void *p2) {
|
||||
int cmp_scis_hz_splid(const void* p1, const void* p2) {
|
||||
const SingleCharItem *s1, *s2;
|
||||
s1 = static_cast<const SingleCharItem *>(p1);
|
||||
s2 = static_cast<const SingleCharItem *>(p2);
|
||||
s1 = static_cast<const SingleCharItem*>(p1);
|
||||
s2 = static_cast<const SingleCharItem*>(p2);
|
||||
|
||||
if (s1->hz < s2->hz) return -1;
|
||||
if (s1->hz > s2->hz) return 1;
|
||||
@@ -74,49 +74,49 @@ int cmp_scis_hz_splid(const void *p1, const void *p2) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
int cmp_lemma_entry_hzs(const void *p1, const void *p2) {
|
||||
size_t size1 = utf16_strlen(((const LemmaEntry *)p1)->hanzi_str);
|
||||
size_t size2 = utf16_strlen(((const LemmaEntry *)p2)->hanzi_str);
|
||||
int cmp_lemma_entry_hzs(const void* p1, const void* p2) {
|
||||
size_t size1 = utf16_strlen(((const LemmaEntry*)p1)->hanzi_str);
|
||||
size_t size2 = utf16_strlen(((const LemmaEntry*)p2)->hanzi_str);
|
||||
if (size1 < size2)
|
||||
return -1;
|
||||
else if (size1 > size2)
|
||||
return 1;
|
||||
|
||||
return utf16_strcmp(((const LemmaEntry *)p1)->hanzi_str, ((const LemmaEntry *)p2)->hanzi_str);
|
||||
return utf16_strcmp(((const LemmaEntry*)p1)->hanzi_str, ((const LemmaEntry*)p2)->hanzi_str);
|
||||
}
|
||||
|
||||
int compare_char16(const void *p1, const void *p2) {
|
||||
if (*((const char16 *)p1) < *((const char16 *)p2)) return -1;
|
||||
if (*((const char16 *)p1) > *((const char16 *)p2)) return 1;
|
||||
int compare_char16(const void* p1, const void* p2) {
|
||||
if (*((const char16*)p1) < *((const char16*)p2)) return -1;
|
||||
if (*((const char16*)p1) > *((const char16*)p2)) return 1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int compare_py(const void *p1, const void *p2) {
|
||||
int ret = utf16_strcmp(((const LemmaEntry *)p1)->spl_idx_arr, ((const LemmaEntry *)p2)->spl_idx_arr);
|
||||
int compare_py(const void* p1, const void* p2) {
|
||||
int ret = utf16_strcmp(((const LemmaEntry*)p1)->spl_idx_arr, ((const LemmaEntry*)p2)->spl_idx_arr);
|
||||
|
||||
if (0 != ret) return ret;
|
||||
|
||||
return static_cast<int>(((const LemmaEntry *)p2)->freq) - static_cast<int>(((const LemmaEntry *)p1)->freq);
|
||||
return static_cast<int>(((const LemmaEntry*)p2)->freq) - static_cast<int>(((const LemmaEntry*)p1)->freq);
|
||||
}
|
||||
|
||||
// First hanzi, if the same, then Pinyin
|
||||
int cmp_lemma_entry_hzspys(const void *p1, const void *p2) {
|
||||
size_t size1 = utf16_strlen(((const LemmaEntry *)p1)->hanzi_str);
|
||||
size_t size2 = utf16_strlen(((const LemmaEntry *)p2)->hanzi_str);
|
||||
int cmp_lemma_entry_hzspys(const void* p1, const void* p2) {
|
||||
size_t size1 = utf16_strlen(((const LemmaEntry*)p1)->hanzi_str);
|
||||
size_t size2 = utf16_strlen(((const LemmaEntry*)p2)->hanzi_str);
|
||||
if (size1 < size2)
|
||||
return -1;
|
||||
else if (size1 > size2)
|
||||
return 1;
|
||||
int ret = utf16_strcmp(((const LemmaEntry *)p1)->hanzi_str, ((const LemmaEntry *)p2)->hanzi_str);
|
||||
int ret = utf16_strcmp(((const LemmaEntry*)p1)->hanzi_str, ((const LemmaEntry*)p2)->hanzi_str);
|
||||
|
||||
if (0 != ret) return ret;
|
||||
|
||||
ret = utf16_strcmp(((const LemmaEntry *)p1)->spl_idx_arr, ((const LemmaEntry *)p2)->spl_idx_arr);
|
||||
ret = utf16_strcmp(((const LemmaEntry*)p1)->spl_idx_arr, ((const LemmaEntry*)p2)->spl_idx_arr);
|
||||
return ret;
|
||||
}
|
||||
|
||||
int compare_splid2(const void *p1, const void *p2) {
|
||||
int ret = utf16_strcmp(((const LemmaEntry *)p1)->spl_idx_arr, ((const LemmaEntry *)p2)->spl_idx_arr);
|
||||
int compare_splid2(const void* p1, const void* p2) {
|
||||
int ret = utf16_strcmp(((const LemmaEntry*)p1)->spl_idx_arr, ((const LemmaEntry*)p2)->spl_idx_arr);
|
||||
return ret;
|
||||
}
|
||||
|
||||
@@ -188,11 +188,11 @@ bool DictBuilder::alloc_resource(size_t lma_num) {
|
||||
return true;
|
||||
}
|
||||
|
||||
char16 *DictBuilder::read_valid_hanzis(const char *fn_validhzs, size_t *num) {
|
||||
char16* DictBuilder::read_valid_hanzis(const char* fn_validhzs, size_t* num) {
|
||||
if (NULL == fn_validhzs || NULL == num) return NULL;
|
||||
|
||||
*num = 0;
|
||||
FILE *fp = fopen(fn_validhzs, "rb");
|
||||
FILE* fp = fopen(fn_validhzs, "rb");
|
||||
if (NULL == fp) return NULL;
|
||||
|
||||
char16 utf16header;
|
||||
@@ -206,7 +206,7 @@ char16 *DictBuilder::read_valid_hanzis(const char *fn_validhzs, size_t *num) {
|
||||
assert(*num >= 1);
|
||||
*num -= 1;
|
||||
|
||||
char16 *hzs = new char16[*num];
|
||||
char16* hzs = new char16[*num];
|
||||
if (NULL == hzs) {
|
||||
fclose(fp);
|
||||
return NULL;
|
||||
@@ -225,11 +225,11 @@ char16 *DictBuilder::read_valid_hanzis(const char *fn_validhzs, size_t *num) {
|
||||
return hzs;
|
||||
}
|
||||
|
||||
bool DictBuilder::hz_in_hanzis_list(const char16 *hzs, size_t hzs_len, char16 hz) {
|
||||
bool DictBuilder::hz_in_hanzis_list(const char16* hzs, size_t hzs_len, char16 hz) {
|
||||
if (NULL == hzs) return false;
|
||||
|
||||
char16 *found;
|
||||
found = static_cast<char16 *>(mybsearch(&hz, hzs, hzs_len, sizeof(char16), compare_char16));
|
||||
char16* found;
|
||||
found = static_cast<char16*>(mybsearch(&hz, hzs, hzs_len, sizeof(char16), compare_char16));
|
||||
if (NULL == found) return false;
|
||||
|
||||
assert(*found == hz);
|
||||
@@ -237,7 +237,7 @@ bool DictBuilder::hz_in_hanzis_list(const char16 *hzs, size_t hzs_len, char16 hz
|
||||
}
|
||||
|
||||
// The caller makes sure that the parameters are valid.
|
||||
bool DictBuilder::str_in_hanzis_list(const char16 *hzs, size_t hzs_len, const char16 *str, size_t str_len) {
|
||||
bool DictBuilder::str_in_hanzis_list(const char16* hzs, size_t hzs_len, const char16* str, size_t str_len) {
|
||||
if (NULL == hzs || NULL == str) return false;
|
||||
|
||||
for (size_t pos = 0; pos < str_len; pos++) {
|
||||
@@ -313,7 +313,7 @@ void DictBuilder::free_resource() {
|
||||
homo_idx_num_gt1_ = 0;
|
||||
}
|
||||
|
||||
size_t DictBuilder::read_raw_dict(const char *fn_raw, const char *fn_validhzs, size_t max_item) {
|
||||
size_t DictBuilder::read_raw_dict(const char* fn_raw, const char* fn_validhzs, size_t max_item) {
|
||||
if (NULL == fn_raw) return 0;
|
||||
|
||||
Utf16Reader utf16_reader;
|
||||
@@ -330,7 +330,7 @@ size_t DictBuilder::read_raw_dict(const char *fn_raw, const char *fn_validhzs, s
|
||||
}
|
||||
|
||||
// Read the valid Hanzi list.
|
||||
char16 *valid_hzs = NULL;
|
||||
char16* valid_hzs = NULL;
|
||||
size_t valid_hzs_num = 0;
|
||||
valid_hzs = read_valid_hanzis(fn_validhzs, &valid_hzs_num);
|
||||
|
||||
@@ -343,8 +343,8 @@ size_t DictBuilder::read_raw_dict(const char *fn_raw, const char *fn_validhzs, s
|
||||
}
|
||||
|
||||
size_t token_size;
|
||||
char16 *token;
|
||||
char16 *to_tokenize = read_buf;
|
||||
char16* token;
|
||||
char16* to_tokenize = read_buf;
|
||||
|
||||
// Get the Hanzi string
|
||||
token = utf16_strtok(to_tokenize, &token_size, &to_tokenize);
|
||||
@@ -443,7 +443,7 @@ size_t DictBuilder::read_raw_dict(const char *fn_raw, const char *fn_validhzs, s
|
||||
return lemma_num;
|
||||
}
|
||||
|
||||
bool DictBuilder::build_dict(const char *fn_raw, const char *fn_validhzs, DictTrie *dict_trie) {
|
||||
bool DictBuilder::build_dict(const char* fn_raw, const char* fn_validhzs, DictTrie* dict_trie) {
|
||||
if (NULL == fn_raw || NULL == dict_trie) return false;
|
||||
|
||||
lemma_num_ = read_raw_dict(fn_raw, fn_validhzs, 240000);
|
||||
@@ -455,14 +455,14 @@ bool DictBuilder::build_dict(const char *fn_raw, const char *fn_validhzs, DictTr
|
||||
// spelling string will be score, and it is also included in spl_item_size.
|
||||
size_t spl_item_size;
|
||||
size_t spl_num;
|
||||
const char *spl_buf;
|
||||
const char* spl_buf;
|
||||
spl_buf = spl_table_->arrange(&spl_item_size, &spl_num);
|
||||
if (NULL == spl_buf) {
|
||||
free_resource();
|
||||
return false;
|
||||
}
|
||||
|
||||
SpellingTrie &spl_trie = SpellingTrie::get_instance();
|
||||
SpellingTrie& spl_trie = SpellingTrie::get_instance();
|
||||
|
||||
if (!spl_trie.construct(spl_buf, spl_item_size, spl_num, spl_table_->get_score_amplifier(), spl_table_->get_average_score())) {
|
||||
free_resource();
|
||||
@@ -500,7 +500,7 @@ bool DictBuilder::build_dict(const char *fn_raw, const char *fn_validhzs, DictTr
|
||||
assert(dl_success);
|
||||
|
||||
// Construct the NGram information
|
||||
NGram &ngram = NGram::get_instance();
|
||||
NGram& ngram = NGram::get_instance();
|
||||
ngram.build_unigram(lemma_arr_, lemma_num_, lemma_arr_[lemma_num_ - 1].idx_by_hz + 1);
|
||||
|
||||
// sort the lemma items according to the spelling idx string
|
||||
@@ -513,7 +513,7 @@ bool DictBuilder::build_dict(const char *fn_raw, const char *fn_validhzs, DictTr
|
||||
#endif
|
||||
|
||||
lma_nds_used_num_le0_ = 1; // The root node
|
||||
bool dt_success = construct_subset(static_cast<void *>(lma_nodes_le0_), lemma_arr_, 0, lemma_num_, 0);
|
||||
bool dt_success = construct_subset(static_cast<void*>(lma_nodes_le0_), lemma_arr_, 0, lemma_num_, 0);
|
||||
if (!dt_success) {
|
||||
free_resource();
|
||||
return false;
|
||||
@@ -561,19 +561,19 @@ bool DictBuilder::build_dict(const char *fn_raw, const char *fn_validhzs, DictTr
|
||||
return dt_success;
|
||||
}
|
||||
|
||||
void DictBuilder::id_to_charbuf(unsigned char *buf, LemmaIdType id) {
|
||||
void DictBuilder::id_to_charbuf(unsigned char* buf, LemmaIdType id) {
|
||||
if (NULL == buf) return;
|
||||
for (size_t pos = 0; pos < kLemmaIdSize; pos++) {
|
||||
(buf)[pos] = (unsigned char)(id >> (pos * 8));
|
||||
}
|
||||
}
|
||||
|
||||
void DictBuilder::set_son_offset(LmaNodeGE1 *node, size_t offset) {
|
||||
void DictBuilder::set_son_offset(LmaNodeGE1* node, size_t offset) {
|
||||
node->son_1st_off_l = static_cast<uint16>(offset);
|
||||
node->son_1st_off_h = static_cast<unsigned char>(offset >> 16);
|
||||
}
|
||||
|
||||
void DictBuilder::set_homo_id_buf_offset(LmaNodeGE1 *node, size_t offset) {
|
||||
void DictBuilder::set_homo_id_buf_offset(LmaNodeGE1* node, size_t offset) {
|
||||
node->homo_idx_buf_off_l = static_cast<uint16>(offset);
|
||||
node->homo_idx_buf_off_h = static_cast<unsigned char>(offset >> 16);
|
||||
}
|
||||
@@ -581,7 +581,7 @@ void DictBuilder::set_homo_id_buf_offset(LmaNodeGE1 *node, size_t offset) {
|
||||
// All spelling strings will be converted to upper case, except that
|
||||
// spellings started with "ZH"/"CH"/"SH" will be converted to
|
||||
// "Zh"/"Ch"/"Sh"
|
||||
void DictBuilder::format_spelling_str(char *spl_str) {
|
||||
void DictBuilder::format_spelling_str(char* spl_str) {
|
||||
if (NULL == spl_str) return;
|
||||
|
||||
uint16 pos = 0;
|
||||
@@ -619,7 +619,7 @@ LemmaIdType DictBuilder::sort_lemmas_by_hz() {
|
||||
size_t DictBuilder::build_scis() {
|
||||
if (NULL == scis_ || lemma_num_ * kMaxLemmaSize > scis_num_) return 0;
|
||||
|
||||
SpellingTrie &spl_trie = SpellingTrie::get_instance();
|
||||
SpellingTrie& spl_trie = SpellingTrie::get_instance();
|
||||
|
||||
// This first one is blank, because id 0 is invalid.
|
||||
scis_[0].freq = 0;
|
||||
@@ -665,8 +665,8 @@ size_t DictBuilder::build_scis() {
|
||||
key.splid.full_splid = lemma_arr_[pos].spl_idx_arr[hzpos];
|
||||
key.splid.half_splid = spl_trie.full_to_half(key.splid.full_splid);
|
||||
|
||||
SingleCharItem *found;
|
||||
found = static_cast<SingleCharItem *>(mybsearch(&key, scis_, unique_scis_num, sizeof(SingleCharItem), cmp_scis_hz_splid));
|
||||
SingleCharItem* found;
|
||||
found = static_cast<SingleCharItem*>(mybsearch(&key, scis_, unique_scis_num, sizeof(SingleCharItem), cmp_scis_hz_splid));
|
||||
|
||||
assert(found);
|
||||
|
||||
@@ -678,7 +678,7 @@ size_t DictBuilder::build_scis() {
|
||||
return scis_num_;
|
||||
}
|
||||
|
||||
bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t item_start, size_t item_end, size_t level) {
|
||||
bool DictBuilder::construct_subset(void* parent, LemmaEntry* lemma_arr, size_t item_start, size_t item_end, size_t level) {
|
||||
if (level >= kMaxLemmaSize || item_end <= item_start) return false;
|
||||
|
||||
// 1. Scan for how many sons
|
||||
@@ -686,12 +686,12 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
|
||||
// LemmaNode *son_1st = NULL;
|
||||
// parent.num_of_son = 0;
|
||||
|
||||
LemmaEntry *lma_last_start = lemma_arr_ + item_start;
|
||||
LemmaEntry* lma_last_start = lemma_arr_ + item_start;
|
||||
uint16 spl_idx_node = lma_last_start->spl_idx_arr[level];
|
||||
|
||||
// Scan for how many sons to be allocaed
|
||||
for (size_t i = item_start + 1; i < item_end; i++) {
|
||||
LemmaEntry *lma_current = lemma_arr + i;
|
||||
LemmaEntry* lma_current = lemma_arr + i;
|
||||
uint16 spl_idx_current = lma_current->spl_idx_arr[level];
|
||||
if (spl_idx_current != spl_idx_node) {
|
||||
parent_son_num++;
|
||||
@@ -719,29 +719,29 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
|
||||
|
||||
// 2. Update the parent's information
|
||||
// Update the parent's son list;
|
||||
LmaNodeLE0 *son_1st_le0 = NULL; // only one of le0 or ge1 is used
|
||||
LmaNodeGE1 *son_1st_ge1 = NULL; // only one of le0 or ge1 is used.
|
||||
LmaNodeLE0* son_1st_le0 = NULL; // only one of le0 or ge1 is used
|
||||
LmaNodeGE1* son_1st_ge1 = NULL; // only one of le0 or ge1 is used.
|
||||
if (0 == level) { // the parent is root
|
||||
(static_cast<LmaNodeLE0 *>(parent))->son_1st_off = lma_nds_used_num_le0_;
|
||||
(static_cast<LmaNodeLE0*>(parent))->son_1st_off = lma_nds_used_num_le0_;
|
||||
son_1st_le0 = lma_nodes_le0_ + lma_nds_used_num_le0_;
|
||||
lma_nds_used_num_le0_ += parent_son_num;
|
||||
|
||||
assert(parent_son_num <= 65535);
|
||||
(static_cast<LmaNodeLE0 *>(parent))->num_of_son = static_cast<uint16>(parent_son_num);
|
||||
(static_cast<LmaNodeLE0*>(parent))->num_of_son = static_cast<uint16>(parent_son_num);
|
||||
} else if (1 == level) { // the parent is a son of root
|
||||
(static_cast<LmaNodeLE0 *>(parent))->son_1st_off = lma_nds_used_num_ge1_;
|
||||
(static_cast<LmaNodeLE0*>(parent))->son_1st_off = lma_nds_used_num_ge1_;
|
||||
son_1st_ge1 = lma_nodes_ge1_ + lma_nds_used_num_ge1_;
|
||||
lma_nds_used_num_ge1_ += parent_son_num;
|
||||
|
||||
assert(parent_son_num <= 65535);
|
||||
(static_cast<LmaNodeLE0 *>(parent))->num_of_son = static_cast<uint16>(parent_son_num);
|
||||
(static_cast<LmaNodeLE0*>(parent))->num_of_son = static_cast<uint16>(parent_son_num);
|
||||
} else {
|
||||
set_son_offset((static_cast<LmaNodeGE1 *>(parent)), lma_nds_used_num_ge1_);
|
||||
set_son_offset((static_cast<LmaNodeGE1*>(parent)), lma_nds_used_num_ge1_);
|
||||
son_1st_ge1 = lma_nodes_ge1_ + lma_nds_used_num_ge1_;
|
||||
lma_nds_used_num_ge1_ += parent_son_num;
|
||||
|
||||
assert(parent_son_num <= 255);
|
||||
(static_cast<LmaNodeGE1 *>(parent))->num_of_son = (unsigned char)parent_son_num;
|
||||
(static_cast<LmaNodeGE1*>(parent))->num_of_son = (unsigned char)parent_son_num;
|
||||
}
|
||||
|
||||
// 3. Now begin to construct the son one by one
|
||||
@@ -756,15 +756,15 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
|
||||
size_t item_start_next = item_start;
|
||||
|
||||
for (size_t i = item_start + 1; i < item_end; i++) {
|
||||
LemmaEntry *lma_current = lemma_arr_ + i;
|
||||
LemmaEntry* lma_current = lemma_arr_ + i;
|
||||
uint16 spl_idx_current = lma_current->spl_idx_arr[level];
|
||||
|
||||
if (spl_idx_current == spl_idx_node) {
|
||||
if (lma_current->spl_idx_arr[level + 1] == 0) homo_num++;
|
||||
} else {
|
||||
// Construct a node
|
||||
LmaNodeLE0 *node_cur_le0 = NULL; // only one of them is valid
|
||||
LmaNodeGE1 *node_cur_ge1 = NULL;
|
||||
LmaNodeLE0* node_cur_le0 = NULL; // only one of them is valid
|
||||
LmaNodeGE1* node_cur_ge1 = NULL;
|
||||
if (0 == level) {
|
||||
node_cur_le0 = son_1st_le0 + son_pos;
|
||||
node_cur_le0->spl_idx = spl_idx_node;
|
||||
@@ -781,7 +781,7 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
|
||||
}
|
||||
|
||||
if (homo_num > 0) {
|
||||
LemmaIdType *idx_buf = homo_idx_buf_ + homo_idx_num_eq1_ + homo_idx_num_gt1_ - homo_num;
|
||||
LemmaIdType* idx_buf = homo_idx_buf_ + homo_idx_num_eq1_ + homo_idx_num_gt1_ - homo_num;
|
||||
if (0 == level) {
|
||||
assert(homo_num <= 65535);
|
||||
node_cur_le0->num_of_homo = static_cast<uint16>(homo_num);
|
||||
@@ -802,11 +802,11 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
|
||||
}
|
||||
|
||||
if (i - item_start_next > homo_num) {
|
||||
void *next_parent;
|
||||
void* next_parent;
|
||||
if (0 == level)
|
||||
next_parent = static_cast<void *>(node_cur_le0);
|
||||
next_parent = static_cast<void*>(node_cur_le0);
|
||||
else
|
||||
next_parent = static_cast<void *>(node_cur_ge1);
|
||||
next_parent = static_cast<void*>(node_cur_ge1);
|
||||
construct_subset(next_parent, lemma_arr, item_start_next + homo_num, i, level + 1);
|
||||
#ifdef ___DO_STATISTICS___
|
||||
|
||||
@@ -827,8 +827,8 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
|
||||
}
|
||||
|
||||
// 4. The last one to construct
|
||||
LmaNodeLE0 *node_cur_le0 = NULL; // only one of them is valid
|
||||
LmaNodeGE1 *node_cur_ge1 = NULL;
|
||||
LmaNodeLE0* node_cur_le0 = NULL; // only one of them is valid
|
||||
LmaNodeGE1* node_cur_ge1 = NULL;
|
||||
if (0 == level) {
|
||||
node_cur_le0 = son_1st_le0 + son_pos;
|
||||
node_cur_le0->spl_idx = spl_idx_node;
|
||||
@@ -845,7 +845,7 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
|
||||
}
|
||||
|
||||
if (homo_num > 0) {
|
||||
LemmaIdType *idx_buf = homo_idx_buf_ + homo_idx_num_eq1_ + homo_idx_num_gt1_ - homo_num;
|
||||
LemmaIdType* idx_buf = homo_idx_buf_ + homo_idx_num_eq1_ + homo_idx_num_gt1_ - homo_num;
|
||||
if (0 == level) {
|
||||
assert(homo_num <= 65535);
|
||||
node_cur_le0->num_of_homo = static_cast<uint16>(homo_num);
|
||||
@@ -866,11 +866,11 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
|
||||
}
|
||||
|
||||
if (item_end - item_start_next > homo_num) {
|
||||
void *next_parent;
|
||||
void* next_parent;
|
||||
if (0 == level)
|
||||
next_parent = static_cast<void *>(node_cur_le0);
|
||||
next_parent = static_cast<void*>(node_cur_le0);
|
||||
else
|
||||
next_parent = static_cast<void *>(node_cur_ge1);
|
||||
next_parent = static_cast<void*>(node_cur_ge1);
|
||||
construct_subset(next_parent, lemma_arr, item_start_next + homo_num, item_end, level + 1);
|
||||
#ifdef ___DO_STATISTICS___
|
||||
|
||||
|
||||
+26
-26
@@ -47,15 +47,15 @@ DictList::~DictList() { free_resource(); }
|
||||
|
||||
bool DictList::alloc_resource(size_t buf_size, size_t scis_num) {
|
||||
// Allocate memory
|
||||
buf_ = static_cast<char16 *>(malloc(buf_size * sizeof(char16)));
|
||||
buf_ = static_cast<char16*>(malloc(buf_size * sizeof(char16)));
|
||||
if (NULL == buf_) return false;
|
||||
|
||||
scis_num_ = scis_num;
|
||||
|
||||
scis_hz_ = static_cast<char16 *>(malloc(scis_num_ * sizeof(char16)));
|
||||
scis_hz_ = static_cast<char16*>(malloc(scis_num_ * sizeof(char16)));
|
||||
if (NULL == scis_hz_) return false;
|
||||
|
||||
scis_splid_ = static_cast<SpellingId *>(malloc(scis_num_ * sizeof(SpellingId)));
|
||||
scis_splid_ = static_cast<SpellingId*>(malloc(scis_num_ * sizeof(SpellingId)));
|
||||
|
||||
if (NULL == scis_splid_) return false;
|
||||
|
||||
@@ -74,7 +74,7 @@ void DictList::free_resource() {
|
||||
}
|
||||
|
||||
#ifdef ___BUILD_MODEL___
|
||||
bool DictList::init_list(const SingleCharItem *scis, size_t scis_num, const LemmaEntry *lemma_arr, size_t lemma_num) {
|
||||
bool DictList::init_list(const SingleCharItem* scis, size_t scis_num, const LemmaEntry* lemma_arr, size_t lemma_num) {
|
||||
if (NULL == scis || 0 == scis_num || NULL == lemma_arr || 0 == lemma_num) return false;
|
||||
|
||||
initialized_ = false;
|
||||
@@ -96,7 +96,7 @@ bool DictList::init_list(const SingleCharItem *scis, size_t scis_num, const Lemm
|
||||
return true;
|
||||
}
|
||||
|
||||
size_t DictList::calculate_size(const LemmaEntry *lemma_arr, size_t lemma_num) {
|
||||
size_t DictList::calculate_size(const LemmaEntry* lemma_arr, size_t lemma_num) {
|
||||
size_t last_hz_len = 0;
|
||||
size_t list_size = 0;
|
||||
size_t id_num = 0;
|
||||
@@ -152,7 +152,7 @@ size_t DictList::calculate_size(const LemmaEntry *lemma_arr, size_t lemma_num) {
|
||||
return start_pos_[kMaxLemmaSize];
|
||||
}
|
||||
|
||||
void DictList::fill_scis(const SingleCharItem *scis, size_t scis_num) {
|
||||
void DictList::fill_scis(const SingleCharItem* scis, size_t scis_num) {
|
||||
assert(scis_num_ == scis_num);
|
||||
|
||||
for (size_t pos = 0; pos < scis_num_; pos++) {
|
||||
@@ -161,7 +161,7 @@ void DictList::fill_scis(const SingleCharItem *scis, size_t scis_num) {
|
||||
}
|
||||
}
|
||||
|
||||
void DictList::fill_list(const LemmaEntry *lemma_arr, size_t lemma_num) {
|
||||
void DictList::fill_list(const LemmaEntry* lemma_arr, size_t lemma_num) {
|
||||
size_t current_pos = 0;
|
||||
|
||||
utf16_strncpy(buf_, lemma_arr[0].hanzi_str, lemma_arr[0].hz_str_len);
|
||||
@@ -181,8 +181,8 @@ void DictList::fill_list(const LemmaEntry *lemma_arr, size_t lemma_num) {
|
||||
assert(id_num == start_id_[kMaxLemmaSize]);
|
||||
}
|
||||
|
||||
char16 *DictList::find_pos2_startedbyhz(char16 hz_char) {
|
||||
char16 *found_2w = static_cast<char16 *>(mybsearch(&hz_char, buf_ + start_pos_[1], (start_pos_[2] - start_pos_[1]) / 2, sizeof(char16) * 2, cmp_hanzis_1));
|
||||
char16* DictList::find_pos2_startedbyhz(char16 hz_char) {
|
||||
char16* found_2w = static_cast<char16*>(mybsearch(&hz_char, buf_ + start_pos_[1], (start_pos_[2] - start_pos_[1]) / 2, sizeof(char16) * 2, cmp_hanzis_1));
|
||||
if (NULL == found_2w) return NULL;
|
||||
|
||||
while (found_2w > buf_ + start_pos_[1] && *found_2w == *(found_2w - 1)) found_2w -= 2;
|
||||
@@ -191,8 +191,8 @@ char16 *DictList::find_pos2_startedbyhz(char16 hz_char) {
|
||||
}
|
||||
#endif // ___BUILD_MODEL___
|
||||
|
||||
char16 *DictList::find_pos_startedbyhzs(const char16 last_hzs[], size_t word_len, int (*cmp_func)(const void *, const void *)) {
|
||||
char16 *found_w = static_cast<char16 *>(mybsearch(last_hzs, buf_ + start_pos_[word_len - 1], (start_pos_[word_len] - start_pos_[word_len - 1]) / word_len, sizeof(char16) * word_len, cmp_func));
|
||||
char16* DictList::find_pos_startedbyhzs(const char16 last_hzs[], size_t word_len, int (*cmp_func)(const void*, const void*)) {
|
||||
char16* found_w = static_cast<char16*>(mybsearch(last_hzs, buf_ + start_pos_[word_len - 1], (start_pos_[word_len] - start_pos_[word_len - 1]) / word_len, sizeof(char16) * word_len, cmp_func));
|
||||
|
||||
if (NULL == found_w) return NULL;
|
||||
|
||||
@@ -201,20 +201,20 @@ char16 *DictList::find_pos_startedbyhzs(const char16 last_hzs[], size_t word_len
|
||||
return found_w;
|
||||
}
|
||||
|
||||
size_t DictList::predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used) {
|
||||
size_t DictList::predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used) {
|
||||
assert(hzs_len <= kMaxPredictSize && hzs_len > 0);
|
||||
|
||||
// 1. Prepare work
|
||||
int (*cmp_func)(const void *, const void *) = cmp_func_[hzs_len - 1];
|
||||
int (*cmp_func)(const void*, const void*) = cmp_func_[hzs_len - 1];
|
||||
|
||||
NGram &ngram = NGram::get_instance();
|
||||
NGram& ngram = NGram::get_instance();
|
||||
|
||||
size_t item_num = 0;
|
||||
|
||||
// 2. Do prediction
|
||||
for (uint16 pre_len = 1; pre_len <= kMaxPredictSize + 1 - hzs_len; pre_len++) {
|
||||
uint16 word_len = hzs_len + pre_len;
|
||||
char16 *w_buf = find_pos_startedbyhzs(last_hzs, word_len, cmp_func);
|
||||
char16* w_buf = find_pos_startedbyhzs(last_hzs, word_len, cmp_func);
|
||||
if (NULL == w_buf) continue;
|
||||
while (w_buf < buf_ + start_pos_[word_len] && cmp_func(w_buf, last_hzs) == 0 && item_num < npre_max) {
|
||||
memset(npre_items + item_num, 0, sizeof(NPredictItem));
|
||||
@@ -243,7 +243,7 @@ size_t DictList::predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *
|
||||
return new_num;
|
||||
}
|
||||
|
||||
uint16 DictList::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max) {
|
||||
uint16 DictList::get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max) {
|
||||
if (!initialized_ || id_lemma >= start_id_[kMaxLemmaSize] || NULL == str_buf || str_max <= 1) return 0;
|
||||
|
||||
// Find the range
|
||||
@@ -252,7 +252,7 @@ uint16 DictList::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str
|
||||
if (start_id_[i] <= id_lemma && start_id_[i + 1] > id_lemma) {
|
||||
size_t id_span = id_lemma - start_id_[i];
|
||||
|
||||
uint16 *buf = buf_ + start_pos_[i] + id_span * (i + 1);
|
||||
uint16* buf = buf_ + start_pos_[i] + id_span * (i + 1);
|
||||
for (uint16 len = 0; len <= i; len++) {
|
||||
str_buf[len] = buf[len];
|
||||
}
|
||||
@@ -263,15 +263,15 @@ uint16 DictList::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint16 DictList::get_splids_for_hanzi(char16 hanzi, uint16 half_splid, uint16 *splids, uint16 max_splids) {
|
||||
char16 *hz_found = static_cast<char16 *>(mybsearch(&hanzi, scis_hz_, scis_num_, sizeof(char16), cmp_hanzis_1));
|
||||
uint16 DictList::get_splids_for_hanzi(char16 hanzi, uint16 half_splid, uint16* splids, uint16 max_splids) {
|
||||
char16* hz_found = static_cast<char16*>(mybsearch(&hanzi, scis_hz_, scis_num_, sizeof(char16), cmp_hanzis_1));
|
||||
assert(NULL != hz_found && hanzi == *hz_found);
|
||||
|
||||
// Move to the first one.
|
||||
while (hz_found > scis_hz_ && hanzi == *(hz_found - 1)) hz_found--;
|
||||
|
||||
// First try to found if strict comparison result is not zero.
|
||||
char16 *hz_f = hz_found;
|
||||
char16* hz_f = hz_found;
|
||||
bool strict = false;
|
||||
while (hz_f < scis_hz_ + scis_num_ && hanzi == *hz_f) {
|
||||
uint16 pos = hz_f - scis_hz_;
|
||||
@@ -295,10 +295,10 @@ uint16 DictList::get_splids_for_hanzi(char16 hanzi, uint16 half_splid, uint16 *s
|
||||
return found_num;
|
||||
}
|
||||
|
||||
LemmaIdType DictList::get_lemma_id(const char16 *str, uint16 str_len) {
|
||||
LemmaIdType DictList::get_lemma_id(const char16* str, uint16 str_len) {
|
||||
if (NULL == str || str_len > kMaxLemmaSize) return 0;
|
||||
|
||||
char16 *found = find_pos_startedbyhzs(str, str_len, cmp_func_[str_len - 1]);
|
||||
char16* found = find_pos_startedbyhzs(str, str_len, cmp_func_[str_len - 1]);
|
||||
if (NULL == found) return 0;
|
||||
|
||||
assert(found > buf_);
|
||||
@@ -306,7 +306,7 @@ LemmaIdType DictList::get_lemma_id(const char16 *str, uint16 str_len) {
|
||||
return static_cast<LemmaIdType>(start_id_[str_len - 1] + (found - buf_ - start_pos_[str_len - 1]) / str_len);
|
||||
}
|
||||
|
||||
void DictList::convert_to_hanzis(char16 *str, uint16 str_len) {
|
||||
void DictList::convert_to_hanzis(char16* str, uint16 str_len) {
|
||||
assert(NULL != str);
|
||||
|
||||
for (uint16 str_pos = 0; str_pos < str_len; str_pos++) {
|
||||
@@ -314,7 +314,7 @@ void DictList::convert_to_hanzis(char16 *str, uint16 str_len) {
|
||||
}
|
||||
}
|
||||
|
||||
void DictList::convert_to_scis_ids(char16 *str, uint16 str_len) {
|
||||
void DictList::convert_to_scis_ids(char16* str, uint16 str_len) {
|
||||
assert(NULL != str);
|
||||
|
||||
for (uint16 str_pos = 0; str_pos < str_len; str_pos++) {
|
||||
@@ -322,7 +322,7 @@ void DictList::convert_to_scis_ids(char16 *str, uint16 str_len) {
|
||||
}
|
||||
}
|
||||
|
||||
bool DictList::save_list(FILE *fp) {
|
||||
bool DictList::save_list(FILE* fp) {
|
||||
if (!initialized_ || NULL == fp) return false;
|
||||
|
||||
if (NULL == buf_ || 0 == start_pos_[kMaxLemmaSize] || NULL == scis_hz_ || NULL == scis_splid_ || 0 == scis_num_) return false;
|
||||
@@ -342,7 +342,7 @@ bool DictList::save_list(FILE *fp) {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool DictList::load_list(FILE *fp) {
|
||||
bool DictList::load_list(FILE* fp) {
|
||||
if (NULL == fp) return false;
|
||||
|
||||
initialized_ = false;
|
||||
|
||||
+77
-77
@@ -75,9 +75,9 @@ void DictTrie::free_resource(bool free_dict_list) {
|
||||
reset_milestones(0, kFirstValidMileStoneHandle);
|
||||
}
|
||||
|
||||
inline size_t DictTrie::get_son_offset(const LmaNodeGE1 *node) { return ((size_t)node->son_1st_off_l + ((size_t)node->son_1st_off_h << 16)); }
|
||||
inline size_t DictTrie::get_son_offset(const LmaNodeGE1* node) { return ((size_t)node->son_1st_off_l + ((size_t)node->son_1st_off_h << 16)); }
|
||||
|
||||
inline size_t DictTrie::get_homo_idx_buf_offset(const LmaNodeGE1 *node) { return ((size_t)node->homo_idx_buf_off_l + ((size_t)node->homo_idx_buf_off_h << 16)); }
|
||||
inline size_t DictTrie::get_homo_idx_buf_offset(const LmaNodeGE1* node) { return ((size_t)node->homo_idx_buf_off_l + ((size_t)node->homo_idx_buf_off_h << 16)); }
|
||||
|
||||
inline LemmaIdType DictTrie::get_lemma_id(size_t id_offset) {
|
||||
LemmaIdType id = 0;
|
||||
@@ -87,15 +87,15 @@ inline LemmaIdType DictTrie::get_lemma_id(size_t id_offset) {
|
||||
}
|
||||
|
||||
#ifdef ___BUILD_MODEL___
|
||||
bool DictTrie::build_dict(const char *fn_raw, const char *fn_validhzs) {
|
||||
DictBuilder *dict_builder = new DictBuilder();
|
||||
bool DictTrie::build_dict(const char* fn_raw, const char* fn_validhzs) {
|
||||
DictBuilder* dict_builder = new DictBuilder();
|
||||
|
||||
free_resource(true);
|
||||
|
||||
return dict_builder->build_dict(fn_raw, fn_validhzs, this);
|
||||
}
|
||||
|
||||
bool DictTrie::save_dict(FILE *fp) {
|
||||
bool DictTrie::save_dict(FILE* fp) {
|
||||
if (NULL == fp) return false;
|
||||
|
||||
if (fwrite(&lma_node_num_le0_, sizeof(uint32), 1, fp) != 1) return false;
|
||||
@@ -115,15 +115,15 @@ bool DictTrie::save_dict(FILE *fp) {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool DictTrie::save_dict(const char *filename) {
|
||||
bool DictTrie::save_dict(const char* filename) {
|
||||
if (NULL == filename) return false;
|
||||
|
||||
if (NULL == root_ || NULL == dict_list_) return false;
|
||||
|
||||
SpellingTrie &spl_trie = SpellingTrie::get_instance();
|
||||
NGram &ngram = NGram::get_instance();
|
||||
SpellingTrie& spl_trie = SpellingTrie::get_instance();
|
||||
NGram& ngram = NGram::get_instance();
|
||||
|
||||
FILE *fp = fopen(filename, "wb");
|
||||
FILE* fp = fopen(filename, "wb");
|
||||
if (NULL == fp) return false;
|
||||
|
||||
if (!spl_trie.save_spl_trie(fp) || !dict_list_->save_list(fp) || !save_dict(fp) || !ngram.save_ngram(fp)) {
|
||||
@@ -136,7 +136,7 @@ bool DictTrie::save_dict(const char *filename) {
|
||||
}
|
||||
#endif // ___BUILD_MODEL___
|
||||
|
||||
bool DictTrie::load_dict(FILE *fp) {
|
||||
bool DictTrie::load_dict(FILE* fp) {
|
||||
if (NULL == fp) return false;
|
||||
if (fread(&lma_node_num_le0_, sizeof(uint32), 1, fp) != 1) return false;
|
||||
|
||||
@@ -148,14 +148,14 @@ bool DictTrie::load_dict(FILE *fp) {
|
||||
|
||||
free_resource(false);
|
||||
|
||||
root_ = static_cast<LmaNodeLE0 *>(malloc(lma_node_num_le0_ * sizeof(LmaNodeLE0)));
|
||||
nodes_ge1_ = static_cast<LmaNodeGE1 *>(malloc(lma_node_num_ge1_ * sizeof(LmaNodeGE1)));
|
||||
lma_idx_buf_ = (unsigned char *)malloc(lma_idx_buf_len_);
|
||||
root_ = static_cast<LmaNodeLE0*>(malloc(lma_node_num_le0_ * sizeof(LmaNodeLE0)));
|
||||
nodes_ge1_ = static_cast<LmaNodeGE1*>(malloc(lma_node_num_ge1_ * sizeof(LmaNodeGE1)));
|
||||
lma_idx_buf_ = (unsigned char*)malloc(lma_idx_buf_len_);
|
||||
total_lma_num_ = lma_idx_buf_len_ / kLemmaIdSize;
|
||||
|
||||
size_t buf_size = SpellingTrie::get_instance().get_spelling_num() + 1;
|
||||
assert(lma_node_num_le0_ <= buf_size);
|
||||
splid_le0_index_ = static_cast<uint16 *>(malloc(buf_size * sizeof(uint16)));
|
||||
splid_le0_index_ = static_cast<uint16*>(malloc(buf_size * sizeof(uint16)));
|
||||
|
||||
// Init the space for parsing.
|
||||
parsing_marks_ = new ParsingMark[kMaxParsingMark];
|
||||
@@ -192,10 +192,10 @@ bool DictTrie::load_dict(FILE *fp) {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool DictTrie::load_dict(const char *filename, LemmaIdType start_id, LemmaIdType end_id) {
|
||||
bool DictTrie::load_dict(const char* filename, LemmaIdType start_id, LemmaIdType end_id) {
|
||||
if (NULL == filename || end_id <= start_id) return false;
|
||||
|
||||
FILE *fp = fopen(filename, "rb");
|
||||
FILE* fp = fopen(filename, "rb");
|
||||
if (NULL == fp) return false;
|
||||
|
||||
free_resource(true);
|
||||
@@ -206,8 +206,8 @@ bool DictTrie::load_dict(const char *filename, LemmaIdType start_id, LemmaIdType
|
||||
return false;
|
||||
}
|
||||
|
||||
SpellingTrie &spl_trie = SpellingTrie::get_instance();
|
||||
NGram &ngram = NGram::get_instance();
|
||||
SpellingTrie& spl_trie = SpellingTrie::get_instance();
|
||||
NGram& ngram = NGram::get_instance();
|
||||
|
||||
if (!spl_trie.load_spl_trie(fp) || !dict_list_->load_list(fp) || !load_dict(fp) || !ngram.load_ngram(fp) || total_lma_num_ > end_id - start_id + 1) {
|
||||
free_resource(true);
|
||||
@@ -222,7 +222,7 @@ bool DictTrie::load_dict(const char *filename, LemmaIdType start_id, LemmaIdType
|
||||
bool DictTrie::load_dict_fd(int sys_fd, long start_offset, long length, LemmaIdType start_id, LemmaIdType end_id) {
|
||||
if (start_offset < 0 || length <= 0 || end_id <= start_id) return false;
|
||||
|
||||
FILE *fp = fdopen(sys_fd, "rb");
|
||||
FILE* fp = fdopen(sys_fd, "rb");
|
||||
if (NULL == fp) return false;
|
||||
|
||||
if (-1 == fseek(fp, start_offset, SEEK_SET)) {
|
||||
@@ -238,8 +238,8 @@ bool DictTrie::load_dict_fd(int sys_fd, long start_offset, long length, LemmaIdT
|
||||
return false;
|
||||
}
|
||||
|
||||
SpellingTrie &spl_trie = SpellingTrie::get_instance();
|
||||
NGram &ngram = NGram::get_instance();
|
||||
SpellingTrie& spl_trie = SpellingTrie::get_instance();
|
||||
NGram& ngram = NGram::get_instance();
|
||||
|
||||
if (!spl_trie.load_spl_trie(fp) || !dict_list_->load_list(fp) || !load_dict(fp) || !ngram.load_ngram(fp) || ftell(fp) < start_offset + length || total_lma_num_ > end_id - start_id + 1) {
|
||||
free_resource(true);
|
||||
@@ -251,9 +251,9 @@ bool DictTrie::load_dict_fd(int sys_fd, long start_offset, long length, LemmaIdT
|
||||
return true;
|
||||
}
|
||||
|
||||
size_t DictTrie::fill_lpi_buffer(LmaPsbItem lpi_items[], size_t lpi_max, LmaNodeLE0 *node) {
|
||||
size_t DictTrie::fill_lpi_buffer(LmaPsbItem lpi_items[], size_t lpi_max, LmaNodeLE0* node) {
|
||||
size_t lpi_num = 0;
|
||||
NGram &ngram = NGram::get_instance();
|
||||
NGram& ngram = NGram::get_instance();
|
||||
for (size_t homo = 0; homo < (size_t)node->num_of_homo; homo++) {
|
||||
lpi_items[lpi_num].id = get_lemma_id(node->homo_idx_buf_off + homo);
|
||||
lpi_items[lpi_num].lma_len = 1;
|
||||
@@ -265,9 +265,9 @@ size_t DictTrie::fill_lpi_buffer(LmaPsbItem lpi_items[], size_t lpi_max, LmaNode
|
||||
return lpi_num;
|
||||
}
|
||||
|
||||
size_t DictTrie::fill_lpi_buffer(LmaPsbItem lpi_items[], size_t lpi_max, size_t homo_buf_off, LmaNodeGE1 *node, uint16 lma_len) {
|
||||
size_t DictTrie::fill_lpi_buffer(LmaPsbItem lpi_items[], size_t lpi_max, size_t homo_buf_off, LmaNodeGE1* node, uint16 lma_len) {
|
||||
size_t lpi_num = 0;
|
||||
NGram &ngram = NGram::get_instance();
|
||||
NGram& ngram = NGram::get_instance();
|
||||
for (size_t homo = 0; homo < (size_t)node->num_of_homo; homo++) {
|
||||
lpi_items[lpi_num].id = get_lemma_id(homo_buf_off + homo);
|
||||
lpi_items[lpi_num].lma_len = lma_len;
|
||||
@@ -287,13 +287,13 @@ void DictTrie::reset_milestones(uint16 from_step, MileStoneHandle from_handle) {
|
||||
if (from_handle > 0 && from_handle < mile_stones_pos_) {
|
||||
mile_stones_pos_ = from_handle;
|
||||
|
||||
MileStone *mile_stone = mile_stones_ + from_handle;
|
||||
MileStone* mile_stone = mile_stones_ + from_handle;
|
||||
parsing_marks_pos_ = mile_stone->mark_start;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
MileStoneHandle DictTrie::extend_dict(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num) {
|
||||
MileStoneHandle DictTrie::extend_dict(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num) {
|
||||
if (NULL == dep) return 0;
|
||||
|
||||
// from LmaNodeLE0 (root) to LmaNodeLE0
|
||||
@@ -309,7 +309,7 @@ MileStoneHandle DictTrie::extend_dict(MileStoneHandle from_handle, const DictExt
|
||||
return extend_dict2(from_handle, dep, lpi_items, lpi_max, lpi_num);
|
||||
}
|
||||
|
||||
MileStoneHandle DictTrie::extend_dict0(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num) {
|
||||
MileStoneHandle DictTrie::extend_dict0(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num) {
|
||||
assert(NULL != dep && 0 == from_handle);
|
||||
*lpi_num = 0;
|
||||
MileStoneHandle ret_handle = 0;
|
||||
@@ -318,17 +318,17 @@ MileStoneHandle DictTrie::extend_dict0(MileStoneHandle from_handle, const DictEx
|
||||
uint16 id_start = dep->id_start;
|
||||
uint16 id_num = dep->id_num;
|
||||
|
||||
LpiCache &lpi_cache = LpiCache::get_instance();
|
||||
LpiCache& lpi_cache = LpiCache::get_instance();
|
||||
bool cached = lpi_cache.is_cached(splid);
|
||||
|
||||
// 2. Begin exgtending
|
||||
// 2.1 Get the LmaPsbItem list
|
||||
LmaNodeLE0 *node = root_;
|
||||
LmaNodeLE0* node = root_;
|
||||
size_t son_start = splid_le0_index_[id_start - kFullSplIdStart];
|
||||
size_t son_end = splid_le0_index_[id_start + id_num - kFullSplIdStart];
|
||||
for (size_t son_pos = son_start; son_pos < son_end; son_pos++) {
|
||||
assert(1 == node->son_1st_off);
|
||||
LmaNodeLE0 *son = root_ + son_pos;
|
||||
LmaNodeLE0* son = root_ + son_pos;
|
||||
assert(son->spl_idx >= id_start && son->spl_idx < id_start + id_num);
|
||||
|
||||
if (!cached && *lpi_num < lpi_max) {
|
||||
@@ -359,7 +359,7 @@ MileStoneHandle DictTrie::extend_dict0(MileStoneHandle from_handle, const DictEx
|
||||
return ret_handle;
|
||||
}
|
||||
|
||||
MileStoneHandle DictTrie::extend_dict1(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num) {
|
||||
MileStoneHandle DictTrie::extend_dict1(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num) {
|
||||
assert(NULL != dep && from_handle > 0 && from_handle < mile_stones_pos_);
|
||||
|
||||
MileStoneHandle ret_handle = 0;
|
||||
@@ -372,18 +372,18 @@ MileStoneHandle DictTrie::extend_dict1(MileStoneHandle from_handle, const DictEx
|
||||
uint16 id_num = dep->id_num;
|
||||
|
||||
// 2. Begin extending.
|
||||
MileStone *mile_stone = mile_stones_ + from_handle;
|
||||
MileStone* mile_stone = mile_stones_ + from_handle;
|
||||
|
||||
for (uint16 h_pos = 0; h_pos < mile_stone->mark_num; h_pos++) {
|
||||
ParsingMark p_mark = parsing_marks_[mile_stone->mark_start + h_pos];
|
||||
uint16 ext_num = p_mark.node_num;
|
||||
for (uint16 ext_pos = 0; ext_pos < ext_num; ext_pos++) {
|
||||
LmaNodeLE0 *node = root_ + p_mark.node_offset + ext_pos;
|
||||
LmaNodeLE0* node = root_ + p_mark.node_offset + ext_pos;
|
||||
size_t found_start = 0;
|
||||
size_t found_num = 0;
|
||||
for (size_t son_pos = 0; son_pos < (size_t)node->num_of_son; son_pos++) {
|
||||
assert(node->son_1st_off <= lma_node_num_ge1_);
|
||||
LmaNodeGE1 *son = nodes_ge1_ + node->son_1st_off + son_pos;
|
||||
LmaNodeGE1* son = nodes_ge1_ + node->son_1st_off + son_pos;
|
||||
if (son->spl_idx >= id_start && son->spl_idx < id_start + id_num) {
|
||||
if (*lpi_num < lpi_max) {
|
||||
size_t homo_buf_off = get_homo_idx_buf_offset(son);
|
||||
@@ -425,7 +425,7 @@ MileStoneHandle DictTrie::extend_dict1(MileStoneHandle from_handle, const DictEx
|
||||
return ret_handle;
|
||||
}
|
||||
|
||||
MileStoneHandle DictTrie::extend_dict2(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num) {
|
||||
MileStoneHandle DictTrie::extend_dict2(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num) {
|
||||
assert(NULL != dep && from_handle > 0 && from_handle < mile_stones_pos_);
|
||||
|
||||
MileStoneHandle ret_handle = 0;
|
||||
@@ -438,19 +438,19 @@ MileStoneHandle DictTrie::extend_dict2(MileStoneHandle from_handle, const DictEx
|
||||
uint16 id_num = dep->id_num;
|
||||
|
||||
// 2. Begin extending.
|
||||
MileStone *mile_stone = mile_stones_ + from_handle;
|
||||
MileStone* mile_stone = mile_stones_ + from_handle;
|
||||
|
||||
for (uint16 h_pos = 0; h_pos < mile_stone->mark_num; h_pos++) {
|
||||
ParsingMark p_mark = parsing_marks_[mile_stone->mark_start + h_pos];
|
||||
uint16 ext_num = p_mark.node_num;
|
||||
for (uint16 ext_pos = 0; ext_pos < ext_num; ext_pos++) {
|
||||
LmaNodeGE1 *node = nodes_ge1_ + p_mark.node_offset + ext_pos;
|
||||
LmaNodeGE1* node = nodes_ge1_ + p_mark.node_offset + ext_pos;
|
||||
size_t found_start = 0;
|
||||
size_t found_num = 0;
|
||||
|
||||
for (size_t son_pos = 0; son_pos < (size_t)node->num_of_son; son_pos++) {
|
||||
assert(node->son_1st_off_l > 0 || node->son_1st_off_h > 0);
|
||||
LmaNodeGE1 *son = nodes_ge1_ + get_son_offset(node) + son_pos;
|
||||
LmaNodeGE1* son = nodes_ge1_ + get_son_offset(node) + son_pos;
|
||||
if (son->spl_idx >= id_start && son->spl_idx < id_start + id_num) {
|
||||
if (*lpi_num < lpi_max) {
|
||||
size_t homo_buf_off = get_homo_idx_buf_offset(son);
|
||||
@@ -491,15 +491,15 @@ MileStoneHandle DictTrie::extend_dict2(MileStoneHandle from_handle, const DictEx
|
||||
return ret_handle;
|
||||
}
|
||||
|
||||
bool DictTrie::try_extend(const uint16 *splids, uint16 splid_num, LemmaIdType id_lemma) {
|
||||
bool DictTrie::try_extend(const uint16* splids, uint16 splid_num, LemmaIdType id_lemma) {
|
||||
if (0 == splid_num || NULL == splids) return false;
|
||||
|
||||
void *node = root_ + splid_le0_index_[splids[0] - kFullSplIdStart];
|
||||
void* node = root_ + splid_le0_index_[splids[0] - kFullSplIdStart];
|
||||
|
||||
for (uint16 pos = 1; pos < splid_num; pos++) {
|
||||
if (1 == pos) {
|
||||
LmaNodeLE0 *node_le0 = reinterpret_cast<LmaNodeLE0 *>(node);
|
||||
LmaNodeGE1 *node_son;
|
||||
LmaNodeLE0* node_le0 = reinterpret_cast<LmaNodeLE0*>(node);
|
||||
LmaNodeGE1* node_son;
|
||||
uint16 son_pos;
|
||||
for (son_pos = 0; son_pos < static_cast<uint16>(node_le0->num_of_son); son_pos++) {
|
||||
assert(node_le0->son_1st_off <= lma_node_num_ge1_);
|
||||
@@ -507,12 +507,12 @@ bool DictTrie::try_extend(const uint16 *splids, uint16 splid_num, LemmaIdType id
|
||||
if (node_son->spl_idx == splids[pos]) break;
|
||||
}
|
||||
if (son_pos < node_le0->num_of_son)
|
||||
node = reinterpret_cast<void *>(node_son);
|
||||
node = reinterpret_cast<void*>(node_son);
|
||||
else
|
||||
return false;
|
||||
} else {
|
||||
LmaNodeGE1 *node_ge1 = reinterpret_cast<LmaNodeGE1 *>(node);
|
||||
LmaNodeGE1 *node_son;
|
||||
LmaNodeGE1* node_ge1 = reinterpret_cast<LmaNodeGE1*>(node);
|
||||
LmaNodeGE1* node_son;
|
||||
uint16 son_pos;
|
||||
for (son_pos = 0; son_pos < static_cast<uint16>(node_ge1->num_of_son); son_pos++) {
|
||||
assert(node_ge1->son_1st_off_l > 0 || node_ge1->son_1st_off_h > 0);
|
||||
@@ -520,14 +520,14 @@ bool DictTrie::try_extend(const uint16 *splids, uint16 splid_num, LemmaIdType id
|
||||
if (node_son->spl_idx == splids[pos]) break;
|
||||
}
|
||||
if (son_pos < node_ge1->num_of_son)
|
||||
node = reinterpret_cast<void *>(node_son);
|
||||
node = reinterpret_cast<void*>(node_son);
|
||||
else
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
if (1 == splid_num) {
|
||||
LmaNodeLE0 *node_le0 = reinterpret_cast<LmaNodeLE0 *>(node);
|
||||
LmaNodeLE0* node_le0 = reinterpret_cast<LmaNodeLE0*>(node);
|
||||
size_t num_of_homo = (size_t)node_le0->num_of_homo;
|
||||
for (size_t homo_pos = 0; homo_pos < num_of_homo; homo_pos++) {
|
||||
LemmaIdType id_this = get_lemma_id(node_le0->homo_idx_buf_off + homo_pos);
|
||||
@@ -536,7 +536,7 @@ bool DictTrie::try_extend(const uint16 *splids, uint16 splid_num, LemmaIdType id
|
||||
if (id_this == id_lemma) return true;
|
||||
}
|
||||
} else {
|
||||
LmaNodeGE1 *node_ge1 = reinterpret_cast<LmaNodeGE1 *>(node);
|
||||
LmaNodeGE1* node_ge1 = reinterpret_cast<LmaNodeGE1*>(node);
|
||||
size_t num_of_homo = (size_t)node_ge1->num_of_homo;
|
||||
for (size_t homo_pos = 0; homo_pos < num_of_homo; homo_pos++) {
|
||||
size_t node_homo_off = get_homo_idx_buf_offset(node_ge1);
|
||||
@@ -547,17 +547,17 @@ bool DictTrie::try_extend(const uint16 *splids, uint16 splid_num, LemmaIdType id
|
||||
return false;
|
||||
}
|
||||
|
||||
size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lma_buf, size_t max_lma_buf) {
|
||||
size_t DictTrie::get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lma_buf, size_t max_lma_buf) {
|
||||
if (splid_str_len > kMaxLemmaSize) return 0;
|
||||
|
||||
#define MAX_EXTENDBUF_LEN 200
|
||||
|
||||
size_t *node_buf1[MAX_EXTENDBUF_LEN]; // use size_t for data alignment
|
||||
size_t *node_buf2[MAX_EXTENDBUF_LEN];
|
||||
LmaNodeLE0 **node_fr_le0 = reinterpret_cast<LmaNodeLE0 **>(node_buf1); // Nodes from.
|
||||
LmaNodeLE0 **node_to_le0 = reinterpret_cast<LmaNodeLE0 **>(node_buf2); // Nodes to.
|
||||
LmaNodeGE1 **node_fr_ge1 = NULL;
|
||||
LmaNodeGE1 **node_to_ge1 = NULL;
|
||||
size_t* node_buf1[MAX_EXTENDBUF_LEN]; // use size_t for data alignment
|
||||
size_t* node_buf2[MAX_EXTENDBUF_LEN];
|
||||
LmaNodeLE0** node_fr_le0 = reinterpret_cast<LmaNodeLE0**>(node_buf1); // Nodes from.
|
||||
LmaNodeLE0** node_to_le0 = reinterpret_cast<LmaNodeLE0**>(node_buf2); // Nodes to.
|
||||
LmaNodeGE1** node_fr_ge1 = NULL;
|
||||
LmaNodeGE1** node_to_ge1 = NULL;
|
||||
size_t node_fr_num = 1;
|
||||
size_t node_to_num = 0;
|
||||
node_fr_le0[0] = root_;
|
||||
@@ -577,13 +577,13 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
|
||||
// Extend the nodes
|
||||
if (0 == spl_pos) { // From LmaNodeLE0 (root) to LmaNodeLE0 nodes
|
||||
for (size_t node_fr_pos = 0; node_fr_pos < node_fr_num; node_fr_pos++) {
|
||||
LmaNodeLE0 *node = node_fr_le0[node_fr_pos];
|
||||
LmaNodeLE0* node = node_fr_le0[node_fr_pos];
|
||||
assert(node == root_ && 1 == node_fr_num);
|
||||
size_t son_start = splid_le0_index_[id_start - kFullSplIdStart];
|
||||
size_t son_end = splid_le0_index_[id_start + id_num - kFullSplIdStart];
|
||||
for (size_t son_pos = son_start; son_pos < son_end; son_pos++) {
|
||||
assert(1 == node->son_1st_off);
|
||||
LmaNodeLE0 *node_son = root_ + son_pos;
|
||||
LmaNodeLE0* node_son = root_ + son_pos;
|
||||
assert(node_son->spl_idx >= id_start && node_son->spl_idx < id_start + id_num);
|
||||
if (node_to_num < MAX_EXTENDBUF_LEN) {
|
||||
node_to_le0[node_to_num] = node_son;
|
||||
@@ -599,16 +599,16 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
|
||||
if (spl_pos >= splid_str_len || node_to_num == 0) break;
|
||||
// Prepare the nodes for next extending
|
||||
// next time, from LmaNodeLE0 to LmaNodeGE1
|
||||
LmaNodeLE0 **node_tmp = node_fr_le0;
|
||||
LmaNodeLE0** node_tmp = node_fr_le0;
|
||||
node_fr_le0 = node_to_le0;
|
||||
node_to_le0 = NULL;
|
||||
node_to_ge1 = reinterpret_cast<LmaNodeGE1 **>(node_tmp);
|
||||
node_to_ge1 = reinterpret_cast<LmaNodeGE1**>(node_tmp);
|
||||
} else if (1 == spl_pos) { // From LmaNodeLE0 to LmaNodeGE1 nodes
|
||||
for (size_t node_fr_pos = 0; node_fr_pos < node_fr_num; node_fr_pos++) {
|
||||
LmaNodeLE0 *node = node_fr_le0[node_fr_pos];
|
||||
LmaNodeLE0* node = node_fr_le0[node_fr_pos];
|
||||
for (size_t son_pos = 0; son_pos < (size_t)node->num_of_son; son_pos++) {
|
||||
assert(node->son_1st_off <= lma_node_num_ge1_);
|
||||
LmaNodeGE1 *node_son = nodes_ge1_ + node->son_1st_off + son_pos;
|
||||
LmaNodeGE1* node_son = nodes_ge1_ + node->son_1st_off + son_pos;
|
||||
if (node_son->spl_idx >= id_start && node_son->spl_idx < id_start + id_num) {
|
||||
if (node_to_num < MAX_EXTENDBUF_LEN) {
|
||||
node_to_ge1[node_to_num] = node_son;
|
||||
@@ -626,15 +626,15 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
|
||||
// Prepare the nodes for next extending
|
||||
// next time, from LmaNodeGE1 to LmaNodeGE1
|
||||
node_fr_ge1 = node_to_ge1;
|
||||
node_to_ge1 = reinterpret_cast<LmaNodeGE1 **>(node_fr_le0);
|
||||
node_to_ge1 = reinterpret_cast<LmaNodeGE1**>(node_fr_le0);
|
||||
node_fr_le0 = NULL;
|
||||
node_to_le0 = NULL;
|
||||
} else { // From LmaNodeGE1 to LmaNodeGE1 nodes
|
||||
for (size_t node_fr_pos = 0; node_fr_pos < node_fr_num; node_fr_pos++) {
|
||||
LmaNodeGE1 *node = node_fr_ge1[node_fr_pos];
|
||||
LmaNodeGE1* node = node_fr_ge1[node_fr_pos];
|
||||
for (size_t son_pos = 0; son_pos < (size_t)node->num_of_son; son_pos++) {
|
||||
assert(node->son_1st_off_l > 0 || node->son_1st_off_h > 0);
|
||||
LmaNodeGE1 *node_son = nodes_ge1_ + get_son_offset(node) + son_pos;
|
||||
LmaNodeGE1* node_son = nodes_ge1_ + get_son_offset(node) + son_pos;
|
||||
if (node_son->spl_idx >= id_start && node_son->spl_idx < id_start + id_num) {
|
||||
if (node_to_num < MAX_EXTENDBUF_LEN) {
|
||||
node_to_ge1[node_to_num] = node_son;
|
||||
@@ -651,7 +651,7 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
|
||||
if (spl_pos >= splid_str_len || node_to_num == 0) break;
|
||||
// Prepare the nodes for next extending
|
||||
// next time, from LmaNodeGE1 to LmaNodeGE1
|
||||
LmaNodeGE1 **node_tmp = node_fr_ge1;
|
||||
LmaNodeGE1** node_tmp = node_fr_ge1;
|
||||
node_fr_ge1 = node_to_ge1;
|
||||
node_to_ge1 = node_tmp;
|
||||
}
|
||||
@@ -663,7 +663,7 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
|
||||
|
||||
if (0 == node_to_num) return 0;
|
||||
|
||||
NGram &ngram = NGram::get_instance();
|
||||
NGram& ngram = NGram::get_instance();
|
||||
size_t lma_num = 0;
|
||||
|
||||
// If the length is 1, and the splid is a one-char Yunmu like 'a', 'o', 'e',
|
||||
@@ -673,7 +673,7 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
|
||||
for (size_t node_pos = 0; node_pos < node_to_num; node_pos++) {
|
||||
size_t num_of_homo = 0;
|
||||
if (spl_pos <= 1) { // Get from LmaNodeLE0 nodes
|
||||
LmaNodeLE0 *node_le0 = node_to_le0[node_pos];
|
||||
LmaNodeLE0* node_le0 = node_to_le0[node_pos];
|
||||
num_of_homo = (size_t)node_le0->num_of_homo;
|
||||
for (size_t homo_pos = 0; homo_pos < num_of_homo; homo_pos++) {
|
||||
size_t ch_pos = lma_num + homo_pos;
|
||||
@@ -684,7 +684,7 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
|
||||
if (lma_num + homo_pos >= max_lma_buf - 1) break;
|
||||
}
|
||||
} else { // Get from LmaNodeGE1 nodes
|
||||
LmaNodeGE1 *node_ge1 = node_to_ge1[node_pos];
|
||||
LmaNodeGE1* node_ge1 = node_to_ge1[node_pos];
|
||||
num_of_homo = (size_t)node_ge1->num_of_homo;
|
||||
for (size_t homo_pos = 0; homo_pos < num_of_homo; homo_pos++) {
|
||||
size_t ch_pos = lma_num + homo_pos;
|
||||
@@ -706,9 +706,9 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
|
||||
return lma_num;
|
||||
}
|
||||
|
||||
uint16 DictTrie::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max) { return dict_list_->get_lemma_str(id_lemma, str_buf, str_max); }
|
||||
uint16 DictTrie::get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max) { return dict_list_->get_lemma_str(id_lemma, str_buf, str_max); }
|
||||
|
||||
uint16 DictTrie::get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid) {
|
||||
uint16 DictTrie::get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid) {
|
||||
char16 lma_str[kMaxLemmaSize + 1];
|
||||
uint16 lma_len = get_lemma_str(id_lemma, lma_str, kMaxLemmaSize + 1);
|
||||
assert((!arg_valid && splids_max >= lma_len) || lma_len == splids_max);
|
||||
@@ -746,13 +746,13 @@ uint16 DictTrie::get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 s
|
||||
}
|
||||
|
||||
void DictTrie::set_total_lemma_count_of_others(size_t count) {
|
||||
NGram &ngram = NGram::get_instance();
|
||||
NGram& ngram = NGram::get_instance();
|
||||
ngram.set_total_freq_none_sys(count);
|
||||
}
|
||||
|
||||
void DictTrie::convert_to_hanzis(char16 *str, uint16 str_len) { return dict_list_->convert_to_hanzis(str, str_len); }
|
||||
void DictTrie::convert_to_hanzis(char16* str, uint16 str_len) { return dict_list_->convert_to_hanzis(str, str_len); }
|
||||
|
||||
void DictTrie::convert_to_scis_ids(char16 *str, uint16 str_len) { return dict_list_->convert_to_scis_ids(str, str_len); }
|
||||
void DictTrie::convert_to_scis_ids(char16* str, uint16 str_len) { return dict_list_->convert_to_scis_ids(str, str_len); }
|
||||
|
||||
LemmaIdType DictTrie::get_lemma_id(const char16 lemma_str[], uint16 lemma_len) {
|
||||
if (NULL == lemma_str || lemma_len > kMaxLemmaSize) return 0;
|
||||
@@ -760,8 +760,8 @@ LemmaIdType DictTrie::get_lemma_id(const char16 lemma_str[], uint16 lemma_len) {
|
||||
return dict_list_->get_lemma_id(lemma_str, lemma_len);
|
||||
}
|
||||
|
||||
size_t DictTrie::predict_top_lmas(size_t his_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used) {
|
||||
NGram &ngram = NGram::get_instance();
|
||||
size_t DictTrie::predict_top_lmas(size_t his_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used) {
|
||||
NGram& ngram = NGram::get_instance();
|
||||
|
||||
size_t item_num = 0;
|
||||
size_t top_lmas_id_offset = lma_idx_buf_len_ / kLemmaIdSize - top_lmas_num_;
|
||||
@@ -780,5 +780,5 @@ size_t DictTrie::predict_top_lmas(size_t his_len, NPredictItem *npre_items, size
|
||||
return item_num;
|
||||
}
|
||||
|
||||
size_t DictTrie::predict(const char16 *last_hzs, uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used) { return dict_list_->predict(last_hzs, hzs_len, npre_items, npre_max, b4_used); }
|
||||
size_t DictTrie::predict(const char16* last_hzs, uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used) { return dict_list_->predict(last_hzs, hzs_len, npre_items, npre_max, b4_used); }
|
||||
} // namespace ime_pinyin
|
||||
|
||||
+44
-44
@@ -69,7 +69,7 @@ bool MatrixSearch::alloc_resource() {
|
||||
free_resource();
|
||||
|
||||
dict_trie_ = new DictTrie();
|
||||
user_dict_ = static_cast<AtomDictBase *>(new UserDict());
|
||||
user_dict_ = static_cast<AtomDictBase*>(new UserDict());
|
||||
spl_parser_ = new SpellingParser();
|
||||
|
||||
size_t mtrx_nd_size = sizeof(MatrixNode) * kMtrxNdPoolSize;
|
||||
@@ -87,13 +87,13 @@ bool MatrixSearch::alloc_resource() {
|
||||
if (NULL == dict_trie_ || NULL == user_dict_ || NULL == spl_parser_ || NULL == share_buf_) return false;
|
||||
|
||||
// The buffers for search are based on the share buffer
|
||||
mtrx_nd_pool_ = reinterpret_cast<MatrixNode *>(share_buf_);
|
||||
dmi_pool_ = reinterpret_cast<DictMatchInfo *>(share_buf_ + mtrx_nd_size);
|
||||
matrix_ = reinterpret_cast<MatrixRow *>(share_buf_ + mtrx_nd_size + dmi_size);
|
||||
dep_ = reinterpret_cast<DictExtPara *>(share_buf_ + mtrx_nd_size + dmi_size + matrix_size);
|
||||
mtrx_nd_pool_ = reinterpret_cast<MatrixNode*>(share_buf_);
|
||||
dmi_pool_ = reinterpret_cast<DictMatchInfo*>(share_buf_ + mtrx_nd_size);
|
||||
matrix_ = reinterpret_cast<MatrixRow*>(share_buf_ + mtrx_nd_size + dmi_size);
|
||||
dep_ = reinterpret_cast<DictExtPara*>(share_buf_ + mtrx_nd_size + dmi_size + matrix_size);
|
||||
|
||||
// The prediction buffer is also based on the share buffer.
|
||||
npre_items_ = reinterpret_cast<NPredictItem *>(share_buf_);
|
||||
npre_items_ = reinterpret_cast<NPredictItem*>(share_buf_);
|
||||
npre_items_len_ = (mtrx_nd_size + dmi_size + matrix_size + dep_size) * sizeof(size_t) / sizeof(NPredictItem);
|
||||
return true;
|
||||
}
|
||||
@@ -110,7 +110,7 @@ void MatrixSearch::free_resource() {
|
||||
reset_pointers_to_null();
|
||||
}
|
||||
|
||||
bool MatrixSearch::init(const char *fn_sys_dict, const char *fn_usr_dict) {
|
||||
bool MatrixSearch::init(const char* fn_sys_dict, const char* fn_usr_dict) {
|
||||
if (NULL == fn_sys_dict || NULL == fn_usr_dict) return false;
|
||||
|
||||
if (!alloc_resource()) return false;
|
||||
@@ -132,7 +132,7 @@ bool MatrixSearch::init(const char *fn_sys_dict, const char *fn_usr_dict) {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool MatrixSearch::init_fd(int sys_fd, long start_offset, long length, const char *fn_usr_dict) {
|
||||
bool MatrixSearch::init_fd(int sys_fd, long start_offset, long length, const char* fn_usr_dict) {
|
||||
if (NULL == fn_usr_dict) return false;
|
||||
|
||||
if (!alloc_resource()) return false;
|
||||
@@ -189,7 +189,7 @@ bool MatrixSearch::reset_search0() {
|
||||
mtrx_nd_pool_used_ += 1;
|
||||
|
||||
// Update the node, and make it to be a starting node
|
||||
MatrixNode *node = mtrx_nd_pool_ + matrix_[0].mtrx_nd_pos;
|
||||
MatrixNode* node = mtrx_nd_pool_ + matrix_[0].mtrx_nd_pos;
|
||||
node->id = 0;
|
||||
node->score = 0;
|
||||
node->from = NULL;
|
||||
@@ -219,7 +219,7 @@ bool MatrixSearch::reset_search(size_t ch_pos, bool clear_fixed_this_step, bool
|
||||
reset_search0();
|
||||
} else {
|
||||
// Prepare mile stones of this step to clear.
|
||||
MileStoneHandle *dict_handles_to_clear = NULL;
|
||||
MileStoneHandle* dict_handles_to_clear = NULL;
|
||||
if (clear_dmi_this_step && matrix_[ch_pos].dmi_num > 0) {
|
||||
dict_handles_to_clear = dmi_pool_[matrix_[ch_pos].dmi_pos].dict_handles;
|
||||
}
|
||||
@@ -274,7 +274,7 @@ bool MatrixSearch::reset_search(size_t ch_pos, bool clear_fixed_this_step, bool
|
||||
// which was previously fixed.
|
||||
//
|
||||
// Prepare mile stones of this step to clear.
|
||||
MileStoneHandle *dict_handles_to_clear = NULL;
|
||||
MileStoneHandle* dict_handles_to_clear = NULL;
|
||||
if (clear_dmi_this_step && ch_pos == fixed_ch_pos && matrix_[fixed_ch_pos].dmi_num > 0) {
|
||||
dict_handles_to_clear = dmi_pool_[matrix_[fixed_ch_pos].dmi_pos].dict_handles;
|
||||
}
|
||||
@@ -365,7 +365,7 @@ void MatrixSearch::del_in_pys(size_t start, size_t len) {
|
||||
}
|
||||
}
|
||||
|
||||
size_t MatrixSearch::search(const char *py, size_t py_len) {
|
||||
size_t MatrixSearch::search(const char* py, size_t py_len) {
|
||||
if (!inited_ || NULL == py) return 0;
|
||||
|
||||
// If the search Pinyin string is too long, it will be truncated.
|
||||
@@ -538,7 +538,7 @@ size_t MatrixSearch::get_candidate_num() {
|
||||
return 1 + lpi_total_;
|
||||
}
|
||||
|
||||
char16 *MatrixSearch::get_candidate(size_t cand_id, char16 *cand_str, size_t max_len) {
|
||||
char16* MatrixSearch::get_candidate(size_t cand_id, char16* cand_str, size_t max_len) {
|
||||
if (!inited_ || 0 == pys_decoded_len_ || NULL == cand_str) return NULL;
|
||||
|
||||
if (0 == cand_id) {
|
||||
@@ -613,13 +613,13 @@ bool MatrixSearch::add_lma_to_userdict(uint16 lma_fr, uint16 lma_to, float score
|
||||
|
||||
assert(spl_id_fr <= kMaxLemmaSize);
|
||||
|
||||
return user_dict_->put_lemma(static_cast<char16 *>(word_str), spl_ids, spl_id_fr, 1);
|
||||
return user_dict_->put_lemma(static_cast<char16*>(word_str), spl_ids, spl_id_fr, 1);
|
||||
}
|
||||
|
||||
void MatrixSearch::debug_print_dmi(PoolPosType dmi_pos, uint16 nest_level) {
|
||||
if (dmi_pos >= dmi_pool_used_) return;
|
||||
|
||||
DictMatchInfo *dmi = dmi_pool_ + dmi_pos;
|
||||
DictMatchInfo* dmi = dmi_pool_ + dmi_pos;
|
||||
|
||||
if (1 == nest_level) {
|
||||
printf("-----------------%d\'th DMI node begin----------->\n", dmi_pos);
|
||||
@@ -809,13 +809,13 @@ size_t MatrixSearch::cancel_last_choice() {
|
||||
size_t step_start = 0;
|
||||
if (fixed_hzs_ > 0) {
|
||||
size_t step_end = spl_start_[fixed_hzs_];
|
||||
MatrixNode *end_node = matrix_[step_end].mtrx_nd_fixed;
|
||||
MatrixNode* end_node = matrix_[step_end].mtrx_nd_fixed;
|
||||
assert(NULL != end_node);
|
||||
|
||||
step_start = end_node->from->step;
|
||||
|
||||
if (step_start > 0) {
|
||||
DictMatchInfo *dmi = dmi_pool_ + end_node->dmi_fr;
|
||||
DictMatchInfo* dmi = dmi_pool_ + end_node->dmi_fr;
|
||||
fixed_hzs_ -= dmi->dict_level;
|
||||
} else {
|
||||
fixed_hzs_ = 0;
|
||||
@@ -847,7 +847,7 @@ bool MatrixSearch::prepare_add_char(char ch) {
|
||||
pys_[pys_decoded_len_] = ch;
|
||||
pys_decoded_len_++;
|
||||
|
||||
MatrixRow *mtrx_this_row = matrix_ + pys_decoded_len_;
|
||||
MatrixRow* mtrx_this_row = matrix_ + pys_decoded_len_;
|
||||
mtrx_this_row->mtrx_nd_pos = mtrx_nd_pool_used_;
|
||||
mtrx_this_row->mtrx_nd_num = 0;
|
||||
mtrx_this_row->dmi_pos = dmi_pool_used_;
|
||||
@@ -859,7 +859,7 @@ bool MatrixSearch::prepare_add_char(char ch) {
|
||||
|
||||
bool MatrixSearch::is_split_at(uint16 pos) { return !spl_parser_->is_valid_to_parse(pys_[pos - 1]); }
|
||||
|
||||
void MatrixSearch::fill_dmi(DictMatchInfo *dmi, MileStoneHandle *handles, PoolPosType dmi_fr, uint16 spl_id, uint16 node_num, unsigned char dict_level, bool splid_end_split, unsigned char splstr_len, unsigned char all_full_id) {
|
||||
void MatrixSearch::fill_dmi(DictMatchInfo* dmi, MileStoneHandle* handles, PoolPosType dmi_fr, uint16 spl_id, uint16 node_num, unsigned char dict_level, bool splid_end_split, unsigned char splstr_len, unsigned char all_full_id) {
|
||||
dmi->dict_handles[0] = handles[0];
|
||||
dmi->dict_handles[1] = handles[1];
|
||||
dmi->dmi_fr = dmi_fr;
|
||||
@@ -920,7 +920,7 @@ bool MatrixSearch::add_char_qwerty() {
|
||||
// 3. Extend the DMI nodes of that old row
|
||||
// + 1 is to extend an extra node from the root
|
||||
for (PoolPosType dmi_pos = matrix_[oldrow].dmi_pos; dmi_pos < matrix_[oldrow].dmi_pos + matrix_[oldrow].dmi_num + 1; dmi_pos++) {
|
||||
DictMatchInfo *dmi = dmi_pool_ + dmi_pos;
|
||||
DictMatchInfo* dmi = dmi_pool_ + dmi_pos;
|
||||
if (dmi_pos == matrix_[oldrow].dmi_pos + matrix_[oldrow].dmi_num) {
|
||||
dmi = NULL; // The last one, NULL means extending from the root.
|
||||
} else {
|
||||
@@ -956,7 +956,7 @@ bool MatrixSearch::add_char_qwerty() {
|
||||
continue;
|
||||
}
|
||||
|
||||
DictMatchInfo *d = dmi;
|
||||
DictMatchInfo* d = dmi;
|
||||
while (d) {
|
||||
dep_->splids[--prev_ids_num] = d->spl_id;
|
||||
if ((PoolPosType)-1 == d->dmi_fr) break;
|
||||
@@ -1001,7 +1001,7 @@ bool MatrixSearch::add_char_qwerty() {
|
||||
fr_row = oldrow - dmi->splstr_len;
|
||||
}
|
||||
for (PoolPosType mtrx_nd_pos = matrix_[fr_row].mtrx_nd_pos; mtrx_nd_pos < matrix_[fr_row].mtrx_nd_pos + matrix_[fr_row].mtrx_nd_num; mtrx_nd_pos++) {
|
||||
MatrixNode *mtrx_nd = mtrx_nd_pool_ + mtrx_nd_pos;
|
||||
MatrixNode* mtrx_nd = mtrx_nd_pool_ + mtrx_nd_pos;
|
||||
|
||||
extend_mtrx_nd(mtrx_nd, lpi_items_, lpi_total_, dmi_pool_used_ - new_dmi_num, pys_decoded_len_);
|
||||
if (longest_ext == 0) longest_ext = ext_len;
|
||||
@@ -1026,7 +1026,7 @@ void MatrixSearch::prepare_candidates() {
|
||||
// If the full sentense candidate's unfixed part may be the same with a normal
|
||||
// lemma. Remove the lemma candidate in this case.
|
||||
char16 fullsent[kMaxLemmaSize + 1];
|
||||
char16 *pfullsent = NULL;
|
||||
char16* pfullsent = NULL;
|
||||
uint16 sent_len;
|
||||
pfullsent = get_candidate0(fullsent, kMaxLemmaSize + 1, &sent_len, true);
|
||||
|
||||
@@ -1070,7 +1070,7 @@ void MatrixSearch::prepare_candidates() {
|
||||
}
|
||||
}
|
||||
|
||||
const char *MatrixSearch::get_pystr(size_t *decoded_len) {
|
||||
const char* MatrixSearch::get_pystr(size_t* decoded_len) {
|
||||
if (!inited_ || NULL == decoded_len) return NULL;
|
||||
|
||||
*decoded_len = pys_decoded_len_;
|
||||
@@ -1116,7 +1116,7 @@ void MatrixSearch::merge_fixed_lmas(size_t del_spl_pos) {
|
||||
if (pos == fixed_lmas_) break;
|
||||
|
||||
uint16 lma_len;
|
||||
char16 *lma_str = c_phrase_.chn_str + c_phrase_.sublma_start[sub_num] + phrase_len;
|
||||
char16* lma_str = c_phrase_.chn_str + c_phrase_.sublma_start[sub_num] + phrase_len;
|
||||
|
||||
lma_len = get_lemma_str(lma_id_[pos], lma_str, kMaxRowNum - phrase_len);
|
||||
assert(lma_len == lma_start_[pos + 1] - lma_start_[pos]);
|
||||
@@ -1144,7 +1144,7 @@ void MatrixSearch::merge_fixed_lmas(size_t del_spl_pos) {
|
||||
// Delete the Chinese character in the merged phrase.
|
||||
// The corresponding elements in spl_ids and spl_start of the
|
||||
// phrase have been deleted.
|
||||
char16 *chn_str = c_phrase_.chn_str + del_spl_pos;
|
||||
char16* chn_str = c_phrase_.chn_str + del_spl_pos;
|
||||
for (uint16 pos = 0; pos < c_phrase_.sublma_start[c_phrase_.sublma_num] - del_spl_pos; pos++) {
|
||||
chn_str[pos] = chn_str[pos + 1];
|
||||
}
|
||||
@@ -1181,7 +1181,7 @@ void MatrixSearch::get_spl_start_id() {
|
||||
lma_id_num_ = fixed_lmas_;
|
||||
spl_id_num_ = fixed_hzs_;
|
||||
|
||||
MatrixNode *mtrx_nd = mtrx_nd_pool_ + matrix_[pys_decoded_len_].mtrx_nd_pos;
|
||||
MatrixNode* mtrx_nd = mtrx_nd_pool_ + matrix_[pys_decoded_len_].mtrx_nd_pos;
|
||||
while (mtrx_nd != mtrx_nd_pool_) {
|
||||
if (fixed_hzs_ > 0) {
|
||||
if (mtrx_nd->step <= spl_start_[fixed_hzs_]) break;
|
||||
@@ -1254,18 +1254,18 @@ void MatrixSearch::get_spl_start_id() {
|
||||
return;
|
||||
}
|
||||
|
||||
size_t MatrixSearch::get_spl_start(const uint16 *&spl_start) {
|
||||
size_t MatrixSearch::get_spl_start(const uint16*& spl_start) {
|
||||
get_spl_start_id();
|
||||
spl_start = spl_start_;
|
||||
return spl_id_num_;
|
||||
}
|
||||
|
||||
size_t MatrixSearch::extend_dmi(DictExtPara *dep, DictMatchInfo *dmi_s) {
|
||||
size_t MatrixSearch::extend_dmi(DictExtPara* dep, DictMatchInfo* dmi_s) {
|
||||
if (dmi_pool_used_ >= kDmiPoolSize) return 0;
|
||||
|
||||
if (dmi_c_phrase_) return extend_dmi_c(dep, dmi_s);
|
||||
|
||||
LpiCache &lpi_cache = LpiCache::get_instance();
|
||||
LpiCache& lpi_cache = LpiCache::get_instance();
|
||||
uint16 splid = dep->splids[dep->splids_extended];
|
||||
|
||||
bool cached = false;
|
||||
@@ -1317,7 +1317,7 @@ size_t MatrixSearch::extend_dmi(DictExtPara *dep, DictMatchInfo *dmi_s) {
|
||||
if (0 != handles[0] || 0 != handles[1]) {
|
||||
if (dmi_pool_used_ >= kDmiPoolSize) return 0;
|
||||
|
||||
DictMatchInfo *dmi_add = dmi_pool_ + dmi_pool_used_;
|
||||
DictMatchInfo* dmi_add = dmi_pool_ + dmi_pool_used_;
|
||||
if (NULL == dmi_s) {
|
||||
fill_dmi(dmi_add, handles, (PoolPosType)-1, splid, 1, 1, dep->splid_end_split, dep->ext_len, spl_trie_->is_half_id(splid) ? 0 : 1);
|
||||
} else {
|
||||
@@ -1344,7 +1344,7 @@ size_t MatrixSearch::extend_dmi(DictExtPara *dep, DictMatchInfo *dmi_s) {
|
||||
return ret_val;
|
||||
}
|
||||
|
||||
size_t MatrixSearch::extend_dmi_c(DictExtPara *dep, DictMatchInfo *dmi_s) {
|
||||
size_t MatrixSearch::extend_dmi_c(DictExtPara* dep, DictMatchInfo* dmi_s) {
|
||||
lpi_total_ = 0;
|
||||
|
||||
uint16 pos = dep->splids_extended;
|
||||
@@ -1353,7 +1353,7 @@ size_t MatrixSearch::extend_dmi_c(DictExtPara *dep, DictMatchInfo *dmi_s) {
|
||||
|
||||
uint16 splid = dep->splids[pos];
|
||||
if (splid == c_phrase_.spl_ids[pos]) {
|
||||
DictMatchInfo *dmi_add = dmi_pool_ + dmi_pool_used_;
|
||||
DictMatchInfo* dmi_add = dmi_pool_ + dmi_pool_used_;
|
||||
MileStoneHandle handles[2]; // Actually never used.
|
||||
if (NULL == dmi_s)
|
||||
fill_dmi(dmi_add, handles, (PoolPosType)-1, splid, 1, 1, dep->splid_end_split, dep->ext_len, spl_trie_->is_half_id(splid) ? 0 : 1);
|
||||
@@ -1370,7 +1370,7 @@ size_t MatrixSearch::extend_dmi_c(DictExtPara *dep, DictMatchInfo *dmi_s) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
size_t MatrixSearch::extend_mtrx_nd(MatrixNode *mtrx_nd, LmaPsbItem lpi_items[], size_t lpi_num, PoolPosType dmi_fr, size_t res_row) {
|
||||
size_t MatrixSearch::extend_mtrx_nd(MatrixNode* mtrx_nd, LmaPsbItem lpi_items[], size_t lpi_num, PoolPosType dmi_fr, size_t res_row) {
|
||||
assert(NULL != mtrx_nd);
|
||||
matrix_[res_row].mtrx_nd_fixed = NULL;
|
||||
|
||||
@@ -1382,14 +1382,14 @@ size_t MatrixSearch::extend_mtrx_nd(MatrixNode *mtrx_nd, LmaPsbItem lpi_items[],
|
||||
if (lpi_num > kMaxNodeARow) lpi_num = kMaxNodeARow;
|
||||
}
|
||||
|
||||
MatrixNode *mtrx_nd_res_min = mtrx_nd_pool_ + matrix_[res_row].mtrx_nd_pos;
|
||||
MatrixNode* mtrx_nd_res_min = mtrx_nd_pool_ + matrix_[res_row].mtrx_nd_pos;
|
||||
for (size_t pos = 0; pos < lpi_num; pos++) {
|
||||
float score = mtrx_nd->score + lpi_items[pos].psb;
|
||||
if (pos > 0 && score - PRUMING_SCORE > mtrx_nd_res_min->score) break;
|
||||
|
||||
// Try to add a new node
|
||||
size_t mtrx_nd_num = matrix_[res_row].mtrx_nd_num;
|
||||
MatrixNode *mtrx_nd_res = mtrx_nd_res_min + mtrx_nd_num;
|
||||
MatrixNode* mtrx_nd_res = mtrx_nd_res_min + mtrx_nd_num;
|
||||
bool replace = false;
|
||||
// Find its position
|
||||
while (mtrx_nd_res > mtrx_nd_res_min && score < (mtrx_nd_res - 1)->score) {
|
||||
@@ -1415,7 +1415,7 @@ PoolPosType MatrixSearch::match_dmi(size_t step_to, uint16 spl_ids[], uint16 spl
|
||||
}
|
||||
|
||||
for (PoolPosType dmi_pos = 0; dmi_pos < matrix_[step_to].dmi_num; dmi_pos++) {
|
||||
DictMatchInfo *dmi = dmi_pool_ + matrix_[step_to].dmi_pos + dmi_pos;
|
||||
DictMatchInfo* dmi = dmi_pool_ + matrix_[step_to].dmi_pos + dmi_pos;
|
||||
|
||||
if (dmi->dict_level != spl_id_num) continue;
|
||||
|
||||
@@ -1436,13 +1436,13 @@ PoolPosType MatrixSearch::match_dmi(size_t step_to, uint16 spl_ids[], uint16 spl
|
||||
return static_cast<PoolPosType>(-1);
|
||||
}
|
||||
|
||||
char16 *MatrixSearch::get_candidate0(char16 *cand_str, size_t max_len, uint16 *retstr_len, bool only_unfixed) {
|
||||
char16* MatrixSearch::get_candidate0(char16* cand_str, size_t max_len, uint16* retstr_len, bool only_unfixed) {
|
||||
if (pys_decoded_len_ == 0 || matrix_[pys_decoded_len_].mtrx_nd_num == 0) return NULL;
|
||||
|
||||
LemmaIdType idxs[kMaxRowNum];
|
||||
size_t id_num = 0;
|
||||
|
||||
MatrixNode *mtrx_nd = mtrx_nd_pool_ + matrix_[pys_decoded_len_].mtrx_nd_pos;
|
||||
MatrixNode* mtrx_nd = mtrx_nd_pool_ + matrix_[pys_decoded_len_].mtrx_nd_pos;
|
||||
|
||||
if (kPrintDebug0) {
|
||||
printf("--- sentence score: %f\n", mtrx_nd->score);
|
||||
@@ -1497,7 +1497,7 @@ char16 *MatrixSearch::get_candidate0(char16 *cand_str, size_t max_len, uint16 *r
|
||||
return cand_str;
|
||||
}
|
||||
|
||||
size_t MatrixSearch::get_lpis(const uint16 *splid_str, size_t splid_str_len, LmaPsbItem *lma_buf, size_t max_lma_buf, const char16 *pfullsent, bool sort_by_psb) {
|
||||
size_t MatrixSearch::get_lpis(const uint16* splid_str, size_t splid_str_len, LmaPsbItem* lma_buf, size_t max_lma_buf, const char16* pfullsent, bool sort_by_psb) {
|
||||
if (splid_str_len > kMaxLemmaSize) return 0;
|
||||
|
||||
size_t num1 = dict_trie_->get_lpis(splid_str, splid_str_len, lma_buf, max_lma_buf);
|
||||
@@ -1512,7 +1512,7 @@ size_t MatrixSearch::get_lpis(const uint16 *splid_str, size_t splid_str_len, Lma
|
||||
|
||||
// Remove repeated items.
|
||||
if (splid_str_len > 1) {
|
||||
LmaPsbStrItem *lpsis = reinterpret_cast<LmaPsbStrItem *>(lma_buf + num);
|
||||
LmaPsbStrItem* lpsis = reinterpret_cast<LmaPsbStrItem*>(lma_buf + num);
|
||||
size_t lpsi_num = (max_lma_buf - num) * sizeof(LmaPsbItem) / sizeof(LmaPsbStrItem);
|
||||
assert(lpsi_num > num);
|
||||
if (num > lpsi_num) num = lpsi_num;
|
||||
@@ -1582,7 +1582,7 @@ size_t MatrixSearch::get_lpis(const uint16 *splid_str, size_t splid_str_len, Lma
|
||||
return num;
|
||||
}
|
||||
|
||||
uint16 MatrixSearch::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max) {
|
||||
uint16 MatrixSearch::get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max) {
|
||||
uint16 str_len = 0;
|
||||
|
||||
if (is_system_lemma(id_lemma)) {
|
||||
@@ -1606,7 +1606,7 @@ uint16 MatrixSearch::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16
|
||||
return str_len;
|
||||
}
|
||||
|
||||
uint16 MatrixSearch::get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid) {
|
||||
uint16 MatrixSearch::get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid) {
|
||||
uint16 splid_num = 0;
|
||||
|
||||
if (arg_valid) {
|
||||
@@ -1638,7 +1638,7 @@ uint16 MatrixSearch::get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint
|
||||
return splid_num;
|
||||
}
|
||||
|
||||
size_t MatrixSearch::inner_predict(const char16 *fixed_buf, uint16 fixed_len, char16 predict_buf[][kMaxPredictSize + 1], size_t buf_len) {
|
||||
size_t MatrixSearch::inner_predict(const char16* fixed_buf, uint16 fixed_len, char16 predict_buf[][kMaxPredictSize + 1], size_t buf_len) {
|
||||
size_t res_total = 0;
|
||||
memset(npre_items_, 0, sizeof(NPredictItem) * npre_items_len_);
|
||||
// In order to shorten the comments, j-character candidates predicted by
|
||||
|
||||
@@ -21,7 +21,7 @@ namespace ime_pinyin {
|
||||
// For debug purpose. You can add a fixed version of qsort and bsearch functions
|
||||
// here so that the output will be totally the same under different platforms.
|
||||
|
||||
void myqsort(void *p, size_t n, size_t es, int (*cmp)(const void *, const void *)) { qsort(p, n, es, cmp); }
|
||||
void myqsort(void* p, size_t n, size_t es, int (*cmp)(const void*, const void*)) { qsort(p, n, es, cmp); }
|
||||
|
||||
void *mybsearch(const void *k, const void *b, size_t n, size_t es, int (*cmp)(const void *, const void *)) { return bsearch(k, b, n, es, cmp); }
|
||||
void* mybsearch(const void* k, const void* b, size_t n, size_t es, int (*cmp)(const void*, const void*)) { return bsearch(k, b, n, es, cmp); }
|
||||
} // namespace ime_pinyin
|
||||
|
||||
+16
-16
@@ -26,9 +26,9 @@ namespace ime_pinyin {
|
||||
|
||||
#define ADD_COUNT 0.3
|
||||
|
||||
int comp_double(const void *p1, const void *p2) {
|
||||
if (*static_cast<const double *>(p1) < *static_cast<const double *>(p2)) return -1;
|
||||
if (*static_cast<const double *>(p1) > *static_cast<const double *>(p2)) return 1;
|
||||
int comp_double(const void* p1, const void* p2) {
|
||||
if (*static_cast<const double*>(p1) < *static_cast<const double*>(p2)) return -1;
|
||||
if (*static_cast<const double*>(p1) > *static_cast<const double*>(p2)) return 1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -54,7 +54,7 @@ int qsearch_nearest(double code_book[], double freq, int start, int end) {
|
||||
return qsearch_nearest(code_book, freq, mid, end);
|
||||
}
|
||||
|
||||
size_t update_code_idx(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE *code_idx) {
|
||||
size_t update_code_idx(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE* code_idx) {
|
||||
size_t changed = 0;
|
||||
for (size_t pos = 0; pos < num; pos++) {
|
||||
CODEBOOK_TYPE idx;
|
||||
@@ -65,14 +65,14 @@ size_t update_code_idx(double freqs[], size_t num, double code_book[], CODEBOOK_
|
||||
return changed;
|
||||
}
|
||||
|
||||
double recalculate_kernel(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE *code_idx) {
|
||||
double recalculate_kernel(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE* code_idx) {
|
||||
double ret = 0;
|
||||
|
||||
size_t *item_num = new size_t[kCodeBookSize];
|
||||
size_t* item_num = new size_t[kCodeBookSize];
|
||||
assert(item_num);
|
||||
memset(item_num, 0, sizeof(size_t) * kCodeBookSize);
|
||||
|
||||
double *cb_new = new double[kCodeBookSize];
|
||||
double* cb_new = new double[kCodeBookSize];
|
||||
assert(cb_new);
|
||||
memset(cb_new, 0, sizeof(double) * kCodeBookSize);
|
||||
|
||||
@@ -94,7 +94,7 @@ double recalculate_kernel(double freqs[], size_t num, double code_book[], CODEBO
|
||||
return ret;
|
||||
}
|
||||
|
||||
void iterate_codes(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE *code_idx) {
|
||||
void iterate_codes(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE* code_idx) {
|
||||
size_t iter_num = 0;
|
||||
double delta_last = 0;
|
||||
do {
|
||||
@@ -112,7 +112,7 @@ void iterate_codes(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE
|
||||
} while (true);
|
||||
}
|
||||
|
||||
NGram *NGram::instance_ = NULL;
|
||||
NGram* NGram::instance_ = NULL;
|
||||
|
||||
NGram::NGram() {
|
||||
initialized_ = false;
|
||||
@@ -136,12 +136,12 @@ NGram::~NGram() {
|
||||
if (NULL != freq_codes_) free(freq_codes_);
|
||||
}
|
||||
|
||||
NGram &NGram::get_instance() {
|
||||
NGram& NGram::get_instance() {
|
||||
if (NULL == instance_) instance_ = new NGram();
|
||||
return *instance_;
|
||||
}
|
||||
|
||||
bool NGram::save_ngram(FILE *fp) {
|
||||
bool NGram::save_ngram(FILE* fp) {
|
||||
if (!initialized_ || NULL == fp) return false;
|
||||
|
||||
if (0 == idx_num_ || NULL == freq_codes_ || NULL == lma_freq_idx_) return false;
|
||||
@@ -155,7 +155,7 @@ bool NGram::save_ngram(FILE *fp) {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool NGram::load_ngram(FILE *fp) {
|
||||
bool NGram::load_ngram(FILE* fp) {
|
||||
if (NULL == fp) return false;
|
||||
|
||||
initialized_ = false;
|
||||
@@ -166,8 +166,8 @@ bool NGram::load_ngram(FILE *fp) {
|
||||
|
||||
if (NULL != freq_codes_) free(freq_codes_);
|
||||
|
||||
lma_freq_idx_ = static_cast<CODEBOOK_TYPE *>(malloc(idx_num_ * sizeof(CODEBOOK_TYPE)));
|
||||
freq_codes_ = static_cast<LmaScoreType *>(malloc(kCodeBookSize * sizeof(LmaScoreType)));
|
||||
lma_freq_idx_ = static_cast<CODEBOOK_TYPE*>(malloc(idx_num_ * sizeof(CODEBOOK_TYPE)));
|
||||
freq_codes_ = static_cast<LmaScoreType*>(malloc(kCodeBookSize * sizeof(LmaScoreType)));
|
||||
|
||||
if (NULL == lma_freq_idx_ || NULL == freq_codes_) return false;
|
||||
|
||||
@@ -203,11 +203,11 @@ float NGram::convert_psb_to_score(double psb) {
|
||||
}
|
||||
|
||||
#ifdef ___BUILD_MODEL___
|
||||
bool NGram::build_unigram(LemmaEntry *lemma_arr, size_t lemma_num, LemmaIdType next_idx_unused) {
|
||||
bool NGram::build_unigram(LemmaEntry* lemma_arr, size_t lemma_num, LemmaIdType next_idx_unused) {
|
||||
if (NULL == lemma_arr || 0 == lemma_num || next_idx_unused <= 1) return false;
|
||||
|
||||
double total_freq = 0;
|
||||
double *freqs = new double[next_idx_unused];
|
||||
double* freqs = new double[next_idx_unused];
|
||||
if (NULL == freqs) return false;
|
||||
|
||||
freqs[0] = ADD_COUNT;
|
||||
|
||||
+11
-11
@@ -29,11 +29,11 @@ using namespace ime_pinyin;
|
||||
static const size_t kMaxPredictNum = 500;
|
||||
|
||||
// Used to search Pinyin string and give the best candidate.
|
||||
MatrixSearch *matrix_search = NULL;
|
||||
MatrixSearch* matrix_search = NULL;
|
||||
|
||||
char16 predict_buf[kMaxPredictNum][kMaxPredictSize + 1];
|
||||
|
||||
bool im_open_decoder(const char *fn_sys_dict, const char *fn_usr_dict) {
|
||||
bool im_open_decoder(const char* fn_sys_dict, const char* fn_usr_dict) {
|
||||
if (NULL != matrix_search) delete matrix_search;
|
||||
|
||||
matrix_search = new MatrixSearch();
|
||||
@@ -44,7 +44,7 @@ bool im_open_decoder(const char *fn_sys_dict, const char *fn_usr_dict) {
|
||||
return matrix_search->init(fn_sys_dict, fn_usr_dict);
|
||||
}
|
||||
|
||||
bool im_open_decoder_fd(int sys_fd, long start_offset, long length, const char *fn_usr_dict) {
|
||||
bool im_open_decoder_fd(int sys_fd, long start_offset, long length, const char* fn_usr_dict) {
|
||||
if (NULL != matrix_search) delete matrix_search;
|
||||
|
||||
matrix_search = new MatrixSearch();
|
||||
@@ -72,7 +72,7 @@ void im_flush_cache() {
|
||||
}
|
||||
|
||||
// To be updated.
|
||||
size_t im_search(const char *pybuf, size_t pylen) {
|
||||
size_t im_search(const char* pybuf, size_t pylen) {
|
||||
if (NULL == matrix_search) return 0;
|
||||
|
||||
matrix_search->search(pybuf, pylen);
|
||||
@@ -94,19 +94,19 @@ void im_reset_search() {
|
||||
// To be removed
|
||||
size_t im_add_letter(char ch) { return 0; }
|
||||
|
||||
const char *im_get_sps_str(size_t *decoded_len) {
|
||||
const char* im_get_sps_str(size_t* decoded_len) {
|
||||
if (NULL == matrix_search) return NULL;
|
||||
|
||||
return matrix_search->get_pystr(decoded_len);
|
||||
}
|
||||
|
||||
char16 *im_get_candidate(size_t cand_id, char16 *cand_str, size_t max_len) {
|
||||
char16* im_get_candidate(size_t cand_id, char16* cand_str, size_t max_len) {
|
||||
if (NULL == matrix_search) return NULL;
|
||||
|
||||
return matrix_search->get_candidate(cand_id, cand_str, max_len);
|
||||
}
|
||||
|
||||
size_t im_get_spl_start_pos(const uint16 *&spl_start) {
|
||||
size_t im_get_spl_start_pos(const uint16*& spl_start) {
|
||||
if (NULL == matrix_search) return 0;
|
||||
|
||||
return matrix_search->get_spl_start(spl_start);
|
||||
@@ -133,11 +133,11 @@ size_t im_get_fixed_len() {
|
||||
// To be removed
|
||||
bool im_cancel_input() { return true; }
|
||||
|
||||
size_t im_get_predicts(const char16 *his_buf, char16 (*&pre_buf)[kMaxPredictSize + 1]) {
|
||||
size_t im_get_predicts(const char16* his_buf, char16 (*&pre_buf)[kMaxPredictSize + 1]) {
|
||||
if (NULL == his_buf) return 0;
|
||||
|
||||
size_t fixed_len = utf16_strlen(his_buf);
|
||||
const char16 *fixed_ptr = his_buf;
|
||||
const char16* fixed_ptr = his_buf;
|
||||
if (fixed_len > kMaxPredictSize) {
|
||||
fixed_ptr += fixed_len - kMaxPredictSize;
|
||||
fixed_len = kMaxPredictSize;
|
||||
@@ -148,12 +148,12 @@ size_t im_get_predicts(const char16 *his_buf, char16 (*&pre_buf)[kMaxPredictSize
|
||||
}
|
||||
|
||||
void im_enable_shm_as_szm(bool enable) {
|
||||
SpellingTrie &spl_trie = SpellingTrie::get_instance();
|
||||
SpellingTrie& spl_trie = SpellingTrie::get_instance();
|
||||
spl_trie.szm_enable_shm(enable);
|
||||
}
|
||||
|
||||
void im_enable_ym_as_szm(bool enable) {
|
||||
SpellingTrie &spl_trie = SpellingTrie::get_instance();
|
||||
SpellingTrie& spl_trie = SpellingTrie::get_instance();
|
||||
spl_trie.szm_enable_ym(enable);
|
||||
}
|
||||
|
||||
|
||||
+36
-36
@@ -26,15 +26,15 @@ bool is_user_lemma(LemmaIdType lma_id) { return (kUserDictIdStart <= lma_id && l
|
||||
|
||||
bool is_composing_lemma(LemmaIdType lma_id) { return (kLemmaIdComposing == lma_id); }
|
||||
|
||||
int cmp_lpi_with_psb(const void *p1, const void *p2) {
|
||||
if ((static_cast<const LmaPsbItem *>(p1))->psb > (static_cast<const LmaPsbItem *>(p2))->psb) return 1;
|
||||
if ((static_cast<const LmaPsbItem *>(p1))->psb < (static_cast<const LmaPsbItem *>(p2))->psb) return -1;
|
||||
int cmp_lpi_with_psb(const void* p1, const void* p2) {
|
||||
if ((static_cast<const LmaPsbItem*>(p1))->psb > (static_cast<const LmaPsbItem*>(p2))->psb) return 1;
|
||||
if ((static_cast<const LmaPsbItem*>(p1))->psb < (static_cast<const LmaPsbItem*>(p2))->psb) return -1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int cmp_lpi_with_unified_psb(const void *p1, const void *p2) {
|
||||
const LmaPsbItem *item1 = static_cast<const LmaPsbItem *>(p1);
|
||||
const LmaPsbItem *item2 = static_cast<const LmaPsbItem *>(p2);
|
||||
int cmp_lpi_with_unified_psb(const void* p1, const void* p2) {
|
||||
const LmaPsbItem* item1 = static_cast<const LmaPsbItem*>(p1);
|
||||
const LmaPsbItem* item2 = static_cast<const LmaPsbItem*>(p2);
|
||||
|
||||
// The real unified psb is psb1 / lma_len1 and psb2 * lma_len2
|
||||
// But we use psb1 * lma_len2 and psb2 * lma_len1 to get better
|
||||
@@ -50,74 +50,74 @@ int cmp_lpi_with_unified_psb(const void *p1, const void *p2) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
int cmp_lpi_with_id(const void *p1, const void *p2) {
|
||||
if ((static_cast<const LmaPsbItem *>(p1))->id < (static_cast<const LmaPsbItem *>(p2))->id) return -1;
|
||||
if ((static_cast<const LmaPsbItem *>(p1))->id > (static_cast<const LmaPsbItem *>(p2))->id) return 1;
|
||||
int cmp_lpi_with_id(const void* p1, const void* p2) {
|
||||
if ((static_cast<const LmaPsbItem*>(p1))->id < (static_cast<const LmaPsbItem*>(p2))->id) return -1;
|
||||
if ((static_cast<const LmaPsbItem*>(p1))->id > (static_cast<const LmaPsbItem*>(p2))->id) return 1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int cmp_lpi_with_hanzi(const void *p1, const void *p2) {
|
||||
if ((static_cast<const LmaPsbItem *>(p1))->hanzi < (static_cast<const LmaPsbItem *>(p2))->hanzi) return -1;
|
||||
if ((static_cast<const LmaPsbItem *>(p1))->hanzi > (static_cast<const LmaPsbItem *>(p2))->hanzi) return 1;
|
||||
int cmp_lpi_with_hanzi(const void* p1, const void* p2) {
|
||||
if ((static_cast<const LmaPsbItem*>(p1))->hanzi < (static_cast<const LmaPsbItem*>(p2))->hanzi) return -1;
|
||||
if ((static_cast<const LmaPsbItem*>(p1))->hanzi > (static_cast<const LmaPsbItem*>(p2))->hanzi) return 1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int cmp_lpsi_with_str(const void *p1, const void *p2) { return utf16_strcmp((static_cast<const LmaPsbStrItem *>(p1))->str, (static_cast<const LmaPsbStrItem *>(p2))->str); }
|
||||
int cmp_lpsi_with_str(const void* p1, const void* p2) { return utf16_strcmp((static_cast<const LmaPsbStrItem*>(p1))->str, (static_cast<const LmaPsbStrItem*>(p2))->str); }
|
||||
|
||||
int cmp_hanzis_1(const void *p1, const void *p2) {
|
||||
if (*static_cast<const char16 *>(p1) < *static_cast<const char16 *>(p2)) return -1;
|
||||
int cmp_hanzis_1(const void* p1, const void* p2) {
|
||||
if (*static_cast<const char16*>(p1) < *static_cast<const char16*>(p2)) return -1;
|
||||
|
||||
if (*static_cast<const char16 *>(p1) > *static_cast<const char16 *>(p2)) return 1;
|
||||
if (*static_cast<const char16*>(p1) > *static_cast<const char16*>(p2)) return 1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int cmp_hanzis_2(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 2); }
|
||||
int cmp_hanzis_2(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 2); }
|
||||
|
||||
int cmp_hanzis_3(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 3); }
|
||||
int cmp_hanzis_3(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 3); }
|
||||
|
||||
int cmp_hanzis_4(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 4); }
|
||||
int cmp_hanzis_4(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 4); }
|
||||
|
||||
int cmp_hanzis_5(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 5); }
|
||||
int cmp_hanzis_5(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 5); }
|
||||
|
||||
int cmp_hanzis_6(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 6); }
|
||||
int cmp_hanzis_6(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 6); }
|
||||
|
||||
int cmp_hanzis_7(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 7); }
|
||||
int cmp_hanzis_7(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 7); }
|
||||
|
||||
int cmp_hanzis_8(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 8); }
|
||||
int cmp_hanzis_8(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 8); }
|
||||
|
||||
int cmp_npre_by_score(const void *p1, const void *p2) {
|
||||
if ((static_cast<const NPredictItem *>(p1))->psb > (static_cast<const NPredictItem *>(p2))->psb) return 1;
|
||||
int cmp_npre_by_score(const void* p1, const void* p2) {
|
||||
if ((static_cast<const NPredictItem*>(p1))->psb > (static_cast<const NPredictItem*>(p2))->psb) return 1;
|
||||
|
||||
if ((static_cast<const NPredictItem *>(p1))->psb < (static_cast<const NPredictItem *>(p2))->psb) return -1;
|
||||
if ((static_cast<const NPredictItem*>(p1))->psb < (static_cast<const NPredictItem*>(p2))->psb) return -1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int cmp_npre_by_hislen_score(const void *p1, const void *p2) {
|
||||
if ((static_cast<const NPredictItem *>(p1))->his_len < (static_cast<const NPredictItem *>(p2))->his_len) return 1;
|
||||
int cmp_npre_by_hislen_score(const void* p1, const void* p2) {
|
||||
if ((static_cast<const NPredictItem*>(p1))->his_len < (static_cast<const NPredictItem*>(p2))->his_len) return 1;
|
||||
|
||||
if ((static_cast<const NPredictItem *>(p1))->his_len > (static_cast<const NPredictItem *>(p2))->his_len) return -1;
|
||||
if ((static_cast<const NPredictItem*>(p1))->his_len > (static_cast<const NPredictItem*>(p2))->his_len) return -1;
|
||||
|
||||
if ((static_cast<const NPredictItem *>(p1))->psb > (static_cast<const NPredictItem *>(p2))->psb) return 1;
|
||||
if ((static_cast<const NPredictItem*>(p1))->psb > (static_cast<const NPredictItem*>(p2))->psb) return 1;
|
||||
|
||||
if ((static_cast<const NPredictItem *>(p1))->psb < (static_cast<const NPredictItem *>(p2))->psb) return -1;
|
||||
if ((static_cast<const NPredictItem*>(p1))->psb < (static_cast<const NPredictItem*>(p2))->psb) return -1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int cmp_npre_by_hanzi_score(const void *p1, const void *p2) {
|
||||
int ret_v = (utf16_strncmp((static_cast<const NPredictItem *>(p1))->pre_hzs, (static_cast<const NPredictItem *>(p2))->pre_hzs, kMaxPredictSize));
|
||||
int cmp_npre_by_hanzi_score(const void* p1, const void* p2) {
|
||||
int ret_v = (utf16_strncmp((static_cast<const NPredictItem*>(p1))->pre_hzs, (static_cast<const NPredictItem*>(p2))->pre_hzs, kMaxPredictSize));
|
||||
if (0 != ret_v) return ret_v;
|
||||
|
||||
if ((static_cast<const NPredictItem *>(p1))->psb > (static_cast<const NPredictItem *>(p2))->psb) return 1;
|
||||
if ((static_cast<const NPredictItem*>(p1))->psb > (static_cast<const NPredictItem*>(p2))->psb) return 1;
|
||||
|
||||
if ((static_cast<const NPredictItem *>(p1))->psb < (static_cast<const NPredictItem *>(p2))->psb) return -1;
|
||||
if ((static_cast<const NPredictItem*>(p1))->psb < (static_cast<const NPredictItem*>(p2))->psb) return -1;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
size_t remove_duplicate_npre(NPredictItem *npre_items, size_t npre_num) {
|
||||
size_t remove_duplicate_npre(NPredictItem* npre_items, size_t npre_num) {
|
||||
if (NULL == npre_items || 0 == npre_num) return 0;
|
||||
|
||||
myqsort(npre_items, npre_num, sizeof(NPredictItem), cmp_npre_by_hanzi_score);
|
||||
|
||||
+27
-27
@@ -27,7 +27,7 @@
|
||||
|
||||
namespace ime_pinyin {
|
||||
|
||||
SpellingTrie *SpellingTrie::instance_ = NULL;
|
||||
SpellingTrie* SpellingTrie::instance_ = NULL;
|
||||
|
||||
// z/c/s is for Zh/Ch/Sh
|
||||
const char SpellingTrie::kHalfId2Sc_[kFullSplIdStart + 1] = "0ABCcDEFGHIJKLMNOPQRSsTUVWXYZz";
|
||||
@@ -45,7 +45,7 @@ unsigned char SpellingTrie::char_flags_[] = {
|
||||
// u v w x y z
|
||||
0x00, 0x00, 0x01, 0x01, 0x01, 0x01};
|
||||
|
||||
int compare_spl(const void *p1, const void *p2) { return strcmp((const char *)(p1), (const char *)(p2)); }
|
||||
int compare_spl(const void* p1, const void* p2) { return strcmp((const char*)(p1), (const char*)(p2)); }
|
||||
|
||||
SpellingTrie::SpellingTrie() {
|
||||
spelling_buf_ = NULL;
|
||||
@@ -101,7 +101,7 @@ SpellingTrie::~SpellingTrie() {
|
||||
if (NULL != f2h_) delete[] f2h_;
|
||||
}
|
||||
|
||||
bool SpellingTrie::if_valid_id_update(uint16 *splid) const {
|
||||
bool SpellingTrie::if_valid_id_update(uint16* splid) const {
|
||||
if (NULL == splid || 0 == *splid) return false;
|
||||
|
||||
if (*splid >= kFullSplIdStart) return true;
|
||||
@@ -193,9 +193,9 @@ void SpellingTrie::szm_enable_ym(bool enable) {
|
||||
|
||||
bool SpellingTrie::is_szm_enabled(char ch) const { return char_flags_[ch - 'A'] & kHalfIdSzmMask; }
|
||||
|
||||
const SpellingTrie *SpellingTrie::get_cpinstance() { return &get_instance(); }
|
||||
const SpellingTrie* SpellingTrie::get_cpinstance() { return &get_instance(); }
|
||||
|
||||
SpellingTrie &SpellingTrie::get_instance() {
|
||||
SpellingTrie& SpellingTrie::get_instance() {
|
||||
if (NULL == instance_) instance_ = new SpellingTrie();
|
||||
|
||||
return *instance_;
|
||||
@@ -206,7 +206,7 @@ uint16 SpellingTrie::half2full_num(uint16 half_id) const {
|
||||
return h2f_num_[half_id];
|
||||
}
|
||||
|
||||
uint16 SpellingTrie::half_to_full(uint16 half_id, uint16 *spl_id_start) const {
|
||||
uint16 SpellingTrie::half_to_full(uint16 half_id, uint16* spl_id_start) const {
|
||||
if (NULL == spl_id_start || NULL == root_ || half_id >= kFullSplIdStart) return 0;
|
||||
|
||||
*spl_id_start = h2f_start_[half_id];
|
||||
@@ -219,7 +219,7 @@ uint16 SpellingTrie::full_to_half(uint16 full_id) const {
|
||||
return f2h_[full_id - kFullSplIdStart];
|
||||
}
|
||||
|
||||
void SpellingTrie::free_son_trie(SpellingNode *node) {
|
||||
void SpellingTrie::free_son_trie(SpellingNode* node) {
|
||||
if (NULL == node) return;
|
||||
|
||||
for (size_t pos = 0; pos < node->num_of_son; pos++) {
|
||||
@@ -229,7 +229,7 @@ void SpellingTrie::free_son_trie(SpellingNode *node) {
|
||||
if (NULL != node->first_son) delete[] node->first_son;
|
||||
}
|
||||
|
||||
bool SpellingTrie::construct(const char *spelling_arr, size_t item_size, size_t item_num, float score_amplifier, unsigned char average_score) {
|
||||
bool SpellingTrie::construct(const char* spelling_arr, size_t item_size, size_t item_num, float score_amplifier, unsigned char average_score) {
|
||||
if (spelling_arr == NULL) return false;
|
||||
|
||||
memset(h2f_start_, 0, sizeof(uint16) * kFullSplIdStart);
|
||||
@@ -277,7 +277,7 @@ bool SpellingTrie::construct(const char *spelling_arr, size_t item_size, size_t
|
||||
memset(splitter_node_, 0, sizeof(SpellingNode));
|
||||
splitter_node_->score = average_score_;
|
||||
|
||||
memset(level1_sons_, 0, sizeof(SpellingNode *) * kValidSplCharNum);
|
||||
memset(level1_sons_, 0, sizeof(SpellingNode*) * kValidSplCharNum);
|
||||
|
||||
root_->first_son = construct_spellings_subset(0, spelling_num_, 0, root_);
|
||||
|
||||
@@ -301,7 +301,7 @@ bool SpellingTrie::construct(const char *spelling_arr, size_t item_size, size_t
|
||||
}
|
||||
|
||||
#ifdef ___BUILD_MODEL___
|
||||
const char *SpellingTrie::get_ym_str(const char *spl_str) {
|
||||
const char* SpellingTrie::get_ym_str(const char* spl_str) {
|
||||
bool start_ZCS = false;
|
||||
if (is_shengmu_char(*spl_str)) {
|
||||
if ('Z' == *spl_str || 'C' == *spl_str || 'S' == *spl_str) start_ZCS = true;
|
||||
@@ -313,13 +313,13 @@ const char *SpellingTrie::get_ym_str(const char *spl_str) {
|
||||
|
||||
bool SpellingTrie::build_ym_info() {
|
||||
bool sucess;
|
||||
SpellingTable *spl_table = new SpellingTable();
|
||||
SpellingTable* spl_table = new SpellingTable();
|
||||
|
||||
sucess = spl_table->init_table(kMaxPinyinSize - 1, 2 * kMaxYmNum, false);
|
||||
assert(sucess);
|
||||
|
||||
for (uint16 pos = 0; pos < spelling_num_; pos++) {
|
||||
const char *spl_str = spelling_buf_ + spelling_size_ * pos;
|
||||
const char* spl_str = spelling_buf_ + spelling_size_ * pos;
|
||||
spl_str = get_ym_str(spl_str);
|
||||
if ('\0' != spl_str[0]) {
|
||||
sucess = spl_table->put_spelling(spl_str, 0);
|
||||
@@ -329,7 +329,7 @@ bool SpellingTrie::build_ym_info() {
|
||||
|
||||
size_t ym_item_size; // '\0' is included
|
||||
size_t ym_num;
|
||||
const char *ym_buf;
|
||||
const char* ym_buf;
|
||||
ym_buf = spl_table->arrange(&ym_item_size, &ym_num);
|
||||
|
||||
if (NULL != ym_buf_) delete[] ym_buf_;
|
||||
@@ -353,7 +353,7 @@ bool SpellingTrie::build_ym_info() {
|
||||
memset(spl_ym_ids_, 0, sizeof(uint8) * (spelling_num_ + kFullSplIdStart));
|
||||
|
||||
for (uint16 id = 1; id < spelling_num_ + kFullSplIdStart; id++) {
|
||||
const char *str = get_spelling_str(id);
|
||||
const char* str = get_spelling_str(id);
|
||||
|
||||
str = get_ym_str(str);
|
||||
if ('\0' != str[0]) {
|
||||
@@ -368,20 +368,20 @@ bool SpellingTrie::build_ym_info() {
|
||||
}
|
||||
#endif
|
||||
|
||||
SpellingNode *SpellingTrie::construct_spellings_subset(size_t item_start, size_t item_end, size_t level, SpellingNode *parent) {
|
||||
SpellingNode* SpellingTrie::construct_spellings_subset(size_t item_start, size_t item_end, size_t level, SpellingNode* parent) {
|
||||
if (level >= spelling_size_ || item_end <= item_start || NULL == parent) return NULL;
|
||||
|
||||
SpellingNode *first_son = NULL;
|
||||
SpellingNode* first_son = NULL;
|
||||
uint16 num_of_son = 0;
|
||||
unsigned char min_son_score = 255;
|
||||
|
||||
const char *spelling_last_start = spelling_buf_ + spelling_size_ * item_start;
|
||||
const char* spelling_last_start = spelling_buf_ + spelling_size_ * item_start;
|
||||
char char_for_node = spelling_last_start[level];
|
||||
assert(char_for_node >= 'A' && char_for_node <= 'Z' || 'h' == char_for_node);
|
||||
|
||||
// Scan the array to find how many sons
|
||||
for (size_t i = item_start + 1; i < item_end; i++) {
|
||||
const char *spelling_current = spelling_buf_ + spelling_size_ * i;
|
||||
const char* spelling_current = spelling_buf_ + spelling_size_ * i;
|
||||
char char_current = spelling_current[level];
|
||||
if (char_current != char_for_node) {
|
||||
num_of_son++;
|
||||
@@ -409,13 +409,13 @@ SpellingNode *SpellingTrie::construct_spellings_subset(size_t item_start, size_t
|
||||
size_t item_start_next = item_start;
|
||||
|
||||
for (size_t i = item_start + 1; i < item_end; i++) {
|
||||
const char *spelling_current = spelling_buf_ + spelling_size_ * i;
|
||||
const char* spelling_current = spelling_buf_ + spelling_size_ * i;
|
||||
char char_current = spelling_current[level];
|
||||
assert(is_valid_spl_char(char_current));
|
||||
|
||||
if (char_current != char_for_node) {
|
||||
// Construct a node
|
||||
SpellingNode *node_current = first_son + son_pos;
|
||||
SpellingNode* node_current = first_son + son_pos;
|
||||
node_current->char_this_node = char_for_node;
|
||||
|
||||
// For quick search in the first level
|
||||
@@ -486,7 +486,7 @@ SpellingNode *SpellingTrie::construct_spellings_subset(size_t item_start, size_t
|
||||
}
|
||||
|
||||
// the last one
|
||||
SpellingNode *node_current = first_son + son_pos;
|
||||
SpellingNode* node_current = first_son + son_pos;
|
||||
node_current->char_this_node = char_for_node;
|
||||
|
||||
// For quick search in the first level
|
||||
@@ -551,7 +551,7 @@ SpellingNode *SpellingTrie::construct_spellings_subset(size_t item_start, size_t
|
||||
return first_son;
|
||||
}
|
||||
|
||||
bool SpellingTrie::save_spl_trie(FILE *fp) {
|
||||
bool SpellingTrie::save_spl_trie(FILE* fp) {
|
||||
if (NULL == fp || NULL == spelling_buf_) return false;
|
||||
|
||||
if (fwrite(&spelling_size_, sizeof(uint32), 1, fp) != 1) return false;
|
||||
@@ -567,7 +567,7 @@ bool SpellingTrie::save_spl_trie(FILE *fp) {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool SpellingTrie::load_spl_trie(FILE *fp) {
|
||||
bool SpellingTrie::load_spl_trie(FILE* fp) {
|
||||
if (NULL == fp) return false;
|
||||
|
||||
if (fread(&spelling_size_, sizeof(uint32), 1, fp) != 1) return false;
|
||||
@@ -602,7 +602,7 @@ bool SpellingTrie::build_f2h() {
|
||||
|
||||
size_t SpellingTrie::get_spelling_num() { return spelling_num_; }
|
||||
|
||||
uint8 SpellingTrie::get_ym_id(const char *ym_str) {
|
||||
uint8 SpellingTrie::get_ym_id(const char* ym_str) {
|
||||
if (NULL == ym_str || NULL == ym_buf_) return 0;
|
||||
|
||||
for (uint8 pos = 0; pos < ym_num_; pos++)
|
||||
@@ -611,7 +611,7 @@ uint8 SpellingTrie::get_ym_id(const char *ym_str) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
const char *SpellingTrie::get_spelling_str(uint16 splid) {
|
||||
const char* SpellingTrie::get_spelling_str(uint16 splid) {
|
||||
splstr_queried_[0] = '\0';
|
||||
|
||||
if (splid >= kFullSplIdStart) {
|
||||
@@ -634,7 +634,7 @@ const char *SpellingTrie::get_spelling_str(uint16 splid) {
|
||||
return splstr_queried_;
|
||||
}
|
||||
|
||||
const char16 *SpellingTrie::get_spelling_str16(uint16 splid) {
|
||||
const char16* SpellingTrie::get_spelling_str16(uint16 splid) {
|
||||
splstr16_queried_[0] = '\0';
|
||||
|
||||
if (splid >= kFullSplIdStart) {
|
||||
@@ -665,7 +665,7 @@ const char16 *SpellingTrie::get_spelling_str16(uint16 splid) {
|
||||
return splstr16_queried_;
|
||||
}
|
||||
|
||||
size_t SpellingTrie::get_spelling_str16(uint16 splid, char16 *splstr16, size_t splstr16_len) {
|
||||
size_t SpellingTrie::get_spelling_str16(uint16 splid, char16* splstr16, size_t splstr16_len) {
|
||||
if (NULL == splstr16 || splstr16_len < kMaxPinyinSize + 1) return 0;
|
||||
|
||||
if (splid >= kFullSplIdStart) {
|
||||
|
||||
+15
-15
@@ -23,14 +23,14 @@ SpellingParser::SpellingParser() { spl_trie_ = SpellingTrie::get_cpinstance(); }
|
||||
|
||||
bool SpellingParser::is_valid_to_parse(char ch) { return SpellingTrie::is_valid_spl_char(ch); }
|
||||
|
||||
uint16 SpellingParser::splstr_to_idxs(const char *splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre) {
|
||||
uint16 SpellingParser::splstr_to_idxs(const char* splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre) {
|
||||
if (NULL == splstr || 0 == max_size || 0 == str_len) return 0;
|
||||
|
||||
if (!SpellingTrie::is_valid_spl_char(splstr[0])) return 0;
|
||||
|
||||
last_is_pre = false;
|
||||
|
||||
const SpellingNode *node_this = spl_trie_->root_;
|
||||
const SpellingNode* node_this = spl_trie_->root_;
|
||||
|
||||
uint16 str_pos = 0;
|
||||
uint16 idx_num = 0;
|
||||
@@ -67,7 +67,7 @@ uint16 SpellingParser::splstr_to_idxs(const char *splstr, uint16 str_len, uint16
|
||||
|
||||
last_is_splitter = false;
|
||||
|
||||
SpellingNode *found_son = NULL;
|
||||
SpellingNode* found_son = NULL;
|
||||
|
||||
if (0 == str_pos) {
|
||||
if (char_this >= 'a')
|
||||
@@ -75,11 +75,11 @@ uint16 SpellingParser::splstr_to_idxs(const char *splstr, uint16 str_len, uint16
|
||||
else
|
||||
found_son = spl_trie_->level1_sons_[char_this - 'A'];
|
||||
} else {
|
||||
SpellingNode *first_son = node_this->first_son;
|
||||
SpellingNode* first_son = node_this->first_son;
|
||||
// Because for Zh/Ch/Sh nodes, they are the last in the buffer and
|
||||
// frequently used, so we scan from the end.
|
||||
for (int i = 0; i < node_this->num_of_son; i++) {
|
||||
SpellingNode *this_son = first_son + i;
|
||||
SpellingNode* this_son = first_son + i;
|
||||
if (SpellingTrie::is_same_spl_char(this_son->char_this_node, char_this)) {
|
||||
found_son = this_son;
|
||||
break;
|
||||
@@ -124,7 +124,7 @@ uint16 SpellingParser::splstr_to_idxs(const char *splstr, uint16 str_len, uint16
|
||||
return idx_num;
|
||||
}
|
||||
|
||||
uint16 SpellingParser::splstr_to_idxs_f(const char *splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre) {
|
||||
uint16 SpellingParser::splstr_to_idxs_f(const char* splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre) {
|
||||
uint16 idx_num = splstr_to_idxs(splstr, str_len, spl_idx, start_pos, max_size, last_is_pre);
|
||||
for (uint16 pos = 0; pos < idx_num; pos++) {
|
||||
if (spl_trie_->is_half_id_yunmu(spl_idx[pos])) {
|
||||
@@ -137,14 +137,14 @@ uint16 SpellingParser::splstr_to_idxs_f(const char *splstr, uint16 str_len, uint
|
||||
return idx_num;
|
||||
}
|
||||
|
||||
uint16 SpellingParser::splstr16_to_idxs(const char16 *splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre) {
|
||||
uint16 SpellingParser::splstr16_to_idxs(const char16* splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre) {
|
||||
if (NULL == splstr || 0 == max_size || 0 == str_len) return 0;
|
||||
|
||||
if (!SpellingTrie::is_valid_spl_char(splstr[0])) return 0;
|
||||
|
||||
last_is_pre = false;
|
||||
|
||||
const SpellingNode *node_this = spl_trie_->root_;
|
||||
const SpellingNode* node_this = spl_trie_->root_;
|
||||
|
||||
uint16 str_pos = 0;
|
||||
uint16 idx_num = 0;
|
||||
@@ -181,7 +181,7 @@ uint16 SpellingParser::splstr16_to_idxs(const char16 *splstr, uint16 str_len, ui
|
||||
|
||||
last_is_splitter = false;
|
||||
|
||||
SpellingNode *found_son = NULL;
|
||||
SpellingNode* found_son = NULL;
|
||||
|
||||
if (0 == str_pos) {
|
||||
if (char_this >= 'a')
|
||||
@@ -189,11 +189,11 @@ uint16 SpellingParser::splstr16_to_idxs(const char16 *splstr, uint16 str_len, ui
|
||||
else
|
||||
found_son = spl_trie_->level1_sons_[char_this - 'A'];
|
||||
} else {
|
||||
SpellingNode *first_son = node_this->first_son;
|
||||
SpellingNode* first_son = node_this->first_son;
|
||||
// Because for Zh/Ch/Sh nodes, they are the last in the buffer and
|
||||
// frequently used, so we scan from the end.
|
||||
for (int i = 0; i < node_this->num_of_son; i++) {
|
||||
SpellingNode *this_son = first_son + i;
|
||||
SpellingNode* this_son = first_son + i;
|
||||
if (SpellingTrie::is_same_spl_char(this_son->char_this_node, char_this)) {
|
||||
found_son = this_son;
|
||||
break;
|
||||
@@ -238,7 +238,7 @@ uint16 SpellingParser::splstr16_to_idxs(const char16 *splstr, uint16 str_len, ui
|
||||
return idx_num;
|
||||
}
|
||||
|
||||
uint16 SpellingParser::splstr16_to_idxs_f(const char16 *splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre) {
|
||||
uint16 SpellingParser::splstr16_to_idxs_f(const char16* splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre) {
|
||||
uint16 idx_num = splstr16_to_idxs(splstr, str_len, spl_idx, start_pos, max_size, last_is_pre);
|
||||
for (uint16 pos = 0; pos < idx_num; pos++) {
|
||||
if (spl_trie_->is_half_id_yunmu(spl_idx[pos])) {
|
||||
@@ -251,7 +251,7 @@ uint16 SpellingParser::splstr16_to_idxs_f(const char16 *splstr, uint16 str_len,
|
||||
return idx_num;
|
||||
}
|
||||
|
||||
uint16 SpellingParser::get_splid_by_str(const char *splstr, uint16 str_len, bool *is_pre) {
|
||||
uint16 SpellingParser::get_splid_by_str(const char* splstr, uint16 str_len, bool* is_pre) {
|
||||
if (NULL == is_pre) return 0;
|
||||
|
||||
uint16 spl_idx[2];
|
||||
@@ -263,7 +263,7 @@ uint16 SpellingParser::get_splid_by_str(const char *splstr, uint16 str_len, bool
|
||||
return spl_idx[0];
|
||||
}
|
||||
|
||||
uint16 SpellingParser::get_splid_by_str_f(const char *splstr, uint16 str_len, bool *is_pre) {
|
||||
uint16 SpellingParser::get_splid_by_str_f(const char* splstr, uint16 str_len, bool* is_pre) {
|
||||
if (NULL == is_pre) return 0;
|
||||
|
||||
uint16 spl_idx[2];
|
||||
@@ -280,7 +280,7 @@ uint16 SpellingParser::get_splid_by_str_f(const char *splstr, uint16 str_len, bo
|
||||
return spl_idx[0];
|
||||
}
|
||||
|
||||
uint16 SpellingParser::get_splids_parallel(const char *splstr, uint16 str_len, uint16 splidx[], uint16 max_size, uint16 &full_id_num, bool &is_pre) {
|
||||
uint16 SpellingParser::get_splids_parallel(const char* splstr, uint16 str_len, uint16 splidx[], uint16 max_size, uint16& full_id_num, bool& is_pre) {
|
||||
if (max_size <= 0 || !is_valid_to_parse(splstr[0])) return 0;
|
||||
|
||||
splidx[0] = get_splid_by_str(splstr, str_len, &is_pre);
|
||||
|
||||
+104
-104
@@ -42,7 +42,7 @@
|
||||
namespace ime_pinyin {
|
||||
|
||||
#ifdef _WIN32
|
||||
static int gettimeofday(struct timeval *tp, void *) {
|
||||
static int gettimeofday(struct timeval* tp, void*) {
|
||||
if (!tp) {
|
||||
return -1;
|
||||
}
|
||||
@@ -101,7 +101,7 @@ static pthread_mutex_t g_mutex_ = PTHREAD_MUTEX_INITIALIZER;
|
||||
#endif
|
||||
static struct timeval g_last_update_ = {0, 0};
|
||||
|
||||
inline uint32 UserDict::get_dict_file_size(UserDictInfo *info) {
|
||||
inline uint32 UserDict::get_dict_file_size(UserDictInfo* info) {
|
||||
return (4 + info->lemma_size + (info->lemma_count << 3)
|
||||
#ifdef ___PREDICT_ENABLED___
|
||||
+ (info->lemma_count << 2)
|
||||
@@ -163,12 +163,12 @@ inline int UserDict::build_score(uint64 lmt, int freq) {
|
||||
return s;
|
||||
}
|
||||
|
||||
inline int64 UserDict::utf16le_atoll(uint16 *s, int len) {
|
||||
inline int64 UserDict::utf16le_atoll(uint16* s, int len) {
|
||||
int64 ret = 0;
|
||||
if (len <= 0) return ret;
|
||||
|
||||
int flag = 1;
|
||||
const uint16 *endp = s + len;
|
||||
const uint16* endp = s + len;
|
||||
if (*s == '-') {
|
||||
flag = -1;
|
||||
s++;
|
||||
@@ -183,9 +183,9 @@ inline int64 UserDict::utf16le_atoll(uint16 *s, int len) {
|
||||
return ret * flag;
|
||||
}
|
||||
|
||||
inline int UserDict::utf16le_lltoa(int64 v, uint16 *s, int size) {
|
||||
inline int UserDict::utf16le_lltoa(int64 v, uint16* s, int size) {
|
||||
if (!s || size <= 0) return 0;
|
||||
uint16 *endp = s + size;
|
||||
uint16* endp = s + size;
|
||||
int ret_len = 0;
|
||||
if (v < 0) {
|
||||
*(s++) = '-';
|
||||
@@ -193,7 +193,7 @@ inline int UserDict::utf16le_lltoa(int64 v, uint16 *s, int size) {
|
||||
v *= -1;
|
||||
}
|
||||
|
||||
uint16 *b = s;
|
||||
uint16* b = s;
|
||||
while (s < endp && v != 0) {
|
||||
*(s++) = '0' + (v % 10);
|
||||
v = v / 10;
|
||||
@@ -227,15 +227,15 @@ inline char UserDict::get_lemma_nchar(uint32 offset) {
|
||||
return (char)(lemmas_[offset + 1]);
|
||||
}
|
||||
|
||||
inline uint16 *UserDict::get_lemma_spell_ids(uint32 offset) {
|
||||
inline uint16* UserDict::get_lemma_spell_ids(uint32 offset) {
|
||||
offset &= kUserDictOffsetMask;
|
||||
return (uint16 *)(lemmas_ + offset + 2);
|
||||
return (uint16*)(lemmas_ + offset + 2);
|
||||
}
|
||||
|
||||
inline uint16 *UserDict::get_lemma_word(uint32 offset) {
|
||||
inline uint16* UserDict::get_lemma_word(uint32 offset) {
|
||||
offset &= kUserDictOffsetMask;
|
||||
uint8 nchar = get_lemma_nchar(offset);
|
||||
return (uint16 *)(lemmas_ + offset + 2 + (nchar << 1));
|
||||
return (uint16*)(lemmas_ + offset + 2 + (nchar << 1));
|
||||
}
|
||||
|
||||
inline LemmaIdType UserDict::get_max_lemma_id() {
|
||||
@@ -282,7 +282,7 @@ UserDict::UserDict()
|
||||
|
||||
UserDict::~UserDict() { close_dict(); }
|
||||
|
||||
bool UserDict::load_dict(const char *file_name, LemmaIdType start_id, LemmaIdType end_id) {
|
||||
bool UserDict::load_dict(const char* file_name, LemmaIdType start_id, LemmaIdType end_id) {
|
||||
#ifdef ___DEBUG_PERF___
|
||||
DEBUG_PERF_BEGIN;
|
||||
#endif
|
||||
@@ -308,7 +308,7 @@ bool UserDict::load_dict(const char *file_name, LemmaIdType start_id, LemmaIdTyp
|
||||
#endif
|
||||
return true;
|
||||
error:
|
||||
free((void *)dict_file_);
|
||||
free((void*)dict_file_);
|
||||
dict_file_ = NULL;
|
||||
start_id_ = 0;
|
||||
return false;
|
||||
@@ -330,7 +330,7 @@ bool UserDict::close_dict() {
|
||||
pthread_mutex_unlock(&g_mutex_);
|
||||
|
||||
out:
|
||||
free((void *)dict_file_);
|
||||
free((void*)dict_file_);
|
||||
free(lemmas_);
|
||||
free(offsets_);
|
||||
free(offsets_by_id_);
|
||||
@@ -367,7 +367,7 @@ size_t UserDict::number_of_lemmas() { return dict_info_.lemma_count; }
|
||||
|
||||
void UserDict::reset_milestones(uint16 from_step, MileStoneHandle from_handle) { return; }
|
||||
|
||||
MileStoneHandle UserDict::extend_dict(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num) {
|
||||
MileStoneHandle UserDict::extend_dict(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num) {
|
||||
if (is_valid_state() == false) return 0;
|
||||
|
||||
bool need_extend = false;
|
||||
@@ -383,10 +383,10 @@ MileStoneHandle UserDict::extend_dict(MileStoneHandle from_handle, const DictExt
|
||||
return ((*lpi_num > 0 || need_extend) ? 1 : 0);
|
||||
}
|
||||
|
||||
int UserDict::is_fuzzy_prefix_spell_id(const uint16 *id1, uint16 len1, const UserDictSearchable *searchable) {
|
||||
int UserDict::is_fuzzy_prefix_spell_id(const uint16* id1, uint16 len1, const UserDictSearchable* searchable) {
|
||||
if (len1 < searchable->splids_len) return 0;
|
||||
|
||||
SpellingTrie &spl_trie = SpellingTrie::get_instance();
|
||||
SpellingTrie& spl_trie = SpellingTrie::get_instance();
|
||||
uint32 i = 0;
|
||||
for (i = 0; i < searchable->splids_len; i++) {
|
||||
const char py1 = *spl_trie.get_spelling_str(id1[i]);
|
||||
@@ -398,11 +398,11 @@ int UserDict::is_fuzzy_prefix_spell_id(const uint16 *id1, uint16 len1, const Use
|
||||
return 1;
|
||||
}
|
||||
|
||||
int UserDict::fuzzy_compare_spell_id(const uint16 *id1, uint16 len1, const UserDictSearchable *searchable) {
|
||||
int UserDict::fuzzy_compare_spell_id(const uint16* id1, uint16 len1, const UserDictSearchable* searchable) {
|
||||
if (len1 < searchable->splids_len) return -1;
|
||||
if (len1 > searchable->splids_len) return 1;
|
||||
|
||||
SpellingTrie &spl_trie = SpellingTrie::get_instance();
|
||||
SpellingTrie& spl_trie = SpellingTrie::get_instance();
|
||||
uint32 i = 0;
|
||||
for (i = 0; i < len1; i++) {
|
||||
const char py1 = *spl_trie.get_spelling_str(id1[i]);
|
||||
@@ -415,7 +415,7 @@ int UserDict::fuzzy_compare_spell_id(const uint16 *id1, uint16 len1, const UserD
|
||||
return 0;
|
||||
}
|
||||
|
||||
bool UserDict::is_prefix_spell_id(const uint16 *fullids, uint16 fulllen, const UserDictSearchable *searchable) {
|
||||
bool UserDict::is_prefix_spell_id(const uint16* fullids, uint16 fulllen, const UserDictSearchable* searchable) {
|
||||
if (fulllen < searchable->splids_len) return false;
|
||||
|
||||
uint32 i = 0;
|
||||
@@ -430,7 +430,7 @@ bool UserDict::is_prefix_spell_id(const uint16 *fullids, uint16 fulllen, const U
|
||||
return true;
|
||||
}
|
||||
|
||||
bool UserDict::equal_spell_id(const uint16 *fullids, uint16 fulllen, const UserDictSearchable *searchable) {
|
||||
bool UserDict::equal_spell_id(const uint16* fullids, uint16 fulllen, const UserDictSearchable* searchable) {
|
||||
if (fulllen != searchable->splids_len) return false;
|
||||
|
||||
uint32 i = 0;
|
||||
@@ -445,7 +445,7 @@ bool UserDict::equal_spell_id(const uint16 *fullids, uint16 fulllen, const UserD
|
||||
return true;
|
||||
}
|
||||
|
||||
int32 UserDict::locate_first_in_offsets(const UserDictSearchable *searchable) {
|
||||
int32 UserDict::locate_first_in_offsets(const UserDictSearchable* searchable) {
|
||||
int32 begin = 0;
|
||||
int32 end = dict_info_.lemma_count - 1;
|
||||
int32 middle = -1;
|
||||
@@ -457,7 +457,7 @@ int32 UserDict::locate_first_in_offsets(const UserDictSearchable *searchable) {
|
||||
middle = (begin + end) >> 1;
|
||||
uint32 offset = offsets_[middle];
|
||||
uint8 nchar = get_lemma_nchar(offset);
|
||||
const uint16 *splids = get_lemma_spell_ids(offset);
|
||||
const uint16* splids = get_lemma_spell_ids(offset);
|
||||
int cmp = fuzzy_compare_spell_id(splids, nchar, searchable);
|
||||
int pre = is_fuzzy_prefix_spell_id(splids, nchar, searchable);
|
||||
|
||||
@@ -476,11 +476,11 @@ int32 UserDict::locate_first_in_offsets(const UserDictSearchable *searchable) {
|
||||
return first_prefix;
|
||||
}
|
||||
|
||||
void UserDict::prepare_locate(UserDictSearchable *searchable, const uint16 *splid_str, uint16 splid_str_len) {
|
||||
void UserDict::prepare_locate(UserDictSearchable* searchable, const uint16* splid_str, uint16 splid_str_len) {
|
||||
searchable->splids_len = splid_str_len;
|
||||
memset(searchable->signature, 0, sizeof(searchable->signature));
|
||||
|
||||
SpellingTrie &spl_trie = SpellingTrie::get_instance();
|
||||
SpellingTrie& spl_trie = SpellingTrie::get_instance();
|
||||
uint32 i = 0;
|
||||
for (; i < splid_str_len; i++) {
|
||||
if (spl_trie.is_half_id(splid_str[i])) {
|
||||
@@ -494,9 +494,9 @@ void UserDict::prepare_locate(UserDictSearchable *searchable, const uint16 *spli
|
||||
}
|
||||
}
|
||||
|
||||
size_t UserDict::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lpi_items, size_t lpi_max) { return _get_lpis(splid_str, splid_str_len, lpi_items, lpi_max, NULL); }
|
||||
size_t UserDict::get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lpi_items, size_t lpi_max) { return _get_lpis(splid_str, splid_str_len, lpi_items, lpi_max, NULL); }
|
||||
|
||||
size_t UserDict::_get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lpi_items, size_t lpi_max, bool *need_extend) {
|
||||
size_t UserDict::_get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lpi_items, size_t lpi_max, bool* need_extend) {
|
||||
bool tmp_extend;
|
||||
if (!need_extend) need_extend = &tmp_extend;
|
||||
|
||||
@@ -555,7 +555,7 @@ size_t UserDict::_get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsb
|
||||
continue;
|
||||
}
|
||||
uint8 nchar = get_lemma_nchar(offset);
|
||||
uint16 *splids = get_lemma_spell_ids(offset);
|
||||
uint16* splids = get_lemma_spell_ids(offset);
|
||||
#ifdef ___CACHE_ENABLED___
|
||||
if (!cached && 0 != fuzzy_compare_spell_id(splids, nchar, &searchable)) {
|
||||
#else
|
||||
@@ -593,12 +593,12 @@ size_t UserDict::_get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsb
|
||||
return lpi_current;
|
||||
}
|
||||
|
||||
uint16 UserDict::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max) {
|
||||
uint16 UserDict::get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max) {
|
||||
if (is_valid_state() == false) return 0;
|
||||
if (is_valid_lemma_id(id_lemma) == false) return 0;
|
||||
uint32 offset = offsets_by_id_[id_lemma - start_id_];
|
||||
uint8 nchar = get_lemma_nchar(offset);
|
||||
char16 *str = get_lemma_word(offset);
|
||||
char16* str = get_lemma_word(offset);
|
||||
uint16 m = nchar < str_max - 1 ? nchar : str_max - 1;
|
||||
int i = 0;
|
||||
for (; i < m; i++) {
|
||||
@@ -608,21 +608,21 @@ uint16 UserDict::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str
|
||||
return m;
|
||||
}
|
||||
|
||||
uint16 UserDict::get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid) {
|
||||
uint16 UserDict::get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid) {
|
||||
if (is_valid_lemma_id(id_lemma) == false) return 0;
|
||||
uint32 offset = offsets_by_id_[id_lemma - start_id_];
|
||||
uint8 nchar = get_lemma_nchar(offset);
|
||||
const uint16 *ids = get_lemma_spell_ids(offset);
|
||||
const uint16* ids = get_lemma_spell_ids(offset);
|
||||
int i = 0;
|
||||
for (; i < nchar && i < splids_max; i++) splids[i] = ids[i];
|
||||
return i;
|
||||
}
|
||||
|
||||
size_t UserDict::predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used) {
|
||||
size_t UserDict::predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used) {
|
||||
uint32 new_added = 0;
|
||||
#ifdef ___PREDICT_ENABLED___
|
||||
int32 end = dict_info_.lemma_count - 1;
|
||||
int j = locate_first_in_predicts((const uint16 *)last_hzs, hzs_len);
|
||||
int j = locate_first_in_predicts((const uint16*)last_hzs, hzs_len);
|
||||
if (j == -1) return 0;
|
||||
|
||||
while (j <= end) {
|
||||
@@ -633,8 +633,8 @@ size_t UserDict::predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *
|
||||
continue;
|
||||
}
|
||||
uint32 nchar = get_lemma_nchar(offset);
|
||||
uint16 *words = get_lemma_word(offset);
|
||||
uint16 *splids = get_lemma_spell_ids(offset);
|
||||
uint16* words = get_lemma_word(offset);
|
||||
uint16* splids = get_lemma_spell_ids(offset);
|
||||
|
||||
if (nchar <= hzs_len) {
|
||||
j++;
|
||||
@@ -693,14 +693,14 @@ int32 UserDict::locate_in_offsets(char16 lemma_str[], uint16 splid_str[], uint16
|
||||
off++;
|
||||
continue;
|
||||
}
|
||||
uint16 *splids = get_lemma_spell_ids(offset);
|
||||
uint16* splids = get_lemma_spell_ids(offset);
|
||||
#ifdef ___CACHE_ENABLED___
|
||||
if (!cached && 0 != fuzzy_compare_spell_id(splids, lemma_len, &searchable)) break;
|
||||
#else
|
||||
if (0 != fuzzy_compare_spell_id(splids, lemma_len, &searchable)) break;
|
||||
#endif
|
||||
if (equal_spell_id(splids, lemma_len, &searchable) == true) {
|
||||
uint16 *str = get_lemma_word(offset);
|
||||
uint16* str = get_lemma_word(offset);
|
||||
uint32 i = 0;
|
||||
for (i = 0; i < lemma_len; i++) {
|
||||
if (str[i] == lemma_str[i]) continue;
|
||||
@@ -728,7 +728,7 @@ int32 UserDict::locate_in_offsets(char16 lemma_str[], uint16 splid_str[], uint16
|
||||
}
|
||||
|
||||
#ifdef ___PREDICT_ENABLED___
|
||||
uint32 UserDict::locate_where_to_insert_in_predicts(const uint16 *words, int lemma_len) {
|
||||
uint32 UserDict::locate_where_to_insert_in_predicts(const uint16* words, int lemma_len) {
|
||||
int32 begin = 0;
|
||||
int32 end = dict_info_.lemma_count - 1;
|
||||
int32 middle = end;
|
||||
@@ -739,7 +739,7 @@ uint32 UserDict::locate_where_to_insert_in_predicts(const uint16 *words, int lem
|
||||
middle = (begin + end) >> 1;
|
||||
uint32 offset = offsets_[middle];
|
||||
uint8 nchar = get_lemma_nchar(offset);
|
||||
const uint16 *ws = get_lemma_word(offset);
|
||||
const uint16* ws = get_lemma_word(offset);
|
||||
|
||||
uint32 minl = nchar < lemma_len ? nchar : lemma_len;
|
||||
uint32 k = 0;
|
||||
@@ -775,7 +775,7 @@ uint32 UserDict::locate_where_to_insert_in_predicts(const uint16 *words, int lem
|
||||
return last_matched;
|
||||
}
|
||||
|
||||
int32 UserDict::locate_first_in_predicts(const uint16 *words, int lemma_len) {
|
||||
int32 UserDict::locate_first_in_predicts(const uint16* words, int lemma_len) {
|
||||
int32 begin = 0;
|
||||
int32 end = dict_info_.lemma_count - 1;
|
||||
int32 middle = -1;
|
||||
@@ -786,7 +786,7 @@ int32 UserDict::locate_first_in_predicts(const uint16 *words, int lemma_len) {
|
||||
middle = (begin + end) >> 1;
|
||||
uint32 offset = offsets_[middle];
|
||||
uint8 nchar = get_lemma_nchar(offset);
|
||||
const uint16 *ws = get_lemma_word(offset);
|
||||
const uint16* ws = get_lemma_word(offset);
|
||||
|
||||
uint32 minl = nchar < lemma_len ? nchar : lemma_len;
|
||||
uint32 k = 0;
|
||||
@@ -851,8 +851,8 @@ int UserDict::_get_lemma_score(LemmaIdType lemma_id) {
|
||||
uint32 offset = offsets_by_id_[lemma_id - start_id_];
|
||||
|
||||
uint32 nchar = get_lemma_nchar(offset);
|
||||
uint16 *spl = get_lemma_spell_ids(offset);
|
||||
uint16 *wrd = get_lemma_word(offset);
|
||||
uint16* spl = get_lemma_spell_ids(offset);
|
||||
uint16* wrd = get_lemma_word(offset);
|
||||
|
||||
int32 off = locate_in_offsets(wrd, spl, nchar);
|
||||
if (off == -1) {
|
||||
@@ -936,8 +936,8 @@ bool UserDict::remove_lemma(LemmaIdType lemma_id) {
|
||||
uint32 offset = offsets_by_id_[lemma_id - start_id_];
|
||||
|
||||
uint32 nchar = get_lemma_nchar(offset);
|
||||
uint16 *spl = get_lemma_spell_ids(offset);
|
||||
uint16 *wrd = get_lemma_word(offset);
|
||||
uint16* spl = get_lemma_spell_ids(offset);
|
||||
uint16* wrd = get_lemma_word(offset);
|
||||
|
||||
int32 off = locate_in_offsets(wrd, spl, nchar);
|
||||
|
||||
@@ -947,19 +947,19 @@ bool UserDict::remove_lemma(LemmaIdType lemma_id) {
|
||||
void UserDict::flush_cache() {
|
||||
LemmaIdType start_id = start_id_;
|
||||
if (!dict_file_) return;
|
||||
const char *file = strdup(dict_file_);
|
||||
const char* file = strdup(dict_file_);
|
||||
if (!file) return;
|
||||
close_dict();
|
||||
load_dict(file, start_id, kUserDictIdEnd);
|
||||
free((void *)file);
|
||||
free((void*)file);
|
||||
#ifdef ___CACHE_ENABLED___
|
||||
cache_init();
|
||||
#endif
|
||||
return;
|
||||
}
|
||||
|
||||
bool UserDict::reset(const char *file) {
|
||||
FILE *fp = fopen(file, "w+");
|
||||
bool UserDict::reset(const char* file) {
|
||||
FILE* fp = fopen(file, "w+");
|
||||
if (!fp) {
|
||||
return false;
|
||||
}
|
||||
@@ -979,10 +979,10 @@ bool UserDict::reset(const char *file) {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool UserDict::validate(const char *file) {
|
||||
bool UserDict::validate(const char* file) {
|
||||
// b is ignored in POSIX compatible os including Linux
|
||||
// while b is important flag for Windows to specify binary mode
|
||||
FILE *fp = fopen(file, "rb");
|
||||
FILE* fp = fopen(file, "rb");
|
||||
if (!fp) {
|
||||
return false;
|
||||
}
|
||||
@@ -1038,13 +1038,13 @@ error:
|
||||
return false;
|
||||
}
|
||||
|
||||
bool UserDict::load(const char *file, LemmaIdType start_id) {
|
||||
bool UserDict::load(const char* file, LemmaIdType start_id) {
|
||||
if (0 != pthread_mutex_trylock(&g_mutex_)) {
|
||||
return false;
|
||||
}
|
||||
// b is ignored in POSIX compatible os including Linux
|
||||
// while b is important flag for Windows to specify binary mode
|
||||
FILE *fp = fopen(file, "rb");
|
||||
FILE* fp = fopen(file, "rb");
|
||||
if (!fp) {
|
||||
pthread_mutex_unlock(&g_mutex_);
|
||||
return false;
|
||||
@@ -1052,16 +1052,16 @@ bool UserDict::load(const char *file, LemmaIdType start_id) {
|
||||
|
||||
size_t readed, toread;
|
||||
UserDictInfo dict_info;
|
||||
uint8 *lemmas = NULL;
|
||||
uint32 *offsets = NULL;
|
||||
uint8* lemmas = NULL;
|
||||
uint32* offsets = NULL;
|
||||
#ifdef ___SYNC_ENABLED___
|
||||
uint32 *syncs = NULL;
|
||||
uint32* syncs = NULL;
|
||||
#endif
|
||||
uint32 *scores = NULL;
|
||||
uint32 *ids = NULL;
|
||||
uint32 *offsets_by_id = NULL;
|
||||
uint32* scores = NULL;
|
||||
uint32* ids = NULL;
|
||||
uint32* offsets_by_id = NULL;
|
||||
#ifdef ___PREDICT_ENABLED___
|
||||
uint32 *predicts = NULL;
|
||||
uint32* predicts = NULL;
|
||||
#endif
|
||||
size_t i;
|
||||
int err;
|
||||
@@ -1072,30 +1072,30 @@ bool UserDict::load(const char *file, LemmaIdType start_id) {
|
||||
readed = fread(&dict_info, 1, sizeof(dict_info), fp);
|
||||
if (readed != sizeof(dict_info)) goto error;
|
||||
|
||||
lemmas = (uint8 *)malloc(dict_info.lemma_size + (kUserDictPreAlloc * (2 + (kUserDictAverageNchar << 2))));
|
||||
lemmas = (uint8*)malloc(dict_info.lemma_size + (kUserDictPreAlloc * (2 + (kUserDictAverageNchar << 2))));
|
||||
|
||||
if (!lemmas) goto error;
|
||||
|
||||
offsets = (uint32 *)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
|
||||
offsets = (uint32*)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
|
||||
if (!offsets) goto error;
|
||||
|
||||
#ifdef ___PREDICT_ENABLED___
|
||||
predicts = (uint32 *)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
|
||||
predicts = (uint32*)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
|
||||
if (!predicts) goto error;
|
||||
#endif
|
||||
|
||||
#ifdef ___SYNC_ENABLED___
|
||||
syncs = (uint32 *)malloc((dict_info.sync_count + kUserDictPreAlloc) << 2);
|
||||
syncs = (uint32*)malloc((dict_info.sync_count + kUserDictPreAlloc) << 2);
|
||||
if (!syncs) goto error;
|
||||
#endif
|
||||
|
||||
scores = (uint32 *)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
|
||||
scores = (uint32*)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
|
||||
if (!scores) goto error;
|
||||
|
||||
ids = (uint32 *)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
|
||||
ids = (uint32*)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
|
||||
if (!ids) goto error;
|
||||
|
||||
offsets_by_id = (uint32 *)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
|
||||
offsets_by_id = (uint32*)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
|
||||
if (!offsets_by_id) goto error;
|
||||
|
||||
err = fseek(fp, 4, SEEK_SET);
|
||||
@@ -1110,7 +1110,7 @@ bool UserDict::load(const char *file, LemmaIdType start_id) {
|
||||
toread = (dict_info.lemma_count << 2);
|
||||
readed = 0;
|
||||
while (readed < toread && !ferror(fp) && !feof(fp)) {
|
||||
readed += fread((((uint8 *)offsets) + readed), 1, toread - readed, fp);
|
||||
readed += fread((((uint8*)offsets) + readed), 1, toread - readed, fp);
|
||||
}
|
||||
if (readed < toread) goto error;
|
||||
|
||||
@@ -1118,14 +1118,14 @@ bool UserDict::load(const char *file, LemmaIdType start_id) {
|
||||
toread = (dict_info.lemma_count << 2);
|
||||
readed = 0;
|
||||
while (readed < toread && !ferror(fp) && !feof(fp)) {
|
||||
readed += fread((((uint8 *)predicts) + readed), 1, toread - readed, fp);
|
||||
readed += fread((((uint8*)predicts) + readed), 1, toread - readed, fp);
|
||||
}
|
||||
if (readed < toread) goto error;
|
||||
#endif
|
||||
|
||||
readed = 0;
|
||||
while (readed < toread && !ferror(fp) && !feof(fp)) {
|
||||
readed += fread((((uint8 *)scores) + readed), 1, toread - readed, fp);
|
||||
readed += fread((((uint8*)scores) + readed), 1, toread - readed, fp);
|
||||
}
|
||||
if (readed < toread) goto error;
|
||||
|
||||
@@ -1133,7 +1133,7 @@ bool UserDict::load(const char *file, LemmaIdType start_id) {
|
||||
toread = (dict_info.sync_count << 2);
|
||||
readed = 0;
|
||||
while (readed < toread && !ferror(fp) && !feof(fp)) {
|
||||
readed += fread((((uint8 *)syncs) + readed), 1, toread - readed, fp);
|
||||
readed += fread((((uint8*)syncs) + readed), 1, toread - readed, fp);
|
||||
}
|
||||
if (readed < toread) goto error;
|
||||
#endif
|
||||
@@ -1302,8 +1302,8 @@ void UserDict::write_back_all(int fd) {
|
||||
}
|
||||
|
||||
#ifdef ___CACHE_ENABLED___
|
||||
bool UserDict::load_cache(UserDictSearchable *searchable, uint32 *offset, uint32 *length) {
|
||||
UserDictCache *cache = &caches_[searchable->splids_len - 1];
|
||||
bool UserDict::load_cache(UserDictSearchable* searchable, uint32* offset, uint32* length) {
|
||||
UserDictCache* cache = &caches_[searchable->splids_len - 1];
|
||||
if (cache->head == cache->tail) return false;
|
||||
|
||||
uint16 j, sig_len = kMaxLemmaSize / 4;
|
||||
@@ -1326,8 +1326,8 @@ bool UserDict::load_cache(UserDictSearchable *searchable, uint32 *offset, uint32
|
||||
return false;
|
||||
}
|
||||
|
||||
void UserDict::save_cache(UserDictSearchable *searchable, uint32 offset, uint32 length) {
|
||||
UserDictCache *cache = &caches_[searchable->splids_len - 1];
|
||||
void UserDict::save_cache(UserDictSearchable* searchable, uint32 offset, uint32 length) {
|
||||
UserDictCache* cache = &caches_[searchable->splids_len - 1];
|
||||
uint16 next = cache->tail;
|
||||
|
||||
cache->offsets[next] = offset;
|
||||
@@ -1352,8 +1352,8 @@ void UserDict::save_cache(UserDictSearchable *searchable, uint32 offset, uint32
|
||||
|
||||
void UserDict::reset_cache() { memset(caches_, 0, sizeof(caches_)); }
|
||||
|
||||
bool UserDict::load_miss_cache(UserDictSearchable *searchable) {
|
||||
UserDictMissCache *cache = &miss_caches_[searchable->splids_len - 1];
|
||||
bool UserDict::load_miss_cache(UserDictSearchable* searchable) {
|
||||
UserDictMissCache* cache = &miss_caches_[searchable->splids_len - 1];
|
||||
if (cache->head == cache->tail) return false;
|
||||
|
||||
uint16 j, sig_len = kMaxLemmaSize / 4;
|
||||
@@ -1374,8 +1374,8 @@ bool UserDict::load_miss_cache(UserDictSearchable *searchable) {
|
||||
return false;
|
||||
}
|
||||
|
||||
void UserDict::save_miss_cache(UserDictSearchable *searchable) {
|
||||
UserDictMissCache *cache = &miss_caches_[searchable->splids_len - 1];
|
||||
void UserDict::save_miss_cache(UserDictSearchable* searchable) {
|
||||
UserDictMissCache* cache = &miss_caches_[searchable->splids_len - 1];
|
||||
uint16 next = cache->tail;
|
||||
|
||||
uint16 sig_len = kMaxLemmaSize / 4;
|
||||
@@ -1403,7 +1403,7 @@ void UserDict::cache_init() {
|
||||
reset_miss_cache();
|
||||
}
|
||||
|
||||
bool UserDict::cache_hit(UserDictSearchable *searchable, uint32 *offset, uint32 *length) {
|
||||
bool UserDict::cache_hit(UserDictSearchable* searchable, uint32* offset, uint32* length) {
|
||||
bool hit = load_miss_cache(searchable);
|
||||
if (hit) {
|
||||
*offset = 0;
|
||||
@@ -1417,7 +1417,7 @@ bool UserDict::cache_hit(UserDictSearchable *searchable, uint32 *offset, uint32
|
||||
return false;
|
||||
}
|
||||
|
||||
void UserDict::cache_push(UserDictCacheType type, UserDictSearchable *searchable, uint32 offset, uint32 length) {
|
||||
void UserDict::cache_push(UserDictCacheType type, UserDictSearchable* searchable, uint32 offset, uint32 length) {
|
||||
switch (type) {
|
||||
case USER_DICT_MISS_CACHE:
|
||||
save_miss_cache(searchable);
|
||||
@@ -1614,7 +1614,7 @@ LemmaIdType UserDict::put_lemma_no_sync(char16 lemma_str[], uint16 splids[], uin
|
||||
int again = 0;
|
||||
begin:
|
||||
LemmaIdType id;
|
||||
uint32 *syncs_bak = syncs_;
|
||||
uint32* syncs_bak = syncs_;
|
||||
syncs_ = NULL;
|
||||
id = _put_lemma(lemma_str, splids, lemma_len, count, lmt);
|
||||
syncs_ = syncs_bak;
|
||||
@@ -1632,26 +1632,26 @@ begin:
|
||||
return id;
|
||||
}
|
||||
|
||||
int UserDict::put_lemmas_no_sync_from_utf16le_string(char16 *lemmas, int len) {
|
||||
int UserDict::put_lemmas_no_sync_from_utf16le_string(char16* lemmas, int len) {
|
||||
int newly_added = 0;
|
||||
|
||||
SpellingParser *spl_parser = new SpellingParser();
|
||||
SpellingParser* spl_parser = new SpellingParser();
|
||||
if (!spl_parser) {
|
||||
return 0;
|
||||
}
|
||||
#ifdef ___DEBUG_PERF___
|
||||
DEBUG_PERF_BEGIN;
|
||||
#endif
|
||||
char16 *ptr = lemmas;
|
||||
char16* ptr = lemmas;
|
||||
|
||||
// Extract pinyin,words,frequence,last_mod_time
|
||||
char16 *p = ptr, *py16 = ptr;
|
||||
char16 *hz16 = NULL;
|
||||
char16* hz16 = NULL;
|
||||
int py16_len = 0;
|
||||
uint16 splid[kMaxLemmaSize];
|
||||
int splid_len = 0;
|
||||
int hz16_len = 0;
|
||||
char16 *fr16 = NULL;
|
||||
char16* fr16 = NULL;
|
||||
int fr16_len = 0;
|
||||
|
||||
while (p - ptr < len) {
|
||||
@@ -1708,7 +1708,7 @@ int UserDict::put_lemmas_no_sync_from_utf16le_string(char16 *lemmas, int len) {
|
||||
return newly_added;
|
||||
}
|
||||
|
||||
int UserDict::get_sync_lemmas_in_utf16le_string_from_beginning(char16 *str, int size, int *count) {
|
||||
int UserDict::get_sync_lemmas_in_utf16le_string_from_beginning(char16* str, int size, int* count) {
|
||||
int len = 0;
|
||||
*count = 0;
|
||||
|
||||
@@ -1716,7 +1716,7 @@ int UserDict::get_sync_lemmas_in_utf16le_string_from_beginning(char16 *str, int
|
||||
|
||||
if (is_valid_state() == false) return len;
|
||||
|
||||
SpellingTrie *spl_trie = &SpellingTrie::get_instance();
|
||||
SpellingTrie* spl_trie = &SpellingTrie::get_instance();
|
||||
if (!spl_trie) {
|
||||
return 0;
|
||||
}
|
||||
@@ -1725,8 +1725,8 @@ int UserDict::get_sync_lemmas_in_utf16le_string_from_beginning(char16 *str, int
|
||||
for (i = 0; i < dict_info_.sync_count; i++) {
|
||||
int offset = syncs_[i];
|
||||
uint32 nchar = get_lemma_nchar(offset);
|
||||
uint16 *spl = get_lemma_spell_ids(offset);
|
||||
uint16 *wrd = get_lemma_word(offset);
|
||||
uint16* spl = get_lemma_spell_ids(offset);
|
||||
uint16* wrd = get_lemma_word(offset);
|
||||
int score = _get_lemma_score(wrd, spl, nchar);
|
||||
|
||||
static char score_temp[32], *pscore_temp = score_temp;
|
||||
@@ -1812,7 +1812,7 @@ int UserDict::get_sync_lemmas_in_utf16le_string_from_beginning(char16 *str, int
|
||||
|
||||
#endif
|
||||
|
||||
bool UserDict::state(UserDictStat *stat) {
|
||||
bool UserDict::state(UserDictStat* stat) {
|
||||
if (is_valid_state() == false) return false;
|
||||
if (!stat) return false;
|
||||
stat->version = version_;
|
||||
@@ -1862,8 +1862,8 @@ void UserDict::reclaim() {
|
||||
uint32 count = dict_info_.lemma_count;
|
||||
int rc = count * dict_info_.reclaim_ratio / 100;
|
||||
|
||||
UserDictScoreOffsetPair *score_offset_pairs = NULL;
|
||||
score_offset_pairs = (UserDictScoreOffsetPair *)malloc(sizeof(UserDictScoreOffsetPair) * rc);
|
||||
UserDictScoreOffsetPair* score_offset_pairs = NULL;
|
||||
score_offset_pairs = (UserDictScoreOffsetPair*)malloc(sizeof(UserDictScoreOffsetPair) * rc);
|
||||
if (score_offset_pairs == NULL) {
|
||||
return;
|
||||
}
|
||||
@@ -1896,7 +1896,7 @@ void UserDict::reclaim() {
|
||||
free(score_offset_pairs);
|
||||
}
|
||||
|
||||
inline void UserDict::swap(UserDictScoreOffsetPair *sop, int i, int j) {
|
||||
inline void UserDict::swap(UserDictScoreOffsetPair* sop, int i, int j) {
|
||||
int s = sop[i].score;
|
||||
int p = sop[i].offset_index;
|
||||
sop[i].score = sop[j].score;
|
||||
@@ -1905,7 +1905,7 @@ inline void UserDict::swap(UserDictScoreOffsetPair *sop, int i, int j) {
|
||||
sop[j].offset_index = p;
|
||||
}
|
||||
|
||||
void UserDict::shift_down(UserDictScoreOffsetPair *sop, int i, int n) {
|
||||
void UserDict::shift_down(UserDictScoreOffsetPair* sop, int i, int n) {
|
||||
int par = i;
|
||||
while (par < n) {
|
||||
int left = par * 2 + 1;
|
||||
@@ -1983,7 +1983,7 @@ void UserDict::queue_lemma_for_sync(LemmaIdType id) {
|
||||
if (dict_info_.sync_count < sync_count_size_) {
|
||||
syncs_[dict_info_.sync_count++] = offsets_by_id_[id - start_id_];
|
||||
} else {
|
||||
uint32 *syncs = (uint32 *)realloc(syncs_, (sync_count_size_ + kUserDictPreAlloc) << 2);
|
||||
uint32* syncs = (uint32*)realloc(syncs_, (sync_count_size_ + kUserDictPreAlloc) << 2);
|
||||
if (syncs) {
|
||||
sync_count_size_ += kUserDictPreAlloc;
|
||||
syncs_ = syncs;
|
||||
@@ -2001,8 +2001,8 @@ LemmaIdType UserDict::update_lemma(LemmaIdType lemma_id, int16 delta_count, bool
|
||||
if (is_valid_lemma_id(lemma_id) == false) return 0;
|
||||
uint32 offset = offsets_by_id_[lemma_id - start_id_];
|
||||
uint8 lemma_len = get_lemma_nchar(offset);
|
||||
char16 *lemma_str = get_lemma_word(offset);
|
||||
uint16 *splids = get_lemma_spell_ids(offset);
|
||||
char16* lemma_str = get_lemma_word(offset);
|
||||
uint16* splids = get_lemma_spell_ids(offset);
|
||||
|
||||
int32 off = locate_in_offsets(lemma_str, splids, lemma_len);
|
||||
if (off != -1) {
|
||||
@@ -2043,8 +2043,8 @@ LemmaIdType UserDict::append_a_lemma(char16 lemma_str[], uint16 splids[], uint16
|
||||
lemmas_[offset] = 0;
|
||||
lemmas_[offset + 1] = (uint8)lemma_len;
|
||||
for (size_t i = 0; i < lemma_len; i++) {
|
||||
*((uint16 *)&lemmas_[offset + 2 + (i << 1)]) = splids[i];
|
||||
*((char16 *)&lemmas_[offset + 2 + (lemma_len << 1) + (i << 1)]) = lemma_str[i];
|
||||
*((uint16*)&lemmas_[offset + 2 + (i << 1)]) = splids[i];
|
||||
*((char16*)&lemmas_[offset + 2 + (lemma_len << 1) + (i << 1)]) = lemma_str[i];
|
||||
}
|
||||
uint32 off = dict_info_.lemma_count;
|
||||
offsets_[off] = offset;
|
||||
@@ -2070,7 +2070,7 @@ LemmaIdType UserDict::append_a_lemma(char16 lemma_str[], uint16 splids[], uint16
|
||||
while (i < off) {
|
||||
offset = offsets_[i];
|
||||
uint32 nchar = get_lemma_nchar(offset);
|
||||
uint16 *spl = get_lemma_spell_ids(offset);
|
||||
uint16* spl = get_lemma_spell_ids(offset);
|
||||
|
||||
if (0 <= fuzzy_compare_spell_id(spl, nchar, &searchable)) break;
|
||||
i++;
|
||||
@@ -2091,7 +2091,7 @@ LemmaIdType UserDict::append_a_lemma(char16 lemma_str[], uint16 splids[], uint16
|
||||
|
||||
#ifdef ___PREDICT_ENABLED___
|
||||
uint32 j = 0;
|
||||
uint16 *words_new = get_lemma_word(predicts_[off]);
|
||||
uint16* words_new = get_lemma_word(predicts_[off]);
|
||||
j = locate_where_to_insert_in_predicts(words_new, lemma_len);
|
||||
if (j != off) {
|
||||
uint32 temp = predicts_[off];
|
||||
|
||||
+13
-13
@@ -23,7 +23,7 @@ namespace ime_pinyin {
|
||||
extern "C" {
|
||||
#endif
|
||||
|
||||
char16 *utf16_strtok(char16 *utf16_str, size_t *token_size, char16 **utf16_str_next) {
|
||||
char16* utf16_strtok(char16* utf16_str, size_t* token_size, char16** utf16_str_next) {
|
||||
if (NULL == utf16_str || NULL == token_size || NULL == utf16_str_next) {
|
||||
return NULL;
|
||||
}
|
||||
@@ -39,7 +39,7 @@ char16 *utf16_strtok(char16 *utf16_str, size_t *token_size, char16 **utf16_str_n
|
||||
pos++;
|
||||
}
|
||||
|
||||
char16 *ret_val = utf16_str;
|
||||
char16* ret_val = utf16_str;
|
||||
if ((char16)'\0' == utf16_str[pos]) {
|
||||
*utf16_str_next = NULL;
|
||||
if (0 == pos) return NULL;
|
||||
@@ -53,7 +53,7 @@ char16 *utf16_strtok(char16 *utf16_str, size_t *token_size, char16 **utf16_str_n
|
||||
return ret_val;
|
||||
}
|
||||
|
||||
int utf16_atoi(const char16 *utf16_str) {
|
||||
int utf16_atoi(const char16* utf16_str) {
|
||||
if (NULL == utf16_str) return 0;
|
||||
|
||||
int value = 0;
|
||||
@@ -73,7 +73,7 @@ int utf16_atoi(const char16 *utf16_str) {
|
||||
return value * sign;
|
||||
}
|
||||
|
||||
float utf16_atof(const char16 *utf16_str) {
|
||||
float utf16_atof(const char16* utf16_str) {
|
||||
// A temporary implemetation.
|
||||
char char8[256];
|
||||
if (utf16_strlen(utf16_str) >= 256) return 0;
|
||||
@@ -82,7 +82,7 @@ float utf16_atof(const char16 *utf16_str) {
|
||||
return atof(char8);
|
||||
}
|
||||
|
||||
size_t utf16_strlen(const char16 *utf16_str) {
|
||||
size_t utf16_strlen(const char16* utf16_str) {
|
||||
if (NULL == utf16_str) return 0;
|
||||
|
||||
size_t size = 0;
|
||||
@@ -90,14 +90,14 @@ size_t utf16_strlen(const char16 *utf16_str) {
|
||||
return size;
|
||||
}
|
||||
|
||||
int utf16_strcmp(const char16 *str1, const char16 *str2) {
|
||||
int utf16_strcmp(const char16* str1, const char16* str2) {
|
||||
size_t pos = 0;
|
||||
while (str1[pos] == str2[pos] && (char16)'\0' != str1[pos]) pos++;
|
||||
|
||||
return static_cast<int>(str1[pos]) - static_cast<int>(str2[pos]);
|
||||
}
|
||||
|
||||
int utf16_strncmp(const char16 *str1, const char16 *str2, size_t size) {
|
||||
int utf16_strncmp(const char16* str1, const char16* str2, size_t size) {
|
||||
size_t pos = 0;
|
||||
while (pos < size && str1[pos] == str2[pos] && (char16)'\0' != str1[pos]) pos++;
|
||||
|
||||
@@ -107,10 +107,10 @@ int utf16_strncmp(const char16 *str1, const char16 *str2, size_t size) {
|
||||
}
|
||||
|
||||
// we do not consider overlapping
|
||||
char16 *utf16_strcpy(char16 *dst, const char16 *src) {
|
||||
char16* utf16_strcpy(char16* dst, const char16* src) {
|
||||
if (NULL == src || NULL == dst) return NULL;
|
||||
|
||||
char16 *cp = dst;
|
||||
char16* cp = dst;
|
||||
|
||||
while ((char16)'\0' != *src) {
|
||||
*cp = *src;
|
||||
@@ -123,12 +123,12 @@ char16 *utf16_strcpy(char16 *dst, const char16 *src) {
|
||||
return dst;
|
||||
}
|
||||
|
||||
char16 *utf16_strncpy(char16 *dst, const char16 *src, size_t size) {
|
||||
char16* utf16_strncpy(char16* dst, const char16* src, size_t size) {
|
||||
if (NULL == src || NULL == dst || 0 == size) return NULL;
|
||||
|
||||
if (src == dst) return dst;
|
||||
|
||||
char16 *cp = dst;
|
||||
char16* cp = dst;
|
||||
|
||||
if (dst < src || (dst > src && dst >= src + size)) {
|
||||
while (size-- && (*cp++ = *src++));
|
||||
@@ -142,10 +142,10 @@ char16 *utf16_strncpy(char16 *dst, const char16 *src, size_t size) {
|
||||
|
||||
// We do not handle complicated cases like overlapping, because in this
|
||||
// codebase, it is not necessary.
|
||||
char *utf16_strcpy_tochar(char *dst, const char16 *src) {
|
||||
char* utf16_strcpy_tochar(char* dst, const char16* src) {
|
||||
if (NULL == src || NULL == dst) return NULL;
|
||||
|
||||
char *cp = dst;
|
||||
char* cp = dst;
|
||||
|
||||
while ((char16)'\0' != *src) {
|
||||
*cp = static_cast<char>(*src);
|
||||
|
||||
@@ -35,7 +35,7 @@ Utf16Reader::~Utf16Reader() {
|
||||
if (NULL != buffer_) delete[] buffer_;
|
||||
}
|
||||
|
||||
bool Utf16Reader::open(const char *filename, size_t buffer_len) {
|
||||
bool Utf16Reader::open(const char* filename, size_t buffer_len) {
|
||||
if (filename == NULL) return false;
|
||||
|
||||
if (buffer_len < MIN_BUF_LEN)
|
||||
@@ -62,7 +62,7 @@ bool Utf16Reader::open(const char *filename, size_t buffer_len) {
|
||||
return true;
|
||||
}
|
||||
|
||||
char16 *Utf16Reader::readline(char16 *read_buf, size_t max_len) {
|
||||
char16* Utf16Reader::readline(char16* read_buf, size_t max_len) {
|
||||
if (NULL == fp_ || NULL == read_buf || 0 == max_len) return NULL;
|
||||
|
||||
size_t ret_len = 0;
|
||||
|
||||
+7
-7
@@ -7,8 +7,8 @@
|
||||
#include <utf8cpp/utf8.h>
|
||||
#include <Windows.h>
|
||||
|
||||
std::string fromUtf16(const ime_pinyin::char16 *buf, size_t len) {
|
||||
const char16_t *utf16Data = reinterpret_cast<const char16_t *>(buf);
|
||||
std::string fromUtf16(const ime_pinyin::char16* buf, size_t len) {
|
||||
const char16_t* utf16Data = reinterpret_cast<const char16_t*>(buf);
|
||||
std::u16string utf16Str(utf16Data, len);
|
||||
|
||||
std::string utf8Result;
|
||||
@@ -17,14 +17,14 @@ std::string fromUtf16(const ime_pinyin::char16 *buf, size_t len) {
|
||||
return utf8Result;
|
||||
}
|
||||
|
||||
void test_pinyin_search_and_segment(const std::string &user_pinyin) {
|
||||
void test_pinyin_search_and_segment(const std::string& user_pinyin) {
|
||||
auto start_time = std::chrono::high_resolution_clock::now();
|
||||
|
||||
std::string pinyin_str = user_pinyin;
|
||||
|
||||
const char *pinyin = pinyin_str.c_str();
|
||||
const char* pinyin = pinyin_str.c_str();
|
||||
size_t cand_cnt = ime_pinyin::im_search(pinyin, strlen(pinyin));
|
||||
const ime_pinyin::uint16 *spl_start = nullptr;
|
||||
const ime_pinyin::uint16* spl_start = nullptr;
|
||||
size_t segment_count = ime_pinyin::im_get_spl_start_pos(spl_start);
|
||||
|
||||
if (spl_start != nullptr && segment_count > 0) {
|
||||
@@ -64,12 +64,12 @@ void test_pinyin_search_and_segment(const std::string &user_pinyin) {
|
||||
std::cout << "========================================" << std::endl;
|
||||
}
|
||||
|
||||
void test_pinyin_search_when_retriving_first_element(const std::string &user_pinyin) {
|
||||
void test_pinyin_search_when_retriving_first_element(const std::string& user_pinyin) {
|
||||
auto start_time = std::chrono::high_resolution_clock::now();
|
||||
|
||||
std::string pinyin_str = user_pinyin;
|
||||
|
||||
const char *pinyin = pinyin_str.c_str();
|
||||
const char* pinyin = pinyin_str.c_str();
|
||||
size_t cand_cnt = ime_pinyin::im_search(pinyin, strlen(pinyin));
|
||||
std::string msg;
|
||||
std::vector<std::string> candidateList;
|
||||
|
||||
Reference in New Issue
Block a user