format codes

This commit is contained in:
fanlumaster
2026-03-11 12:24:00 +08:00
parent 4001efb63a
commit b0700b2a21
27 changed files with 647 additions and 647 deletions
+6 -6
View File
@@ -61,7 +61,7 @@ class AtomDictBase {
* parameter.
* @return True if succeed.
*/
virtual bool load_dict(const char *file_name, LemmaIdType start_id, LemmaIdType end_id) = 0;
virtual bool load_dict(const char* file_name, LemmaIdType start_id, LemmaIdType end_id) = 0;
/**
* Close this atom dictionary.
@@ -119,7 +119,7 @@ class AtomDictBase {
* @param lpi_num Used to return the newly added items.
* @return The new mile stone for this extending. 0 if fail.
*/
virtual MileStoneHandle extend_dict(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num) = 0;
virtual MileStoneHandle extend_dict(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num) = 0;
/**
* Get lemma items with scores according to a spelling id stream.
@@ -131,7 +131,7 @@ class AtomDictBase {
* @param lpi_max The maximum size of the buffer to return result.
* @return The number of matched items which have been filled in to lpi_items.
*/
virtual size_t get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lpi_items, size_t lpi_max) = 0;
virtual size_t get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lpi_items, size_t lpi_max) = 0;
/**
* Get a lemma string (The Chinese string) by the given lemma id.
@@ -141,7 +141,7 @@ class AtomDictBase {
* @param str_max The maximum size of the buffer.
* @return The length of the string, 0 if fail.
*/
virtual uint16 get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max) = 0;
virtual uint16 get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max) = 0;
/**
* Get the full spelling ids for the given lemma id.
@@ -155,7 +155,7 @@ class AtomDictBase {
* case, splids_max is the number of valid ids in splids.
* @return The number of ids in the buffer.
*/
virtual uint16 get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid) = 0;
virtual uint16 get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid) = 0;
/**
* Function used for prediction.
@@ -170,7 +170,7 @@ class AtomDictBase {
* from other atom dictionaries. A atom ditionary can just ignore it.
* @return The number of prediction result from this atom dictionary.
*/
virtual size_t predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used) = 0;
virtual size_t predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used) = 0;
/**
* Add a lemma to the dictionary. If the dictionary allows to add new
+18 -18
View File
@@ -36,38 +36,38 @@ class DictTrie;
class DictBuilder {
private:
// The raw lemma array buffer.
LemmaEntry *lemma_arr_;
LemmaEntry* lemma_arr_;
size_t lemma_num_;
// Used to store all possible single char items.
// Two items may have the same Hanzi while their spelling ids are different.
SingleCharItem *scis_;
SingleCharItem* scis_;
size_t scis_num_;
// In the tree, root's level is -1.
// Lemma nodes for root, and level 0
LmaNodeLE0 *lma_nodes_le0_;
LmaNodeLE0* lma_nodes_le0_;
// Lemma nodes for layers whose levels are deeper than 0
LmaNodeGE1 *lma_nodes_ge1_;
LmaNodeGE1* lma_nodes_ge1_;
// Number of used lemma nodes
size_t lma_nds_used_num_le0_;
size_t lma_nds_used_num_ge1_;
// Used to store homophonies' ids.
LemmaIdType *homo_idx_buf_;
LemmaIdType* homo_idx_buf_;
// Number of homophonies each of which only contains one Chinese character.
size_t homo_idx_num_eq1_;
// Number of homophonies each of which contains more than one character.
size_t homo_idx_num_gt1_;
// The items with highest scores.
LemmaEntry *top_lmas_;
LemmaEntry* top_lmas_;
size_t top_lmas_num_;
SpellingTable *spl_table_;
SpellingParser *spl_parser_;
SpellingTable* spl_table_;
SpellingParser* spl_parser_;
#ifdef ___DO_STATISTICS___
size_t max_sonbuf_len_[kMaxLemmaSize];
@@ -96,21 +96,21 @@ class DictBuilder {
// Build dictionary trie from the file fn_raw. File fn_validhzs provides
// valid chars. If fn_validhzs is NULL, only chars in GB2312 will be
// included.
bool build_dict(const char *fn_raw, const char *fn_validhzs, DictTrie *dict_trie);
bool build_dict(const char* fn_raw, const char* fn_validhzs, DictTrie* dict_trie);
private:
// Fill in the buffer with id. The caller guarantees that the paramters are
// vaild.
void id_to_charbuf(unsigned char *buf, LemmaIdType id);
void id_to_charbuf(unsigned char* buf, LemmaIdType id);
// Update the offset of sons for a node.
void set_son_offset(LmaNodeGE1 *node, size_t offset);
void set_son_offset(LmaNodeGE1* node, size_t offset);
// Update the offset of homophonies' ids for a node.
void set_homo_id_buf_offset(LmaNodeGE1 *node, size_t offset);
void set_homo_id_buf_offset(LmaNodeGE1* node, size_t offset);
// Format a speling string.
void format_spelling_str(char *spl_str);
void format_spelling_str(char* spl_str);
// Sort the lemma_arr by the hanzi string, and give each of unique items
// a id. Why we need to sort the lemma list according to their Hanzi string
@@ -130,23 +130,23 @@ class DictBuilder {
// item_star to item_end)
// parent is the parent node to update the necessary information
// parent can be a member of LmaNodeLE0 or LmaNodeGE1
bool construct_subset(void *parent, LemmaEntry *lemma_arr, size_t item_start, size_t item_end, size_t level);
bool construct_subset(void* parent, LemmaEntry* lemma_arr, size_t item_start, size_t item_end, size_t level);
// Read valid Chinese Hanzis from the given file.
// num is used to return number of chars.
// The return buffer is sorted and caller needs to free the returned buffer.
char16 *read_valid_hanzis(const char *fn_validhzs, size_t *num);
char16* read_valid_hanzis(const char* fn_validhzs, size_t* num);
// Read a raw dictionary. max_item is the maximum number of items. If there
// are more items in the ditionary, only the first max_item will be read.
// Returned value is the number of items successfully read from the file.
size_t read_raw_dict(const char *fn_raw, const char *fn_validhzs, size_t max_item);
size_t read_raw_dict(const char* fn_raw, const char* fn_validhzs, size_t max_item);
// Try to find if a character is in hzs buffer.
bool hz_in_hanzis_list(const char16 *hzs, size_t hzs_len, char16 hz);
bool hz_in_hanzis_list(const char16* hzs, size_t hzs_len, char16 hz);
// Try to find if all characters in str are in hzs buffer.
bool str_in_hanzis_list(const char16 *hzs, size_t hzs_len, const char16 *str, size_t str_len);
bool str_in_hanzis_list(const char16* hzs, size_t hzs_len, const char16* str, size_t str_len);
// Get these lemmas with toppest scores.
void get_top_lemmas();
+19 -19
View File
@@ -30,15 +30,15 @@ class DictList {
private:
bool initialized_;
const SpellingTrie *spl_trie_;
const SpellingTrie* spl_trie_;
// Number of SingCharItem. The first is blank, because id 0 is invalid.
size_t scis_num_;
char16 *scis_hz_;
SpellingId *scis_splid_;
char16* scis_hz_;
SpellingId* scis_splid_;
// The large memory block to store the word list.
char16 *buf_;
char16* buf_;
// Starting position of those words whose lengths are i+1, counted in
// char16
@@ -46,7 +46,7 @@ class DictList {
uint32 start_id_[kMaxLemmaSize + 1];
int (*cmp_func_[kMaxLemmaSize])(const void *, const void *);
int (*cmp_func_[kMaxLemmaSize])(const void*, const void*);
bool alloc_resource(size_t buf_size, size_t scim_num);
@@ -54,44 +54,44 @@ class DictList {
#ifdef ___BUILD_MODEL___
// Calculate the requsted memory, including the start_pos[] buffer.
size_t calculate_size(const LemmaEntry *lemma_arr, size_t lemma_num);
size_t calculate_size(const LemmaEntry* lemma_arr, size_t lemma_num);
void fill_scis(const SingleCharItem *scis, size_t scis_num);
void fill_scis(const SingleCharItem* scis, size_t scis_num);
// Copy the related content to the inner buffer
// It should be called after calculate_size()
void fill_list(const LemmaEntry *lemma_arr, size_t lemma_num);
void fill_list(const LemmaEntry* lemma_arr, size_t lemma_num);
// Find the starting position for the buffer of those 2-character Chinese word
// whose first character is the given Chinese character.
char16 *find_pos2_startedbyhz(char16 hz_char);
char16* find_pos2_startedbyhz(char16 hz_char);
#endif
// Find the starting position for the buffer of those words whose lengths are
// word_len. The given parameter cmp_func decides how many characters from
// beginning will be used to compare.
char16 *find_pos_startedbyhzs(const char16 last_hzs[], size_t word_Len, int (*cmp_func)(const void *, const void *));
char16* find_pos_startedbyhzs(const char16 last_hzs[], size_t word_Len, int (*cmp_func)(const void*, const void*));
public:
DictList();
~DictList();
bool save_list(FILE *fp);
bool load_list(FILE *fp);
bool save_list(FILE* fp);
bool load_list(FILE* fp);
#ifdef ___BUILD_MODEL___
// Init the list from the LemmaEntry array.
// lemma_arr should have been sorted by the hanzi_str, and have been given
// ids from 1
bool init_list(const SingleCharItem *scis, size_t scis_num, const LemmaEntry *lemma_arr, size_t lemma_num);
bool init_list(const SingleCharItem* scis, size_t scis_num, const LemmaEntry* lemma_arr, size_t lemma_num);
#endif
// Get the hanzi string for the given id
uint16 get_lemma_str(LemmaIdType id_hz, char16 *str_buf, uint16 str_max);
uint16 get_lemma_str(LemmaIdType id_hz, char16* str_buf, uint16 str_max);
void convert_to_hanzis(char16 *str, uint16 str_len);
void convert_to_hanzis(char16* str, uint16 str_len);
void convert_to_scis_ids(char16 *str, uint16 str_len);
void convert_to_scis_ids(char16* str, uint16 str_len);
// last_hzs stores the last n Chinese characters history, its length should be
// less or equal than kMaxPredictSize.
@@ -100,13 +100,13 @@ class DictList {
// buf_len specifies the buffer length.
// b4_used specifies how many items before predict_buf have been used.
// Returned value is the number of newly added items.
size_t predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used);
size_t predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used);
// If half_splid is a valid half spelling id, return those full spelling
// ids which share this half id.
uint16 get_splids_for_hanzi(char16 hanzi, uint16 half_splid, uint16 *splids, uint16 max_splids);
uint16 get_splids_for_hanzi(char16 hanzi, uint16 half_splid, uint16* splids, uint16 max_splids);
LemmaIdType get_lemma_id(const char16 *str, uint16 str_len);
LemmaIdType get_lemma_id(const char16* str, uint16 str_len);
};
} // namespace ime_pinyin
+29 -29
View File
@@ -55,12 +55,12 @@ class DictTrie : AtomDictBase {
uint16 mark_num;
};
DictList *dict_list_;
DictList* dict_list_;
const SpellingTrie *spl_trie_;
const SpellingTrie* spl_trie_;
LmaNodeLE0 *root_; // Nodes for root and the first layer.
LmaNodeGE1 *nodes_ge1_; // Nodes for other layers.
LmaNodeLE0* root_; // Nodes for root and the first layer.
LmaNodeGE1* nodes_ge1_; // Nodes for other layers.
// An quick index from spelling id to the LmaNodeLE0 node buffer, or
// to the root_ buffer.
@@ -71,69 +71,69 @@ class DictTrie : AtomDictBase {
// corresponding full ids.
// So, given an id splid, the son is:
// root_[splid_le0_index_[splid - kFullSplIdStart]]
uint16 *splid_le0_index_;
uint16* splid_le0_index_;
size_t lma_node_num_le0_;
size_t lma_node_num_ge1_;
// The first part is for homophnies, and the last top_lma_num_ items are
// lemmas with highest scores.
unsigned char *lma_idx_buf_;
unsigned char* lma_idx_buf_;
size_t lma_idx_buf_len_; // The total size of lma_idx_buf_ in byte.
size_t total_lma_num_; // Total number of lemmas in this dictionary.
size_t top_lmas_num_; // Number of lemma with highest scores.
// Parsing mark list used to mark the detailed extended statuses.
ParsingMark *parsing_marks_;
ParsingMark* parsing_marks_;
// The position for next available mark.
uint16 parsing_marks_pos_;
// Mile stone list used to mark the extended status.
MileStone *mile_stones_;
MileStone* mile_stones_;
// The position for the next available mile stone. We use positions (except 0)
// as handles.
MileStoneHandle mile_stones_pos_;
// Get the offset of sons for a node.
inline size_t get_son_offset(const LmaNodeGE1 *node);
inline size_t get_son_offset(const LmaNodeGE1* node);
// Get the offset of homonious ids for a node.
inline size_t get_homo_idx_buf_offset(const LmaNodeGE1 *node);
inline size_t get_homo_idx_buf_offset(const LmaNodeGE1* node);
// Get the lemma id by the offset.
inline LemmaIdType get_lemma_id(size_t id_offset);
void free_resource(bool free_dict_list);
bool load_dict(FILE *fp);
bool load_dict(FILE* fp);
// Given a LmaNodeLE0 node, extract the lemmas specified by it, and fill
// them into the lpi_items buffer.
// This function is called by the search engine.
size_t fill_lpi_buffer(LmaPsbItem lpi_items[], size_t max_size, LmaNodeLE0 *node);
size_t fill_lpi_buffer(LmaPsbItem lpi_items[], size_t max_size, LmaNodeLE0* node);
// Given a LmaNodeGE1 node, extract the lemmas specified by it, and fill
// them into the lpi_items buffer.
// This function is called by inner functions extend_dict0(), extend_dict1()
// and extend_dict2().
size_t fill_lpi_buffer(LmaPsbItem lpi_items[], size_t max_size, size_t homo_buf_off, LmaNodeGE1 *node, uint16 lma_len);
size_t fill_lpi_buffer(LmaPsbItem lpi_items[], size_t max_size, size_t homo_buf_off, LmaNodeGE1* node, uint16 lma_len);
// Extend in the trie from level 0.
MileStoneHandle extend_dict0(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num);
MileStoneHandle extend_dict0(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num);
// Extend in the trie from level 1.
MileStoneHandle extend_dict1(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num);
MileStoneHandle extend_dict1(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num);
// Extend in the trie from level 2.
MileStoneHandle extend_dict2(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num);
MileStoneHandle extend_dict2(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num);
// Try to extend the given spelling id buffer, and if the given id_lemma can
// be successfully gotten, return true;
// The given spelling ids are all valid full ids.
bool try_extend(const uint16 *splids, uint16 splid_num, LemmaIdType id_lemma);
bool try_extend(const uint16* splids, uint16 splid_num, LemmaIdType id_lemma);
#ifdef ___BUILD_MODEL___
bool save_dict(FILE *fp);
bool save_dict(FILE* fp);
#endif // ___BUILD_MODEL___
// Increase limits to support longer Pinyin input.
@@ -153,35 +153,35 @@ class DictTrie : AtomDictBase {
// Construct the tree from the file fn_raw.
// fn_validhzs provide the valid hanzi list. If fn_validhzs is
// NULL, only chars in GB2312 will be included.
bool build_dict(const char *fn_raw, const char *fn_validhzs);
bool build_dict(const char* fn_raw, const char* fn_validhzs);
// Save the binary dictionary
// Actually, the SpellingTrie/DictList instance will be also saved.
bool save_dict(const char *filename);
bool save_dict(const char* filename);
#endif // ___BUILD_MODEL___
void convert_to_hanzis(char16 *str, uint16 str_len);
void convert_to_hanzis(char16* str, uint16 str_len);
void convert_to_scis_ids(char16 *str, uint16 str_len);
void convert_to_scis_ids(char16* str, uint16 str_len);
// Load a binary dictionary
// The SpellingTrie instance/DictList will be also loaded
bool load_dict(const char *filename, LemmaIdType start_id, LemmaIdType end_id);
bool load_dict(const char* filename, LemmaIdType start_id, LemmaIdType end_id);
bool load_dict_fd(int sys_fd, long start_offset, long length, LemmaIdType start_id, LemmaIdType end_id);
bool close_dict() { return true; }
size_t number_of_lemmas() { return 0; }
void reset_milestones(uint16 from_step, MileStoneHandle from_handle);
MileStoneHandle extend_dict(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num);
MileStoneHandle extend_dict(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num);
size_t get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lpi_items, size_t lpi_max);
size_t get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lpi_items, size_t lpi_max);
uint16 get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max);
uint16 get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max);
uint16 get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid);
uint16 get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid);
size_t predict(const char16 *last_hzs, uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used);
size_t predict(const char16* last_hzs, uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used);
LemmaIdType put_lemma(char16 /*lemma_str*/[], uint16 /*splids*/[], uint16 /*lemma_len*/, uint16 /*count*/) { return 0; }
@@ -204,7 +204,7 @@ class DictTrie : AtomDictBase {
// Fill the lemmas with highest scores to the prediction buffer.
// his_len is the history length to fill in the prediction buffer.
size_t predict_top_lmas(size_t his_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used);
size_t predict_top_lmas(size_t his_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used);
};
} // namespace ime_pinyin
+4 -4
View File
@@ -26,17 +26,17 @@ namespace ime_pinyin {
// Used to cache LmaPsbItem list for half spelling ids.
class LpiCache {
private:
static LpiCache *instance_;
static LpiCache* instance_;
static const int kMaxLpiCachePerId = 15;
LmaPsbItem *lpi_cache_;
uint16 *lpi_cache_len_;
LmaPsbItem* lpi_cache_;
uint16* lpi_cache_len_;
public:
LpiCache();
~LpiCache();
static LpiCache &get_instance();
static LpiCache& get_instance();
// Test if the LPI list of the given splid has been cached.
// If splid is a full spelling id, it returns false, because we only cache
+26 -26
View File
@@ -57,7 +57,7 @@ typedef struct {
typedef struct MatrixNode {
LemmaIdType id;
float score;
MatrixNode *from;
MatrixNode* from;
// From which DMI node. Used to trace the spelling segmentation.
PoolPosType dmi_fr;
uint16 step;
@@ -98,7 +98,7 @@ typedef struct {
uint16 dmi_has_full_id : 1;
// Points to a MatrixNode of the current step to indicate which choice the
// user selects.
MatrixNode *mtrx_nd_fixed;
MatrixNode* mtrx_nd_fixed;
} MatrixRow, *PMatrixRow;
// When user inputs and selects candidates, the fixed lemma ids are stored in
@@ -159,7 +159,7 @@ class MatrixSearch {
bool inited_;
// Spelling trie.
const SpellingTrie *spl_trie_;
const SpellingTrie* spl_trie_;
// Used to indicate this switcher status: when "xian" is parseed, should
// "xi an" also be extended. Default is false.
@@ -170,13 +170,13 @@ class MatrixSearch {
bool xi_an_enabled_;
// System dictionary.
DictTrie *dict_trie_;
DictTrie* dict_trie_;
// User dictionary.
AtomDictBase *user_dict_;
AtomDictBase* user_dict_;
// Spelling parser.
SpellingParser *spl_parser_;
SpellingParser* spl_parser_;
// The maximum allowed length of spelling string (such as a Pinyin string).
size_t max_sps_len_;
@@ -191,18 +191,18 @@ class MatrixSearch {
size_t pys_decoded_len_;
// Shared buffer for multiple purposes.
size_t *share_buf_;
size_t* share_buf_;
MatrixNode *mtrx_nd_pool_;
MatrixNode* mtrx_nd_pool_;
PoolPosType mtrx_nd_pool_used_; // How many nodes used in the pool
DictMatchInfo *dmi_pool_;
DictMatchInfo* dmi_pool_;
PoolPosType dmi_pool_used_; // How many items used in the pool
MatrixRow *matrix_; // The first row is for starting
MatrixRow* matrix_; // The first row is for starting
DictExtPara *dep_; // Parameter used to extend DMI nodes.
DictExtPara* dep_; // Parameter used to extend DMI nodes.
NPredictItem *npre_items_; // Used to do prediction
NPredictItem* npre_items_; // Used to do prediction
size_t npre_items_len_;
// The starting positions and lemma ids for the full sentence candidate.
@@ -287,11 +287,11 @@ class MatrixSearch {
// If pfullsent is not NULL, means the full sentence candidate may be the
// same with the coming lemma string, if so, remove that lemma.
// The result is sorted in descendant order by the frequency score.
size_t get_lpis(const uint16 *splid_str, size_t splid_str_len, LmaPsbItem *lma_buf, size_t max_lma_buf, const char16 *pfullsent, bool sort_by_psb);
size_t get_lpis(const uint16* splid_str, size_t splid_str_len, LmaPsbItem* lma_buf, size_t max_lma_buf, const char16* pfullsent, bool sort_by_psb);
uint16 get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max);
uint16 get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max);
uint16 get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid);
uint16 get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid);
// Extend a DMI node with a spelling id. ext_len is the length of the rows
// to extend, actually, it is the size of the spelling string of splid.
@@ -312,17 +312,17 @@ class MatrixSearch {
// calling this function if necessary.
//
// The caller should guarantees that NULL != dep.
size_t extend_dmi(DictExtPara *dep, DictMatchInfo *dmi_s);
size_t extend_dmi(DictExtPara* dep, DictMatchInfo* dmi_s);
// Extend dmi for the composing phrase.
size_t extend_dmi_c(DictExtPara *dep, DictMatchInfo *dmi_s);
size_t extend_dmi_c(DictExtPara* dep, DictMatchInfo* dmi_s);
// Extend a MatrixNode with the give LmaPsbItem list.
// res_row is the destination row number.
// This function does not change mtrx_nd_pool_used_. Please change it after
// calling this function if necessary.
// return 0 always.
size_t extend_mtrx_nd(MatrixNode *mtrx_nd, LmaPsbItem lpi_items[], size_t lpi_num, PoolPosType dmi_fr, size_t res_row);
size_t extend_mtrx_nd(MatrixNode* mtrx_nd, LmaPsbItem lpi_items[], size_t lpi_num, PoolPosType dmi_fr, size_t res_row);
// Try to find a dmi node at step_to position, and the found dmi node should
// match the given spelling id strings.
@@ -341,7 +341,7 @@ class MatrixSearch {
// The caller guarantees that the position is valid.
bool is_split_at(uint16 pos);
void fill_dmi(DictMatchInfo *dmi, MileStoneHandle *handles, PoolPosType dmi_fr, uint16 spl_id, uint16 node_num, unsigned char dict_level, bool splid_end_split, unsigned char splstr_len, unsigned char all_full_id);
void fill_dmi(DictMatchInfo* dmi, MileStoneHandle* handles, PoolPosType dmi_fr, uint16 spl_id, uint16 node_num, unsigned char dict_level, bool splid_end_split, unsigned char splstr_len, unsigned char all_full_id);
size_t inner_predict(const char16 fixed_scis_ids[], uint16 scis_num, char16 predict_buf[][kMaxPredictSize + 1], size_t buf_len);
@@ -364,9 +364,9 @@ class MatrixSearch {
MatrixSearch();
~MatrixSearch();
bool init(const char *fn_sys_dict, const char *fn_usr_dict);
bool init(const char* fn_sys_dict, const char* fn_usr_dict);
bool init_fd(int sys_fd, long start_offset, long length, const char *fn_usr_dict);
bool init_fd(int sys_fd, long start_offset, long length, const char* fn_usr_dict);
void set_max_lens(size_t max_sps_len, size_t max_hzs_len);
@@ -384,7 +384,7 @@ class MatrixSearch {
// Search a Pinyin string.
// Return value is the position successfully parsed.
size_t search(const char *py, size_t py_len);
size_t search(const char* py, size_t py_len);
// Used to delete something in the Pinyin string kept by the engine, and do
// a re-search.
@@ -406,23 +406,23 @@ class MatrixSearch {
// Get the Pinyin string stored by the engine.
// *decoded_len returns the length of the successfully decoded string.
const char *get_pystr(size_t *decoded_len);
const char* get_pystr(size_t* decoded_len);
// Get the spelling boundaries for the first sentence candidate.
// Number of spellings will be returned. The number of valid elements in
// spl_start is one more than the return value because the last one is used
// to indicate the beginning of the next un-input speling.
// For a Pinyin "women", the returned value is 2, spl_start is [0, 2, 5] .
size_t get_spl_start(const uint16 *&spl_start);
size_t get_spl_start(const uint16*& spl_start);
// Get one candiate string. If full sentence candidate is available, it will
// be the first one.
char16 *get_candidate(size_t cand_id, char16 *cand_str, size_t max_len);
char16* get_candidate(size_t cand_id, char16* cand_str, size_t max_len);
// Get the first candiate, which is a "full sentence".
// retstr_len is not NULL, it will be used to return the string length.
// If only_unfixed is true, only unfixed part will be fetched.
char16 *get_candidate0(char16 *cand_str, size_t max_len, uint16 *retstr_len, bool only_unfixed);
char16* get_candidate0(char16* cand_str, size_t max_len, uint16* retstr_len, bool only_unfixed);
// Choose a candidate. The decoder will do a search after the fixed position.
size_t choose(size_t cand_id);
+2 -2
View File
@@ -21,9 +21,9 @@
namespace ime_pinyin {
void myqsort(void *p, size_t n, size_t es, int (*cmp)(const void *, const void *));
void myqsort(void* p, size_t n, size_t es, int (*cmp)(const void*, const void*));
void *mybsearch(const void *key, const void *base, size_t nmemb, size_t size, int (*compar)(const void *, const void *));
void* mybsearch(const void* key, const void* base, size_t nmemb, size_t size, int (*compar)(const void*, const void*));
} // namespace ime_pinyin
#endif // PINYINIME_INCLUDE_MYSTDLIB_H__
+8 -8
View File
@@ -45,7 +45,7 @@ class NGram {
static const size_t kSysDictTotalFreq = 100000000;
private:
static NGram *instance_;
static NGram* instance_;
bool initialized_;
size_t idx_num_;
@@ -58,19 +58,19 @@ class NGram {
float sys_score_compensation_;
#ifdef ___BUILD_MODEL___
double *freq_codes_df_;
double* freq_codes_df_;
#endif
LmaScoreType *freq_codes_;
CODEBOOK_TYPE *lma_freq_idx_;
LmaScoreType* freq_codes_;
CODEBOOK_TYPE* lma_freq_idx_;
public:
NGram();
~NGram();
static NGram &get_instance();
static NGram& get_instance();
bool save_ngram(FILE *fp);
bool load_ngram(FILE *fp);
bool save_ngram(FILE* fp);
bool load_ngram(FILE* fp);
// Set the total frequency of all none system dictionaries.
void set_total_freq_none_sys(size_t freq_none_sys);
@@ -86,7 +86,7 @@ class NGram {
#ifdef ___BUILD_MODEL___
// For constructing the unigram mode model.
bool build_unigram(LemmaEntry *lemma_arr, size_t num, LemmaIdType next_idx_unused);
bool build_unigram(LemmaEntry* lemma_arr, size_t num, LemmaIdType next_idx_unused);
#endif
};
} // namespace ime_pinyin
+7 -7
View File
@@ -33,7 +33,7 @@ namespace ime_pinyin {
* @param fn_usr_dict The file name of the user dictionary.
* @return true if open the decoder engine successfully.
*/
bool im_open_decoder(const char *fn_sys_dict, const char *fn_usr_dict);
bool im_open_decoder(const char* fn_sys_dict, const char* fn_usr_dict);
/**
* Open the decoder engine via the system dictionary FD and user dictionary
@@ -47,7 +47,7 @@ bool im_open_decoder(const char *fn_sys_dict, const char *fn_usr_dict);
* counted in byte.
* @return true if succeed.
*/
bool im_open_decoder_fd(int sys_fd, long start_offset, long length, const char *fn_usr_dict);
bool im_open_decoder_fd(int sys_fd, long start_offset, long length, const char* fn_usr_dict);
/**
* Close the decoder engine.
@@ -86,7 +86,7 @@ void im_flush_cache();
* @param sps_len The length of the spelling string buffer.
* @return The number of candidates.
*/
size_t im_search(const char *sps_buf, size_t sps_len);
size_t im_search(const char* sps_buf, size_t sps_len);
/**
* Make a delete operation in the current search result, and make research if
@@ -122,7 +122,7 @@ size_t im_add_letter(char ch);
* string is successfully parsed.
* @return The spelling string kept by the decoder.
*/
const char *im_get_sps_str(size_t *decoded_len);
const char* im_get_sps_str(size_t* decoded_len);
/**
* Get a candidate(or choice) string.
@@ -133,7 +133,7 @@ const char *im_get_sps_str(size_t *decoded_len);
* @param max_len The maximum length of the buffer.
* @return cand_str if succeeds, otherwise NULL.
*/
char16 *im_get_candidate(size_t cand_id, char16 *cand_str, size_t max_len);
char16* im_get_candidate(size_t cand_id, char16* cand_str, size_t max_len);
/**
* Get the segmentation information(the starting positions) of the spelling
@@ -144,7 +144,7 @@ char16 *im_get_candidate(size_t cand_id, char16 *cand_str, size_t max_len);
* elements in spl_start, and spl_start[L] is the posistion after the end of
* the last spelling id.
*/
size_t im_get_spl_start_pos(const uint16 *&spl_start);
size_t im_get_spl_start_pos(const uint16*& spl_start);
/**
* Choose a candidate and make it fixed. If the candidate does not match
@@ -187,7 +187,7 @@ bool im_cancel_input();
* @param pre_buf Used to return prediction result list.
* @return The number of predicted result string.
*/
size_t im_get_predicts(const char16 *his_buf, char16 (*&pre_buf)[kMaxPredictSize + 1]);
size_t im_get_predicts(const char16* his_buf, char16 (*&pre_buf)[kMaxPredictSize + 1]);
/**
* Enable Shengmus in ShouZiMu mode.
+17 -17
View File
@@ -111,27 +111,27 @@ bool is_system_lemma(LemmaIdType lma_id);
bool is_user_lemma(LemmaIdType lma_id);
bool is_composing_lemma(LemmaIdType lma_id);
int cmp_lpi_with_psb(const void *p1, const void *p2);
int cmp_lpi_with_unified_psb(const void *p1, const void *p2);
int cmp_lpi_with_id(const void *p1, const void *p2);
int cmp_lpi_with_hanzi(const void *p1, const void *p2);
int cmp_lpi_with_psb(const void* p1, const void* p2);
int cmp_lpi_with_unified_psb(const void* p1, const void* p2);
int cmp_lpi_with_id(const void* p1, const void* p2);
int cmp_lpi_with_hanzi(const void* p1, const void* p2);
int cmp_lpsi_with_str(const void *p1, const void *p2);
int cmp_lpsi_with_str(const void* p1, const void* p2);
int cmp_hanzis_1(const void *p1, const void *p2);
int cmp_hanzis_2(const void *p1, const void *p2);
int cmp_hanzis_3(const void *p1, const void *p2);
int cmp_hanzis_4(const void *p1, const void *p2);
int cmp_hanzis_5(const void *p1, const void *p2);
int cmp_hanzis_6(const void *p1, const void *p2);
int cmp_hanzis_7(const void *p1, const void *p2);
int cmp_hanzis_8(const void *p1, const void *p2);
int cmp_hanzis_1(const void* p1, const void* p2);
int cmp_hanzis_2(const void* p1, const void* p2);
int cmp_hanzis_3(const void* p1, const void* p2);
int cmp_hanzis_4(const void* p1, const void* p2);
int cmp_hanzis_5(const void* p1, const void* p2);
int cmp_hanzis_6(const void* p1, const void* p2);
int cmp_hanzis_7(const void* p1, const void* p2);
int cmp_hanzis_8(const void* p1, const void* p2);
int cmp_npre_by_score(const void *p1, const void *p2);
int cmp_npre_by_hislen_score(const void *p1, const void *p2);
int cmp_npre_by_hanzi_score(const void *p1, const void *p2);
int cmp_npre_by_score(const void* p1, const void* p2);
int cmp_npre_by_hislen_score(const void* p1, const void* p2);
int cmp_npre_by_hanzi_score(const void* p1, const void* p2);
size_t remove_duplicate_npre(NPredictItem *npre_items, size_t npre_num);
size_t remove_duplicate_npre(NPredictItem* npre_items, size_t npre_num);
size_t align_to_size_t(size_t size);
+8 -8
View File
@@ -24,7 +24,7 @@ namespace ime_pinyin {
class SpellingParser {
protected:
const SpellingTrie *spl_trie_;
const SpellingTrie* spl_trie_;
public:
SpellingParser();
@@ -39,27 +39,27 @@ class SpellingParser {
// If splstr starts with a character not in ['a'-z'] (it is a split char),
// return 0.
// Split char can only appear in the middle of the string or at the end.
uint16 splstr_to_idxs(const char *splstr, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre);
uint16 splstr_to_idxs(const char* splstr, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre);
// Similar to splstr_to_idxs(), the only difference is that splstr_to_idxs()
// convert single-character Yunmus into half ids, while this function converts
// them into full ids.
uint16 splstr_to_idxs_f(const char *splstr, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre);
uint16 splstr_to_idxs_f(const char* splstr, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre);
// Similar to splstr_to_idxs(), the only difference is that this function
// uses char16 instead of char8.
uint16 splstr16_to_idxs(const char16 *splstr, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre);
uint16 splstr16_to_idxs(const char16* splstr, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre);
// Similar to splstr_to_idxs_f(), the only difference is that this function
// uses char16 instead of char8.
uint16 splstr16_to_idxs_f(const char16 *splstr16, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre);
uint16 splstr16_to_idxs_f(const char16* splstr16, uint16 str_len, uint16 splidx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre);
// If the given string is a spelling, return the id, others, return 0.
// If the give string is a single char Yunmus like "A", and the char is
// enabled in ShouZiMu mode, the returned spelling id will be a half id.
// When the returned spelling id is a half id, *is_pre returns whether it
// is a prefix of a full spelling string.
uint16 get_splid_by_str(const char *splstr, uint16 str_len, bool *is_pre);
uint16 get_splid_by_str(const char* splstr, uint16 str_len, bool* is_pre);
// If the given string is a spelling, return the id, others, return 0.
// If the give string is a single char Yunmus like "a", no matter the char
@@ -67,7 +67,7 @@ class SpellingParser {
// a full id.
// When the returned spelling id is a half id, *p_is_pre returns whether it
// is a prefix of a full spelling string.
uint16 get_splid_by_str_f(const char *splstr, uint16 str_len, bool *is_pre);
uint16 get_splid_by_str_f(const char* splstr, uint16 str_len, bool* is_pre);
// Splitter chars are not included.
bool is_valid_to_parse(char ch);
@@ -82,7 +82,7 @@ class SpellingParser {
// return 0.
// Split char can only appear in the middle of the string or at the end.
// The caller should guarantee NULL != splstr && str_len > 0 && NULL != splidx
uint16 get_splids_parallel(const char *splstr, uint16 str_len, uint16 splidx[], uint16 max_size, uint16 &full_id_num, bool &is_pre);
uint16 get_splids_parallel(const char* splstr, uint16 str_len, uint16 splidx[], uint16 max_size, uint16& full_id_num, bool& is_pre);
};
} // namespace ime_pinyin
+44 -44
View File
@@ -26,7 +26,7 @@
#ifdef _WIN32
#include <time.h>
#include <winsock.h> // timeval
#include <winsock.h> // timeval
#else
#include <pthread.h>
#include <sys/time.h>
@@ -40,7 +40,7 @@ class UserDict : public AtomDictBase {
UserDict();
~UserDict();
bool load_dict(const char *file_name, LemmaIdType start_id, LemmaIdType end_id);
bool load_dict(const char* file_name, LemmaIdType start_id, LemmaIdType end_id);
bool close_dict();
@@ -48,15 +48,15 @@ class UserDict : public AtomDictBase {
void reset_milestones(uint16 from_step, MileStoneHandle from_handle);
MileStoneHandle extend_dict(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num);
MileStoneHandle extend_dict(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num);
size_t get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lpi_items, size_t lpi_max);
size_t get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lpi_items, size_t lpi_max);
uint16 get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max);
uint16 get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max);
uint16 get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid);
uint16 get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid);
size_t predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used);
size_t predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used);
// Full spelling ids are required
LemmaIdType put_lemma(char16 lemma_str[], uint16 splids[], uint16 lemma_len, uint16 count);
@@ -95,7 +95,7 @@ class UserDict : public AtomDictBase {
* @param len length of lemmas string in UTF-16LE
* @return newly added lemma count
*/
int put_lemmas_no_sync_from_utf16le_string(char16 *lemmas, int len);
int put_lemmas_no_sync_from_utf16le_string(char16* lemmas, int len);
/**
* Get lemmas need sync to a UTF-16LE string of above format.
@@ -107,13 +107,13 @@ class UserDict : public AtomDictBase {
* @param count output value of lemma returned
* @return UTF-16LE string length
*/
int get_sync_lemmas_in_utf16le_string_from_beginning(char16 *str, int size, int *count);
int get_sync_lemmas_in_utf16le_string_from_beginning(char16* str, int size, int* count);
#endif
struct UserDictStat {
uint32 version;
const char *file_name;
const char* file_name;
struct timeval load_time;
struct timeval last_update;
uint32 disk_size;
@@ -129,19 +129,19 @@ class UserDict : public AtomDictBase {
uint32 limit_lemma_size;
};
bool state(UserDictStat *stat);
bool state(UserDictStat* stat);
private:
uint32 total_other_nfreq_;
struct timeval load_time_;
LemmaIdType start_id_;
uint32 version_;
uint8 *lemmas_;
uint8* lemmas_;
// In-Memory-Only flag for each lemma
static const uint8 kUserDictLemmaFlagRemove = 1;
// Inuse lemmas' offset
uint32 *offsets_;
uint32* offsets_;
// Highest bit in offset tells whether corresponding lemma is removed
static const uint32 kUserDictOffsetFlagRemove = (1 << 31);
// Maximum possible for the offset
@@ -157,22 +157,22 @@ class UserDict : public AtomDictBase {
static const uint64 kUserDictLMTSince = COARSE_UTC(2009, 1, 1, 0, 0, 0);
// Correspond to offsets_
uint32 *scores_;
uint32* scores_;
// Following two fields are only valid in memory
uint32 *ids_;
uint32* ids_;
#ifdef ___PREDICT_ENABLED___
uint32 *predicts_;
uint32* predicts_;
#endif
#ifdef ___SYNC_ENABLED___
uint32 *syncs_;
uint32* syncs_;
size_t sync_count_size_;
#endif
uint32 *offsets_by_id_;
uint32* offsets_by_id_;
size_t lemma_count_left_;
size_t lemma_size_left_;
const char *dict_file_;
const char* dict_file_;
// Be sure size is 4xN
struct UserDictInfo {
@@ -250,19 +250,19 @@ class UserDict : public AtomDictBase {
void cache_init();
void cache_push(UserDictCacheType type, UserDictSearchable *searchable, uint32 offset, uint32 length);
void cache_push(UserDictCacheType type, UserDictSearchable* searchable, uint32 offset, uint32 length);
bool cache_hit(UserDictSearchable *searchable, uint32 *offset, uint32 *length);
bool cache_hit(UserDictSearchable* searchable, uint32* offset, uint32* length);
bool load_cache(UserDictSearchable *searchable, uint32 *offset, uint32 *length);
bool load_cache(UserDictSearchable* searchable, uint32* offset, uint32* length);
void save_cache(UserDictSearchable *searchable, uint32 offset, uint32 length);
void save_cache(UserDictSearchable* searchable, uint32 offset, uint32 length);
void reset_cache();
bool load_miss_cache(UserDictSearchable *searchable);
bool load_miss_cache(UserDictSearchable* searchable);
void save_miss_cache(UserDictSearchable *searchable);
void save_miss_cache(UserDictSearchable* searchable);
void reset_miss_cache();
#endif
@@ -275,29 +275,29 @@ class UserDict : public AtomDictBase {
inline int build_score(uint64 lmt, int freq);
inline int64 utf16le_atoll(uint16 *s, int len);
inline int64 utf16le_atoll(uint16* s, int len);
inline int utf16le_lltoa(int64 v, uint16 *s, int size);
inline int utf16le_lltoa(int64 v, uint16* s, int size);
LemmaIdType _put_lemma(char16 lemma_str[], uint16 splids[], uint16 lemma_len, uint16 count, uint64 lmt);
size_t _get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lpi_items, size_t lpi_max, bool *need_extend);
size_t _get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lpi_items, size_t lpi_max, bool* need_extend);
int _get_lemma_score(char16 lemma_str[], uint16 splids[], uint16 lemma_len);
int _get_lemma_score(LemmaIdType lemma_id);
int is_fuzzy_prefix_spell_id(const uint16 *id1, uint16 len1, const UserDictSearchable *searchable);
int is_fuzzy_prefix_spell_id(const uint16* id1, uint16 len1, const UserDictSearchable* searchable);
bool is_prefix_spell_id(const uint16 *fullids, uint16 fulllen, const UserDictSearchable *searchable);
bool is_prefix_spell_id(const uint16* fullids, uint16 fulllen, const UserDictSearchable* searchable);
uint32 get_dict_file_size(UserDictInfo *info);
uint32 get_dict_file_size(UserDictInfo* info);
bool reset(const char *file);
bool reset(const char* file);
bool validate(const char *file);
bool validate(const char* file);
bool load(const char *file, LemmaIdType start_id);
bool load(const char* file, LemmaIdType start_id);
bool is_valid_state();
@@ -311,22 +311,22 @@ class UserDict : public AtomDictBase {
char get_lemma_nchar(uint32 offset);
uint16 *get_lemma_spell_ids(uint32 offset);
uint16* get_lemma_spell_ids(uint32 offset);
uint16 *get_lemma_word(uint32 offset);
uint16* get_lemma_word(uint32 offset);
// Prepare searchable to fasten locate process
void prepare_locate(UserDictSearchable *searchable, const uint16 *splids, uint16 len);
void prepare_locate(UserDictSearchable* searchable, const uint16* splids, uint16 len);
// Compare initial letters only
int32 fuzzy_compare_spell_id(const uint16 *id1, uint16 len1, const UserDictSearchable *searchable);
int32 fuzzy_compare_spell_id(const uint16* id1, uint16 len1, const UserDictSearchable* searchable);
// Compare exactly two spell ids
// First argument must be a full id spell id
bool equal_spell_id(const uint16 *fullids, uint16 fulllen, const UserDictSearchable *searchable);
bool equal_spell_id(const uint16* fullids, uint16 fulllen, const UserDictSearchable* searchable);
// Find first item by initial letters
int32 locate_first_in_offsets(const UserDictSearchable *searchable);
int32 locate_first_in_offsets(const UserDictSearchable* searchable);
LemmaIdType append_a_lemma(char16 lemma_str[], uint16 splids[], uint16 lemma_len, uint16 count, uint64 lmt);
@@ -335,9 +335,9 @@ class UserDict : public AtomDictBase {
bool remove_lemma_by_offset_index(int offset_index);
#ifdef ___PREDICT_ENABLED___
uint32 locate_where_to_insert_in_predicts(const uint16 *words, int lemma_len);
uint32 locate_where_to_insert_in_predicts(const uint16* words, int lemma_len);
int32 locate_first_in_predicts(const uint16 *words, int lemma_len);
int32 locate_first_in_predicts(const uint16* words, int lemma_len);
void remove_lemma_from_predict_list(uint32 offset);
#endif
@@ -359,9 +359,9 @@ class UserDict : public AtomDictBase {
uint32 offset_index;
};
inline void swap(UserDictScoreOffsetPair *sop, int i, int j);
inline void swap(UserDictScoreOffsetPair* sop, int i, int j);
void shift_down(UserDictScoreOffsetPair *sop, int i, int n);
void shift_down(UserDictScoreOffsetPair* sop, int i, int n);
// On-disk format for each lemma
// +-------------+
+9 -9
View File
@@ -30,21 +30,21 @@ typedef unsigned short char16;
// Get a token from utf16_str,
// Returned pointer is a '\0'-terminated utf16 string, or NULL
// *utf16_str_next returns the next part of the string for further tokenizing
char16 *utf16_strtok(char16 *utf16_str, size_t *token_size, char16 **utf16_str_next);
char16* utf16_strtok(char16* utf16_str, size_t* token_size, char16** utf16_str_next);
int utf16_atoi(const char16 *utf16_str);
int utf16_atoi(const char16* utf16_str);
float utf16_atof(const char16 *utf16_str);
float utf16_atof(const char16* utf16_str);
size_t utf16_strlen(const char16 *utf16_str);
size_t utf16_strlen(const char16* utf16_str);
int utf16_strcmp(const char16 *str1, const char16 *str2);
int utf16_strncmp(const char16 *str1, const char16 *str2, size_t size);
int utf16_strcmp(const char16* str1, const char16* str2);
int utf16_strncmp(const char16* str1, const char16* str2, size_t size);
char16 *utf16_strcpy(char16 *dst, const char16 *src);
char16 *utf16_strncpy(char16 *dst, const char16 *src, size_t size);
char16* utf16_strcpy(char16* dst, const char16* src);
char16* utf16_strncpy(char16* dst, const char16* src, size_t size);
char *utf16_strcpy_tochar(char *dst, const char16 *src);
char* utf16_strcpy_tochar(char* dst, const char16* src);
#ifdef __cplusplus
}
+70 -70
View File
@@ -38,10 +38,10 @@ static const size_t kSplTableHashLen = 2000;
// Compare a SingleCharItem, first by Hanzis, then by spelling ids, then by
// frequencies.
int cmp_scis_hz_splid_freq(const void *p1, const void *p2) {
int cmp_scis_hz_splid_freq(const void* p1, const void* p2) {
const SingleCharItem *s1, *s2;
s1 = static_cast<const SingleCharItem *>(p1);
s2 = static_cast<const SingleCharItem *>(p2);
s1 = static_cast<const SingleCharItem*>(p1);
s2 = static_cast<const SingleCharItem*>(p2);
if (s1->hz < s2->hz) return -1;
if (s1->hz > s2->hz) return 1;
@@ -57,10 +57,10 @@ int cmp_scis_hz_splid_freq(const void *p1, const void *p2) {
return 0;
}
int cmp_scis_hz_splid(const void *p1, const void *p2) {
int cmp_scis_hz_splid(const void* p1, const void* p2) {
const SingleCharItem *s1, *s2;
s1 = static_cast<const SingleCharItem *>(p1);
s2 = static_cast<const SingleCharItem *>(p2);
s1 = static_cast<const SingleCharItem*>(p1);
s2 = static_cast<const SingleCharItem*>(p2);
if (s1->hz < s2->hz) return -1;
if (s1->hz > s2->hz) return 1;
@@ -74,49 +74,49 @@ int cmp_scis_hz_splid(const void *p1, const void *p2) {
return 0;
}
int cmp_lemma_entry_hzs(const void *p1, const void *p2) {
size_t size1 = utf16_strlen(((const LemmaEntry *)p1)->hanzi_str);
size_t size2 = utf16_strlen(((const LemmaEntry *)p2)->hanzi_str);
int cmp_lemma_entry_hzs(const void* p1, const void* p2) {
size_t size1 = utf16_strlen(((const LemmaEntry*)p1)->hanzi_str);
size_t size2 = utf16_strlen(((const LemmaEntry*)p2)->hanzi_str);
if (size1 < size2)
return -1;
else if (size1 > size2)
return 1;
return utf16_strcmp(((const LemmaEntry *)p1)->hanzi_str, ((const LemmaEntry *)p2)->hanzi_str);
return utf16_strcmp(((const LemmaEntry*)p1)->hanzi_str, ((const LemmaEntry*)p2)->hanzi_str);
}
int compare_char16(const void *p1, const void *p2) {
if (*((const char16 *)p1) < *((const char16 *)p2)) return -1;
if (*((const char16 *)p1) > *((const char16 *)p2)) return 1;
int compare_char16(const void* p1, const void* p2) {
if (*((const char16*)p1) < *((const char16*)p2)) return -1;
if (*((const char16*)p1) > *((const char16*)p2)) return 1;
return 0;
}
int compare_py(const void *p1, const void *p2) {
int ret = utf16_strcmp(((const LemmaEntry *)p1)->spl_idx_arr, ((const LemmaEntry *)p2)->spl_idx_arr);
int compare_py(const void* p1, const void* p2) {
int ret = utf16_strcmp(((const LemmaEntry*)p1)->spl_idx_arr, ((const LemmaEntry*)p2)->spl_idx_arr);
if (0 != ret) return ret;
return static_cast<int>(((const LemmaEntry *)p2)->freq) - static_cast<int>(((const LemmaEntry *)p1)->freq);
return static_cast<int>(((const LemmaEntry*)p2)->freq) - static_cast<int>(((const LemmaEntry*)p1)->freq);
}
// First hanzi, if the same, then Pinyin
int cmp_lemma_entry_hzspys(const void *p1, const void *p2) {
size_t size1 = utf16_strlen(((const LemmaEntry *)p1)->hanzi_str);
size_t size2 = utf16_strlen(((const LemmaEntry *)p2)->hanzi_str);
int cmp_lemma_entry_hzspys(const void* p1, const void* p2) {
size_t size1 = utf16_strlen(((const LemmaEntry*)p1)->hanzi_str);
size_t size2 = utf16_strlen(((const LemmaEntry*)p2)->hanzi_str);
if (size1 < size2)
return -1;
else if (size1 > size2)
return 1;
int ret = utf16_strcmp(((const LemmaEntry *)p1)->hanzi_str, ((const LemmaEntry *)p2)->hanzi_str);
int ret = utf16_strcmp(((const LemmaEntry*)p1)->hanzi_str, ((const LemmaEntry*)p2)->hanzi_str);
if (0 != ret) return ret;
ret = utf16_strcmp(((const LemmaEntry *)p1)->spl_idx_arr, ((const LemmaEntry *)p2)->spl_idx_arr);
ret = utf16_strcmp(((const LemmaEntry*)p1)->spl_idx_arr, ((const LemmaEntry*)p2)->spl_idx_arr);
return ret;
}
int compare_splid2(const void *p1, const void *p2) {
int ret = utf16_strcmp(((const LemmaEntry *)p1)->spl_idx_arr, ((const LemmaEntry *)p2)->spl_idx_arr);
int compare_splid2(const void* p1, const void* p2) {
int ret = utf16_strcmp(((const LemmaEntry*)p1)->spl_idx_arr, ((const LemmaEntry*)p2)->spl_idx_arr);
return ret;
}
@@ -188,11 +188,11 @@ bool DictBuilder::alloc_resource(size_t lma_num) {
return true;
}
char16 *DictBuilder::read_valid_hanzis(const char *fn_validhzs, size_t *num) {
char16* DictBuilder::read_valid_hanzis(const char* fn_validhzs, size_t* num) {
if (NULL == fn_validhzs || NULL == num) return NULL;
*num = 0;
FILE *fp = fopen(fn_validhzs, "rb");
FILE* fp = fopen(fn_validhzs, "rb");
if (NULL == fp) return NULL;
char16 utf16header;
@@ -206,7 +206,7 @@ char16 *DictBuilder::read_valid_hanzis(const char *fn_validhzs, size_t *num) {
assert(*num >= 1);
*num -= 1;
char16 *hzs = new char16[*num];
char16* hzs = new char16[*num];
if (NULL == hzs) {
fclose(fp);
return NULL;
@@ -225,11 +225,11 @@ char16 *DictBuilder::read_valid_hanzis(const char *fn_validhzs, size_t *num) {
return hzs;
}
bool DictBuilder::hz_in_hanzis_list(const char16 *hzs, size_t hzs_len, char16 hz) {
bool DictBuilder::hz_in_hanzis_list(const char16* hzs, size_t hzs_len, char16 hz) {
if (NULL == hzs) return false;
char16 *found;
found = static_cast<char16 *>(mybsearch(&hz, hzs, hzs_len, sizeof(char16), compare_char16));
char16* found;
found = static_cast<char16*>(mybsearch(&hz, hzs, hzs_len, sizeof(char16), compare_char16));
if (NULL == found) return false;
assert(*found == hz);
@@ -237,7 +237,7 @@ bool DictBuilder::hz_in_hanzis_list(const char16 *hzs, size_t hzs_len, char16 hz
}
// The caller makes sure that the parameters are valid.
bool DictBuilder::str_in_hanzis_list(const char16 *hzs, size_t hzs_len, const char16 *str, size_t str_len) {
bool DictBuilder::str_in_hanzis_list(const char16* hzs, size_t hzs_len, const char16* str, size_t str_len) {
if (NULL == hzs || NULL == str) return false;
for (size_t pos = 0; pos < str_len; pos++) {
@@ -313,7 +313,7 @@ void DictBuilder::free_resource() {
homo_idx_num_gt1_ = 0;
}
size_t DictBuilder::read_raw_dict(const char *fn_raw, const char *fn_validhzs, size_t max_item) {
size_t DictBuilder::read_raw_dict(const char* fn_raw, const char* fn_validhzs, size_t max_item) {
if (NULL == fn_raw) return 0;
Utf16Reader utf16_reader;
@@ -330,7 +330,7 @@ size_t DictBuilder::read_raw_dict(const char *fn_raw, const char *fn_validhzs, s
}
// Read the valid Hanzi list.
char16 *valid_hzs = NULL;
char16* valid_hzs = NULL;
size_t valid_hzs_num = 0;
valid_hzs = read_valid_hanzis(fn_validhzs, &valid_hzs_num);
@@ -343,8 +343,8 @@ size_t DictBuilder::read_raw_dict(const char *fn_raw, const char *fn_validhzs, s
}
size_t token_size;
char16 *token;
char16 *to_tokenize = read_buf;
char16* token;
char16* to_tokenize = read_buf;
// Get the Hanzi string
token = utf16_strtok(to_tokenize, &token_size, &to_tokenize);
@@ -443,7 +443,7 @@ size_t DictBuilder::read_raw_dict(const char *fn_raw, const char *fn_validhzs, s
return lemma_num;
}
bool DictBuilder::build_dict(const char *fn_raw, const char *fn_validhzs, DictTrie *dict_trie) {
bool DictBuilder::build_dict(const char* fn_raw, const char* fn_validhzs, DictTrie* dict_trie) {
if (NULL == fn_raw || NULL == dict_trie) return false;
lemma_num_ = read_raw_dict(fn_raw, fn_validhzs, 240000);
@@ -455,14 +455,14 @@ bool DictBuilder::build_dict(const char *fn_raw, const char *fn_validhzs, DictTr
// spelling string will be score, and it is also included in spl_item_size.
size_t spl_item_size;
size_t spl_num;
const char *spl_buf;
const char* spl_buf;
spl_buf = spl_table_->arrange(&spl_item_size, &spl_num);
if (NULL == spl_buf) {
free_resource();
return false;
}
SpellingTrie &spl_trie = SpellingTrie::get_instance();
SpellingTrie& spl_trie = SpellingTrie::get_instance();
if (!spl_trie.construct(spl_buf, spl_item_size, spl_num, spl_table_->get_score_amplifier(), spl_table_->get_average_score())) {
free_resource();
@@ -500,7 +500,7 @@ bool DictBuilder::build_dict(const char *fn_raw, const char *fn_validhzs, DictTr
assert(dl_success);
// Construct the NGram information
NGram &ngram = NGram::get_instance();
NGram& ngram = NGram::get_instance();
ngram.build_unigram(lemma_arr_, lemma_num_, lemma_arr_[lemma_num_ - 1].idx_by_hz + 1);
// sort the lemma items according to the spelling idx string
@@ -513,7 +513,7 @@ bool DictBuilder::build_dict(const char *fn_raw, const char *fn_validhzs, DictTr
#endif
lma_nds_used_num_le0_ = 1; // The root node
bool dt_success = construct_subset(static_cast<void *>(lma_nodes_le0_), lemma_arr_, 0, lemma_num_, 0);
bool dt_success = construct_subset(static_cast<void*>(lma_nodes_le0_), lemma_arr_, 0, lemma_num_, 0);
if (!dt_success) {
free_resource();
return false;
@@ -561,19 +561,19 @@ bool DictBuilder::build_dict(const char *fn_raw, const char *fn_validhzs, DictTr
return dt_success;
}
void DictBuilder::id_to_charbuf(unsigned char *buf, LemmaIdType id) {
void DictBuilder::id_to_charbuf(unsigned char* buf, LemmaIdType id) {
if (NULL == buf) return;
for (size_t pos = 0; pos < kLemmaIdSize; pos++) {
(buf)[pos] = (unsigned char)(id >> (pos * 8));
}
}
void DictBuilder::set_son_offset(LmaNodeGE1 *node, size_t offset) {
void DictBuilder::set_son_offset(LmaNodeGE1* node, size_t offset) {
node->son_1st_off_l = static_cast<uint16>(offset);
node->son_1st_off_h = static_cast<unsigned char>(offset >> 16);
}
void DictBuilder::set_homo_id_buf_offset(LmaNodeGE1 *node, size_t offset) {
void DictBuilder::set_homo_id_buf_offset(LmaNodeGE1* node, size_t offset) {
node->homo_idx_buf_off_l = static_cast<uint16>(offset);
node->homo_idx_buf_off_h = static_cast<unsigned char>(offset >> 16);
}
@@ -581,7 +581,7 @@ void DictBuilder::set_homo_id_buf_offset(LmaNodeGE1 *node, size_t offset) {
// All spelling strings will be converted to upper case, except that
// spellings started with "ZH"/"CH"/"SH" will be converted to
// "Zh"/"Ch"/"Sh"
void DictBuilder::format_spelling_str(char *spl_str) {
void DictBuilder::format_spelling_str(char* spl_str) {
if (NULL == spl_str) return;
uint16 pos = 0;
@@ -619,7 +619,7 @@ LemmaIdType DictBuilder::sort_lemmas_by_hz() {
size_t DictBuilder::build_scis() {
if (NULL == scis_ || lemma_num_ * kMaxLemmaSize > scis_num_) return 0;
SpellingTrie &spl_trie = SpellingTrie::get_instance();
SpellingTrie& spl_trie = SpellingTrie::get_instance();
// This first one is blank, because id 0 is invalid.
scis_[0].freq = 0;
@@ -665,8 +665,8 @@ size_t DictBuilder::build_scis() {
key.splid.full_splid = lemma_arr_[pos].spl_idx_arr[hzpos];
key.splid.half_splid = spl_trie.full_to_half(key.splid.full_splid);
SingleCharItem *found;
found = static_cast<SingleCharItem *>(mybsearch(&key, scis_, unique_scis_num, sizeof(SingleCharItem), cmp_scis_hz_splid));
SingleCharItem* found;
found = static_cast<SingleCharItem*>(mybsearch(&key, scis_, unique_scis_num, sizeof(SingleCharItem), cmp_scis_hz_splid));
assert(found);
@@ -678,7 +678,7 @@ size_t DictBuilder::build_scis() {
return scis_num_;
}
bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t item_start, size_t item_end, size_t level) {
bool DictBuilder::construct_subset(void* parent, LemmaEntry* lemma_arr, size_t item_start, size_t item_end, size_t level) {
if (level >= kMaxLemmaSize || item_end <= item_start) return false;
// 1. Scan for how many sons
@@ -686,12 +686,12 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
// LemmaNode *son_1st = NULL;
// parent.num_of_son = 0;
LemmaEntry *lma_last_start = lemma_arr_ + item_start;
LemmaEntry* lma_last_start = lemma_arr_ + item_start;
uint16 spl_idx_node = lma_last_start->spl_idx_arr[level];
// Scan for how many sons to be allocaed
for (size_t i = item_start + 1; i < item_end; i++) {
LemmaEntry *lma_current = lemma_arr + i;
LemmaEntry* lma_current = lemma_arr + i;
uint16 spl_idx_current = lma_current->spl_idx_arr[level];
if (spl_idx_current != spl_idx_node) {
parent_son_num++;
@@ -719,29 +719,29 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
// 2. Update the parent's information
// Update the parent's son list;
LmaNodeLE0 *son_1st_le0 = NULL; // only one of le0 or ge1 is used
LmaNodeGE1 *son_1st_ge1 = NULL; // only one of le0 or ge1 is used.
LmaNodeLE0* son_1st_le0 = NULL; // only one of le0 or ge1 is used
LmaNodeGE1* son_1st_ge1 = NULL; // only one of le0 or ge1 is used.
if (0 == level) { // the parent is root
(static_cast<LmaNodeLE0 *>(parent))->son_1st_off = lma_nds_used_num_le0_;
(static_cast<LmaNodeLE0*>(parent))->son_1st_off = lma_nds_used_num_le0_;
son_1st_le0 = lma_nodes_le0_ + lma_nds_used_num_le0_;
lma_nds_used_num_le0_ += parent_son_num;
assert(parent_son_num <= 65535);
(static_cast<LmaNodeLE0 *>(parent))->num_of_son = static_cast<uint16>(parent_son_num);
(static_cast<LmaNodeLE0*>(parent))->num_of_son = static_cast<uint16>(parent_son_num);
} else if (1 == level) { // the parent is a son of root
(static_cast<LmaNodeLE0 *>(parent))->son_1st_off = lma_nds_used_num_ge1_;
(static_cast<LmaNodeLE0*>(parent))->son_1st_off = lma_nds_used_num_ge1_;
son_1st_ge1 = lma_nodes_ge1_ + lma_nds_used_num_ge1_;
lma_nds_used_num_ge1_ += parent_son_num;
assert(parent_son_num <= 65535);
(static_cast<LmaNodeLE0 *>(parent))->num_of_son = static_cast<uint16>(parent_son_num);
(static_cast<LmaNodeLE0*>(parent))->num_of_son = static_cast<uint16>(parent_son_num);
} else {
set_son_offset((static_cast<LmaNodeGE1 *>(parent)), lma_nds_used_num_ge1_);
set_son_offset((static_cast<LmaNodeGE1*>(parent)), lma_nds_used_num_ge1_);
son_1st_ge1 = lma_nodes_ge1_ + lma_nds_used_num_ge1_;
lma_nds_used_num_ge1_ += parent_son_num;
assert(parent_son_num <= 255);
(static_cast<LmaNodeGE1 *>(parent))->num_of_son = (unsigned char)parent_son_num;
(static_cast<LmaNodeGE1*>(parent))->num_of_son = (unsigned char)parent_son_num;
}
// 3. Now begin to construct the son one by one
@@ -756,15 +756,15 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
size_t item_start_next = item_start;
for (size_t i = item_start + 1; i < item_end; i++) {
LemmaEntry *lma_current = lemma_arr_ + i;
LemmaEntry* lma_current = lemma_arr_ + i;
uint16 spl_idx_current = lma_current->spl_idx_arr[level];
if (spl_idx_current == spl_idx_node) {
if (lma_current->spl_idx_arr[level + 1] == 0) homo_num++;
} else {
// Construct a node
LmaNodeLE0 *node_cur_le0 = NULL; // only one of them is valid
LmaNodeGE1 *node_cur_ge1 = NULL;
LmaNodeLE0* node_cur_le0 = NULL; // only one of them is valid
LmaNodeGE1* node_cur_ge1 = NULL;
if (0 == level) {
node_cur_le0 = son_1st_le0 + son_pos;
node_cur_le0->spl_idx = spl_idx_node;
@@ -781,7 +781,7 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
}
if (homo_num > 0) {
LemmaIdType *idx_buf = homo_idx_buf_ + homo_idx_num_eq1_ + homo_idx_num_gt1_ - homo_num;
LemmaIdType* idx_buf = homo_idx_buf_ + homo_idx_num_eq1_ + homo_idx_num_gt1_ - homo_num;
if (0 == level) {
assert(homo_num <= 65535);
node_cur_le0->num_of_homo = static_cast<uint16>(homo_num);
@@ -802,11 +802,11 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
}
if (i - item_start_next > homo_num) {
void *next_parent;
void* next_parent;
if (0 == level)
next_parent = static_cast<void *>(node_cur_le0);
next_parent = static_cast<void*>(node_cur_le0);
else
next_parent = static_cast<void *>(node_cur_ge1);
next_parent = static_cast<void*>(node_cur_ge1);
construct_subset(next_parent, lemma_arr, item_start_next + homo_num, i, level + 1);
#ifdef ___DO_STATISTICS___
@@ -827,8 +827,8 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
}
// 4. The last one to construct
LmaNodeLE0 *node_cur_le0 = NULL; // only one of them is valid
LmaNodeGE1 *node_cur_ge1 = NULL;
LmaNodeLE0* node_cur_le0 = NULL; // only one of them is valid
LmaNodeGE1* node_cur_ge1 = NULL;
if (0 == level) {
node_cur_le0 = son_1st_le0 + son_pos;
node_cur_le0->spl_idx = spl_idx_node;
@@ -845,7 +845,7 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
}
if (homo_num > 0) {
LemmaIdType *idx_buf = homo_idx_buf_ + homo_idx_num_eq1_ + homo_idx_num_gt1_ - homo_num;
LemmaIdType* idx_buf = homo_idx_buf_ + homo_idx_num_eq1_ + homo_idx_num_gt1_ - homo_num;
if (0 == level) {
assert(homo_num <= 65535);
node_cur_le0->num_of_homo = static_cast<uint16>(homo_num);
@@ -866,11 +866,11 @@ bool DictBuilder::construct_subset(void *parent, LemmaEntry *lemma_arr, size_t i
}
if (item_end - item_start_next > homo_num) {
void *next_parent;
void* next_parent;
if (0 == level)
next_parent = static_cast<void *>(node_cur_le0);
next_parent = static_cast<void*>(node_cur_le0);
else
next_parent = static_cast<void *>(node_cur_ge1);
next_parent = static_cast<void*>(node_cur_ge1);
construct_subset(next_parent, lemma_arr, item_start_next + homo_num, item_end, level + 1);
#ifdef ___DO_STATISTICS___
+26 -26
View File
@@ -47,15 +47,15 @@ DictList::~DictList() { free_resource(); }
bool DictList::alloc_resource(size_t buf_size, size_t scis_num) {
// Allocate memory
buf_ = static_cast<char16 *>(malloc(buf_size * sizeof(char16)));
buf_ = static_cast<char16*>(malloc(buf_size * sizeof(char16)));
if (NULL == buf_) return false;
scis_num_ = scis_num;
scis_hz_ = static_cast<char16 *>(malloc(scis_num_ * sizeof(char16)));
scis_hz_ = static_cast<char16*>(malloc(scis_num_ * sizeof(char16)));
if (NULL == scis_hz_) return false;
scis_splid_ = static_cast<SpellingId *>(malloc(scis_num_ * sizeof(SpellingId)));
scis_splid_ = static_cast<SpellingId*>(malloc(scis_num_ * sizeof(SpellingId)));
if (NULL == scis_splid_) return false;
@@ -74,7 +74,7 @@ void DictList::free_resource() {
}
#ifdef ___BUILD_MODEL___
bool DictList::init_list(const SingleCharItem *scis, size_t scis_num, const LemmaEntry *lemma_arr, size_t lemma_num) {
bool DictList::init_list(const SingleCharItem* scis, size_t scis_num, const LemmaEntry* lemma_arr, size_t lemma_num) {
if (NULL == scis || 0 == scis_num || NULL == lemma_arr || 0 == lemma_num) return false;
initialized_ = false;
@@ -96,7 +96,7 @@ bool DictList::init_list(const SingleCharItem *scis, size_t scis_num, const Lemm
return true;
}
size_t DictList::calculate_size(const LemmaEntry *lemma_arr, size_t lemma_num) {
size_t DictList::calculate_size(const LemmaEntry* lemma_arr, size_t lemma_num) {
size_t last_hz_len = 0;
size_t list_size = 0;
size_t id_num = 0;
@@ -152,7 +152,7 @@ size_t DictList::calculate_size(const LemmaEntry *lemma_arr, size_t lemma_num) {
return start_pos_[kMaxLemmaSize];
}
void DictList::fill_scis(const SingleCharItem *scis, size_t scis_num) {
void DictList::fill_scis(const SingleCharItem* scis, size_t scis_num) {
assert(scis_num_ == scis_num);
for (size_t pos = 0; pos < scis_num_; pos++) {
@@ -161,7 +161,7 @@ void DictList::fill_scis(const SingleCharItem *scis, size_t scis_num) {
}
}
void DictList::fill_list(const LemmaEntry *lemma_arr, size_t lemma_num) {
void DictList::fill_list(const LemmaEntry* lemma_arr, size_t lemma_num) {
size_t current_pos = 0;
utf16_strncpy(buf_, lemma_arr[0].hanzi_str, lemma_arr[0].hz_str_len);
@@ -181,8 +181,8 @@ void DictList::fill_list(const LemmaEntry *lemma_arr, size_t lemma_num) {
assert(id_num == start_id_[kMaxLemmaSize]);
}
char16 *DictList::find_pos2_startedbyhz(char16 hz_char) {
char16 *found_2w = static_cast<char16 *>(mybsearch(&hz_char, buf_ + start_pos_[1], (start_pos_[2] - start_pos_[1]) / 2, sizeof(char16) * 2, cmp_hanzis_1));
char16* DictList::find_pos2_startedbyhz(char16 hz_char) {
char16* found_2w = static_cast<char16*>(mybsearch(&hz_char, buf_ + start_pos_[1], (start_pos_[2] - start_pos_[1]) / 2, sizeof(char16) * 2, cmp_hanzis_1));
if (NULL == found_2w) return NULL;
while (found_2w > buf_ + start_pos_[1] && *found_2w == *(found_2w - 1)) found_2w -= 2;
@@ -191,8 +191,8 @@ char16 *DictList::find_pos2_startedbyhz(char16 hz_char) {
}
#endif // ___BUILD_MODEL___
char16 *DictList::find_pos_startedbyhzs(const char16 last_hzs[], size_t word_len, int (*cmp_func)(const void *, const void *)) {
char16 *found_w = static_cast<char16 *>(mybsearch(last_hzs, buf_ + start_pos_[word_len - 1], (start_pos_[word_len] - start_pos_[word_len - 1]) / word_len, sizeof(char16) * word_len, cmp_func));
char16* DictList::find_pos_startedbyhzs(const char16 last_hzs[], size_t word_len, int (*cmp_func)(const void*, const void*)) {
char16* found_w = static_cast<char16*>(mybsearch(last_hzs, buf_ + start_pos_[word_len - 1], (start_pos_[word_len] - start_pos_[word_len - 1]) / word_len, sizeof(char16) * word_len, cmp_func));
if (NULL == found_w) return NULL;
@@ -201,20 +201,20 @@ char16 *DictList::find_pos_startedbyhzs(const char16 last_hzs[], size_t word_len
return found_w;
}
size_t DictList::predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used) {
size_t DictList::predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used) {
assert(hzs_len <= kMaxPredictSize && hzs_len > 0);
// 1. Prepare work
int (*cmp_func)(const void *, const void *) = cmp_func_[hzs_len - 1];
int (*cmp_func)(const void*, const void*) = cmp_func_[hzs_len - 1];
NGram &ngram = NGram::get_instance();
NGram& ngram = NGram::get_instance();
size_t item_num = 0;
// 2. Do prediction
for (uint16 pre_len = 1; pre_len <= kMaxPredictSize + 1 - hzs_len; pre_len++) {
uint16 word_len = hzs_len + pre_len;
char16 *w_buf = find_pos_startedbyhzs(last_hzs, word_len, cmp_func);
char16* w_buf = find_pos_startedbyhzs(last_hzs, word_len, cmp_func);
if (NULL == w_buf) continue;
while (w_buf < buf_ + start_pos_[word_len] && cmp_func(w_buf, last_hzs) == 0 && item_num < npre_max) {
memset(npre_items + item_num, 0, sizeof(NPredictItem));
@@ -243,7 +243,7 @@ size_t DictList::predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *
return new_num;
}
uint16 DictList::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max) {
uint16 DictList::get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max) {
if (!initialized_ || id_lemma >= start_id_[kMaxLemmaSize] || NULL == str_buf || str_max <= 1) return 0;
// Find the range
@@ -252,7 +252,7 @@ uint16 DictList::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str
if (start_id_[i] <= id_lemma && start_id_[i + 1] > id_lemma) {
size_t id_span = id_lemma - start_id_[i];
uint16 *buf = buf_ + start_pos_[i] + id_span * (i + 1);
uint16* buf = buf_ + start_pos_[i] + id_span * (i + 1);
for (uint16 len = 0; len <= i; len++) {
str_buf[len] = buf[len];
}
@@ -263,15 +263,15 @@ uint16 DictList::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str
return 0;
}
uint16 DictList::get_splids_for_hanzi(char16 hanzi, uint16 half_splid, uint16 *splids, uint16 max_splids) {
char16 *hz_found = static_cast<char16 *>(mybsearch(&hanzi, scis_hz_, scis_num_, sizeof(char16), cmp_hanzis_1));
uint16 DictList::get_splids_for_hanzi(char16 hanzi, uint16 half_splid, uint16* splids, uint16 max_splids) {
char16* hz_found = static_cast<char16*>(mybsearch(&hanzi, scis_hz_, scis_num_, sizeof(char16), cmp_hanzis_1));
assert(NULL != hz_found && hanzi == *hz_found);
// Move to the first one.
while (hz_found > scis_hz_ && hanzi == *(hz_found - 1)) hz_found--;
// First try to found if strict comparison result is not zero.
char16 *hz_f = hz_found;
char16* hz_f = hz_found;
bool strict = false;
while (hz_f < scis_hz_ + scis_num_ && hanzi == *hz_f) {
uint16 pos = hz_f - scis_hz_;
@@ -295,10 +295,10 @@ uint16 DictList::get_splids_for_hanzi(char16 hanzi, uint16 half_splid, uint16 *s
return found_num;
}
LemmaIdType DictList::get_lemma_id(const char16 *str, uint16 str_len) {
LemmaIdType DictList::get_lemma_id(const char16* str, uint16 str_len) {
if (NULL == str || str_len > kMaxLemmaSize) return 0;
char16 *found = find_pos_startedbyhzs(str, str_len, cmp_func_[str_len - 1]);
char16* found = find_pos_startedbyhzs(str, str_len, cmp_func_[str_len - 1]);
if (NULL == found) return 0;
assert(found > buf_);
@@ -306,7 +306,7 @@ LemmaIdType DictList::get_lemma_id(const char16 *str, uint16 str_len) {
return static_cast<LemmaIdType>(start_id_[str_len - 1] + (found - buf_ - start_pos_[str_len - 1]) / str_len);
}
void DictList::convert_to_hanzis(char16 *str, uint16 str_len) {
void DictList::convert_to_hanzis(char16* str, uint16 str_len) {
assert(NULL != str);
for (uint16 str_pos = 0; str_pos < str_len; str_pos++) {
@@ -314,7 +314,7 @@ void DictList::convert_to_hanzis(char16 *str, uint16 str_len) {
}
}
void DictList::convert_to_scis_ids(char16 *str, uint16 str_len) {
void DictList::convert_to_scis_ids(char16* str, uint16 str_len) {
assert(NULL != str);
for (uint16 str_pos = 0; str_pos < str_len; str_pos++) {
@@ -322,7 +322,7 @@ void DictList::convert_to_scis_ids(char16 *str, uint16 str_len) {
}
}
bool DictList::save_list(FILE *fp) {
bool DictList::save_list(FILE* fp) {
if (!initialized_ || NULL == fp) return false;
if (NULL == buf_ || 0 == start_pos_[kMaxLemmaSize] || NULL == scis_hz_ || NULL == scis_splid_ || 0 == scis_num_) return false;
@@ -342,7 +342,7 @@ bool DictList::save_list(FILE *fp) {
return true;
}
bool DictList::load_list(FILE *fp) {
bool DictList::load_list(FILE* fp) {
if (NULL == fp) return false;
initialized_ = false;
+77 -77
View File
@@ -75,9 +75,9 @@ void DictTrie::free_resource(bool free_dict_list) {
reset_milestones(0, kFirstValidMileStoneHandle);
}
inline size_t DictTrie::get_son_offset(const LmaNodeGE1 *node) { return ((size_t)node->son_1st_off_l + ((size_t)node->son_1st_off_h << 16)); }
inline size_t DictTrie::get_son_offset(const LmaNodeGE1* node) { return ((size_t)node->son_1st_off_l + ((size_t)node->son_1st_off_h << 16)); }
inline size_t DictTrie::get_homo_idx_buf_offset(const LmaNodeGE1 *node) { return ((size_t)node->homo_idx_buf_off_l + ((size_t)node->homo_idx_buf_off_h << 16)); }
inline size_t DictTrie::get_homo_idx_buf_offset(const LmaNodeGE1* node) { return ((size_t)node->homo_idx_buf_off_l + ((size_t)node->homo_idx_buf_off_h << 16)); }
inline LemmaIdType DictTrie::get_lemma_id(size_t id_offset) {
LemmaIdType id = 0;
@@ -87,15 +87,15 @@ inline LemmaIdType DictTrie::get_lemma_id(size_t id_offset) {
}
#ifdef ___BUILD_MODEL___
bool DictTrie::build_dict(const char *fn_raw, const char *fn_validhzs) {
DictBuilder *dict_builder = new DictBuilder();
bool DictTrie::build_dict(const char* fn_raw, const char* fn_validhzs) {
DictBuilder* dict_builder = new DictBuilder();
free_resource(true);
return dict_builder->build_dict(fn_raw, fn_validhzs, this);
}
bool DictTrie::save_dict(FILE *fp) {
bool DictTrie::save_dict(FILE* fp) {
if (NULL == fp) return false;
if (fwrite(&lma_node_num_le0_, sizeof(uint32), 1, fp) != 1) return false;
@@ -115,15 +115,15 @@ bool DictTrie::save_dict(FILE *fp) {
return true;
}
bool DictTrie::save_dict(const char *filename) {
bool DictTrie::save_dict(const char* filename) {
if (NULL == filename) return false;
if (NULL == root_ || NULL == dict_list_) return false;
SpellingTrie &spl_trie = SpellingTrie::get_instance();
NGram &ngram = NGram::get_instance();
SpellingTrie& spl_trie = SpellingTrie::get_instance();
NGram& ngram = NGram::get_instance();
FILE *fp = fopen(filename, "wb");
FILE* fp = fopen(filename, "wb");
if (NULL == fp) return false;
if (!spl_trie.save_spl_trie(fp) || !dict_list_->save_list(fp) || !save_dict(fp) || !ngram.save_ngram(fp)) {
@@ -136,7 +136,7 @@ bool DictTrie::save_dict(const char *filename) {
}
#endif // ___BUILD_MODEL___
bool DictTrie::load_dict(FILE *fp) {
bool DictTrie::load_dict(FILE* fp) {
if (NULL == fp) return false;
if (fread(&lma_node_num_le0_, sizeof(uint32), 1, fp) != 1) return false;
@@ -148,14 +148,14 @@ bool DictTrie::load_dict(FILE *fp) {
free_resource(false);
root_ = static_cast<LmaNodeLE0 *>(malloc(lma_node_num_le0_ * sizeof(LmaNodeLE0)));
nodes_ge1_ = static_cast<LmaNodeGE1 *>(malloc(lma_node_num_ge1_ * sizeof(LmaNodeGE1)));
lma_idx_buf_ = (unsigned char *)malloc(lma_idx_buf_len_);
root_ = static_cast<LmaNodeLE0*>(malloc(lma_node_num_le0_ * sizeof(LmaNodeLE0)));
nodes_ge1_ = static_cast<LmaNodeGE1*>(malloc(lma_node_num_ge1_ * sizeof(LmaNodeGE1)));
lma_idx_buf_ = (unsigned char*)malloc(lma_idx_buf_len_);
total_lma_num_ = lma_idx_buf_len_ / kLemmaIdSize;
size_t buf_size = SpellingTrie::get_instance().get_spelling_num() + 1;
assert(lma_node_num_le0_ <= buf_size);
splid_le0_index_ = static_cast<uint16 *>(malloc(buf_size * sizeof(uint16)));
splid_le0_index_ = static_cast<uint16*>(malloc(buf_size * sizeof(uint16)));
// Init the space for parsing.
parsing_marks_ = new ParsingMark[kMaxParsingMark];
@@ -192,10 +192,10 @@ bool DictTrie::load_dict(FILE *fp) {
return true;
}
bool DictTrie::load_dict(const char *filename, LemmaIdType start_id, LemmaIdType end_id) {
bool DictTrie::load_dict(const char* filename, LemmaIdType start_id, LemmaIdType end_id) {
if (NULL == filename || end_id <= start_id) return false;
FILE *fp = fopen(filename, "rb");
FILE* fp = fopen(filename, "rb");
if (NULL == fp) return false;
free_resource(true);
@@ -206,8 +206,8 @@ bool DictTrie::load_dict(const char *filename, LemmaIdType start_id, LemmaIdType
return false;
}
SpellingTrie &spl_trie = SpellingTrie::get_instance();
NGram &ngram = NGram::get_instance();
SpellingTrie& spl_trie = SpellingTrie::get_instance();
NGram& ngram = NGram::get_instance();
if (!spl_trie.load_spl_trie(fp) || !dict_list_->load_list(fp) || !load_dict(fp) || !ngram.load_ngram(fp) || total_lma_num_ > end_id - start_id + 1) {
free_resource(true);
@@ -222,7 +222,7 @@ bool DictTrie::load_dict(const char *filename, LemmaIdType start_id, LemmaIdType
bool DictTrie::load_dict_fd(int sys_fd, long start_offset, long length, LemmaIdType start_id, LemmaIdType end_id) {
if (start_offset < 0 || length <= 0 || end_id <= start_id) return false;
FILE *fp = fdopen(sys_fd, "rb");
FILE* fp = fdopen(sys_fd, "rb");
if (NULL == fp) return false;
if (-1 == fseek(fp, start_offset, SEEK_SET)) {
@@ -238,8 +238,8 @@ bool DictTrie::load_dict_fd(int sys_fd, long start_offset, long length, LemmaIdT
return false;
}
SpellingTrie &spl_trie = SpellingTrie::get_instance();
NGram &ngram = NGram::get_instance();
SpellingTrie& spl_trie = SpellingTrie::get_instance();
NGram& ngram = NGram::get_instance();
if (!spl_trie.load_spl_trie(fp) || !dict_list_->load_list(fp) || !load_dict(fp) || !ngram.load_ngram(fp) || ftell(fp) < start_offset + length || total_lma_num_ > end_id - start_id + 1) {
free_resource(true);
@@ -251,9 +251,9 @@ bool DictTrie::load_dict_fd(int sys_fd, long start_offset, long length, LemmaIdT
return true;
}
size_t DictTrie::fill_lpi_buffer(LmaPsbItem lpi_items[], size_t lpi_max, LmaNodeLE0 *node) {
size_t DictTrie::fill_lpi_buffer(LmaPsbItem lpi_items[], size_t lpi_max, LmaNodeLE0* node) {
size_t lpi_num = 0;
NGram &ngram = NGram::get_instance();
NGram& ngram = NGram::get_instance();
for (size_t homo = 0; homo < (size_t)node->num_of_homo; homo++) {
lpi_items[lpi_num].id = get_lemma_id(node->homo_idx_buf_off + homo);
lpi_items[lpi_num].lma_len = 1;
@@ -265,9 +265,9 @@ size_t DictTrie::fill_lpi_buffer(LmaPsbItem lpi_items[], size_t lpi_max, LmaNode
return lpi_num;
}
size_t DictTrie::fill_lpi_buffer(LmaPsbItem lpi_items[], size_t lpi_max, size_t homo_buf_off, LmaNodeGE1 *node, uint16 lma_len) {
size_t DictTrie::fill_lpi_buffer(LmaPsbItem lpi_items[], size_t lpi_max, size_t homo_buf_off, LmaNodeGE1* node, uint16 lma_len) {
size_t lpi_num = 0;
NGram &ngram = NGram::get_instance();
NGram& ngram = NGram::get_instance();
for (size_t homo = 0; homo < (size_t)node->num_of_homo; homo++) {
lpi_items[lpi_num].id = get_lemma_id(homo_buf_off + homo);
lpi_items[lpi_num].lma_len = lma_len;
@@ -287,13 +287,13 @@ void DictTrie::reset_milestones(uint16 from_step, MileStoneHandle from_handle) {
if (from_handle > 0 && from_handle < mile_stones_pos_) {
mile_stones_pos_ = from_handle;
MileStone *mile_stone = mile_stones_ + from_handle;
MileStone* mile_stone = mile_stones_ + from_handle;
parsing_marks_pos_ = mile_stone->mark_start;
}
}
}
MileStoneHandle DictTrie::extend_dict(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num) {
MileStoneHandle DictTrie::extend_dict(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num) {
if (NULL == dep) return 0;
// from LmaNodeLE0 (root) to LmaNodeLE0
@@ -309,7 +309,7 @@ MileStoneHandle DictTrie::extend_dict(MileStoneHandle from_handle, const DictExt
return extend_dict2(from_handle, dep, lpi_items, lpi_max, lpi_num);
}
MileStoneHandle DictTrie::extend_dict0(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num) {
MileStoneHandle DictTrie::extend_dict0(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num) {
assert(NULL != dep && 0 == from_handle);
*lpi_num = 0;
MileStoneHandle ret_handle = 0;
@@ -318,17 +318,17 @@ MileStoneHandle DictTrie::extend_dict0(MileStoneHandle from_handle, const DictEx
uint16 id_start = dep->id_start;
uint16 id_num = dep->id_num;
LpiCache &lpi_cache = LpiCache::get_instance();
LpiCache& lpi_cache = LpiCache::get_instance();
bool cached = lpi_cache.is_cached(splid);
// 2. Begin exgtending
// 2.1 Get the LmaPsbItem list
LmaNodeLE0 *node = root_;
LmaNodeLE0* node = root_;
size_t son_start = splid_le0_index_[id_start - kFullSplIdStart];
size_t son_end = splid_le0_index_[id_start + id_num - kFullSplIdStart];
for (size_t son_pos = son_start; son_pos < son_end; son_pos++) {
assert(1 == node->son_1st_off);
LmaNodeLE0 *son = root_ + son_pos;
LmaNodeLE0* son = root_ + son_pos;
assert(son->spl_idx >= id_start && son->spl_idx < id_start + id_num);
if (!cached && *lpi_num < lpi_max) {
@@ -359,7 +359,7 @@ MileStoneHandle DictTrie::extend_dict0(MileStoneHandle from_handle, const DictEx
return ret_handle;
}
MileStoneHandle DictTrie::extend_dict1(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num) {
MileStoneHandle DictTrie::extend_dict1(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num) {
assert(NULL != dep && from_handle > 0 && from_handle < mile_stones_pos_);
MileStoneHandle ret_handle = 0;
@@ -372,18 +372,18 @@ MileStoneHandle DictTrie::extend_dict1(MileStoneHandle from_handle, const DictEx
uint16 id_num = dep->id_num;
// 2. Begin extending.
MileStone *mile_stone = mile_stones_ + from_handle;
MileStone* mile_stone = mile_stones_ + from_handle;
for (uint16 h_pos = 0; h_pos < mile_stone->mark_num; h_pos++) {
ParsingMark p_mark = parsing_marks_[mile_stone->mark_start + h_pos];
uint16 ext_num = p_mark.node_num;
for (uint16 ext_pos = 0; ext_pos < ext_num; ext_pos++) {
LmaNodeLE0 *node = root_ + p_mark.node_offset + ext_pos;
LmaNodeLE0* node = root_ + p_mark.node_offset + ext_pos;
size_t found_start = 0;
size_t found_num = 0;
for (size_t son_pos = 0; son_pos < (size_t)node->num_of_son; son_pos++) {
assert(node->son_1st_off <= lma_node_num_ge1_);
LmaNodeGE1 *son = nodes_ge1_ + node->son_1st_off + son_pos;
LmaNodeGE1* son = nodes_ge1_ + node->son_1st_off + son_pos;
if (son->spl_idx >= id_start && son->spl_idx < id_start + id_num) {
if (*lpi_num < lpi_max) {
size_t homo_buf_off = get_homo_idx_buf_offset(son);
@@ -425,7 +425,7 @@ MileStoneHandle DictTrie::extend_dict1(MileStoneHandle from_handle, const DictEx
return ret_handle;
}
MileStoneHandle DictTrie::extend_dict2(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num) {
MileStoneHandle DictTrie::extend_dict2(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num) {
assert(NULL != dep && from_handle > 0 && from_handle < mile_stones_pos_);
MileStoneHandle ret_handle = 0;
@@ -438,19 +438,19 @@ MileStoneHandle DictTrie::extend_dict2(MileStoneHandle from_handle, const DictEx
uint16 id_num = dep->id_num;
// 2. Begin extending.
MileStone *mile_stone = mile_stones_ + from_handle;
MileStone* mile_stone = mile_stones_ + from_handle;
for (uint16 h_pos = 0; h_pos < mile_stone->mark_num; h_pos++) {
ParsingMark p_mark = parsing_marks_[mile_stone->mark_start + h_pos];
uint16 ext_num = p_mark.node_num;
for (uint16 ext_pos = 0; ext_pos < ext_num; ext_pos++) {
LmaNodeGE1 *node = nodes_ge1_ + p_mark.node_offset + ext_pos;
LmaNodeGE1* node = nodes_ge1_ + p_mark.node_offset + ext_pos;
size_t found_start = 0;
size_t found_num = 0;
for (size_t son_pos = 0; son_pos < (size_t)node->num_of_son; son_pos++) {
assert(node->son_1st_off_l > 0 || node->son_1st_off_h > 0);
LmaNodeGE1 *son = nodes_ge1_ + get_son_offset(node) + son_pos;
LmaNodeGE1* son = nodes_ge1_ + get_son_offset(node) + son_pos;
if (son->spl_idx >= id_start && son->spl_idx < id_start + id_num) {
if (*lpi_num < lpi_max) {
size_t homo_buf_off = get_homo_idx_buf_offset(son);
@@ -491,15 +491,15 @@ MileStoneHandle DictTrie::extend_dict2(MileStoneHandle from_handle, const DictEx
return ret_handle;
}
bool DictTrie::try_extend(const uint16 *splids, uint16 splid_num, LemmaIdType id_lemma) {
bool DictTrie::try_extend(const uint16* splids, uint16 splid_num, LemmaIdType id_lemma) {
if (0 == splid_num || NULL == splids) return false;
void *node = root_ + splid_le0_index_[splids[0] - kFullSplIdStart];
void* node = root_ + splid_le0_index_[splids[0] - kFullSplIdStart];
for (uint16 pos = 1; pos < splid_num; pos++) {
if (1 == pos) {
LmaNodeLE0 *node_le0 = reinterpret_cast<LmaNodeLE0 *>(node);
LmaNodeGE1 *node_son;
LmaNodeLE0* node_le0 = reinterpret_cast<LmaNodeLE0*>(node);
LmaNodeGE1* node_son;
uint16 son_pos;
for (son_pos = 0; son_pos < static_cast<uint16>(node_le0->num_of_son); son_pos++) {
assert(node_le0->son_1st_off <= lma_node_num_ge1_);
@@ -507,12 +507,12 @@ bool DictTrie::try_extend(const uint16 *splids, uint16 splid_num, LemmaIdType id
if (node_son->spl_idx == splids[pos]) break;
}
if (son_pos < node_le0->num_of_son)
node = reinterpret_cast<void *>(node_son);
node = reinterpret_cast<void*>(node_son);
else
return false;
} else {
LmaNodeGE1 *node_ge1 = reinterpret_cast<LmaNodeGE1 *>(node);
LmaNodeGE1 *node_son;
LmaNodeGE1* node_ge1 = reinterpret_cast<LmaNodeGE1*>(node);
LmaNodeGE1* node_son;
uint16 son_pos;
for (son_pos = 0; son_pos < static_cast<uint16>(node_ge1->num_of_son); son_pos++) {
assert(node_ge1->son_1st_off_l > 0 || node_ge1->son_1st_off_h > 0);
@@ -520,14 +520,14 @@ bool DictTrie::try_extend(const uint16 *splids, uint16 splid_num, LemmaIdType id
if (node_son->spl_idx == splids[pos]) break;
}
if (son_pos < node_ge1->num_of_son)
node = reinterpret_cast<void *>(node_son);
node = reinterpret_cast<void*>(node_son);
else
return false;
}
}
if (1 == splid_num) {
LmaNodeLE0 *node_le0 = reinterpret_cast<LmaNodeLE0 *>(node);
LmaNodeLE0* node_le0 = reinterpret_cast<LmaNodeLE0*>(node);
size_t num_of_homo = (size_t)node_le0->num_of_homo;
for (size_t homo_pos = 0; homo_pos < num_of_homo; homo_pos++) {
LemmaIdType id_this = get_lemma_id(node_le0->homo_idx_buf_off + homo_pos);
@@ -536,7 +536,7 @@ bool DictTrie::try_extend(const uint16 *splids, uint16 splid_num, LemmaIdType id
if (id_this == id_lemma) return true;
}
} else {
LmaNodeGE1 *node_ge1 = reinterpret_cast<LmaNodeGE1 *>(node);
LmaNodeGE1* node_ge1 = reinterpret_cast<LmaNodeGE1*>(node);
size_t num_of_homo = (size_t)node_ge1->num_of_homo;
for (size_t homo_pos = 0; homo_pos < num_of_homo; homo_pos++) {
size_t node_homo_off = get_homo_idx_buf_offset(node_ge1);
@@ -547,17 +547,17 @@ bool DictTrie::try_extend(const uint16 *splids, uint16 splid_num, LemmaIdType id
return false;
}
size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lma_buf, size_t max_lma_buf) {
size_t DictTrie::get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lma_buf, size_t max_lma_buf) {
if (splid_str_len > kMaxLemmaSize) return 0;
#define MAX_EXTENDBUF_LEN 200
size_t *node_buf1[MAX_EXTENDBUF_LEN]; // use size_t for data alignment
size_t *node_buf2[MAX_EXTENDBUF_LEN];
LmaNodeLE0 **node_fr_le0 = reinterpret_cast<LmaNodeLE0 **>(node_buf1); // Nodes from.
LmaNodeLE0 **node_to_le0 = reinterpret_cast<LmaNodeLE0 **>(node_buf2); // Nodes to.
LmaNodeGE1 **node_fr_ge1 = NULL;
LmaNodeGE1 **node_to_ge1 = NULL;
size_t* node_buf1[MAX_EXTENDBUF_LEN]; // use size_t for data alignment
size_t* node_buf2[MAX_EXTENDBUF_LEN];
LmaNodeLE0** node_fr_le0 = reinterpret_cast<LmaNodeLE0**>(node_buf1); // Nodes from.
LmaNodeLE0** node_to_le0 = reinterpret_cast<LmaNodeLE0**>(node_buf2); // Nodes to.
LmaNodeGE1** node_fr_ge1 = NULL;
LmaNodeGE1** node_to_ge1 = NULL;
size_t node_fr_num = 1;
size_t node_to_num = 0;
node_fr_le0[0] = root_;
@@ -577,13 +577,13 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
// Extend the nodes
if (0 == spl_pos) { // From LmaNodeLE0 (root) to LmaNodeLE0 nodes
for (size_t node_fr_pos = 0; node_fr_pos < node_fr_num; node_fr_pos++) {
LmaNodeLE0 *node = node_fr_le0[node_fr_pos];
LmaNodeLE0* node = node_fr_le0[node_fr_pos];
assert(node == root_ && 1 == node_fr_num);
size_t son_start = splid_le0_index_[id_start - kFullSplIdStart];
size_t son_end = splid_le0_index_[id_start + id_num - kFullSplIdStart];
for (size_t son_pos = son_start; son_pos < son_end; son_pos++) {
assert(1 == node->son_1st_off);
LmaNodeLE0 *node_son = root_ + son_pos;
LmaNodeLE0* node_son = root_ + son_pos;
assert(node_son->spl_idx >= id_start && node_son->spl_idx < id_start + id_num);
if (node_to_num < MAX_EXTENDBUF_LEN) {
node_to_le0[node_to_num] = node_son;
@@ -599,16 +599,16 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
if (spl_pos >= splid_str_len || node_to_num == 0) break;
// Prepare the nodes for next extending
// next time, from LmaNodeLE0 to LmaNodeGE1
LmaNodeLE0 **node_tmp = node_fr_le0;
LmaNodeLE0** node_tmp = node_fr_le0;
node_fr_le0 = node_to_le0;
node_to_le0 = NULL;
node_to_ge1 = reinterpret_cast<LmaNodeGE1 **>(node_tmp);
node_to_ge1 = reinterpret_cast<LmaNodeGE1**>(node_tmp);
} else if (1 == spl_pos) { // From LmaNodeLE0 to LmaNodeGE1 nodes
for (size_t node_fr_pos = 0; node_fr_pos < node_fr_num; node_fr_pos++) {
LmaNodeLE0 *node = node_fr_le0[node_fr_pos];
LmaNodeLE0* node = node_fr_le0[node_fr_pos];
for (size_t son_pos = 0; son_pos < (size_t)node->num_of_son; son_pos++) {
assert(node->son_1st_off <= lma_node_num_ge1_);
LmaNodeGE1 *node_son = nodes_ge1_ + node->son_1st_off + son_pos;
LmaNodeGE1* node_son = nodes_ge1_ + node->son_1st_off + son_pos;
if (node_son->spl_idx >= id_start && node_son->spl_idx < id_start + id_num) {
if (node_to_num < MAX_EXTENDBUF_LEN) {
node_to_ge1[node_to_num] = node_son;
@@ -626,15 +626,15 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
// Prepare the nodes for next extending
// next time, from LmaNodeGE1 to LmaNodeGE1
node_fr_ge1 = node_to_ge1;
node_to_ge1 = reinterpret_cast<LmaNodeGE1 **>(node_fr_le0);
node_to_ge1 = reinterpret_cast<LmaNodeGE1**>(node_fr_le0);
node_fr_le0 = NULL;
node_to_le0 = NULL;
} else { // From LmaNodeGE1 to LmaNodeGE1 nodes
for (size_t node_fr_pos = 0; node_fr_pos < node_fr_num; node_fr_pos++) {
LmaNodeGE1 *node = node_fr_ge1[node_fr_pos];
LmaNodeGE1* node = node_fr_ge1[node_fr_pos];
for (size_t son_pos = 0; son_pos < (size_t)node->num_of_son; son_pos++) {
assert(node->son_1st_off_l > 0 || node->son_1st_off_h > 0);
LmaNodeGE1 *node_son = nodes_ge1_ + get_son_offset(node) + son_pos;
LmaNodeGE1* node_son = nodes_ge1_ + get_son_offset(node) + son_pos;
if (node_son->spl_idx >= id_start && node_son->spl_idx < id_start + id_num) {
if (node_to_num < MAX_EXTENDBUF_LEN) {
node_to_ge1[node_to_num] = node_son;
@@ -651,7 +651,7 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
if (spl_pos >= splid_str_len || node_to_num == 0) break;
// Prepare the nodes for next extending
// next time, from LmaNodeGE1 to LmaNodeGE1
LmaNodeGE1 **node_tmp = node_fr_ge1;
LmaNodeGE1** node_tmp = node_fr_ge1;
node_fr_ge1 = node_to_ge1;
node_to_ge1 = node_tmp;
}
@@ -663,7 +663,7 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
if (0 == node_to_num) return 0;
NGram &ngram = NGram::get_instance();
NGram& ngram = NGram::get_instance();
size_t lma_num = 0;
// If the length is 1, and the splid is a one-char Yunmu like 'a', 'o', 'e',
@@ -673,7 +673,7 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
for (size_t node_pos = 0; node_pos < node_to_num; node_pos++) {
size_t num_of_homo = 0;
if (spl_pos <= 1) { // Get from LmaNodeLE0 nodes
LmaNodeLE0 *node_le0 = node_to_le0[node_pos];
LmaNodeLE0* node_le0 = node_to_le0[node_pos];
num_of_homo = (size_t)node_le0->num_of_homo;
for (size_t homo_pos = 0; homo_pos < num_of_homo; homo_pos++) {
size_t ch_pos = lma_num + homo_pos;
@@ -684,7 +684,7 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
if (lma_num + homo_pos >= max_lma_buf - 1) break;
}
} else { // Get from LmaNodeGE1 nodes
LmaNodeGE1 *node_ge1 = node_to_ge1[node_pos];
LmaNodeGE1* node_ge1 = node_to_ge1[node_pos];
num_of_homo = (size_t)node_ge1->num_of_homo;
for (size_t homo_pos = 0; homo_pos < num_of_homo; homo_pos++) {
size_t ch_pos = lma_num + homo_pos;
@@ -706,9 +706,9 @@ size_t DictTrie::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbI
return lma_num;
}
uint16 DictTrie::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max) { return dict_list_->get_lemma_str(id_lemma, str_buf, str_max); }
uint16 DictTrie::get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max) { return dict_list_->get_lemma_str(id_lemma, str_buf, str_max); }
uint16 DictTrie::get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid) {
uint16 DictTrie::get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid) {
char16 lma_str[kMaxLemmaSize + 1];
uint16 lma_len = get_lemma_str(id_lemma, lma_str, kMaxLemmaSize + 1);
assert((!arg_valid && splids_max >= lma_len) || lma_len == splids_max);
@@ -746,13 +746,13 @@ uint16 DictTrie::get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 s
}
void DictTrie::set_total_lemma_count_of_others(size_t count) {
NGram &ngram = NGram::get_instance();
NGram& ngram = NGram::get_instance();
ngram.set_total_freq_none_sys(count);
}
void DictTrie::convert_to_hanzis(char16 *str, uint16 str_len) { return dict_list_->convert_to_hanzis(str, str_len); }
void DictTrie::convert_to_hanzis(char16* str, uint16 str_len) { return dict_list_->convert_to_hanzis(str, str_len); }
void DictTrie::convert_to_scis_ids(char16 *str, uint16 str_len) { return dict_list_->convert_to_scis_ids(str, str_len); }
void DictTrie::convert_to_scis_ids(char16* str, uint16 str_len) { return dict_list_->convert_to_scis_ids(str, str_len); }
LemmaIdType DictTrie::get_lemma_id(const char16 lemma_str[], uint16 lemma_len) {
if (NULL == lemma_str || lemma_len > kMaxLemmaSize) return 0;
@@ -760,8 +760,8 @@ LemmaIdType DictTrie::get_lemma_id(const char16 lemma_str[], uint16 lemma_len) {
return dict_list_->get_lemma_id(lemma_str, lemma_len);
}
size_t DictTrie::predict_top_lmas(size_t his_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used) {
NGram &ngram = NGram::get_instance();
size_t DictTrie::predict_top_lmas(size_t his_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used) {
NGram& ngram = NGram::get_instance();
size_t item_num = 0;
size_t top_lmas_id_offset = lma_idx_buf_len_ / kLemmaIdSize - top_lmas_num_;
@@ -780,5 +780,5 @@ size_t DictTrie::predict_top_lmas(size_t his_len, NPredictItem *npre_items, size
return item_num;
}
size_t DictTrie::predict(const char16 *last_hzs, uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used) { return dict_list_->predict(last_hzs, hzs_len, npre_items, npre_max, b4_used); }
size_t DictTrie::predict(const char16* last_hzs, uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used) { return dict_list_->predict(last_hzs, hzs_len, npre_items, npre_max, b4_used); }
} // namespace ime_pinyin
+44 -44
View File
@@ -69,7 +69,7 @@ bool MatrixSearch::alloc_resource() {
free_resource();
dict_trie_ = new DictTrie();
user_dict_ = static_cast<AtomDictBase *>(new UserDict());
user_dict_ = static_cast<AtomDictBase*>(new UserDict());
spl_parser_ = new SpellingParser();
size_t mtrx_nd_size = sizeof(MatrixNode) * kMtrxNdPoolSize;
@@ -87,13 +87,13 @@ bool MatrixSearch::alloc_resource() {
if (NULL == dict_trie_ || NULL == user_dict_ || NULL == spl_parser_ || NULL == share_buf_) return false;
// The buffers for search are based on the share buffer
mtrx_nd_pool_ = reinterpret_cast<MatrixNode *>(share_buf_);
dmi_pool_ = reinterpret_cast<DictMatchInfo *>(share_buf_ + mtrx_nd_size);
matrix_ = reinterpret_cast<MatrixRow *>(share_buf_ + mtrx_nd_size + dmi_size);
dep_ = reinterpret_cast<DictExtPara *>(share_buf_ + mtrx_nd_size + dmi_size + matrix_size);
mtrx_nd_pool_ = reinterpret_cast<MatrixNode*>(share_buf_);
dmi_pool_ = reinterpret_cast<DictMatchInfo*>(share_buf_ + mtrx_nd_size);
matrix_ = reinterpret_cast<MatrixRow*>(share_buf_ + mtrx_nd_size + dmi_size);
dep_ = reinterpret_cast<DictExtPara*>(share_buf_ + mtrx_nd_size + dmi_size + matrix_size);
// The prediction buffer is also based on the share buffer.
npre_items_ = reinterpret_cast<NPredictItem *>(share_buf_);
npre_items_ = reinterpret_cast<NPredictItem*>(share_buf_);
npre_items_len_ = (mtrx_nd_size + dmi_size + matrix_size + dep_size) * sizeof(size_t) / sizeof(NPredictItem);
return true;
}
@@ -110,7 +110,7 @@ void MatrixSearch::free_resource() {
reset_pointers_to_null();
}
bool MatrixSearch::init(const char *fn_sys_dict, const char *fn_usr_dict) {
bool MatrixSearch::init(const char* fn_sys_dict, const char* fn_usr_dict) {
if (NULL == fn_sys_dict || NULL == fn_usr_dict) return false;
if (!alloc_resource()) return false;
@@ -132,7 +132,7 @@ bool MatrixSearch::init(const char *fn_sys_dict, const char *fn_usr_dict) {
return true;
}
bool MatrixSearch::init_fd(int sys_fd, long start_offset, long length, const char *fn_usr_dict) {
bool MatrixSearch::init_fd(int sys_fd, long start_offset, long length, const char* fn_usr_dict) {
if (NULL == fn_usr_dict) return false;
if (!alloc_resource()) return false;
@@ -189,7 +189,7 @@ bool MatrixSearch::reset_search0() {
mtrx_nd_pool_used_ += 1;
// Update the node, and make it to be a starting node
MatrixNode *node = mtrx_nd_pool_ + matrix_[0].mtrx_nd_pos;
MatrixNode* node = mtrx_nd_pool_ + matrix_[0].mtrx_nd_pos;
node->id = 0;
node->score = 0;
node->from = NULL;
@@ -219,7 +219,7 @@ bool MatrixSearch::reset_search(size_t ch_pos, bool clear_fixed_this_step, bool
reset_search0();
} else {
// Prepare mile stones of this step to clear.
MileStoneHandle *dict_handles_to_clear = NULL;
MileStoneHandle* dict_handles_to_clear = NULL;
if (clear_dmi_this_step && matrix_[ch_pos].dmi_num > 0) {
dict_handles_to_clear = dmi_pool_[matrix_[ch_pos].dmi_pos].dict_handles;
}
@@ -274,7 +274,7 @@ bool MatrixSearch::reset_search(size_t ch_pos, bool clear_fixed_this_step, bool
// which was previously fixed.
//
// Prepare mile stones of this step to clear.
MileStoneHandle *dict_handles_to_clear = NULL;
MileStoneHandle* dict_handles_to_clear = NULL;
if (clear_dmi_this_step && ch_pos == fixed_ch_pos && matrix_[fixed_ch_pos].dmi_num > 0) {
dict_handles_to_clear = dmi_pool_[matrix_[fixed_ch_pos].dmi_pos].dict_handles;
}
@@ -365,7 +365,7 @@ void MatrixSearch::del_in_pys(size_t start, size_t len) {
}
}
size_t MatrixSearch::search(const char *py, size_t py_len) {
size_t MatrixSearch::search(const char* py, size_t py_len) {
if (!inited_ || NULL == py) return 0;
// If the search Pinyin string is too long, it will be truncated.
@@ -538,7 +538,7 @@ size_t MatrixSearch::get_candidate_num() {
return 1 + lpi_total_;
}
char16 *MatrixSearch::get_candidate(size_t cand_id, char16 *cand_str, size_t max_len) {
char16* MatrixSearch::get_candidate(size_t cand_id, char16* cand_str, size_t max_len) {
if (!inited_ || 0 == pys_decoded_len_ || NULL == cand_str) return NULL;
if (0 == cand_id) {
@@ -613,13 +613,13 @@ bool MatrixSearch::add_lma_to_userdict(uint16 lma_fr, uint16 lma_to, float score
assert(spl_id_fr <= kMaxLemmaSize);
return user_dict_->put_lemma(static_cast<char16 *>(word_str), spl_ids, spl_id_fr, 1);
return user_dict_->put_lemma(static_cast<char16*>(word_str), spl_ids, spl_id_fr, 1);
}
void MatrixSearch::debug_print_dmi(PoolPosType dmi_pos, uint16 nest_level) {
if (dmi_pos >= dmi_pool_used_) return;
DictMatchInfo *dmi = dmi_pool_ + dmi_pos;
DictMatchInfo* dmi = dmi_pool_ + dmi_pos;
if (1 == nest_level) {
printf("-----------------%d\'th DMI node begin----------->\n", dmi_pos);
@@ -809,13 +809,13 @@ size_t MatrixSearch::cancel_last_choice() {
size_t step_start = 0;
if (fixed_hzs_ > 0) {
size_t step_end = spl_start_[fixed_hzs_];
MatrixNode *end_node = matrix_[step_end].mtrx_nd_fixed;
MatrixNode* end_node = matrix_[step_end].mtrx_nd_fixed;
assert(NULL != end_node);
step_start = end_node->from->step;
if (step_start > 0) {
DictMatchInfo *dmi = dmi_pool_ + end_node->dmi_fr;
DictMatchInfo* dmi = dmi_pool_ + end_node->dmi_fr;
fixed_hzs_ -= dmi->dict_level;
} else {
fixed_hzs_ = 0;
@@ -847,7 +847,7 @@ bool MatrixSearch::prepare_add_char(char ch) {
pys_[pys_decoded_len_] = ch;
pys_decoded_len_++;
MatrixRow *mtrx_this_row = matrix_ + pys_decoded_len_;
MatrixRow* mtrx_this_row = matrix_ + pys_decoded_len_;
mtrx_this_row->mtrx_nd_pos = mtrx_nd_pool_used_;
mtrx_this_row->mtrx_nd_num = 0;
mtrx_this_row->dmi_pos = dmi_pool_used_;
@@ -859,7 +859,7 @@ bool MatrixSearch::prepare_add_char(char ch) {
bool MatrixSearch::is_split_at(uint16 pos) { return !spl_parser_->is_valid_to_parse(pys_[pos - 1]); }
void MatrixSearch::fill_dmi(DictMatchInfo *dmi, MileStoneHandle *handles, PoolPosType dmi_fr, uint16 spl_id, uint16 node_num, unsigned char dict_level, bool splid_end_split, unsigned char splstr_len, unsigned char all_full_id) {
void MatrixSearch::fill_dmi(DictMatchInfo* dmi, MileStoneHandle* handles, PoolPosType dmi_fr, uint16 spl_id, uint16 node_num, unsigned char dict_level, bool splid_end_split, unsigned char splstr_len, unsigned char all_full_id) {
dmi->dict_handles[0] = handles[0];
dmi->dict_handles[1] = handles[1];
dmi->dmi_fr = dmi_fr;
@@ -920,7 +920,7 @@ bool MatrixSearch::add_char_qwerty() {
// 3. Extend the DMI nodes of that old row
// + 1 is to extend an extra node from the root
for (PoolPosType dmi_pos = matrix_[oldrow].dmi_pos; dmi_pos < matrix_[oldrow].dmi_pos + matrix_[oldrow].dmi_num + 1; dmi_pos++) {
DictMatchInfo *dmi = dmi_pool_ + dmi_pos;
DictMatchInfo* dmi = dmi_pool_ + dmi_pos;
if (dmi_pos == matrix_[oldrow].dmi_pos + matrix_[oldrow].dmi_num) {
dmi = NULL; // The last one, NULL means extending from the root.
} else {
@@ -956,7 +956,7 @@ bool MatrixSearch::add_char_qwerty() {
continue;
}
DictMatchInfo *d = dmi;
DictMatchInfo* d = dmi;
while (d) {
dep_->splids[--prev_ids_num] = d->spl_id;
if ((PoolPosType)-1 == d->dmi_fr) break;
@@ -1001,7 +1001,7 @@ bool MatrixSearch::add_char_qwerty() {
fr_row = oldrow - dmi->splstr_len;
}
for (PoolPosType mtrx_nd_pos = matrix_[fr_row].mtrx_nd_pos; mtrx_nd_pos < matrix_[fr_row].mtrx_nd_pos + matrix_[fr_row].mtrx_nd_num; mtrx_nd_pos++) {
MatrixNode *mtrx_nd = mtrx_nd_pool_ + mtrx_nd_pos;
MatrixNode* mtrx_nd = mtrx_nd_pool_ + mtrx_nd_pos;
extend_mtrx_nd(mtrx_nd, lpi_items_, lpi_total_, dmi_pool_used_ - new_dmi_num, pys_decoded_len_);
if (longest_ext == 0) longest_ext = ext_len;
@@ -1026,7 +1026,7 @@ void MatrixSearch::prepare_candidates() {
// If the full sentense candidate's unfixed part may be the same with a normal
// lemma. Remove the lemma candidate in this case.
char16 fullsent[kMaxLemmaSize + 1];
char16 *pfullsent = NULL;
char16* pfullsent = NULL;
uint16 sent_len;
pfullsent = get_candidate0(fullsent, kMaxLemmaSize + 1, &sent_len, true);
@@ -1070,7 +1070,7 @@ void MatrixSearch::prepare_candidates() {
}
}
const char *MatrixSearch::get_pystr(size_t *decoded_len) {
const char* MatrixSearch::get_pystr(size_t* decoded_len) {
if (!inited_ || NULL == decoded_len) return NULL;
*decoded_len = pys_decoded_len_;
@@ -1116,7 +1116,7 @@ void MatrixSearch::merge_fixed_lmas(size_t del_spl_pos) {
if (pos == fixed_lmas_) break;
uint16 lma_len;
char16 *lma_str = c_phrase_.chn_str + c_phrase_.sublma_start[sub_num] + phrase_len;
char16* lma_str = c_phrase_.chn_str + c_phrase_.sublma_start[sub_num] + phrase_len;
lma_len = get_lemma_str(lma_id_[pos], lma_str, kMaxRowNum - phrase_len);
assert(lma_len == lma_start_[pos + 1] - lma_start_[pos]);
@@ -1144,7 +1144,7 @@ void MatrixSearch::merge_fixed_lmas(size_t del_spl_pos) {
// Delete the Chinese character in the merged phrase.
// The corresponding elements in spl_ids and spl_start of the
// phrase have been deleted.
char16 *chn_str = c_phrase_.chn_str + del_spl_pos;
char16* chn_str = c_phrase_.chn_str + del_spl_pos;
for (uint16 pos = 0; pos < c_phrase_.sublma_start[c_phrase_.sublma_num] - del_spl_pos; pos++) {
chn_str[pos] = chn_str[pos + 1];
}
@@ -1181,7 +1181,7 @@ void MatrixSearch::get_spl_start_id() {
lma_id_num_ = fixed_lmas_;
spl_id_num_ = fixed_hzs_;
MatrixNode *mtrx_nd = mtrx_nd_pool_ + matrix_[pys_decoded_len_].mtrx_nd_pos;
MatrixNode* mtrx_nd = mtrx_nd_pool_ + matrix_[pys_decoded_len_].mtrx_nd_pos;
while (mtrx_nd != mtrx_nd_pool_) {
if (fixed_hzs_ > 0) {
if (mtrx_nd->step <= spl_start_[fixed_hzs_]) break;
@@ -1254,18 +1254,18 @@ void MatrixSearch::get_spl_start_id() {
return;
}
size_t MatrixSearch::get_spl_start(const uint16 *&spl_start) {
size_t MatrixSearch::get_spl_start(const uint16*& spl_start) {
get_spl_start_id();
spl_start = spl_start_;
return spl_id_num_;
}
size_t MatrixSearch::extend_dmi(DictExtPara *dep, DictMatchInfo *dmi_s) {
size_t MatrixSearch::extend_dmi(DictExtPara* dep, DictMatchInfo* dmi_s) {
if (dmi_pool_used_ >= kDmiPoolSize) return 0;
if (dmi_c_phrase_) return extend_dmi_c(dep, dmi_s);
LpiCache &lpi_cache = LpiCache::get_instance();
LpiCache& lpi_cache = LpiCache::get_instance();
uint16 splid = dep->splids[dep->splids_extended];
bool cached = false;
@@ -1317,7 +1317,7 @@ size_t MatrixSearch::extend_dmi(DictExtPara *dep, DictMatchInfo *dmi_s) {
if (0 != handles[0] || 0 != handles[1]) {
if (dmi_pool_used_ >= kDmiPoolSize) return 0;
DictMatchInfo *dmi_add = dmi_pool_ + dmi_pool_used_;
DictMatchInfo* dmi_add = dmi_pool_ + dmi_pool_used_;
if (NULL == dmi_s) {
fill_dmi(dmi_add, handles, (PoolPosType)-1, splid, 1, 1, dep->splid_end_split, dep->ext_len, spl_trie_->is_half_id(splid) ? 0 : 1);
} else {
@@ -1344,7 +1344,7 @@ size_t MatrixSearch::extend_dmi(DictExtPara *dep, DictMatchInfo *dmi_s) {
return ret_val;
}
size_t MatrixSearch::extend_dmi_c(DictExtPara *dep, DictMatchInfo *dmi_s) {
size_t MatrixSearch::extend_dmi_c(DictExtPara* dep, DictMatchInfo* dmi_s) {
lpi_total_ = 0;
uint16 pos = dep->splids_extended;
@@ -1353,7 +1353,7 @@ size_t MatrixSearch::extend_dmi_c(DictExtPara *dep, DictMatchInfo *dmi_s) {
uint16 splid = dep->splids[pos];
if (splid == c_phrase_.spl_ids[pos]) {
DictMatchInfo *dmi_add = dmi_pool_ + dmi_pool_used_;
DictMatchInfo* dmi_add = dmi_pool_ + dmi_pool_used_;
MileStoneHandle handles[2]; // Actually never used.
if (NULL == dmi_s)
fill_dmi(dmi_add, handles, (PoolPosType)-1, splid, 1, 1, dep->splid_end_split, dep->ext_len, spl_trie_->is_half_id(splid) ? 0 : 1);
@@ -1370,7 +1370,7 @@ size_t MatrixSearch::extend_dmi_c(DictExtPara *dep, DictMatchInfo *dmi_s) {
return 0;
}
size_t MatrixSearch::extend_mtrx_nd(MatrixNode *mtrx_nd, LmaPsbItem lpi_items[], size_t lpi_num, PoolPosType dmi_fr, size_t res_row) {
size_t MatrixSearch::extend_mtrx_nd(MatrixNode* mtrx_nd, LmaPsbItem lpi_items[], size_t lpi_num, PoolPosType dmi_fr, size_t res_row) {
assert(NULL != mtrx_nd);
matrix_[res_row].mtrx_nd_fixed = NULL;
@@ -1382,14 +1382,14 @@ size_t MatrixSearch::extend_mtrx_nd(MatrixNode *mtrx_nd, LmaPsbItem lpi_items[],
if (lpi_num > kMaxNodeARow) lpi_num = kMaxNodeARow;
}
MatrixNode *mtrx_nd_res_min = mtrx_nd_pool_ + matrix_[res_row].mtrx_nd_pos;
MatrixNode* mtrx_nd_res_min = mtrx_nd_pool_ + matrix_[res_row].mtrx_nd_pos;
for (size_t pos = 0; pos < lpi_num; pos++) {
float score = mtrx_nd->score + lpi_items[pos].psb;
if (pos > 0 && score - PRUMING_SCORE > mtrx_nd_res_min->score) break;
// Try to add a new node
size_t mtrx_nd_num = matrix_[res_row].mtrx_nd_num;
MatrixNode *mtrx_nd_res = mtrx_nd_res_min + mtrx_nd_num;
MatrixNode* mtrx_nd_res = mtrx_nd_res_min + mtrx_nd_num;
bool replace = false;
// Find its position
while (mtrx_nd_res > mtrx_nd_res_min && score < (mtrx_nd_res - 1)->score) {
@@ -1415,7 +1415,7 @@ PoolPosType MatrixSearch::match_dmi(size_t step_to, uint16 spl_ids[], uint16 spl
}
for (PoolPosType dmi_pos = 0; dmi_pos < matrix_[step_to].dmi_num; dmi_pos++) {
DictMatchInfo *dmi = dmi_pool_ + matrix_[step_to].dmi_pos + dmi_pos;
DictMatchInfo* dmi = dmi_pool_ + matrix_[step_to].dmi_pos + dmi_pos;
if (dmi->dict_level != spl_id_num) continue;
@@ -1436,13 +1436,13 @@ PoolPosType MatrixSearch::match_dmi(size_t step_to, uint16 spl_ids[], uint16 spl
return static_cast<PoolPosType>(-1);
}
char16 *MatrixSearch::get_candidate0(char16 *cand_str, size_t max_len, uint16 *retstr_len, bool only_unfixed) {
char16* MatrixSearch::get_candidate0(char16* cand_str, size_t max_len, uint16* retstr_len, bool only_unfixed) {
if (pys_decoded_len_ == 0 || matrix_[pys_decoded_len_].mtrx_nd_num == 0) return NULL;
LemmaIdType idxs[kMaxRowNum];
size_t id_num = 0;
MatrixNode *mtrx_nd = mtrx_nd_pool_ + matrix_[pys_decoded_len_].mtrx_nd_pos;
MatrixNode* mtrx_nd = mtrx_nd_pool_ + matrix_[pys_decoded_len_].mtrx_nd_pos;
if (kPrintDebug0) {
printf("--- sentence score: %f\n", mtrx_nd->score);
@@ -1497,7 +1497,7 @@ char16 *MatrixSearch::get_candidate0(char16 *cand_str, size_t max_len, uint16 *r
return cand_str;
}
size_t MatrixSearch::get_lpis(const uint16 *splid_str, size_t splid_str_len, LmaPsbItem *lma_buf, size_t max_lma_buf, const char16 *pfullsent, bool sort_by_psb) {
size_t MatrixSearch::get_lpis(const uint16* splid_str, size_t splid_str_len, LmaPsbItem* lma_buf, size_t max_lma_buf, const char16* pfullsent, bool sort_by_psb) {
if (splid_str_len > kMaxLemmaSize) return 0;
size_t num1 = dict_trie_->get_lpis(splid_str, splid_str_len, lma_buf, max_lma_buf);
@@ -1512,7 +1512,7 @@ size_t MatrixSearch::get_lpis(const uint16 *splid_str, size_t splid_str_len, Lma
// Remove repeated items.
if (splid_str_len > 1) {
LmaPsbStrItem *lpsis = reinterpret_cast<LmaPsbStrItem *>(lma_buf + num);
LmaPsbStrItem* lpsis = reinterpret_cast<LmaPsbStrItem*>(lma_buf + num);
size_t lpsi_num = (max_lma_buf - num) * sizeof(LmaPsbItem) / sizeof(LmaPsbStrItem);
assert(lpsi_num > num);
if (num > lpsi_num) num = lpsi_num;
@@ -1582,7 +1582,7 @@ size_t MatrixSearch::get_lpis(const uint16 *splid_str, size_t splid_str_len, Lma
return num;
}
uint16 MatrixSearch::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max) {
uint16 MatrixSearch::get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max) {
uint16 str_len = 0;
if (is_system_lemma(id_lemma)) {
@@ -1606,7 +1606,7 @@ uint16 MatrixSearch::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16
return str_len;
}
uint16 MatrixSearch::get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid) {
uint16 MatrixSearch::get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid) {
uint16 splid_num = 0;
if (arg_valid) {
@@ -1638,7 +1638,7 @@ uint16 MatrixSearch::get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint
return splid_num;
}
size_t MatrixSearch::inner_predict(const char16 *fixed_buf, uint16 fixed_len, char16 predict_buf[][kMaxPredictSize + 1], size_t buf_len) {
size_t MatrixSearch::inner_predict(const char16* fixed_buf, uint16 fixed_len, char16 predict_buf[][kMaxPredictSize + 1], size_t buf_len) {
size_t res_total = 0;
memset(npre_items_, 0, sizeof(NPredictItem) * npre_items_len_);
// In order to shorten the comments, j-character candidates predicted by
+2 -2
View File
@@ -21,7 +21,7 @@ namespace ime_pinyin {
// For debug purpose. You can add a fixed version of qsort and bsearch functions
// here so that the output will be totally the same under different platforms.
void myqsort(void *p, size_t n, size_t es, int (*cmp)(const void *, const void *)) { qsort(p, n, es, cmp); }
void myqsort(void* p, size_t n, size_t es, int (*cmp)(const void*, const void*)) { qsort(p, n, es, cmp); }
void *mybsearch(const void *k, const void *b, size_t n, size_t es, int (*cmp)(const void *, const void *)) { return bsearch(k, b, n, es, cmp); }
void* mybsearch(const void* k, const void* b, size_t n, size_t es, int (*cmp)(const void*, const void*)) { return bsearch(k, b, n, es, cmp); }
} // namespace ime_pinyin
+16 -16
View File
@@ -26,9 +26,9 @@ namespace ime_pinyin {
#define ADD_COUNT 0.3
int comp_double(const void *p1, const void *p2) {
if (*static_cast<const double *>(p1) < *static_cast<const double *>(p2)) return -1;
if (*static_cast<const double *>(p1) > *static_cast<const double *>(p2)) return 1;
int comp_double(const void* p1, const void* p2) {
if (*static_cast<const double*>(p1) < *static_cast<const double*>(p2)) return -1;
if (*static_cast<const double*>(p1) > *static_cast<const double*>(p2)) return 1;
return 0;
}
@@ -54,7 +54,7 @@ int qsearch_nearest(double code_book[], double freq, int start, int end) {
return qsearch_nearest(code_book, freq, mid, end);
}
size_t update_code_idx(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE *code_idx) {
size_t update_code_idx(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE* code_idx) {
size_t changed = 0;
for (size_t pos = 0; pos < num; pos++) {
CODEBOOK_TYPE idx;
@@ -65,14 +65,14 @@ size_t update_code_idx(double freqs[], size_t num, double code_book[], CODEBOOK_
return changed;
}
double recalculate_kernel(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE *code_idx) {
double recalculate_kernel(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE* code_idx) {
double ret = 0;
size_t *item_num = new size_t[kCodeBookSize];
size_t* item_num = new size_t[kCodeBookSize];
assert(item_num);
memset(item_num, 0, sizeof(size_t) * kCodeBookSize);
double *cb_new = new double[kCodeBookSize];
double* cb_new = new double[kCodeBookSize];
assert(cb_new);
memset(cb_new, 0, sizeof(double) * kCodeBookSize);
@@ -94,7 +94,7 @@ double recalculate_kernel(double freqs[], size_t num, double code_book[], CODEBO
return ret;
}
void iterate_codes(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE *code_idx) {
void iterate_codes(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE* code_idx) {
size_t iter_num = 0;
double delta_last = 0;
do {
@@ -112,7 +112,7 @@ void iterate_codes(double freqs[], size_t num, double code_book[], CODEBOOK_TYPE
} while (true);
}
NGram *NGram::instance_ = NULL;
NGram* NGram::instance_ = NULL;
NGram::NGram() {
initialized_ = false;
@@ -136,12 +136,12 @@ NGram::~NGram() {
if (NULL != freq_codes_) free(freq_codes_);
}
NGram &NGram::get_instance() {
NGram& NGram::get_instance() {
if (NULL == instance_) instance_ = new NGram();
return *instance_;
}
bool NGram::save_ngram(FILE *fp) {
bool NGram::save_ngram(FILE* fp) {
if (!initialized_ || NULL == fp) return false;
if (0 == idx_num_ || NULL == freq_codes_ || NULL == lma_freq_idx_) return false;
@@ -155,7 +155,7 @@ bool NGram::save_ngram(FILE *fp) {
return true;
}
bool NGram::load_ngram(FILE *fp) {
bool NGram::load_ngram(FILE* fp) {
if (NULL == fp) return false;
initialized_ = false;
@@ -166,8 +166,8 @@ bool NGram::load_ngram(FILE *fp) {
if (NULL != freq_codes_) free(freq_codes_);
lma_freq_idx_ = static_cast<CODEBOOK_TYPE *>(malloc(idx_num_ * sizeof(CODEBOOK_TYPE)));
freq_codes_ = static_cast<LmaScoreType *>(malloc(kCodeBookSize * sizeof(LmaScoreType)));
lma_freq_idx_ = static_cast<CODEBOOK_TYPE*>(malloc(idx_num_ * sizeof(CODEBOOK_TYPE)));
freq_codes_ = static_cast<LmaScoreType*>(malloc(kCodeBookSize * sizeof(LmaScoreType)));
if (NULL == lma_freq_idx_ || NULL == freq_codes_) return false;
@@ -203,11 +203,11 @@ float NGram::convert_psb_to_score(double psb) {
}
#ifdef ___BUILD_MODEL___
bool NGram::build_unigram(LemmaEntry *lemma_arr, size_t lemma_num, LemmaIdType next_idx_unused) {
bool NGram::build_unigram(LemmaEntry* lemma_arr, size_t lemma_num, LemmaIdType next_idx_unused) {
if (NULL == lemma_arr || 0 == lemma_num || next_idx_unused <= 1) return false;
double total_freq = 0;
double *freqs = new double[next_idx_unused];
double* freqs = new double[next_idx_unused];
if (NULL == freqs) return false;
freqs[0] = ADD_COUNT;
+11 -11
View File
@@ -29,11 +29,11 @@ using namespace ime_pinyin;
static const size_t kMaxPredictNum = 500;
// Used to search Pinyin string and give the best candidate.
MatrixSearch *matrix_search = NULL;
MatrixSearch* matrix_search = NULL;
char16 predict_buf[kMaxPredictNum][kMaxPredictSize + 1];
bool im_open_decoder(const char *fn_sys_dict, const char *fn_usr_dict) {
bool im_open_decoder(const char* fn_sys_dict, const char* fn_usr_dict) {
if (NULL != matrix_search) delete matrix_search;
matrix_search = new MatrixSearch();
@@ -44,7 +44,7 @@ bool im_open_decoder(const char *fn_sys_dict, const char *fn_usr_dict) {
return matrix_search->init(fn_sys_dict, fn_usr_dict);
}
bool im_open_decoder_fd(int sys_fd, long start_offset, long length, const char *fn_usr_dict) {
bool im_open_decoder_fd(int sys_fd, long start_offset, long length, const char* fn_usr_dict) {
if (NULL != matrix_search) delete matrix_search;
matrix_search = new MatrixSearch();
@@ -72,7 +72,7 @@ void im_flush_cache() {
}
// To be updated.
size_t im_search(const char *pybuf, size_t pylen) {
size_t im_search(const char* pybuf, size_t pylen) {
if (NULL == matrix_search) return 0;
matrix_search->search(pybuf, pylen);
@@ -94,19 +94,19 @@ void im_reset_search() {
// To be removed
size_t im_add_letter(char ch) { return 0; }
const char *im_get_sps_str(size_t *decoded_len) {
const char* im_get_sps_str(size_t* decoded_len) {
if (NULL == matrix_search) return NULL;
return matrix_search->get_pystr(decoded_len);
}
char16 *im_get_candidate(size_t cand_id, char16 *cand_str, size_t max_len) {
char16* im_get_candidate(size_t cand_id, char16* cand_str, size_t max_len) {
if (NULL == matrix_search) return NULL;
return matrix_search->get_candidate(cand_id, cand_str, max_len);
}
size_t im_get_spl_start_pos(const uint16 *&spl_start) {
size_t im_get_spl_start_pos(const uint16*& spl_start) {
if (NULL == matrix_search) return 0;
return matrix_search->get_spl_start(spl_start);
@@ -133,11 +133,11 @@ size_t im_get_fixed_len() {
// To be removed
bool im_cancel_input() { return true; }
size_t im_get_predicts(const char16 *his_buf, char16 (*&pre_buf)[kMaxPredictSize + 1]) {
size_t im_get_predicts(const char16* his_buf, char16 (*&pre_buf)[kMaxPredictSize + 1]) {
if (NULL == his_buf) return 0;
size_t fixed_len = utf16_strlen(his_buf);
const char16 *fixed_ptr = his_buf;
const char16* fixed_ptr = his_buf;
if (fixed_len > kMaxPredictSize) {
fixed_ptr += fixed_len - kMaxPredictSize;
fixed_len = kMaxPredictSize;
@@ -148,12 +148,12 @@ size_t im_get_predicts(const char16 *his_buf, char16 (*&pre_buf)[kMaxPredictSize
}
void im_enable_shm_as_szm(bool enable) {
SpellingTrie &spl_trie = SpellingTrie::get_instance();
SpellingTrie& spl_trie = SpellingTrie::get_instance();
spl_trie.szm_enable_shm(enable);
}
void im_enable_ym_as_szm(bool enable) {
SpellingTrie &spl_trie = SpellingTrie::get_instance();
SpellingTrie& spl_trie = SpellingTrie::get_instance();
spl_trie.szm_enable_ym(enable);
}
+36 -36
View File
@@ -26,15 +26,15 @@ bool is_user_lemma(LemmaIdType lma_id) { return (kUserDictIdStart <= lma_id && l
bool is_composing_lemma(LemmaIdType lma_id) { return (kLemmaIdComposing == lma_id); }
int cmp_lpi_with_psb(const void *p1, const void *p2) {
if ((static_cast<const LmaPsbItem *>(p1))->psb > (static_cast<const LmaPsbItem *>(p2))->psb) return 1;
if ((static_cast<const LmaPsbItem *>(p1))->psb < (static_cast<const LmaPsbItem *>(p2))->psb) return -1;
int cmp_lpi_with_psb(const void* p1, const void* p2) {
if ((static_cast<const LmaPsbItem*>(p1))->psb > (static_cast<const LmaPsbItem*>(p2))->psb) return 1;
if ((static_cast<const LmaPsbItem*>(p1))->psb < (static_cast<const LmaPsbItem*>(p2))->psb) return -1;
return 0;
}
int cmp_lpi_with_unified_psb(const void *p1, const void *p2) {
const LmaPsbItem *item1 = static_cast<const LmaPsbItem *>(p1);
const LmaPsbItem *item2 = static_cast<const LmaPsbItem *>(p2);
int cmp_lpi_with_unified_psb(const void* p1, const void* p2) {
const LmaPsbItem* item1 = static_cast<const LmaPsbItem*>(p1);
const LmaPsbItem* item2 = static_cast<const LmaPsbItem*>(p2);
// The real unified psb is psb1 / lma_len1 and psb2 * lma_len2
// But we use psb1 * lma_len2 and psb2 * lma_len1 to get better
@@ -50,74 +50,74 @@ int cmp_lpi_with_unified_psb(const void *p1, const void *p2) {
return 0;
}
int cmp_lpi_with_id(const void *p1, const void *p2) {
if ((static_cast<const LmaPsbItem *>(p1))->id < (static_cast<const LmaPsbItem *>(p2))->id) return -1;
if ((static_cast<const LmaPsbItem *>(p1))->id > (static_cast<const LmaPsbItem *>(p2))->id) return 1;
int cmp_lpi_with_id(const void* p1, const void* p2) {
if ((static_cast<const LmaPsbItem*>(p1))->id < (static_cast<const LmaPsbItem*>(p2))->id) return -1;
if ((static_cast<const LmaPsbItem*>(p1))->id > (static_cast<const LmaPsbItem*>(p2))->id) return 1;
return 0;
}
int cmp_lpi_with_hanzi(const void *p1, const void *p2) {
if ((static_cast<const LmaPsbItem *>(p1))->hanzi < (static_cast<const LmaPsbItem *>(p2))->hanzi) return -1;
if ((static_cast<const LmaPsbItem *>(p1))->hanzi > (static_cast<const LmaPsbItem *>(p2))->hanzi) return 1;
int cmp_lpi_with_hanzi(const void* p1, const void* p2) {
if ((static_cast<const LmaPsbItem*>(p1))->hanzi < (static_cast<const LmaPsbItem*>(p2))->hanzi) return -1;
if ((static_cast<const LmaPsbItem*>(p1))->hanzi > (static_cast<const LmaPsbItem*>(p2))->hanzi) return 1;
return 0;
}
int cmp_lpsi_with_str(const void *p1, const void *p2) { return utf16_strcmp((static_cast<const LmaPsbStrItem *>(p1))->str, (static_cast<const LmaPsbStrItem *>(p2))->str); }
int cmp_lpsi_with_str(const void* p1, const void* p2) { return utf16_strcmp((static_cast<const LmaPsbStrItem*>(p1))->str, (static_cast<const LmaPsbStrItem*>(p2))->str); }
int cmp_hanzis_1(const void *p1, const void *p2) {
if (*static_cast<const char16 *>(p1) < *static_cast<const char16 *>(p2)) return -1;
int cmp_hanzis_1(const void* p1, const void* p2) {
if (*static_cast<const char16*>(p1) < *static_cast<const char16*>(p2)) return -1;
if (*static_cast<const char16 *>(p1) > *static_cast<const char16 *>(p2)) return 1;
if (*static_cast<const char16*>(p1) > *static_cast<const char16*>(p2)) return 1;
return 0;
}
int cmp_hanzis_2(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 2); }
int cmp_hanzis_2(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 2); }
int cmp_hanzis_3(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 3); }
int cmp_hanzis_3(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 3); }
int cmp_hanzis_4(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 4); }
int cmp_hanzis_4(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 4); }
int cmp_hanzis_5(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 5); }
int cmp_hanzis_5(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 5); }
int cmp_hanzis_6(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 6); }
int cmp_hanzis_6(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 6); }
int cmp_hanzis_7(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 7); }
int cmp_hanzis_7(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 7); }
int cmp_hanzis_8(const void *p1, const void *p2) { return utf16_strncmp(static_cast<const char16 *>(p1), static_cast<const char16 *>(p2), 8); }
int cmp_hanzis_8(const void* p1, const void* p2) { return utf16_strncmp(static_cast<const char16*>(p1), static_cast<const char16*>(p2), 8); }
int cmp_npre_by_score(const void *p1, const void *p2) {
if ((static_cast<const NPredictItem *>(p1))->psb > (static_cast<const NPredictItem *>(p2))->psb) return 1;
int cmp_npre_by_score(const void* p1, const void* p2) {
if ((static_cast<const NPredictItem*>(p1))->psb > (static_cast<const NPredictItem*>(p2))->psb) return 1;
if ((static_cast<const NPredictItem *>(p1))->psb < (static_cast<const NPredictItem *>(p2))->psb) return -1;
if ((static_cast<const NPredictItem*>(p1))->psb < (static_cast<const NPredictItem*>(p2))->psb) return -1;
return 0;
}
int cmp_npre_by_hislen_score(const void *p1, const void *p2) {
if ((static_cast<const NPredictItem *>(p1))->his_len < (static_cast<const NPredictItem *>(p2))->his_len) return 1;
int cmp_npre_by_hislen_score(const void* p1, const void* p2) {
if ((static_cast<const NPredictItem*>(p1))->his_len < (static_cast<const NPredictItem*>(p2))->his_len) return 1;
if ((static_cast<const NPredictItem *>(p1))->his_len > (static_cast<const NPredictItem *>(p2))->his_len) return -1;
if ((static_cast<const NPredictItem*>(p1))->his_len > (static_cast<const NPredictItem*>(p2))->his_len) return -1;
if ((static_cast<const NPredictItem *>(p1))->psb > (static_cast<const NPredictItem *>(p2))->psb) return 1;
if ((static_cast<const NPredictItem*>(p1))->psb > (static_cast<const NPredictItem*>(p2))->psb) return 1;
if ((static_cast<const NPredictItem *>(p1))->psb < (static_cast<const NPredictItem *>(p2))->psb) return -1;
if ((static_cast<const NPredictItem*>(p1))->psb < (static_cast<const NPredictItem*>(p2))->psb) return -1;
return 0;
}
int cmp_npre_by_hanzi_score(const void *p1, const void *p2) {
int ret_v = (utf16_strncmp((static_cast<const NPredictItem *>(p1))->pre_hzs, (static_cast<const NPredictItem *>(p2))->pre_hzs, kMaxPredictSize));
int cmp_npre_by_hanzi_score(const void* p1, const void* p2) {
int ret_v = (utf16_strncmp((static_cast<const NPredictItem*>(p1))->pre_hzs, (static_cast<const NPredictItem*>(p2))->pre_hzs, kMaxPredictSize));
if (0 != ret_v) return ret_v;
if ((static_cast<const NPredictItem *>(p1))->psb > (static_cast<const NPredictItem *>(p2))->psb) return 1;
if ((static_cast<const NPredictItem*>(p1))->psb > (static_cast<const NPredictItem*>(p2))->psb) return 1;
if ((static_cast<const NPredictItem *>(p1))->psb < (static_cast<const NPredictItem *>(p2))->psb) return -1;
if ((static_cast<const NPredictItem*>(p1))->psb < (static_cast<const NPredictItem*>(p2))->psb) return -1;
return 0;
}
size_t remove_duplicate_npre(NPredictItem *npre_items, size_t npre_num) {
size_t remove_duplicate_npre(NPredictItem* npre_items, size_t npre_num) {
if (NULL == npre_items || 0 == npre_num) return 0;
myqsort(npre_items, npre_num, sizeof(NPredictItem), cmp_npre_by_hanzi_score);
+27 -27
View File
@@ -27,7 +27,7 @@
namespace ime_pinyin {
SpellingTrie *SpellingTrie::instance_ = NULL;
SpellingTrie* SpellingTrie::instance_ = NULL;
// z/c/s is for Zh/Ch/Sh
const char SpellingTrie::kHalfId2Sc_[kFullSplIdStart + 1] = "0ABCcDEFGHIJKLMNOPQRSsTUVWXYZz";
@@ -45,7 +45,7 @@ unsigned char SpellingTrie::char_flags_[] = {
// u v w x y z
0x00, 0x00, 0x01, 0x01, 0x01, 0x01};
int compare_spl(const void *p1, const void *p2) { return strcmp((const char *)(p1), (const char *)(p2)); }
int compare_spl(const void* p1, const void* p2) { return strcmp((const char*)(p1), (const char*)(p2)); }
SpellingTrie::SpellingTrie() {
spelling_buf_ = NULL;
@@ -101,7 +101,7 @@ SpellingTrie::~SpellingTrie() {
if (NULL != f2h_) delete[] f2h_;
}
bool SpellingTrie::if_valid_id_update(uint16 *splid) const {
bool SpellingTrie::if_valid_id_update(uint16* splid) const {
if (NULL == splid || 0 == *splid) return false;
if (*splid >= kFullSplIdStart) return true;
@@ -193,9 +193,9 @@ void SpellingTrie::szm_enable_ym(bool enable) {
bool SpellingTrie::is_szm_enabled(char ch) const { return char_flags_[ch - 'A'] & kHalfIdSzmMask; }
const SpellingTrie *SpellingTrie::get_cpinstance() { return &get_instance(); }
const SpellingTrie* SpellingTrie::get_cpinstance() { return &get_instance(); }
SpellingTrie &SpellingTrie::get_instance() {
SpellingTrie& SpellingTrie::get_instance() {
if (NULL == instance_) instance_ = new SpellingTrie();
return *instance_;
@@ -206,7 +206,7 @@ uint16 SpellingTrie::half2full_num(uint16 half_id) const {
return h2f_num_[half_id];
}
uint16 SpellingTrie::half_to_full(uint16 half_id, uint16 *spl_id_start) const {
uint16 SpellingTrie::half_to_full(uint16 half_id, uint16* spl_id_start) const {
if (NULL == spl_id_start || NULL == root_ || half_id >= kFullSplIdStart) return 0;
*spl_id_start = h2f_start_[half_id];
@@ -219,7 +219,7 @@ uint16 SpellingTrie::full_to_half(uint16 full_id) const {
return f2h_[full_id - kFullSplIdStart];
}
void SpellingTrie::free_son_trie(SpellingNode *node) {
void SpellingTrie::free_son_trie(SpellingNode* node) {
if (NULL == node) return;
for (size_t pos = 0; pos < node->num_of_son; pos++) {
@@ -229,7 +229,7 @@ void SpellingTrie::free_son_trie(SpellingNode *node) {
if (NULL != node->first_son) delete[] node->first_son;
}
bool SpellingTrie::construct(const char *spelling_arr, size_t item_size, size_t item_num, float score_amplifier, unsigned char average_score) {
bool SpellingTrie::construct(const char* spelling_arr, size_t item_size, size_t item_num, float score_amplifier, unsigned char average_score) {
if (spelling_arr == NULL) return false;
memset(h2f_start_, 0, sizeof(uint16) * kFullSplIdStart);
@@ -277,7 +277,7 @@ bool SpellingTrie::construct(const char *spelling_arr, size_t item_size, size_t
memset(splitter_node_, 0, sizeof(SpellingNode));
splitter_node_->score = average_score_;
memset(level1_sons_, 0, sizeof(SpellingNode *) * kValidSplCharNum);
memset(level1_sons_, 0, sizeof(SpellingNode*) * kValidSplCharNum);
root_->first_son = construct_spellings_subset(0, spelling_num_, 0, root_);
@@ -301,7 +301,7 @@ bool SpellingTrie::construct(const char *spelling_arr, size_t item_size, size_t
}
#ifdef ___BUILD_MODEL___
const char *SpellingTrie::get_ym_str(const char *spl_str) {
const char* SpellingTrie::get_ym_str(const char* spl_str) {
bool start_ZCS = false;
if (is_shengmu_char(*spl_str)) {
if ('Z' == *spl_str || 'C' == *spl_str || 'S' == *spl_str) start_ZCS = true;
@@ -313,13 +313,13 @@ const char *SpellingTrie::get_ym_str(const char *spl_str) {
bool SpellingTrie::build_ym_info() {
bool sucess;
SpellingTable *spl_table = new SpellingTable();
SpellingTable* spl_table = new SpellingTable();
sucess = spl_table->init_table(kMaxPinyinSize - 1, 2 * kMaxYmNum, false);
assert(sucess);
for (uint16 pos = 0; pos < spelling_num_; pos++) {
const char *spl_str = spelling_buf_ + spelling_size_ * pos;
const char* spl_str = spelling_buf_ + spelling_size_ * pos;
spl_str = get_ym_str(spl_str);
if ('\0' != spl_str[0]) {
sucess = spl_table->put_spelling(spl_str, 0);
@@ -329,7 +329,7 @@ bool SpellingTrie::build_ym_info() {
size_t ym_item_size; // '\0' is included
size_t ym_num;
const char *ym_buf;
const char* ym_buf;
ym_buf = spl_table->arrange(&ym_item_size, &ym_num);
if (NULL != ym_buf_) delete[] ym_buf_;
@@ -353,7 +353,7 @@ bool SpellingTrie::build_ym_info() {
memset(spl_ym_ids_, 0, sizeof(uint8) * (spelling_num_ + kFullSplIdStart));
for (uint16 id = 1; id < spelling_num_ + kFullSplIdStart; id++) {
const char *str = get_spelling_str(id);
const char* str = get_spelling_str(id);
str = get_ym_str(str);
if ('\0' != str[0]) {
@@ -368,20 +368,20 @@ bool SpellingTrie::build_ym_info() {
}
#endif
SpellingNode *SpellingTrie::construct_spellings_subset(size_t item_start, size_t item_end, size_t level, SpellingNode *parent) {
SpellingNode* SpellingTrie::construct_spellings_subset(size_t item_start, size_t item_end, size_t level, SpellingNode* parent) {
if (level >= spelling_size_ || item_end <= item_start || NULL == parent) return NULL;
SpellingNode *first_son = NULL;
SpellingNode* first_son = NULL;
uint16 num_of_son = 0;
unsigned char min_son_score = 255;
const char *spelling_last_start = spelling_buf_ + spelling_size_ * item_start;
const char* spelling_last_start = spelling_buf_ + spelling_size_ * item_start;
char char_for_node = spelling_last_start[level];
assert(char_for_node >= 'A' && char_for_node <= 'Z' || 'h' == char_for_node);
// Scan the array to find how many sons
for (size_t i = item_start + 1; i < item_end; i++) {
const char *spelling_current = spelling_buf_ + spelling_size_ * i;
const char* spelling_current = spelling_buf_ + spelling_size_ * i;
char char_current = spelling_current[level];
if (char_current != char_for_node) {
num_of_son++;
@@ -409,13 +409,13 @@ SpellingNode *SpellingTrie::construct_spellings_subset(size_t item_start, size_t
size_t item_start_next = item_start;
for (size_t i = item_start + 1; i < item_end; i++) {
const char *spelling_current = spelling_buf_ + spelling_size_ * i;
const char* spelling_current = spelling_buf_ + spelling_size_ * i;
char char_current = spelling_current[level];
assert(is_valid_spl_char(char_current));
if (char_current != char_for_node) {
// Construct a node
SpellingNode *node_current = first_son + son_pos;
SpellingNode* node_current = first_son + son_pos;
node_current->char_this_node = char_for_node;
// For quick search in the first level
@@ -486,7 +486,7 @@ SpellingNode *SpellingTrie::construct_spellings_subset(size_t item_start, size_t
}
// the last one
SpellingNode *node_current = first_son + son_pos;
SpellingNode* node_current = first_son + son_pos;
node_current->char_this_node = char_for_node;
// For quick search in the first level
@@ -551,7 +551,7 @@ SpellingNode *SpellingTrie::construct_spellings_subset(size_t item_start, size_t
return first_son;
}
bool SpellingTrie::save_spl_trie(FILE *fp) {
bool SpellingTrie::save_spl_trie(FILE* fp) {
if (NULL == fp || NULL == spelling_buf_) return false;
if (fwrite(&spelling_size_, sizeof(uint32), 1, fp) != 1) return false;
@@ -567,7 +567,7 @@ bool SpellingTrie::save_spl_trie(FILE *fp) {
return true;
}
bool SpellingTrie::load_spl_trie(FILE *fp) {
bool SpellingTrie::load_spl_trie(FILE* fp) {
if (NULL == fp) return false;
if (fread(&spelling_size_, sizeof(uint32), 1, fp) != 1) return false;
@@ -602,7 +602,7 @@ bool SpellingTrie::build_f2h() {
size_t SpellingTrie::get_spelling_num() { return spelling_num_; }
uint8 SpellingTrie::get_ym_id(const char *ym_str) {
uint8 SpellingTrie::get_ym_id(const char* ym_str) {
if (NULL == ym_str || NULL == ym_buf_) return 0;
for (uint8 pos = 0; pos < ym_num_; pos++)
@@ -611,7 +611,7 @@ uint8 SpellingTrie::get_ym_id(const char *ym_str) {
return 0;
}
const char *SpellingTrie::get_spelling_str(uint16 splid) {
const char* SpellingTrie::get_spelling_str(uint16 splid) {
splstr_queried_[0] = '\0';
if (splid >= kFullSplIdStart) {
@@ -634,7 +634,7 @@ const char *SpellingTrie::get_spelling_str(uint16 splid) {
return splstr_queried_;
}
const char16 *SpellingTrie::get_spelling_str16(uint16 splid) {
const char16* SpellingTrie::get_spelling_str16(uint16 splid) {
splstr16_queried_[0] = '\0';
if (splid >= kFullSplIdStart) {
@@ -665,7 +665,7 @@ const char16 *SpellingTrie::get_spelling_str16(uint16 splid) {
return splstr16_queried_;
}
size_t SpellingTrie::get_spelling_str16(uint16 splid, char16 *splstr16, size_t splstr16_len) {
size_t SpellingTrie::get_spelling_str16(uint16 splid, char16* splstr16, size_t splstr16_len) {
if (NULL == splstr16 || splstr16_len < kMaxPinyinSize + 1) return 0;
if (splid >= kFullSplIdStart) {
+15 -15
View File
@@ -23,14 +23,14 @@ SpellingParser::SpellingParser() { spl_trie_ = SpellingTrie::get_cpinstance(); }
bool SpellingParser::is_valid_to_parse(char ch) { return SpellingTrie::is_valid_spl_char(ch); }
uint16 SpellingParser::splstr_to_idxs(const char *splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre) {
uint16 SpellingParser::splstr_to_idxs(const char* splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre) {
if (NULL == splstr || 0 == max_size || 0 == str_len) return 0;
if (!SpellingTrie::is_valid_spl_char(splstr[0])) return 0;
last_is_pre = false;
const SpellingNode *node_this = spl_trie_->root_;
const SpellingNode* node_this = spl_trie_->root_;
uint16 str_pos = 0;
uint16 idx_num = 0;
@@ -67,7 +67,7 @@ uint16 SpellingParser::splstr_to_idxs(const char *splstr, uint16 str_len, uint16
last_is_splitter = false;
SpellingNode *found_son = NULL;
SpellingNode* found_son = NULL;
if (0 == str_pos) {
if (char_this >= 'a')
@@ -75,11 +75,11 @@ uint16 SpellingParser::splstr_to_idxs(const char *splstr, uint16 str_len, uint16
else
found_son = spl_trie_->level1_sons_[char_this - 'A'];
} else {
SpellingNode *first_son = node_this->first_son;
SpellingNode* first_son = node_this->first_son;
// Because for Zh/Ch/Sh nodes, they are the last in the buffer and
// frequently used, so we scan from the end.
for (int i = 0; i < node_this->num_of_son; i++) {
SpellingNode *this_son = first_son + i;
SpellingNode* this_son = first_son + i;
if (SpellingTrie::is_same_spl_char(this_son->char_this_node, char_this)) {
found_son = this_son;
break;
@@ -124,7 +124,7 @@ uint16 SpellingParser::splstr_to_idxs(const char *splstr, uint16 str_len, uint16
return idx_num;
}
uint16 SpellingParser::splstr_to_idxs_f(const char *splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre) {
uint16 SpellingParser::splstr_to_idxs_f(const char* splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre) {
uint16 idx_num = splstr_to_idxs(splstr, str_len, spl_idx, start_pos, max_size, last_is_pre);
for (uint16 pos = 0; pos < idx_num; pos++) {
if (spl_trie_->is_half_id_yunmu(spl_idx[pos])) {
@@ -137,14 +137,14 @@ uint16 SpellingParser::splstr_to_idxs_f(const char *splstr, uint16 str_len, uint
return idx_num;
}
uint16 SpellingParser::splstr16_to_idxs(const char16 *splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre) {
uint16 SpellingParser::splstr16_to_idxs(const char16* splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre) {
if (NULL == splstr || 0 == max_size || 0 == str_len) return 0;
if (!SpellingTrie::is_valid_spl_char(splstr[0])) return 0;
last_is_pre = false;
const SpellingNode *node_this = spl_trie_->root_;
const SpellingNode* node_this = spl_trie_->root_;
uint16 str_pos = 0;
uint16 idx_num = 0;
@@ -181,7 +181,7 @@ uint16 SpellingParser::splstr16_to_idxs(const char16 *splstr, uint16 str_len, ui
last_is_splitter = false;
SpellingNode *found_son = NULL;
SpellingNode* found_son = NULL;
if (0 == str_pos) {
if (char_this >= 'a')
@@ -189,11 +189,11 @@ uint16 SpellingParser::splstr16_to_idxs(const char16 *splstr, uint16 str_len, ui
else
found_son = spl_trie_->level1_sons_[char_this - 'A'];
} else {
SpellingNode *first_son = node_this->first_son;
SpellingNode* first_son = node_this->first_son;
// Because for Zh/Ch/Sh nodes, they are the last in the buffer and
// frequently used, so we scan from the end.
for (int i = 0; i < node_this->num_of_son; i++) {
SpellingNode *this_son = first_son + i;
SpellingNode* this_son = first_son + i;
if (SpellingTrie::is_same_spl_char(this_son->char_this_node, char_this)) {
found_son = this_son;
break;
@@ -238,7 +238,7 @@ uint16 SpellingParser::splstr16_to_idxs(const char16 *splstr, uint16 str_len, ui
return idx_num;
}
uint16 SpellingParser::splstr16_to_idxs_f(const char16 *splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool &last_is_pre) {
uint16 SpellingParser::splstr16_to_idxs_f(const char16* splstr, uint16 str_len, uint16 spl_idx[], uint16 start_pos[], uint16 max_size, bool& last_is_pre) {
uint16 idx_num = splstr16_to_idxs(splstr, str_len, spl_idx, start_pos, max_size, last_is_pre);
for (uint16 pos = 0; pos < idx_num; pos++) {
if (spl_trie_->is_half_id_yunmu(spl_idx[pos])) {
@@ -251,7 +251,7 @@ uint16 SpellingParser::splstr16_to_idxs_f(const char16 *splstr, uint16 str_len,
return idx_num;
}
uint16 SpellingParser::get_splid_by_str(const char *splstr, uint16 str_len, bool *is_pre) {
uint16 SpellingParser::get_splid_by_str(const char* splstr, uint16 str_len, bool* is_pre) {
if (NULL == is_pre) return 0;
uint16 spl_idx[2];
@@ -263,7 +263,7 @@ uint16 SpellingParser::get_splid_by_str(const char *splstr, uint16 str_len, bool
return spl_idx[0];
}
uint16 SpellingParser::get_splid_by_str_f(const char *splstr, uint16 str_len, bool *is_pre) {
uint16 SpellingParser::get_splid_by_str_f(const char* splstr, uint16 str_len, bool* is_pre) {
if (NULL == is_pre) return 0;
uint16 spl_idx[2];
@@ -280,7 +280,7 @@ uint16 SpellingParser::get_splid_by_str_f(const char *splstr, uint16 str_len, bo
return spl_idx[0];
}
uint16 SpellingParser::get_splids_parallel(const char *splstr, uint16 str_len, uint16 splidx[], uint16 max_size, uint16 &full_id_num, bool &is_pre) {
uint16 SpellingParser::get_splids_parallel(const char* splstr, uint16 str_len, uint16 splidx[], uint16 max_size, uint16& full_id_num, bool& is_pre) {
if (max_size <= 0 || !is_valid_to_parse(splstr[0])) return 0;
splidx[0] = get_splid_by_str(splstr, str_len, &is_pre);
+104 -104
View File
@@ -42,7 +42,7 @@
namespace ime_pinyin {
#ifdef _WIN32
static int gettimeofday(struct timeval *tp, void *) {
static int gettimeofday(struct timeval* tp, void*) {
if (!tp) {
return -1;
}
@@ -101,7 +101,7 @@ static pthread_mutex_t g_mutex_ = PTHREAD_MUTEX_INITIALIZER;
#endif
static struct timeval g_last_update_ = {0, 0};
inline uint32 UserDict::get_dict_file_size(UserDictInfo *info) {
inline uint32 UserDict::get_dict_file_size(UserDictInfo* info) {
return (4 + info->lemma_size + (info->lemma_count << 3)
#ifdef ___PREDICT_ENABLED___
+ (info->lemma_count << 2)
@@ -163,12 +163,12 @@ inline int UserDict::build_score(uint64 lmt, int freq) {
return s;
}
inline int64 UserDict::utf16le_atoll(uint16 *s, int len) {
inline int64 UserDict::utf16le_atoll(uint16* s, int len) {
int64 ret = 0;
if (len <= 0) return ret;
int flag = 1;
const uint16 *endp = s + len;
const uint16* endp = s + len;
if (*s == '-') {
flag = -1;
s++;
@@ -183,9 +183,9 @@ inline int64 UserDict::utf16le_atoll(uint16 *s, int len) {
return ret * flag;
}
inline int UserDict::utf16le_lltoa(int64 v, uint16 *s, int size) {
inline int UserDict::utf16le_lltoa(int64 v, uint16* s, int size) {
if (!s || size <= 0) return 0;
uint16 *endp = s + size;
uint16* endp = s + size;
int ret_len = 0;
if (v < 0) {
*(s++) = '-';
@@ -193,7 +193,7 @@ inline int UserDict::utf16le_lltoa(int64 v, uint16 *s, int size) {
v *= -1;
}
uint16 *b = s;
uint16* b = s;
while (s < endp && v != 0) {
*(s++) = '0' + (v % 10);
v = v / 10;
@@ -227,15 +227,15 @@ inline char UserDict::get_lemma_nchar(uint32 offset) {
return (char)(lemmas_[offset + 1]);
}
inline uint16 *UserDict::get_lemma_spell_ids(uint32 offset) {
inline uint16* UserDict::get_lemma_spell_ids(uint32 offset) {
offset &= kUserDictOffsetMask;
return (uint16 *)(lemmas_ + offset + 2);
return (uint16*)(lemmas_ + offset + 2);
}
inline uint16 *UserDict::get_lemma_word(uint32 offset) {
inline uint16* UserDict::get_lemma_word(uint32 offset) {
offset &= kUserDictOffsetMask;
uint8 nchar = get_lemma_nchar(offset);
return (uint16 *)(lemmas_ + offset + 2 + (nchar << 1));
return (uint16*)(lemmas_ + offset + 2 + (nchar << 1));
}
inline LemmaIdType UserDict::get_max_lemma_id() {
@@ -282,7 +282,7 @@ UserDict::UserDict()
UserDict::~UserDict() { close_dict(); }
bool UserDict::load_dict(const char *file_name, LemmaIdType start_id, LemmaIdType end_id) {
bool UserDict::load_dict(const char* file_name, LemmaIdType start_id, LemmaIdType end_id) {
#ifdef ___DEBUG_PERF___
DEBUG_PERF_BEGIN;
#endif
@@ -308,7 +308,7 @@ bool UserDict::load_dict(const char *file_name, LemmaIdType start_id, LemmaIdTyp
#endif
return true;
error:
free((void *)dict_file_);
free((void*)dict_file_);
dict_file_ = NULL;
start_id_ = 0;
return false;
@@ -330,7 +330,7 @@ bool UserDict::close_dict() {
pthread_mutex_unlock(&g_mutex_);
out:
free((void *)dict_file_);
free((void*)dict_file_);
free(lemmas_);
free(offsets_);
free(offsets_by_id_);
@@ -367,7 +367,7 @@ size_t UserDict::number_of_lemmas() { return dict_info_.lemma_count; }
void UserDict::reset_milestones(uint16 from_step, MileStoneHandle from_handle) { return; }
MileStoneHandle UserDict::extend_dict(MileStoneHandle from_handle, const DictExtPara *dep, LmaPsbItem *lpi_items, size_t lpi_max, size_t *lpi_num) {
MileStoneHandle UserDict::extend_dict(MileStoneHandle from_handle, const DictExtPara* dep, LmaPsbItem* lpi_items, size_t lpi_max, size_t* lpi_num) {
if (is_valid_state() == false) return 0;
bool need_extend = false;
@@ -383,10 +383,10 @@ MileStoneHandle UserDict::extend_dict(MileStoneHandle from_handle, const DictExt
return ((*lpi_num > 0 || need_extend) ? 1 : 0);
}
int UserDict::is_fuzzy_prefix_spell_id(const uint16 *id1, uint16 len1, const UserDictSearchable *searchable) {
int UserDict::is_fuzzy_prefix_spell_id(const uint16* id1, uint16 len1, const UserDictSearchable* searchable) {
if (len1 < searchable->splids_len) return 0;
SpellingTrie &spl_trie = SpellingTrie::get_instance();
SpellingTrie& spl_trie = SpellingTrie::get_instance();
uint32 i = 0;
for (i = 0; i < searchable->splids_len; i++) {
const char py1 = *spl_trie.get_spelling_str(id1[i]);
@@ -398,11 +398,11 @@ int UserDict::is_fuzzy_prefix_spell_id(const uint16 *id1, uint16 len1, const Use
return 1;
}
int UserDict::fuzzy_compare_spell_id(const uint16 *id1, uint16 len1, const UserDictSearchable *searchable) {
int UserDict::fuzzy_compare_spell_id(const uint16* id1, uint16 len1, const UserDictSearchable* searchable) {
if (len1 < searchable->splids_len) return -1;
if (len1 > searchable->splids_len) return 1;
SpellingTrie &spl_trie = SpellingTrie::get_instance();
SpellingTrie& spl_trie = SpellingTrie::get_instance();
uint32 i = 0;
for (i = 0; i < len1; i++) {
const char py1 = *spl_trie.get_spelling_str(id1[i]);
@@ -415,7 +415,7 @@ int UserDict::fuzzy_compare_spell_id(const uint16 *id1, uint16 len1, const UserD
return 0;
}
bool UserDict::is_prefix_spell_id(const uint16 *fullids, uint16 fulllen, const UserDictSearchable *searchable) {
bool UserDict::is_prefix_spell_id(const uint16* fullids, uint16 fulllen, const UserDictSearchable* searchable) {
if (fulllen < searchable->splids_len) return false;
uint32 i = 0;
@@ -430,7 +430,7 @@ bool UserDict::is_prefix_spell_id(const uint16 *fullids, uint16 fulllen, const U
return true;
}
bool UserDict::equal_spell_id(const uint16 *fullids, uint16 fulllen, const UserDictSearchable *searchable) {
bool UserDict::equal_spell_id(const uint16* fullids, uint16 fulllen, const UserDictSearchable* searchable) {
if (fulllen != searchable->splids_len) return false;
uint32 i = 0;
@@ -445,7 +445,7 @@ bool UserDict::equal_spell_id(const uint16 *fullids, uint16 fulllen, const UserD
return true;
}
int32 UserDict::locate_first_in_offsets(const UserDictSearchable *searchable) {
int32 UserDict::locate_first_in_offsets(const UserDictSearchable* searchable) {
int32 begin = 0;
int32 end = dict_info_.lemma_count - 1;
int32 middle = -1;
@@ -457,7 +457,7 @@ int32 UserDict::locate_first_in_offsets(const UserDictSearchable *searchable) {
middle = (begin + end) >> 1;
uint32 offset = offsets_[middle];
uint8 nchar = get_lemma_nchar(offset);
const uint16 *splids = get_lemma_spell_ids(offset);
const uint16* splids = get_lemma_spell_ids(offset);
int cmp = fuzzy_compare_spell_id(splids, nchar, searchable);
int pre = is_fuzzy_prefix_spell_id(splids, nchar, searchable);
@@ -476,11 +476,11 @@ int32 UserDict::locate_first_in_offsets(const UserDictSearchable *searchable) {
return first_prefix;
}
void UserDict::prepare_locate(UserDictSearchable *searchable, const uint16 *splid_str, uint16 splid_str_len) {
void UserDict::prepare_locate(UserDictSearchable* searchable, const uint16* splid_str, uint16 splid_str_len) {
searchable->splids_len = splid_str_len;
memset(searchable->signature, 0, sizeof(searchable->signature));
SpellingTrie &spl_trie = SpellingTrie::get_instance();
SpellingTrie& spl_trie = SpellingTrie::get_instance();
uint32 i = 0;
for (; i < splid_str_len; i++) {
if (spl_trie.is_half_id(splid_str[i])) {
@@ -494,9 +494,9 @@ void UserDict::prepare_locate(UserDictSearchable *searchable, const uint16 *spli
}
}
size_t UserDict::get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lpi_items, size_t lpi_max) { return _get_lpis(splid_str, splid_str_len, lpi_items, lpi_max, NULL); }
size_t UserDict::get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lpi_items, size_t lpi_max) { return _get_lpis(splid_str, splid_str_len, lpi_items, lpi_max, NULL); }
size_t UserDict::_get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsbItem *lpi_items, size_t lpi_max, bool *need_extend) {
size_t UserDict::_get_lpis(const uint16* splid_str, uint16 splid_str_len, LmaPsbItem* lpi_items, size_t lpi_max, bool* need_extend) {
bool tmp_extend;
if (!need_extend) need_extend = &tmp_extend;
@@ -555,7 +555,7 @@ size_t UserDict::_get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsb
continue;
}
uint8 nchar = get_lemma_nchar(offset);
uint16 *splids = get_lemma_spell_ids(offset);
uint16* splids = get_lemma_spell_ids(offset);
#ifdef ___CACHE_ENABLED___
if (!cached && 0 != fuzzy_compare_spell_id(splids, nchar, &searchable)) {
#else
@@ -593,12 +593,12 @@ size_t UserDict::_get_lpis(const uint16 *splid_str, uint16 splid_str_len, LmaPsb
return lpi_current;
}
uint16 UserDict::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str_max) {
uint16 UserDict::get_lemma_str(LemmaIdType id_lemma, char16* str_buf, uint16 str_max) {
if (is_valid_state() == false) return 0;
if (is_valid_lemma_id(id_lemma) == false) return 0;
uint32 offset = offsets_by_id_[id_lemma - start_id_];
uint8 nchar = get_lemma_nchar(offset);
char16 *str = get_lemma_word(offset);
char16* str = get_lemma_word(offset);
uint16 m = nchar < str_max - 1 ? nchar : str_max - 1;
int i = 0;
for (; i < m; i++) {
@@ -608,21 +608,21 @@ uint16 UserDict::get_lemma_str(LemmaIdType id_lemma, char16 *str_buf, uint16 str
return m;
}
uint16 UserDict::get_lemma_splids(LemmaIdType id_lemma, uint16 *splids, uint16 splids_max, bool arg_valid) {
uint16 UserDict::get_lemma_splids(LemmaIdType id_lemma, uint16* splids, uint16 splids_max, bool arg_valid) {
if (is_valid_lemma_id(id_lemma) == false) return 0;
uint32 offset = offsets_by_id_[id_lemma - start_id_];
uint8 nchar = get_lemma_nchar(offset);
const uint16 *ids = get_lemma_spell_ids(offset);
const uint16* ids = get_lemma_spell_ids(offset);
int i = 0;
for (; i < nchar && i < splids_max; i++) splids[i] = ids[i];
return i;
}
size_t UserDict::predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *npre_items, size_t npre_max, size_t b4_used) {
size_t UserDict::predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem* npre_items, size_t npre_max, size_t b4_used) {
uint32 new_added = 0;
#ifdef ___PREDICT_ENABLED___
int32 end = dict_info_.lemma_count - 1;
int j = locate_first_in_predicts((const uint16 *)last_hzs, hzs_len);
int j = locate_first_in_predicts((const uint16*)last_hzs, hzs_len);
if (j == -1) return 0;
while (j <= end) {
@@ -633,8 +633,8 @@ size_t UserDict::predict(const char16 last_hzs[], uint16 hzs_len, NPredictItem *
continue;
}
uint32 nchar = get_lemma_nchar(offset);
uint16 *words = get_lemma_word(offset);
uint16 *splids = get_lemma_spell_ids(offset);
uint16* words = get_lemma_word(offset);
uint16* splids = get_lemma_spell_ids(offset);
if (nchar <= hzs_len) {
j++;
@@ -693,14 +693,14 @@ int32 UserDict::locate_in_offsets(char16 lemma_str[], uint16 splid_str[], uint16
off++;
continue;
}
uint16 *splids = get_lemma_spell_ids(offset);
uint16* splids = get_lemma_spell_ids(offset);
#ifdef ___CACHE_ENABLED___
if (!cached && 0 != fuzzy_compare_spell_id(splids, lemma_len, &searchable)) break;
#else
if (0 != fuzzy_compare_spell_id(splids, lemma_len, &searchable)) break;
#endif
if (equal_spell_id(splids, lemma_len, &searchable) == true) {
uint16 *str = get_lemma_word(offset);
uint16* str = get_lemma_word(offset);
uint32 i = 0;
for (i = 0; i < lemma_len; i++) {
if (str[i] == lemma_str[i]) continue;
@@ -728,7 +728,7 @@ int32 UserDict::locate_in_offsets(char16 lemma_str[], uint16 splid_str[], uint16
}
#ifdef ___PREDICT_ENABLED___
uint32 UserDict::locate_where_to_insert_in_predicts(const uint16 *words, int lemma_len) {
uint32 UserDict::locate_where_to_insert_in_predicts(const uint16* words, int lemma_len) {
int32 begin = 0;
int32 end = dict_info_.lemma_count - 1;
int32 middle = end;
@@ -739,7 +739,7 @@ uint32 UserDict::locate_where_to_insert_in_predicts(const uint16 *words, int lem
middle = (begin + end) >> 1;
uint32 offset = offsets_[middle];
uint8 nchar = get_lemma_nchar(offset);
const uint16 *ws = get_lemma_word(offset);
const uint16* ws = get_lemma_word(offset);
uint32 minl = nchar < lemma_len ? nchar : lemma_len;
uint32 k = 0;
@@ -775,7 +775,7 @@ uint32 UserDict::locate_where_to_insert_in_predicts(const uint16 *words, int lem
return last_matched;
}
int32 UserDict::locate_first_in_predicts(const uint16 *words, int lemma_len) {
int32 UserDict::locate_first_in_predicts(const uint16* words, int lemma_len) {
int32 begin = 0;
int32 end = dict_info_.lemma_count - 1;
int32 middle = -1;
@@ -786,7 +786,7 @@ int32 UserDict::locate_first_in_predicts(const uint16 *words, int lemma_len) {
middle = (begin + end) >> 1;
uint32 offset = offsets_[middle];
uint8 nchar = get_lemma_nchar(offset);
const uint16 *ws = get_lemma_word(offset);
const uint16* ws = get_lemma_word(offset);
uint32 minl = nchar < lemma_len ? nchar : lemma_len;
uint32 k = 0;
@@ -851,8 +851,8 @@ int UserDict::_get_lemma_score(LemmaIdType lemma_id) {
uint32 offset = offsets_by_id_[lemma_id - start_id_];
uint32 nchar = get_lemma_nchar(offset);
uint16 *spl = get_lemma_spell_ids(offset);
uint16 *wrd = get_lemma_word(offset);
uint16* spl = get_lemma_spell_ids(offset);
uint16* wrd = get_lemma_word(offset);
int32 off = locate_in_offsets(wrd, spl, nchar);
if (off == -1) {
@@ -936,8 +936,8 @@ bool UserDict::remove_lemma(LemmaIdType lemma_id) {
uint32 offset = offsets_by_id_[lemma_id - start_id_];
uint32 nchar = get_lemma_nchar(offset);
uint16 *spl = get_lemma_spell_ids(offset);
uint16 *wrd = get_lemma_word(offset);
uint16* spl = get_lemma_spell_ids(offset);
uint16* wrd = get_lemma_word(offset);
int32 off = locate_in_offsets(wrd, spl, nchar);
@@ -947,19 +947,19 @@ bool UserDict::remove_lemma(LemmaIdType lemma_id) {
void UserDict::flush_cache() {
LemmaIdType start_id = start_id_;
if (!dict_file_) return;
const char *file = strdup(dict_file_);
const char* file = strdup(dict_file_);
if (!file) return;
close_dict();
load_dict(file, start_id, kUserDictIdEnd);
free((void *)file);
free((void*)file);
#ifdef ___CACHE_ENABLED___
cache_init();
#endif
return;
}
bool UserDict::reset(const char *file) {
FILE *fp = fopen(file, "w+");
bool UserDict::reset(const char* file) {
FILE* fp = fopen(file, "w+");
if (!fp) {
return false;
}
@@ -979,10 +979,10 @@ bool UserDict::reset(const char *file) {
return true;
}
bool UserDict::validate(const char *file) {
bool UserDict::validate(const char* file) {
// b is ignored in POSIX compatible os including Linux
// while b is important flag for Windows to specify binary mode
FILE *fp = fopen(file, "rb");
FILE* fp = fopen(file, "rb");
if (!fp) {
return false;
}
@@ -1038,13 +1038,13 @@ error:
return false;
}
bool UserDict::load(const char *file, LemmaIdType start_id) {
bool UserDict::load(const char* file, LemmaIdType start_id) {
if (0 != pthread_mutex_trylock(&g_mutex_)) {
return false;
}
// b is ignored in POSIX compatible os including Linux
// while b is important flag for Windows to specify binary mode
FILE *fp = fopen(file, "rb");
FILE* fp = fopen(file, "rb");
if (!fp) {
pthread_mutex_unlock(&g_mutex_);
return false;
@@ -1052,16 +1052,16 @@ bool UserDict::load(const char *file, LemmaIdType start_id) {
size_t readed, toread;
UserDictInfo dict_info;
uint8 *lemmas = NULL;
uint32 *offsets = NULL;
uint8* lemmas = NULL;
uint32* offsets = NULL;
#ifdef ___SYNC_ENABLED___
uint32 *syncs = NULL;
uint32* syncs = NULL;
#endif
uint32 *scores = NULL;
uint32 *ids = NULL;
uint32 *offsets_by_id = NULL;
uint32* scores = NULL;
uint32* ids = NULL;
uint32* offsets_by_id = NULL;
#ifdef ___PREDICT_ENABLED___
uint32 *predicts = NULL;
uint32* predicts = NULL;
#endif
size_t i;
int err;
@@ -1072,30 +1072,30 @@ bool UserDict::load(const char *file, LemmaIdType start_id) {
readed = fread(&dict_info, 1, sizeof(dict_info), fp);
if (readed != sizeof(dict_info)) goto error;
lemmas = (uint8 *)malloc(dict_info.lemma_size + (kUserDictPreAlloc * (2 + (kUserDictAverageNchar << 2))));
lemmas = (uint8*)malloc(dict_info.lemma_size + (kUserDictPreAlloc * (2 + (kUserDictAverageNchar << 2))));
if (!lemmas) goto error;
offsets = (uint32 *)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
offsets = (uint32*)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
if (!offsets) goto error;
#ifdef ___PREDICT_ENABLED___
predicts = (uint32 *)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
predicts = (uint32*)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
if (!predicts) goto error;
#endif
#ifdef ___SYNC_ENABLED___
syncs = (uint32 *)malloc((dict_info.sync_count + kUserDictPreAlloc) << 2);
syncs = (uint32*)malloc((dict_info.sync_count + kUserDictPreAlloc) << 2);
if (!syncs) goto error;
#endif
scores = (uint32 *)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
scores = (uint32*)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
if (!scores) goto error;
ids = (uint32 *)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
ids = (uint32*)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
if (!ids) goto error;
offsets_by_id = (uint32 *)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
offsets_by_id = (uint32*)malloc((dict_info.lemma_count + kUserDictPreAlloc) << 2);
if (!offsets_by_id) goto error;
err = fseek(fp, 4, SEEK_SET);
@@ -1110,7 +1110,7 @@ bool UserDict::load(const char *file, LemmaIdType start_id) {
toread = (dict_info.lemma_count << 2);
readed = 0;
while (readed < toread && !ferror(fp) && !feof(fp)) {
readed += fread((((uint8 *)offsets) + readed), 1, toread - readed, fp);
readed += fread((((uint8*)offsets) + readed), 1, toread - readed, fp);
}
if (readed < toread) goto error;
@@ -1118,14 +1118,14 @@ bool UserDict::load(const char *file, LemmaIdType start_id) {
toread = (dict_info.lemma_count << 2);
readed = 0;
while (readed < toread && !ferror(fp) && !feof(fp)) {
readed += fread((((uint8 *)predicts) + readed), 1, toread - readed, fp);
readed += fread((((uint8*)predicts) + readed), 1, toread - readed, fp);
}
if (readed < toread) goto error;
#endif
readed = 0;
while (readed < toread && !ferror(fp) && !feof(fp)) {
readed += fread((((uint8 *)scores) + readed), 1, toread - readed, fp);
readed += fread((((uint8*)scores) + readed), 1, toread - readed, fp);
}
if (readed < toread) goto error;
@@ -1133,7 +1133,7 @@ bool UserDict::load(const char *file, LemmaIdType start_id) {
toread = (dict_info.sync_count << 2);
readed = 0;
while (readed < toread && !ferror(fp) && !feof(fp)) {
readed += fread((((uint8 *)syncs) + readed), 1, toread - readed, fp);
readed += fread((((uint8*)syncs) + readed), 1, toread - readed, fp);
}
if (readed < toread) goto error;
#endif
@@ -1302,8 +1302,8 @@ void UserDict::write_back_all(int fd) {
}
#ifdef ___CACHE_ENABLED___
bool UserDict::load_cache(UserDictSearchable *searchable, uint32 *offset, uint32 *length) {
UserDictCache *cache = &caches_[searchable->splids_len - 1];
bool UserDict::load_cache(UserDictSearchable* searchable, uint32* offset, uint32* length) {
UserDictCache* cache = &caches_[searchable->splids_len - 1];
if (cache->head == cache->tail) return false;
uint16 j, sig_len = kMaxLemmaSize / 4;
@@ -1326,8 +1326,8 @@ bool UserDict::load_cache(UserDictSearchable *searchable, uint32 *offset, uint32
return false;
}
void UserDict::save_cache(UserDictSearchable *searchable, uint32 offset, uint32 length) {
UserDictCache *cache = &caches_[searchable->splids_len - 1];
void UserDict::save_cache(UserDictSearchable* searchable, uint32 offset, uint32 length) {
UserDictCache* cache = &caches_[searchable->splids_len - 1];
uint16 next = cache->tail;
cache->offsets[next] = offset;
@@ -1352,8 +1352,8 @@ void UserDict::save_cache(UserDictSearchable *searchable, uint32 offset, uint32
void UserDict::reset_cache() { memset(caches_, 0, sizeof(caches_)); }
bool UserDict::load_miss_cache(UserDictSearchable *searchable) {
UserDictMissCache *cache = &miss_caches_[searchable->splids_len - 1];
bool UserDict::load_miss_cache(UserDictSearchable* searchable) {
UserDictMissCache* cache = &miss_caches_[searchable->splids_len - 1];
if (cache->head == cache->tail) return false;
uint16 j, sig_len = kMaxLemmaSize / 4;
@@ -1374,8 +1374,8 @@ bool UserDict::load_miss_cache(UserDictSearchable *searchable) {
return false;
}
void UserDict::save_miss_cache(UserDictSearchable *searchable) {
UserDictMissCache *cache = &miss_caches_[searchable->splids_len - 1];
void UserDict::save_miss_cache(UserDictSearchable* searchable) {
UserDictMissCache* cache = &miss_caches_[searchable->splids_len - 1];
uint16 next = cache->tail;
uint16 sig_len = kMaxLemmaSize / 4;
@@ -1403,7 +1403,7 @@ void UserDict::cache_init() {
reset_miss_cache();
}
bool UserDict::cache_hit(UserDictSearchable *searchable, uint32 *offset, uint32 *length) {
bool UserDict::cache_hit(UserDictSearchable* searchable, uint32* offset, uint32* length) {
bool hit = load_miss_cache(searchable);
if (hit) {
*offset = 0;
@@ -1417,7 +1417,7 @@ bool UserDict::cache_hit(UserDictSearchable *searchable, uint32 *offset, uint32
return false;
}
void UserDict::cache_push(UserDictCacheType type, UserDictSearchable *searchable, uint32 offset, uint32 length) {
void UserDict::cache_push(UserDictCacheType type, UserDictSearchable* searchable, uint32 offset, uint32 length) {
switch (type) {
case USER_DICT_MISS_CACHE:
save_miss_cache(searchable);
@@ -1614,7 +1614,7 @@ LemmaIdType UserDict::put_lemma_no_sync(char16 lemma_str[], uint16 splids[], uin
int again = 0;
begin:
LemmaIdType id;
uint32 *syncs_bak = syncs_;
uint32* syncs_bak = syncs_;
syncs_ = NULL;
id = _put_lemma(lemma_str, splids, lemma_len, count, lmt);
syncs_ = syncs_bak;
@@ -1632,26 +1632,26 @@ begin:
return id;
}
int UserDict::put_lemmas_no_sync_from_utf16le_string(char16 *lemmas, int len) {
int UserDict::put_lemmas_no_sync_from_utf16le_string(char16* lemmas, int len) {
int newly_added = 0;
SpellingParser *spl_parser = new SpellingParser();
SpellingParser* spl_parser = new SpellingParser();
if (!spl_parser) {
return 0;
}
#ifdef ___DEBUG_PERF___
DEBUG_PERF_BEGIN;
#endif
char16 *ptr = lemmas;
char16* ptr = lemmas;
// Extract pinyin,words,frequence,last_mod_time
char16 *p = ptr, *py16 = ptr;
char16 *hz16 = NULL;
char16* hz16 = NULL;
int py16_len = 0;
uint16 splid[kMaxLemmaSize];
int splid_len = 0;
int hz16_len = 0;
char16 *fr16 = NULL;
char16* fr16 = NULL;
int fr16_len = 0;
while (p - ptr < len) {
@@ -1708,7 +1708,7 @@ int UserDict::put_lemmas_no_sync_from_utf16le_string(char16 *lemmas, int len) {
return newly_added;
}
int UserDict::get_sync_lemmas_in_utf16le_string_from_beginning(char16 *str, int size, int *count) {
int UserDict::get_sync_lemmas_in_utf16le_string_from_beginning(char16* str, int size, int* count) {
int len = 0;
*count = 0;
@@ -1716,7 +1716,7 @@ int UserDict::get_sync_lemmas_in_utf16le_string_from_beginning(char16 *str, int
if (is_valid_state() == false) return len;
SpellingTrie *spl_trie = &SpellingTrie::get_instance();
SpellingTrie* spl_trie = &SpellingTrie::get_instance();
if (!spl_trie) {
return 0;
}
@@ -1725,8 +1725,8 @@ int UserDict::get_sync_lemmas_in_utf16le_string_from_beginning(char16 *str, int
for (i = 0; i < dict_info_.sync_count; i++) {
int offset = syncs_[i];
uint32 nchar = get_lemma_nchar(offset);
uint16 *spl = get_lemma_spell_ids(offset);
uint16 *wrd = get_lemma_word(offset);
uint16* spl = get_lemma_spell_ids(offset);
uint16* wrd = get_lemma_word(offset);
int score = _get_lemma_score(wrd, spl, nchar);
static char score_temp[32], *pscore_temp = score_temp;
@@ -1812,7 +1812,7 @@ int UserDict::get_sync_lemmas_in_utf16le_string_from_beginning(char16 *str, int
#endif
bool UserDict::state(UserDictStat *stat) {
bool UserDict::state(UserDictStat* stat) {
if (is_valid_state() == false) return false;
if (!stat) return false;
stat->version = version_;
@@ -1862,8 +1862,8 @@ void UserDict::reclaim() {
uint32 count = dict_info_.lemma_count;
int rc = count * dict_info_.reclaim_ratio / 100;
UserDictScoreOffsetPair *score_offset_pairs = NULL;
score_offset_pairs = (UserDictScoreOffsetPair *)malloc(sizeof(UserDictScoreOffsetPair) * rc);
UserDictScoreOffsetPair* score_offset_pairs = NULL;
score_offset_pairs = (UserDictScoreOffsetPair*)malloc(sizeof(UserDictScoreOffsetPair) * rc);
if (score_offset_pairs == NULL) {
return;
}
@@ -1896,7 +1896,7 @@ void UserDict::reclaim() {
free(score_offset_pairs);
}
inline void UserDict::swap(UserDictScoreOffsetPair *sop, int i, int j) {
inline void UserDict::swap(UserDictScoreOffsetPair* sop, int i, int j) {
int s = sop[i].score;
int p = sop[i].offset_index;
sop[i].score = sop[j].score;
@@ -1905,7 +1905,7 @@ inline void UserDict::swap(UserDictScoreOffsetPair *sop, int i, int j) {
sop[j].offset_index = p;
}
void UserDict::shift_down(UserDictScoreOffsetPair *sop, int i, int n) {
void UserDict::shift_down(UserDictScoreOffsetPair* sop, int i, int n) {
int par = i;
while (par < n) {
int left = par * 2 + 1;
@@ -1983,7 +1983,7 @@ void UserDict::queue_lemma_for_sync(LemmaIdType id) {
if (dict_info_.sync_count < sync_count_size_) {
syncs_[dict_info_.sync_count++] = offsets_by_id_[id - start_id_];
} else {
uint32 *syncs = (uint32 *)realloc(syncs_, (sync_count_size_ + kUserDictPreAlloc) << 2);
uint32* syncs = (uint32*)realloc(syncs_, (sync_count_size_ + kUserDictPreAlloc) << 2);
if (syncs) {
sync_count_size_ += kUserDictPreAlloc;
syncs_ = syncs;
@@ -2001,8 +2001,8 @@ LemmaIdType UserDict::update_lemma(LemmaIdType lemma_id, int16 delta_count, bool
if (is_valid_lemma_id(lemma_id) == false) return 0;
uint32 offset = offsets_by_id_[lemma_id - start_id_];
uint8 lemma_len = get_lemma_nchar(offset);
char16 *lemma_str = get_lemma_word(offset);
uint16 *splids = get_lemma_spell_ids(offset);
char16* lemma_str = get_lemma_word(offset);
uint16* splids = get_lemma_spell_ids(offset);
int32 off = locate_in_offsets(lemma_str, splids, lemma_len);
if (off != -1) {
@@ -2043,8 +2043,8 @@ LemmaIdType UserDict::append_a_lemma(char16 lemma_str[], uint16 splids[], uint16
lemmas_[offset] = 0;
lemmas_[offset + 1] = (uint8)lemma_len;
for (size_t i = 0; i < lemma_len; i++) {
*((uint16 *)&lemmas_[offset + 2 + (i << 1)]) = splids[i];
*((char16 *)&lemmas_[offset + 2 + (lemma_len << 1) + (i << 1)]) = lemma_str[i];
*((uint16*)&lemmas_[offset + 2 + (i << 1)]) = splids[i];
*((char16*)&lemmas_[offset + 2 + (lemma_len << 1) + (i << 1)]) = lemma_str[i];
}
uint32 off = dict_info_.lemma_count;
offsets_[off] = offset;
@@ -2070,7 +2070,7 @@ LemmaIdType UserDict::append_a_lemma(char16 lemma_str[], uint16 splids[], uint16
while (i < off) {
offset = offsets_[i];
uint32 nchar = get_lemma_nchar(offset);
uint16 *spl = get_lemma_spell_ids(offset);
uint16* spl = get_lemma_spell_ids(offset);
if (0 <= fuzzy_compare_spell_id(spl, nchar, &searchable)) break;
i++;
@@ -2091,7 +2091,7 @@ LemmaIdType UserDict::append_a_lemma(char16 lemma_str[], uint16 splids[], uint16
#ifdef ___PREDICT_ENABLED___
uint32 j = 0;
uint16 *words_new = get_lemma_word(predicts_[off]);
uint16* words_new = get_lemma_word(predicts_[off]);
j = locate_where_to_insert_in_predicts(words_new, lemma_len);
if (j != off) {
uint32 temp = predicts_[off];
+13 -13
View File
@@ -23,7 +23,7 @@ namespace ime_pinyin {
extern "C" {
#endif
char16 *utf16_strtok(char16 *utf16_str, size_t *token_size, char16 **utf16_str_next) {
char16* utf16_strtok(char16* utf16_str, size_t* token_size, char16** utf16_str_next) {
if (NULL == utf16_str || NULL == token_size || NULL == utf16_str_next) {
return NULL;
}
@@ -39,7 +39,7 @@ char16 *utf16_strtok(char16 *utf16_str, size_t *token_size, char16 **utf16_str_n
pos++;
}
char16 *ret_val = utf16_str;
char16* ret_val = utf16_str;
if ((char16)'\0' == utf16_str[pos]) {
*utf16_str_next = NULL;
if (0 == pos) return NULL;
@@ -53,7 +53,7 @@ char16 *utf16_strtok(char16 *utf16_str, size_t *token_size, char16 **utf16_str_n
return ret_val;
}
int utf16_atoi(const char16 *utf16_str) {
int utf16_atoi(const char16* utf16_str) {
if (NULL == utf16_str) return 0;
int value = 0;
@@ -73,7 +73,7 @@ int utf16_atoi(const char16 *utf16_str) {
return value * sign;
}
float utf16_atof(const char16 *utf16_str) {
float utf16_atof(const char16* utf16_str) {
// A temporary implemetation.
char char8[256];
if (utf16_strlen(utf16_str) >= 256) return 0;
@@ -82,7 +82,7 @@ float utf16_atof(const char16 *utf16_str) {
return atof(char8);
}
size_t utf16_strlen(const char16 *utf16_str) {
size_t utf16_strlen(const char16* utf16_str) {
if (NULL == utf16_str) return 0;
size_t size = 0;
@@ -90,14 +90,14 @@ size_t utf16_strlen(const char16 *utf16_str) {
return size;
}
int utf16_strcmp(const char16 *str1, const char16 *str2) {
int utf16_strcmp(const char16* str1, const char16* str2) {
size_t pos = 0;
while (str1[pos] == str2[pos] && (char16)'\0' != str1[pos]) pos++;
return static_cast<int>(str1[pos]) - static_cast<int>(str2[pos]);
}
int utf16_strncmp(const char16 *str1, const char16 *str2, size_t size) {
int utf16_strncmp(const char16* str1, const char16* str2, size_t size) {
size_t pos = 0;
while (pos < size && str1[pos] == str2[pos] && (char16)'\0' != str1[pos]) pos++;
@@ -107,10 +107,10 @@ int utf16_strncmp(const char16 *str1, const char16 *str2, size_t size) {
}
// we do not consider overlapping
char16 *utf16_strcpy(char16 *dst, const char16 *src) {
char16* utf16_strcpy(char16* dst, const char16* src) {
if (NULL == src || NULL == dst) return NULL;
char16 *cp = dst;
char16* cp = dst;
while ((char16)'\0' != *src) {
*cp = *src;
@@ -123,12 +123,12 @@ char16 *utf16_strcpy(char16 *dst, const char16 *src) {
return dst;
}
char16 *utf16_strncpy(char16 *dst, const char16 *src, size_t size) {
char16* utf16_strncpy(char16* dst, const char16* src, size_t size) {
if (NULL == src || NULL == dst || 0 == size) return NULL;
if (src == dst) return dst;
char16 *cp = dst;
char16* cp = dst;
if (dst < src || (dst > src && dst >= src + size)) {
while (size-- && (*cp++ = *src++));
@@ -142,10 +142,10 @@ char16 *utf16_strncpy(char16 *dst, const char16 *src, size_t size) {
// We do not handle complicated cases like overlapping, because in this
// codebase, it is not necessary.
char *utf16_strcpy_tochar(char *dst, const char16 *src) {
char* utf16_strcpy_tochar(char* dst, const char16* src) {
if (NULL == src || NULL == dst) return NULL;
char *cp = dst;
char* cp = dst;
while ((char16)'\0' != *src) {
*cp = static_cast<char>(*src);
+2 -2
View File
@@ -35,7 +35,7 @@ Utf16Reader::~Utf16Reader() {
if (NULL != buffer_) delete[] buffer_;
}
bool Utf16Reader::open(const char *filename, size_t buffer_len) {
bool Utf16Reader::open(const char* filename, size_t buffer_len) {
if (filename == NULL) return false;
if (buffer_len < MIN_BUF_LEN)
@@ -62,7 +62,7 @@ bool Utf16Reader::open(const char *filename, size_t buffer_len) {
return true;
}
char16 *Utf16Reader::readline(char16 *read_buf, size_t max_len) {
char16* Utf16Reader::readline(char16* read_buf, size_t max_len) {
if (NULL == fp_ || NULL == read_buf || 0 == max_len) return NULL;
size_t ret_len = 0;
+7 -7
View File
@@ -7,8 +7,8 @@
#include <utf8cpp/utf8.h>
#include <Windows.h>
std::string fromUtf16(const ime_pinyin::char16 *buf, size_t len) {
const char16_t *utf16Data = reinterpret_cast<const char16_t *>(buf);
std::string fromUtf16(const ime_pinyin::char16* buf, size_t len) {
const char16_t* utf16Data = reinterpret_cast<const char16_t*>(buf);
std::u16string utf16Str(utf16Data, len);
std::string utf8Result;
@@ -17,14 +17,14 @@ std::string fromUtf16(const ime_pinyin::char16 *buf, size_t len) {
return utf8Result;
}
void test_pinyin_search_and_segment(const std::string &user_pinyin) {
void test_pinyin_search_and_segment(const std::string& user_pinyin) {
auto start_time = std::chrono::high_resolution_clock::now();
std::string pinyin_str = user_pinyin;
const char *pinyin = pinyin_str.c_str();
const char* pinyin = pinyin_str.c_str();
size_t cand_cnt = ime_pinyin::im_search(pinyin, strlen(pinyin));
const ime_pinyin::uint16 *spl_start = nullptr;
const ime_pinyin::uint16* spl_start = nullptr;
size_t segment_count = ime_pinyin::im_get_spl_start_pos(spl_start);
if (spl_start != nullptr && segment_count > 0) {
@@ -64,12 +64,12 @@ void test_pinyin_search_and_segment(const std::string &user_pinyin) {
std::cout << "========================================" << std::endl;
}
void test_pinyin_search_when_retriving_first_element(const std::string &user_pinyin) {
void test_pinyin_search_when_retriving_first_element(const std::string& user_pinyin) {
auto start_time = std::chrono::high_resolution_clock::now();
std::string pinyin_str = user_pinyin;
const char *pinyin = pinyin_str.c_str();
const char* pinyin = pinyin_str.c_str();
size_t cand_cnt = ime_pinyin::im_search(pinyin, strlen(pinyin));
std::string msg;
std::vector<std::string> candidateList;