From da7877336cc1dcd69eb32400202758a9becf24b4 Mon Sep 17 00:00:00 2001 From: nonstriater <510495266@qq.com> Date: Fri, 16 May 2014 20:11:50 +0800 Subject: [PATCH] =?UTF-8?q?trie=E6=A0=91?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- BAT iOS面试题.md | 2 +- 二叉树/1-二叉树 /btree/bintree.c | 349 +++++++++++++++++++++++++++++++ 二叉树/1-二叉树 /btree/btree | Bin 0 -> 10336 bytes 二叉树/1-二叉树 /btree/public.h | 34 +++ 二叉树/1-二叉树 /btree/队列.h | 175 ++++++++++++++++ 二叉树/4-字典树Trie/trie.c | 193 ++++++++++------- 二叉树/README.md | 256 +++++++++++++++++++++++ 排序算法/README.md | 8 - 算法问题选编/数值问题.md | 20 +- 算法问题选编/数组数列问题.md | 3 + 10 files changed, 952 insertions(+), 88 deletions(-) create mode 100644 二叉树/1-二叉树 /btree/bintree.c create mode 100755 二叉树/1-二叉树 /btree/btree create mode 100644 二叉树/1-二叉树 /btree/public.h create mode 100644 二叉树/1-二叉树 /btree/队列.h diff --git a/BAT iOS面试题.md b/BAT iOS面试题.md index 5bc42f9..8e5d410 100644 --- a/BAT iOS面试题.md +++ b/BAT iOS面试题.md @@ -55,7 +55,7 @@ http://archive.cnblogs.com/a/1886332/ ###数据处理 Core Data:中多线程中处理大量数据同步时的操作。 -缓存设计 +缓存设计LRU Cache sqlite中插入特殊字符的方法和接收到处理方法。 diff --git a/二叉树/1-二叉树 /btree/bintree.c b/二叉树/1-二叉树 /btree/bintree.c new file mode 100644 index 0000000..8297aa2 --- /dev/null +++ b/二叉树/1-二叉树 /btree/bintree.c @@ -0,0 +1,349 @@ +/** +* 第一次题目 :构建二叉树(链表存储方式),空格表示空树,实现二叉树的基本操作:创建,遍历(先,中,后,按层)二叉树 +* 提交邮箱 :510495266@qq.com to learn_algorithm@163.com +* 邮件题目 :第一期第一次作业 +**/ + +#include +#include + + +typedef struct BiTNode{ + + char item; + struct BiTNode *lChild,*rChild; + +}BiTNode,*BiTree; + + + + + +// ========================================================================== + + +#define OK 1 +#define ERROR 0 +typedef int bool; + + +typedef BiTree ElemType ; // 也可能是一个复杂的复合类型 +typedef int Status; + + + +//队列的顺序存储 +typedef struct Node{ + + ElemType *elem; + ElemType *head; + ElemType *tail; + int length; // 当前队列的长度 + int size; // 队列容器的长度,在队列慢得时候可以扩容 + + +}SqQueue; + +// +// 在队列采用顺序存储时,有一个毛病,就是队列操作一段时间后,头指针到了队列容器的尾部,而头指针前面的容器内存不可用了, +//造成内存极大的浪费,这个问题可以通过循环队列来解决。但是在 +// 链式队列上则不存在这样的问题 +// + + +typedef struct QNode{ + + struct QNode *next; + ElemType elem; + +}LinkQueue; + + +// 队列需要维护两个 指针 (队头指针,队尾指针) + +typedef struct{ + + LinkQueue *head; + LinkQueue *tail; + int length; + +}Queue; + + +Status initQueue(Queue *queue);// 带头结点,没有引用传值,就用指针的指针吧 +bool isEmpty(Queue *q); +int length(Queue *q); + + +// 在头部插入,尾部删除 +Status getHead(Queue *q,ElemType *e); +Status enQueue(Queue *q,ElemType e); +Status deQueue(Queue *q,ElemType *e); + +void traveser(Queue *q); + + +Status initQueue(Queue *queue){// 带头结点,没有引用传值,就用指针的指针吧 + + LinkQueue *lq = (LinkQueue *)malloc(sizeof(LinkQueue)); + if (!lq) { + return ERROR; + } + + lq->elem = NULL; + lq->next = NULL; + + (queue)->head = (queue)->tail = lq; // -> 优先级高于 * ,老老实实打上括号 + (queue)->length = 0; + + return OK; + +} + +bool isEmpty(Queue *q){ + + return (q->head == q->tail); + +} + +int length(Queue *q){ + +// int ret; +// LinkQueue *p = q->head; +// while (p != q->tail) { +// ret++; +// p = p->next; +// +// } + + return q->length; + +} + +Status getHead(Queue *q,ElemType *e){ + + LinkQueue *p = q->head->next; + if (!p) { + *e=NULL; + return ERROR; + } + + *e = p->elem; + return OK; + +} + +// 入队(加入到队尾) +Status enQueue(Queue *q,ElemType e){ + + LinkQueue *newNode = (LinkQueue *)malloc(sizeof(LinkQueue)); + if (!newNode) { + return ERROR; + } + newNode->elem = e; + newNode->next = NULL; + q->tail->next = newNode; + q->tail = newNode; + + q->length++; + + return OK; + +} + +//出队(队头). 注意,可能只有1个元素,造成队尾指针丢失 +Status deQueue(Queue *q,ElemType *e){ + + LinkQueue *p = q->head->next; + if (!p) {// 队列空 + return ERROR; + } + + if (e) { + *e = p->elem ; + } + + LinkQueue *temp = p->next; + q->head->next = temp; + if (p == q->tail) { + q->tail = q->head; + } + free(p); + + + q->length--; + + return OK; + +} + + +void traveser(Queue *q){ + + LinkQueue *p = q->head->next; + while (NULL != p) { + //printf("%d\n",p->elem); + p=p->next; + } + +} + + +// ================================================== + + + +/* +int CreateBiTree(BiTree *T) +{ + + *T = (BiTNode *)malloc(sizeof(BiTNode)); + if (*T==NULL) + { + printf("memery malloc failure !\n"); + return -1; + } + + printf("enter data to create node:\n"); + scanf("%c",&((*T)->item)); + if((*T)->item=='#'){ + *T=NULL; + } + if(*T){ + printf("创建左子树:\n"); + CreateBiTree( &((*T)->lChild) ); + printf("创建右子树:\n"); + CreateBiTree( &((*T)->rChild) ); + } + + return 0; +} +*/ + +// BiTree CreateBiTree() + +// 先序遍历方式创建二叉树 +BiTree CreateBiTree(){ + + char c; + BiTNode *tree; + + // 过滤回车键 + + + scanf("%c",&c); + if (c==' ') + { + printf("创建空节点\n"); + tree = NULL; + }else{ + printf("创建节点 %c\n",c); + tree = (BiTNode *)malloc(sizeof(BiTNode)); + tree->item = c; + tree->lChild = CreateBiTree(); + tree->rChild = CreateBiTree(); + } + + return tree; + +} + + +int PreOrderTraverse(BiTree T){ + + if(T){ + printf("%c\n",T->item ); + PreOrderTraverse(T->lChild); + PreOrderTraverse(T->rChild); + } + + return 0; +} + +int InOrderTraverse(BiTree T){ + + if (T) + { + InOrderTraverse(T->lChild); + printf("%c\n", T->item); + InOrderTraverse(T->rChild); + } + + return 0; +} + +int PostOrderTraverse(BiTree T){ + + if (T) + { + PostOrderTraverse(T->lChild); + PostOrderTraverse(T->rChild); + printf("%c\n",T->item ); + } + + return 0; +} + +// 广度优先遍历 (队列实现) +int LevelOrderTraverse(BiTree T){ + + if (T) + { + Queue queue; + initQueue(&queue); + + BiTree u; + u=(BiTree)malloc(sizeof(BiTNode)); + enQueue(&queue, T); + while(!isEmpty(&queue)) + { + deQueue(&queue, &u); + printf("%c",u->item); + if(u->lChild) + enQueue(&queue, u->lChild); + if(u->rChild) + enQueue(&queue, u->rChild); + } + } + + return 0; +} + + + + int main(int argc, char const *argv[]) +{ + /* code */ + BiTree binaryTree; + + printf("创建二叉树,输入\"空格\"创建空节点(先序方式建立二叉树):\n"); + binaryTree = CreateBiTree(); + if(binaryTree==NULL){ + printf("创建空的二叉树\n"); + } + + + printf("前序遍历:\n"); + PreOrderTraverse(binaryTree); + printf("\n\n"); + + printf("中序遍历:\n"); + InOrderTraverse(binaryTree); + printf("\n\n"); + + printf("后序遍历:\n"); + PostOrderTraverse(binaryTree); + printf("\n\n"); + + printf("层序遍历:\n"); + LevelOrderTraverse(binaryTree); + printf("\n\n"); + + + return 0; +} + + + + + diff --git a/二叉树/1-二叉树 /btree/btree b/二叉树/1-二叉树 /btree/btree new file mode 100755 index 0000000000000000000000000000000000000000..2ab2e85980bbdc3ffb20c6fe5d5389d6526e6716 GIT binary patch literal 10336 zcmeHNeQaA-6~C`zt6Mtkz`Az8sOj3N3=wAZMJmH&@@&(=i}vrI!>^i zE{PU3X_HacO&i$g#z0^J19hmPgNc3+m4sF#6G*G3O%yauL2BBUsndcqK`GtB?|0w5 z_Op{zllTWDU**1cfA^f~tKO8^5dhMML<_lrmBt*|VAw+=nW+D1T5G^6@Ks8Y< zt7^mM@MGbJA7XEQG4kQ8h;~-Nfn`;PAFa+;k$gJ#7KIU;Kse$SC9|xg-I2^G7TO!; zU?4uB2*e-BsT1^gp=BkK?X`()SE0R*kGb}iDh-V_!Kduy;-fvQ$&R+!aVQnq8&vkz zDg(sKd$!DouP%{{H^&;4LZQ7eW$!g5C-#;++Yne*d+dqkSiRL8YiN}?|9L#`cgM?T z&O&^nVp(>R)ex^~v0bgg_#9L8;`8**8_SB;+{j)4e8v|H~u^qL43j1!n8;&bn`BlF-wu^j8RjP0NhRPo`ia}6khz?^H3;=U9_=7;jkXNA~` zd&k|#H~K${b`>fM^;tQU9zf;#o2yiTzi<{Ixk1!LsG;^mJQQuN4b^u>>w#H5g~~YI z+V%Lg1IO2%`sw)BtA7&t^>ZuGhtvtN09H}&(U^TGt1w>jBDYGIz0eP1ri^1WjAM9w zM$R|<@90bzjXIfI*bX~pZ#Oqo;qT`f84X`2nEWy>W*XAb3c0hmZHnCPbR~>z z@R{;%Ksq&AAw9r=0i|^Ky1DOMaOc?uSxWuO`XpJ#lj^=|rdjU?@v6DL!{>iyq~~vj z|2HFwFspugb6_re2&UaJ)9q7cdJX?l?c*5NI9>Ytm*DD&Uk~wiw|m@7S8yDTnKE+0 z!;yd4CL!j^AG{%lk7MgY4n4lSIozWCrYpzYWI=NjoiZHwMGn(jeDI6tTqsi-o=R5* zdVkL}rm6zDm^=y1&(FRqW$sg*I|tslx)^|d>U})GUH{oMxR4{efF>#fUEqy#{RW@E zUojzr0X-iu5G1A?n=7bOX*19}GL4cl^9NG4^K34^-uZC99LY%pC5yQ}Jmv4>%%VAZ zN`bkiIQ>(!{r@@jJbxr@|MVR`$UAfy!ydvft2C`D zR-_3e?LE+Om-@Rm>(CMF))x2X(`=)!GC5=D*Pt|_zX>_yDyl%d*q&%3Xl2w zZqrP;5ZZjs+whSJpTl#5Vu#%GW@rVRb5d(~(nc~SVM6G~Rw1{r!u4Ba4&1yK?$zGJ zWTdc}rA86f`(ShamAsFz^ET`f8C&@eK!n~KQg0k1`?1?`j^cE;lM27-BuO49e-;uX z4%$C=xhdK4Iy#PIwa+_0bG4hLkg0L=SN;vzF@}z&toAwQf~&n=3Wp(7t7)t37(&Oo ztoBJ~;RiC3Tcz-22=9dO6ela$<}M^TgBEGS}W_s#QyJ3{Py(aqoBEP z9ji?6zuGV2&dUEtS=h!FfKS{8KCzN5i)wJ1vlv|u3!@`o1QyNr4Q7>G8Lt^wWA}^d z{<6BCQTLo?u~p1KF$2X66f;oFKrsWw3=}g^%s?>%#S9cPP|QFv1OG1>C|Mq;W~?O@ zd^smyhFi6GKa7{x+v8$T!~-Q2c)x4gY%fK-r6w9}ts@Eowc*utvVmw7P@=9TM!;fN zcyGAKYH70U4!f>BY2%%E4b47nG>d1As2yueHi;8Pqn$J%5JQF?+um-s+hWA1x82qs zjAXoKx19j%kH%Nxc1_aW*j&wE&Kp(Y4P#5}p*T!dQz@RXMen?A_HH|x-siI^8_PP)dA8HsR9Te4I93O{}r_T#&g^8s9XN^r^W((`e)K7q^lZEP(DJ8tJ2 zzizZ9lKEXv&4anyNmzmlURHuHFZ|XDjs}ggU?gJ<8x+wZhHVt@-=3Y{0u4!G5Hw+6k_RL@%gQ(^slhy52K$&^M@tM@gsPBP1%o7$=6u( zD?Ix2@5R!;;vZG~jKZ3qDd2lC`4xDfPu&x!^slkzN6^ngO#cxA6k_RL@m~R!<5&2@ z^7-PQx_?!CjXx|uqWJG9zQ&p_jYCGv_`R6%^Y;O=8yA^>jcE^c356)Uk?+OzN=Es7 zwEd_$|BE$N`M6tQ9p5^I^>`~4*7>(24$!ppolx?OnqN<9zQX-luGYgLh4uRWp2B*4 z^3?$2)$8p=g>^oTYpmweX<+wHh9rgy@K^ypUx5EyfFFRF&8Q~o7SykzqEr<22Sw-+ z?CXh8OHEy9lbzU=Y;6nGCb74N1kN8eO6x>ucTGIh&>F=ip9p;^9%{R9ZR48NwX5&# z>aw~zcSqMWc6RN0qQ0&Sr7;$&4k_vv?)zF|m8ryt%<;hnr2n Zjk4qhnoEfMG@w5R=uZIZ?4R)=;y=lI8lwOJ literal 0 HcmV?d00001 diff --git a/二叉树/1-二叉树 /btree/public.h b/二叉树/1-二叉树 /btree/public.h new file mode 100644 index 0000000..27b6d1f --- /dev/null +++ b/二叉树/1-二叉树 /btree/public.h @@ -0,0 +1,34 @@ +// +// public.h +// DS +// +// Created by mac on 13-9-8. +// Copyright (c) 2013年 xiaoran. All rights reserved. +// + +#ifndef DS_public_h +#define DS_public_h + +//#include +#include "stdlib.h" // 后来,换到mac环境下,#include 应该为 或 #include "stdlib.h" +#include +#include // for memset + + +typedef char ElemType ; // 也可能是一个复杂的复合类型 +typedef int Status; + + + +#define OK 1 +#define ERROR 0 + +typedef int bool; +#define YES 1 +#define NO 0 + +#define DEFAULT_SIZE 10 +#define OUT_OF_BOUND -1 + + +#endif diff --git a/二叉树/1-二叉树 /btree/队列.h b/二叉树/1-二叉树 /btree/队列.h new file mode 100644 index 0000000..220ae46 --- /dev/null +++ b/二叉树/1-二叉树 /btree/队列.h @@ -0,0 +1,175 @@ +// +// 队列.h +// 只在一段进行插入,另一端删除元素 +// +// Created by mac on 13-9-8. +// Copyright (c) 2013年 xiaoran. All rights reserved. +// + +#ifndef DS____h +#define DS____h + +#include "public.h" +#include "bintree.c" + +typedef BiTNode ElemType; + +//队列的顺序存储 +typedef struct Node{ + + ElemType *elem; + ElemType *head; + ElemType *tail; + int length; // 当前队列的长度 + int size; // 队列容器的长度,在队列慢得时候可以扩容 + + +}SqQueue; + +// +// 在队列采用顺序存储时,有一个毛病,就是队列操作一段时间后,头指针到了队列容器的尾部,而头指针前面的容器内存不可用了, +//造成内存极大的浪费,这个问题可以通过循环队列来解决。但是在 +// 链式队列上则不存在这样的问题 +// + + +typedef struct QNode{ + + struct QNode *next; + ElemType elem; + +}LinkQueue; + + +// 队列需要维护两个 指针 (队头指针,队尾指针) + +typedef struct{ + + LinkQueue *head; + LinkQueue *tail; + int length; + +}Queue; + + +Status initQueue(Queue *queue);// 带头结点,没有引用传值,就用指针的指针吧 +bool isEmpty(Queue *q); +int length(Queue *q); + + +// 在头部插入,尾部删除 +Status getHead(Queue *q,ElemType *e); +Status enQueue(Queue *q,ElemType e); +Status deQueue(Queue *q,ElemType *e); + +void traveser(Queue *q); + + +Status initQueue(Queue *queue){// 带头结点,没有引用传值,就用指针的指针吧 + + LinkQueue *lq = (LinkQueue *)malloc(sizeof(LinkQueue)); + if (!lq) { + return ERROR; + } + + lq->elem = 0; + lq->next = NULL; + + (queue)->head = (queue)->tail = lq; // -> 优先级高于 * ,老老实实打上括号 + (queue)->length = 0; + + return OK; + +} + +bool isEmpty(Queue *q){ + + return (q->head == q->tail); + +} + +int length(Queue *q){ + +// int ret; +// LinkQueue *p = q->head; +// while (p != q->tail) { +// ret++; +// p = p->next; +// +// } + + return q->length; + +} + +Status getHead(Queue *q,ElemType *e){ + + LinkQueue *p = q->head->next; + if (!p) { + *e=0; + return ERROR; + } + + *e = p->elem; + return OK; + +} + +// 入队(加入到队尾) +Status enQueue(Queue *q,ElemType e){ + + LinkQueue *newNode = (LinkQueue *)malloc(sizeof(LinkQueue)); + if (!newNode) { + return ERROR; + } + newNode->elem = e; + newNode->next = NULL; + q->tail->next = newNode; + q->tail = newNode; + + q->length++; + + return OK; + +} + +//出队(队头). 注意,可能只有1个元素,造成队尾指针丢失 +Status deQueue(Queue *q,ElemType *e){ + + LinkQueue *p = q->head->next; + if (!p) {// 队列空 + return ERROR; + } + + if (e) { + *e = p->elem ; + } + + LinkQueue *temp = p->next; + q->head->next = temp; + if (p == q->tail) { + q->tail = q->head; + } + free(p); + + + q->length--; + + return OK; + +} + + +void traveser(Queue *q){ + + LinkQueue *p = q->head->next; + while (NULL != p) { + printf("%d\n",p->elem); + p=p->next; + } + +} + + + +#endif diff --git a/二叉树/4-字典树Trie/trie.c b/二叉树/4-字典树Trie/trie.c index 22a9d77..d2f42c3 100644 --- a/二叉树/4-字典树Trie/trie.c +++ b/二叉树/4-字典树Trie/trie.c @@ -1,110 +1,127 @@ /* - -### trie基本 - -字典树,英文名Trie树,Trie一词来自retrieve,发音为/tri:/ “tree”,也有人读为/traɪ/ “try”, -又称单词查找树,Trie树,是一种树形结构。 - -trie,又称为前缀树或字典树,是一种有序树,用于保存关联数组。 - -1) 一个节点的所有子孙都有相同的前缀,即为这个节点对应的字符串 -2) 根节点对应空字符串,即不包含字符 -3) 不是所有节点都有对应值,只有叶子节点和部分内部节点所对应的键才有相关的值 -4) 键不是保存在节点中,而是由节点在树中的位置决定。 - - -### trie应用 - -典型应用是: - -用于统计,排序和保存大量的字符串(但不仅限于字符串), -所以经常被搜索引擎系统用于文本词频统计。 -排序大量字符串 -用于索引结构 -敏感词过滤 - - -它的优点是: -节省空间:利用字符串的公共前缀来节约存储空间 -最大限度地减少无谓的字符串比较 - - - - -查询效率比hash table 更优?? - - - -### trie 存储结构设计 - - 最简单实现 ---- 26个字母表 a-z 。 - -子树用数组存储,浪费空间; -如果用链表存储,查询时需要遍历链表,降低查询效率 - - - -DATrie libdatrie - -double-array trie -参考作者的这篇论文 http://linux.thai.net/~thep/datrie/datrie.html - - -trie树的增加和删除都比较麻烦,但索引本身就是写少读多,是否考虑添加删除的复杂度上升,依靠具体场景决定 - - - -### trie 问题 - - 中文支持 - - +trie树 */ #include #include -#define ALPHABET_SIZE 26 +#define ALPHABET_SIZE 26 // 256 + +typedef int bool; +#define YES 1 +#define NO 0 typedef struct node { int count; //count 如果为0,则代表非黄色点,count>0代表是黄色点,同时表示出现次数; char value; //字符 struct node * subtries[ALPHABET_SIZE]; //子树 -} *Trie; +} Trie; -//插入字符串,建立字典树. 返回值 < 0 表示插入失败 -int trie_insert(Trie *trie,char *c){ - if (NULL==trie) +Trie *trie_create(Trie **trie){ + + if (NULL==*trie) { - *trie = (Trie)malloc(sizeof(struct node)); - if (NULL==trie) + *trie = (Trie *)malloc(sizeof(struct node)); + if (NULL==*trie) { printf("malloc failure...\n"); } + + for (int i = 0; i < ALPHABET_SIZE; ++i) + { + (*trie)->subtries[i]=NULL; + } + } + + return *trie; + +} + + +//插入字符串(插入一个单词),建立字典树. 返回值 < 0 表示插入失败 +int trie_insert(Trie *trie,char *c){ + + if (trie==NULL || c==NULL) // 对NULL指针解引用会崩溃 + { + return -1; } char *p = c; - while(*p!='\0'){ - if (*p) + Trie *temp = trie; + while(*p != '\0'){ + + if (temp->subtries==NULL) { - /* code */ + //temp->subtries = (struct node *)malloc(sizeof(struct node)*ALPHABET_SIZE); + if (temp->subtries==NULL) + { + return -1; + } } + if (temp->subtries[*p-'a']==NULL) + { + struct node *newNode = (struct node *)malloc(sizeof(struct node)); + if (!newNode) + { + printf("create new node fail \n"); + return -1; + } + newNode->value = *p; + //newNode->subtries = NULL; + temp->subtries[*p-'a'] = newNode; + } + + + temp = temp->subtries[*p-'a']; p++; + } + return 0; } // 字符串查找,返回值<0表示没有查找到 -int trie_query(Trie *trie,char *c){ +bool trie_query(Trie *trie,char *c){ + if (trie==NULL ) + { + return NO; + } + + char *p = c; + Trie *temp = trie; + bool ret=NO; + if (temp->subtries == NULL) + { + return NO; + } + + while (*p!='\0') + { + if (temp->subtries[*p-'a']!=NULL && temp->subtries[*p-'a']->value==*p)//匹配 + { + temp = temp->subtries[*p-'a']; + p++; + continue; + } + + break; + } + + if (*p =='\0') + { + ret = YES; + } + + return ret; } @@ -117,11 +134,16 @@ void trie_remove(){} int main(){ - Trie trie = NULL; - char *dict[10]={"int","integer","float","char","nonstriater","weibo"}; - for (int i = 0; i < sizeof(dict)/sizeof(int); ++i) + Trie *trie = NULL; + if (!trie_create(&trie)) { - if (trie_insert(&trie,dict[i])<0) + printf("trie init fail...\n"); + } + + char *dict[10]={"int","integer","float","char","nonstriater","weibo"}; + for (int i = 0; i < 6; ++i) + { + if (trie_insert(trie,dict[i])<0) { printf("%s 插入失败\n", dict[i]); } @@ -130,12 +152,33 @@ int main(){ // 查询cha printf("查询cha \n"); + if (trie_query(trie,"cha")) + { + printf("YES\n"); + }else{ + printf("NO\n"); + + } // 查询char printf("查询char \n"); + if (trie_query(trie,"char")) + { + printf("YES\n"); + }else{ + printf("NO\n"); + + } // 查询hello printf("查询hello \n"); + if (trie_query(trie,"hello")) + { + printf("YES\n"); + }else{ + printf("NO\n"); + + } return 0; diff --git a/二叉树/README.md b/二叉树/README.md index e69de29..65b5222 100644 --- a/二叉树/README.md +++ b/二叉树/README.md @@ -0,0 +1,256 @@ + + +## 二叉查找树 + + 二叉查找树(Binary search tree),也叫有序二叉树(Ordered binary tree),排序二叉树(Sorted binary tree)。是指一个空树或者具有下列性质的二叉树: + + 1. 若任意节点的左子树不为空,则左子树上所有的节点值小于它的根节点值 + 2. 若任意节点的右子树不为空,则右子树上所有节点的值均大于它的根节点的值 + 3. 任意节点左右子树也为二叉查找树 + 4. 没有键值相等的节点 + + + typedef int ElemType; + typedef struct BiSearchTree{ + ElemType key; + struct BiSearchTree *lChild; + struct BiSearchTree *rChild; + }BiSearchTree; + BiSearchTree *bisearch_tree_insert(BiSearchTree *tree,ElemType node); + int bisearch_tree_delete(BiSearchTree **tree,ElemType node); + int bisearch_tree_search(BiSearchTree *tree,ElemType node); + + + +删除节点,需要重建排序树 + + 1) 删除节点是叶子节点(分支为0),结构不破坏 + 2)删除节点只有一个分支(分支为1),结构也不破坏 + 3)删除节点有2个分支,此时删除节点 + +思路一: 选左子树的最大节点,或右子树最小节点替换 + + + +int bisearch_tree_delete(BiSearchTree **tree,ElemType node){ + + if (NULL==tree) { + return -1; + } + // 查找删除目标节点 + BiSearchTree *target=*tree,*parent=NULL; + while (NULL!=target) { + if (nodekey) { + parent=target; + target=target->lChild; + }else if(node==target->key){ + break; + }else{ + parent=target; + target=target->rChild; + } + } + + if (NULL==target) { + printf("树为空,或想要删除的节点不存在\n"); + return -1; + } + //该节点为叶子节点,直接删除 + if (!target->rChild && !target->lChild) + { + if (NULL==parent) {////只有一个节点的二叉查找树 + *tree=NULL; + }else{ + if (target->key>parent->key) { + parent->rChild=NULL; + }else{ + parent->lChild=NULL; + } + + } + free(target);//父节点处理,不然野指针,造成崩溃 + } + + else if(!target->rChild){ //右子树空则只需重接它的左子树,用左子树替换掉当前要删除的节点 + BiSearchTree *del=target->lChild; + target->key = target->lChild->key; + target->lChild=target->lChild->lChild; + target->rChild=target->lChild->rChild; + + free(del); + } + else if(!target->lChild){ //左子树空只需重接它的右子树 + BiSearchTree *del=target->rChild; + target->key = target->rChild->key; + target->lChild=target->rChild->lChild; + target->rChild=target->rChild->rChild; + + free(del); + } + else{ //左右子树均不空,p,t 2个指针一前以后,将左子树最大的节点(肯定是一个最右的节点)替换到删除的节点后,还需要处理左子树最大节点的左子树 + + BiSearchTree *p=target,*t=target->lChild; + while (t->rChild) { + p = t; + t=t->rChild; + }// 找到左子树最大的,是删除节点的直接“前驱” + + target->key = t->key; + + if (p!=target) { + p->rChild = t->lChild; + }else{ + target->lChild = t->lChild; + } + + free(t); + } + return 0; + } + + + +## 伸展树 (splay tree) + +是一种自平衡的二叉排序树。为什么需要这些自平衡的二叉排序树? + +n个节点的完全二叉树,其查找,删除的复杂度都是O(logN),但是如果频繁的插入删除,导致二叉树退化成一个n个节点的单链表,也就是插入,查找复杂度趋于O(N),为了克服这个缺点,出现了很多二叉查找树的变形,如AVL树,红黑树,以及接下来介绍的 伸展树(splay tree) + + + +## AVL树 + + +## 红黑树 + + +## Treap 树 + + + + +## B树 + + + +## B+树 + + +## B*树 + + +## R树 + + +## 赫夫曼编码 + + + + + + +## 字典树trie(前缀树,单词查找树) + +### trie基本 + +字典树,英文名Trie树,Trie一词来自retrieve,发音为/tri:/ “tree”,也有人读为/traɪ/ “try”, +又称单词查找树,Trie树,是一种树形结构(多叉树)。 + +trie,又称为前缀树或字典树,是一种有序树,用于保存关联数组。 + +1. 除根节点不包含字符,每个节点都包含一个字符 +2. 从根节点到某一个节点,路径上经过的字符连接起来,为该节点对应的字符串 +3. 每个节点的所有子节点包含的字符都不相同(保证每个节点对应的字符串都不一样) + +### trie树存储结构和基本操作 + +最简单实现 ---- 26个字母表 a-z (没有考虑数字,大小写,其他字符) + +子树用数组存储,浪费空间;如果系统中存在大量字符串,且这些字符串基本没有公共前缀,trie树将消耗大量内存 +如果用链表存储,查询时需要遍历链表,查询效率有所降低 + +define ALPHABET_NUM 26 +typedef struct Node{ + char value; + struct Node *subTries[ALPHABET]; +}*Trie + +int Trie_insert(Trie *trie,char *word); // 插入一个单词 +int Trie_search(Trie *trie,char *word);// 查找一个单词 +int Trie_delete(Trie *trie,char *word);// 删除一个单词 + + +trie树的增加和删除都比较麻烦,但索引本身就是写少读多,是否考虑添加删除的复杂度上升,依靠具体场景决定 + + + +DATrie libdatrie 双数组可以减少内存的使用量 +double-array trie +参考作者的这篇论文 http://linux.thai.net/~thep/datrie/datrie.html + + + +### trie 问题 + +它的优点是: +节省空间:利用字符串的公共前缀来节约存储空间 +最大限度地减少无谓的字符串比较 + + +中文支持 + +查询效率比hash table 更优?? + + +### trie应用 + +典型应用是:前缀查询 , 字符串查询,排序 + +用于统计,排序和保存大量的字符串(但不仅限于字符串) +经常被搜索引擎系统用于文本词频统计 +排序大量字符串 +用于索引结构 +敏感词过滤 + +实际应用问题: +给你100000个长度不超过10的单词。对于每一个单词,我们要判断他出没出现过,如果出现了,求第一次出现在第几个位置 +分析: +思路一:trie树 ,找到这个字符串查询操作就可以了,如何知道出现的第一个位置呢?我们可以在trie树中加一个字段来记录当前字符串第一次出现的位置。 + + +已知n个由小写字母构成的平均长度为10的单词,判断其中是否存在某个串为另一个串的前缀子串 +分析: + +给出N 个单词组成的熟词表,以及一篇全用小写英文书写的文章,请你按最早出现的顺序写出所有不在熟词表中的生词。 +分析:trie树查询单词的应用。先建立N个熟词的前缀树,然后按文章的单词一次查询。 + +给出一个词典,其中的单词为不良单词。单词均为小写字母。再给出一段文本,文本的每一行也由小写字母构成。判断文本中是否含有任何不良单词。例如,若rob是不良单词,那么文本problem含有不良单词。 +分析:先用不良单词建立trie树,然后过滤文本(每个单词都在trie树上查询,查询的复杂度O(1),效率非常高),这正是`敏感词过滤系统(或垃圾评论系统)`的原理。 + + +给你N 个互不相同的仅由一个单词构成的英文名,让你将它们按字典序从小到大排序输出 +分析:这是trie树排序的典型应用,建立N个单词的trie树,然后线序遍历整个树,就可以达到效果。 + + +## 后缀树(suffix tree) + + + +###后缀树的应用 + +可以解决很多字符串的问题 + +1. 查找字符串S1是否在字符串S中 +2. 指定字符串S1在字符串S中出现的次数 +3. 字符串S中的最长重复子串 +4. 2个字符串的最长公共部分 + + + + + + + + + + diff --git a/排序算法/README.md b/排序算法/README.md index 374cea2..84ed276 100644 --- a/排序算法/README.md +++ b/排序算法/README.md @@ -244,13 +244,5 @@ - - - - - - - - diff --git a/算法问题选编/数值问题.md b/算法问题选编/数值问题.md index ee5a074..edbe1b5 100644 --- a/算法问题选编/数值问题.md +++ b/算法问题选编/数值问题.md @@ -1,7 +1,10 @@ -写一个函数,检查字符是否是整数,如果是,返回其整数值。 -(或者:怎样只用4行代码编写出一个从字符串到长整形的函数?) +写一个函数,检查字符是否是整数,如果是,返回其整数值 + +类似的题目还有: +怎样只用4行代码编写出一个从字符串到长整形的函数? +请编写能直接实现int atoi(const char * pstr)函数功能的代码。 将字符串转出整数 分析: 使用递归实现 @@ -10,8 +13,7 @@ int string_to_integer(char *s,int length) 1. s是空字符串 2. s传参为非数字字符 3. 超出整数所能表示范围 -类似的题目还有: -请编写能直接实现int atoi(const char * pstr)函数功能的代码。 将字符串转出整数 + @@ -212,6 +214,16 @@ int one_appear_count(int n); +给你5个球,每个球被抽到的可能性为30、50、20、40、10,设计一个随机算法,该算法的输出结果为本次执行的结果。输出A,B,C,D,E即可。 + + +2.已知一随机发生器,产生0的概率是p,产生1的概率是1-p,现在要你构造一个发生器, +使得它构造0和1的概率均为1/2;构造一个发生器,使得它构造1、2、3的概率均为1/3;..., +构造一个发生器,使得它构造1、2、3、...n的概率均为1/n,要求复杂度最低。 + + + + diff --git a/算法问题选编/数组数列问题.md b/算法问题选编/数组数列问题.md index 8f636ba..e9e7154 100644 --- a/算法问题选编/数组数列问题.md +++ b/算法问题选编/数组数列问题.md @@ -572,6 +572,9 @@ b[0] *= a[1]; +四对括号可以有多少种匹配排列方式?比如两对括号可以有两种:()()和(()) + +