This commit is contained in:
nonstriater
2014-05-16 20:11:50 +08:00
parent 3290e495d2
commit da7877336c
10 changed files with 952 additions and 88 deletions
+1 -1
View File
@@ -55,7 +55,7 @@ http://archive.cnblogs.com/a/1886332/
###数据处理
Core Data:中多线程中处理大量数据同步时的操作。
缓存设计
缓存设计LRU Cache
sqlite中插入特殊字符的方法和接收到处理方法。
+349
View File
@@ -0,0 +1,349 @@
/**
* 第一次题目 :构建二叉树(链表存储方式),空格表示空树,实现二叉树的基本操作:创建,遍历(先,中,后,按层)二叉树
* 提交邮箱 :[email protected] to [email protected]
* 邮件题目 :第一期第一次作业
**/
#include <stdlib.h>
#include <stdio.h>
typedef struct BiTNode{
char item;
struct BiTNode *lChild,*rChild;
}BiTNode,*BiTree;
// ==========================================================================
#define OK 1
#define ERROR 0
typedef int bool;
typedef BiTree ElemType ; // 也可能是一个复杂的复合类型
typedef int Status;
//队列的顺序存储
typedef struct Node{
ElemType *elem;
ElemType *head;
ElemType *tail;
int length; // 当前队列的长度
int size; // 队列容器的长度,在队列慢得时候可以扩容
}SqQueue;
//
// 在队列采用顺序存储时,有一个毛病,就是队列操作一段时间后,头指针到了队列容器的尾部,而头指针前面的容器内存不可用了,
//造成内存极大的浪费,这个问题可以通过循环队列来解决。但是在
// 链式队列上则不存在这样的问题
//
typedef struct QNode{
struct QNode *next;
ElemType elem;
}LinkQueue;
// 队列需要维护两个 指针 (队头指针,队尾指针)
typedef struct{
LinkQueue *head;
LinkQueue *tail;
int length;
}Queue;
Status initQueue(Queue *queue);// 带头结点,没有引用传值,就用指针的指针吧
bool isEmpty(Queue *q);
int length(Queue *q);
// 在头部插入,尾部删除
Status getHead(Queue *q,ElemType *e);
Status enQueue(Queue *q,ElemType e);
Status deQueue(Queue *q,ElemType *e);
void traveser(Queue *q);
Status initQueue(Queue *queue){// 带头结点,没有引用传值,就用指针的指针吧
LinkQueue *lq = (LinkQueue *)malloc(sizeof(LinkQueue));
if (!lq) {
return ERROR;
}
lq->elem = NULL;
lq->next = NULL;
(queue)->head = (queue)->tail = lq; // -> 优先级高于 * ,老老实实打上括号
(queue)->length = 0;
return OK;
}
bool isEmpty(Queue *q){
return (q->head == q->tail);
}
int length(Queue *q){
// int ret;
// LinkQueue *p = q->head;
// while (p != q->tail) {
// ret++;
// p = p->next;
//
// }
return q->length;
}
Status getHead(Queue *q,ElemType *e){
LinkQueue *p = q->head->next;
if (!p) {
*e=NULL;
return ERROR;
}
*e = p->elem;
return OK;
}
// 入队(加入到队尾)
Status enQueue(Queue *q,ElemType e){
LinkQueue *newNode = (LinkQueue *)malloc(sizeof(LinkQueue));
if (!newNode) {
return ERROR;
}
newNode->elem = e;
newNode->next = NULL;
q->tail->next = newNode;
q->tail = newNode;
q->length++;
return OK;
}
//出队(队头). 注意,可能只有1个元素,造成队尾指针丢失
Status deQueue(Queue *q,ElemType *e){
LinkQueue *p = q->head->next;
if (!p) {// 队列空
return ERROR;
}
if (e) {
*e = p->elem ;
}
LinkQueue *temp = p->next;
q->head->next = temp;
if (p == q->tail) {
q->tail = q->head;
}
free(p);
q->length--;
return OK;
}
void traveser(Queue *q){
LinkQueue *p = q->head->next;
while (NULL != p) {
//printf("%d\n",p->elem);
p=p->next;
}
}
// ==================================================
/*
int CreateBiTree(BiTree *T)
{
*T = (BiTNode *)malloc(sizeof(BiTNode));
if (*T==NULL)
{
printf("memery malloc failure !\n");
return -1;
}
printf("enter data to create node:\n");
scanf("%c",&((*T)->item));
if((*T)->item=='#'){
*T=NULL;
}
if(*T){
printf("创建左子树:\n");
CreateBiTree( &((*T)->lChild) );
printf("创建右子树:\n");
CreateBiTree( &((*T)->rChild) );
}
return 0;
}
*/
// BiTree CreateBiTree()
// 先序遍历方式创建二叉树
BiTree CreateBiTree(){
char c;
BiTNode *tree;
// 过滤回车键
scanf("%c",&c);
if (c==' ')
{
printf("创建空节点\n");
tree = NULL;
}else{
printf("创建节点 %c\n",c);
tree = (BiTNode *)malloc(sizeof(BiTNode));
tree->item = c;
tree->lChild = CreateBiTree();
tree->rChild = CreateBiTree();
}
return tree;
}
int PreOrderTraverse(BiTree T){
if(T){
printf("%c\n",T->item );
PreOrderTraverse(T->lChild);
PreOrderTraverse(T->rChild);
}
return 0;
}
int InOrderTraverse(BiTree T){
if (T)
{
InOrderTraverse(T->lChild);
printf("%c\n", T->item);
InOrderTraverse(T->rChild);
}
return 0;
}
int PostOrderTraverse(BiTree T){
if (T)
{
PostOrderTraverse(T->lChild);
PostOrderTraverse(T->rChild);
printf("%c\n",T->item );
}
return 0;
}
// 广度优先遍历 (队列实现)
int LevelOrderTraverse(BiTree T){
if (T)
{
Queue queue;
initQueue(&queue);
BiTree u;
u=(BiTree)malloc(sizeof(BiTNode));
enQueue(&queue, T);
while(!isEmpty(&queue))
{
deQueue(&queue, &u);
printf("%c",u->item);
if(u->lChild)
enQueue(&queue, u->lChild);
if(u->rChild)
enQueue(&queue, u->rChild);
}
}
return 0;
}
int main(int argc, char const *argv[])
{
/* code */
BiTree binaryTree;
printf("创建二叉树,输入\"空格\"创建空节点(先序方式建立二叉树):\n");
binaryTree = CreateBiTree();
if(binaryTree==NULL){
printf("创建空的二叉树\n");
}
printf("前序遍历:\n");
PreOrderTraverse(binaryTree);
printf("\n\n");
printf("中序遍历:\n");
InOrderTraverse(binaryTree);
printf("\n\n");
printf("后序遍历:\n");
PostOrderTraverse(binaryTree);
printf("\n\n");
printf("层序遍历:\n");
LevelOrderTraverse(binaryTree);
printf("\n\n");
return 0;
}
BIN
View File
Binary file not shown.
+34
View File
@@ -0,0 +1,34 @@
//
// public.h
// DS
//
// Created by mac on 13-9-8.
// Copyright (c) 2013年 xiaoran. All rights reserved.
//
#ifndef DS_public_h
#define DS_public_h
//#include <malloc.h>
#include "stdlib.h" // 后来,换到mac环境下,#include <malloc.h> 应该为<malloc/malloc.h> 或 #include "stdlib.h"
#include <stdio.h>
#include <string.h> // for memset
typedef char ElemType ; // 也可能是一个复杂的复合类型
typedef int Status;
#define OK 1
#define ERROR 0
typedef int bool;
#define YES 1
#define NO 0
#define DEFAULT_SIZE 10
#define OUT_OF_BOUND -1
#endif
+175
View File
@@ -0,0 +1,175 @@
//
// 队列.h
// 只在一段进行插入,另一端删除元素
//
// Created by mac on 13-9-8.
// Copyright (c) 2013年 xiaoran. All rights reserved.
//
#ifndef DS____h
#define DS____h
#include "public.h"
#include "bintree.c"
typedef BiTNode ElemType;
//队列的顺序存储
typedef struct Node{
ElemType *elem;
ElemType *head;
ElemType *tail;
int length; // 当前队列的长度
int size; // 队列容器的长度,在队列慢得时候可以扩容
}SqQueue;
//
// 在队列采用顺序存储时,有一个毛病,就是队列操作一段时间后,头指针到了队列容器的尾部,而头指针前面的容器内存不可用了,
//造成内存极大的浪费,这个问题可以通过循环队列来解决。但是在
// 链式队列上则不存在这样的问题
//
typedef struct QNode{
struct QNode *next;
ElemType elem;
}LinkQueue;
// 队列需要维护两个 指针 (队头指针,队尾指针)
typedef struct{
LinkQueue *head;
LinkQueue *tail;
int length;
}Queue;
Status initQueue(Queue *queue);// 带头结点,没有引用传值,就用指针的指针吧
bool isEmpty(Queue *q);
int length(Queue *q);
// 在头部插入,尾部删除
Status getHead(Queue *q,ElemType *e);
Status enQueue(Queue *q,ElemType e);
Status deQueue(Queue *q,ElemType *e);
void traveser(Queue *q);
Status initQueue(Queue *queue){// 带头结点,没有引用传值,就用指针的指针吧
LinkQueue *lq = (LinkQueue *)malloc(sizeof(LinkQueue));
if (!lq) {
return ERROR;
}
lq->elem = 0;
lq->next = NULL;
(queue)->head = (queue)->tail = lq; // -> 优先级高于 * ,老老实实打上括号
(queue)->length = 0;
return OK;
}
bool isEmpty(Queue *q){
return (q->head == q->tail);
}
int length(Queue *q){
// int ret;
// LinkQueue *p = q->head;
// while (p != q->tail) {
// ret++;
// p = p->next;
//
// }
return q->length;
}
Status getHead(Queue *q,ElemType *e){
LinkQueue *p = q->head->next;
if (!p) {
*e=0;
return ERROR;
}
*e = p->elem;
return OK;
}
// 入队(加入到队尾)
Status enQueue(Queue *q,ElemType e){
LinkQueue *newNode = (LinkQueue *)malloc(sizeof(LinkQueue));
if (!newNode) {
return ERROR;
}
newNode->elem = e;
newNode->next = NULL;
q->tail->next = newNode;
q->tail = newNode;
q->length++;
return OK;
}
//出队(队头). 注意,可能只有1个元素,造成队尾指针丢失
Status deQueue(Queue *q,ElemType *e){
LinkQueue *p = q->head->next;
if (!p) {// 队列空
return ERROR;
}
if (e) {
*e = p->elem ;
}
LinkQueue *temp = p->next;
q->head->next = temp;
if (p == q->tail) {
q->tail = q->head;
}
free(p);
q->length--;
return OK;
}
void traveser(Queue *q){
LinkQueue *p = q->head->next;
while (NULL != p) {
printf("%d\n",p->elem);
p=p->next;
}
}
#endif
+118 -75
View File
@@ -1,110 +1,127 @@
/*
### trie基本
字典树,英文名Trie树,Trie一词来自retrieve,发音为/tri:/ “tree”,也有人读为/traɪ/ “try”,
又称单词查找树,Trie树,是一种树形结构。
trie,又称为前缀树或字典树,是一种有序树,用于保存关联数组。
1) 一个节点的所有子孙都有相同的前缀,即为这个节点对应的字符串
2) 根节点对应空字符串,即不包含字符
3) 不是所有节点都有对应值,只有叶子节点和部分内部节点所对应的键才有相关的值
4) 键不是保存在节点中,而是由节点在树中的位置决定。
### trie应用
典型应用是:
用于统计,排序和保存大量的字符串(但不仅限于字符串),
所以经常被搜索引擎系统用于文本词频统计。
排序大量字符串
用于索引结构
敏感词过滤
它的优点是:
节省空间:利用字符串的公共前缀来节约存储空间
最大限度地减少无谓的字符串比较
查询效率比hash table 更优??
### trie 存储结构设计
最简单实现 ---- 26个字母表 a-z 。
子树用数组存储,浪费空间;
如果用链表存储,查询时需要遍历链表,降低查询效率
DATrie libdatrie
double-array trie
参考作者的这篇论文 http://linux.thai.net/~thep/datrie/datrie.html
trie树的增加和删除都比较麻烦,但索引本身就是写少读多,是否考虑添加删除的复杂度上升,依靠具体场景决定
### trie 问题
中文支持
trie树
*/
#include <stdio.h>
#include <stdlib.h>
#define ALPHABET_SIZE 26
#define ALPHABET_SIZE 26 // 256
typedef int bool;
#define YES 1
#define NO 0
typedef struct node
{
int count; //count 如果为0,则代表非黄色点,count>0代表是黄色点,同时表示出现次数;
char value; //字符
struct node * subtries[ALPHABET_SIZE]; //子树
} *Trie;
} Trie;
//插入字符串,建立字典树. 返回值 < 0 表示插入失败
int trie_insert(Trie *trie,char *c){
if (NULL==trie)
Trie *trie_create(Trie **trie){
if (NULL==*trie)
{
*trie = (Trie)malloc(sizeof(struct node));
if (NULL==trie)
*trie = (Trie *)malloc(sizeof(struct node));
if (NULL==*trie)
{
printf("malloc failure...\n");
}
for (int i = 0; i < ALPHABET_SIZE; ++i)
{
(*trie)->subtries[i]=NULL;
}
}
return *trie;
}
//插入字符串(插入一个单词),建立字典树. 返回值 < 0 表示插入失败
int trie_insert(Trie *trie,char *c){
if (trie==NULL || c==NULL) // 对NULL指针解引用会崩溃
{
return -1;
}
char *p = c;
while(*p!='\0'){
if (*p)
Trie *temp = trie;
while(*p != '\0'){
if (temp->subtries==NULL)
{
/* code */
//temp->subtries = (struct node *)malloc(sizeof(struct node)*ALPHABET_SIZE);
if (temp->subtries==NULL)
{
return -1;
}
}
if (temp->subtries[*p-'a']==NULL)
{
struct node *newNode = (struct node *)malloc(sizeof(struct node));
if (!newNode)
{
printf("create new node fail \n");
return -1;
}
newNode->value = *p;
//newNode->subtries = NULL;
temp->subtries[*p-'a'] = newNode;
}
temp = temp->subtries[*p-'a'];
p++;
}
return 0;
}
// 字符串查找,返回值<0表示没有查找到
int trie_query(Trie *trie,char *c){
bool trie_query(Trie *trie,char *c){
if (trie==NULL )
{
return NO;
}
char *p = c;
Trie *temp = trie;
bool ret=NO;
if (temp->subtries == NULL)
{
return NO;
}
while (*p!='\0')
{
if (temp->subtries[*p-'a']!=NULL && temp->subtries[*p-'a']->value==*p)//匹配
{
temp = temp->subtries[*p-'a'];
p++;
continue;
}
break;
}
if (*p =='\0')
{
ret = YES;
}
return ret;
}
@@ -117,11 +134,16 @@ void trie_remove(){}
int main(){
Trie trie = NULL;
char *dict[10]={"int","integer","float","char","nonstriater","weibo"};
for (int i = 0; i < sizeof(dict)/sizeof(int); ++i)
Trie *trie = NULL;
if (!trie_create(&trie))
{
if (trie_insert(&trie,dict[i])<0)
printf("trie init fail...\n");
}
char *dict[10]={"int","integer","float","char","nonstriater","weibo"};
for (int i = 0; i < 6; ++i)
{
if (trie_insert(trie,dict[i])<0)
{
printf("%s 插入失败\n", dict[i]);
}
@@ -130,12 +152,33 @@ int main(){
// 查询cha
printf("查询cha \n");
if (trie_query(trie,"cha"))
{
printf("YES\n");
}else{
printf("NO\n");
}
// 查询char
printf("查询char \n");
if (trie_query(trie,"char"))
{
printf("YES\n");
}else{
printf("NO\n");
}
// 查询hello
printf("查询hello \n");
if (trie_query(trie,"hello"))
{
printf("YES\n");
}else{
printf("NO\n");
}
return 0;
+256
View File
@@ -0,0 +1,256 @@
## 二叉查找树
二叉查找树(Binary search tree),也叫有序二叉树(Ordered binary tree),排序二叉树(Sorted binary tree)。是指一个空树或者具有下列性质的二叉树:
1. 若任意节点的左子树不为空,则左子树上所有的节点值小于它的根节点值
2. 若任意节点的右子树不为空,则右子树上所有节点的值均大于它的根节点的值
3. 任意节点左右子树也为二叉查找树
4. 没有键值相等的节点
typedef int ElemType;
typedef struct BiSearchTree{
ElemType key;
struct BiSearchTree *lChild;
struct BiSearchTree *rChild;
}BiSearchTree;
BiSearchTree *bisearch_tree_insert(BiSearchTree *tree,ElemType node);
int bisearch_tree_delete(BiSearchTree **tree,ElemType node);
int bisearch_tree_search(BiSearchTree *tree,ElemType node);
删除节点,需要重建排序树
1) 删除节点是叶子节点(分支为0),结构不破坏
2)删除节点只有一个分支(分支为1),结构也不破坏
3)删除节点有2个分支,此时删除节点
思路一: 选左子树的最大节点,或右子树最小节点替换
int bisearch_tree_delete(BiSearchTree **tree,ElemType node){
if (NULL==tree) {
return -1;
}
// 查找删除目标节点
BiSearchTree *target=*tree,*parent=NULL;
while (NULL!=target) {
if (node<target->key) {
parent=target;
target=target->lChild;
}else if(node==target->key){
break;
}else{
parent=target;
target=target->rChild;
}
}
if (NULL==target) {
printf("树为空,或想要删除的节点不存在\n");
return -1;
}
//该节点为叶子节点,直接删除
if (!target->rChild && !target->lChild)
{
if (NULL==parent) {////只有一个节点的二叉查找树
*tree=NULL;
}else{
if (target->key>parent->key) {
parent->rChild=NULL;
}else{
parent->lChild=NULL;
}
}
free(target);//父节点处理,不然野指针,造成崩溃
}
else if(!target->rChild){ //右子树空则只需重接它的左子树,用左子树替换掉当前要删除的节点
BiSearchTree *del=target->lChild;
target->key = target->lChild->key;
target->lChild=target->lChild->lChild;
target->rChild=target->lChild->rChild;
free(del);
}
else if(!target->lChild){ //左子树空只需重接它的右子树
BiSearchTree *del=target->rChild;
target->key = target->rChild->key;
target->lChild=target->rChild->lChild;
target->rChild=target->rChild->rChild;
free(del);
}
else{ //左右子树均不空,p,t 2个指针一前以后,将左子树最大的节点(肯定是一个最右的节点)替换到删除的节点后,还需要处理左子树最大节点的左子树
BiSearchTree *p=target,*t=target->lChild;
while (t->rChild) {
p = t;
t=t->rChild;
}// 找到左子树最大的,是删除节点的直接“前驱”
target->key = t->key;
if (p!=target) {
p->rChild = t->lChild;
}else{
target->lChild = t->lChild;
}
free(t);
}
return 0;
}
## 伸展树 (splay tree)
是一种自平衡的二叉排序树。为什么需要这些自平衡的二叉排序树?
n个节点的完全二叉树,其查找,删除的复杂度都是O(logN),但是如果频繁的插入删除,导致二叉树退化成一个n个节点的单链表,也就是插入,查找复杂度趋于O(N),为了克服这个缺点,出现了很多二叉查找树的变形,如AVL树,红黑树,以及接下来介绍的 伸展树(splay tree)
## AVL树
## 红黑树
## Treap 树
## B树
## B+树
## B*树
## R树
## 赫夫曼编码
## 字典树trie(前缀树,单词查找树)
### trie基本
字典树,英文名Trie树,Trie一词来自retrieve,发音为/tri:/ “tree”,也有人读为/traɪ/ “try”,
又称单词查找树,Trie树,是一种树形结构(多叉树)。
trie,又称为前缀树或字典树,是一种有序树,用于保存关联数组。
1. 除根节点不包含字符,每个节点都包含一个字符
2. 从根节点到某一个节点,路径上经过的字符连接起来,为该节点对应的字符串
3. 每个节点的所有子节点包含的字符都不相同(保证每个节点对应的字符串都不一样)
### trie树存储结构和基本操作
最简单实现 ---- 26个字母表 a-z (没有考虑数字,大小写,其他字符)
子树用数组存储,浪费空间;如果系统中存在大量字符串,且这些字符串基本没有公共前缀,trie树将消耗大量内存
如果用链表存储,查询时需要遍历链表,查询效率有所降低
define ALPHABET_NUM 26
typedef struct Node{
char value;
struct Node *subTries[ALPHABET];
}*Trie
int Trie_insert(Trie *trie,char *word); // 插入一个单词
int Trie_search(Trie *trie,char *word);// 查找一个单词
int Trie_delete(Trie *trie,char *word);// 删除一个单词
trie树的增加和删除都比较麻烦,但索引本身就是写少读多,是否考虑添加删除的复杂度上升,依靠具体场景决定
DATrie libdatrie 双数组可以减少内存的使用量
double-array trie
参考作者的这篇论文 http://linux.thai.net/~thep/datrie/datrie.html
### trie 问题
它的优点是:
节省空间:利用字符串的公共前缀来节约存储空间
最大限度地减少无谓的字符串比较
中文支持
查询效率比hash table 更优??
### trie应用
典型应用是:前缀查询 , 字符串查询,排序
用于统计,排序和保存大量的字符串(但不仅限于字符串)
经常被搜索引擎系统用于文本词频统计
排序大量字符串
用于索引结构
敏感词过滤
实际应用问题:
给你100000个长度不超过10的单词。对于每一个单词,我们要判断他出没出现过,如果出现了,求第一次出现在第几个位置
分析:
思路一:trie树 ,找到这个字符串查询操作就可以了,如何知道出现的第一个位置呢?我们可以在trie树中加一个字段来记录当前字符串第一次出现的位置。
已知n个由小写字母构成的平均长度为10的单词,判断其中是否存在某个串为另一个串的前缀子串
分析:
给出N 个单词组成的熟词表,以及一篇全用小写英文书写的文章,请你按最早出现的顺序写出所有不在熟词表中的生词。
分析:trie树查询单词的应用。先建立N个熟词的前缀树,然后按文章的单词一次查询。
给出一个词典,其中的单词为不良单词。单词均为小写字母。再给出一段文本,文本的每一行也由小写字母构成。判断文本中是否含有任何不良单词。例如,若rob是不良单词,那么文本problem含有不良单词。
分析:先用不良单词建立trie树,然后过滤文本(每个单词都在trie树上查询,查询的复杂度O(1),效率非常高),这正是`敏感词过滤系统(或垃圾评论系统)`的原理。
给你N 个互不相同的仅由一个单词构成的英文名,让你将它们按字典序从小到大排序输出
分析:这是trie树排序的典型应用,建立N个单词的trie树,然后线序遍历整个树,就可以达到效果。
## 后缀树(suffix tree)
###后缀树的应用
可以解决很多字符串的问题
1. 查找字符串S1是否在字符串S中
2. 指定字符串S1在字符串S中出现的次数
3. 字符串S中的最长重复子串
4. 2个字符串的最长公共部分
-8
View File
@@ -244,13 +244,5 @@
+16 -4
View File
@@ -1,7 +1,10 @@
写一个函数,检查字符是否是整数,如果是,返回其整数值。
(或者:怎样只用4行代码编写出一个从字符串到长整形的函数?)
写一个函数,检查字符是否是整数,如果是,返回其整数值
类似的题目还有:
怎样只用4行代码编写出一个从字符串到长整形的函数?
请编写能直接实现int atoi(const char * pstr)函数功能的代码。 将字符串转出整数
分析:
使用递归实现
@@ -10,8 +13,7 @@ int string_to_integer(char *s,int length)
1. s是空字符串
2. s传参为非数字字符
3. 超出整数所能表示范围
类似的题目还有:
请编写能直接实现int atoi(const char * pstr)函数功能的代码。 将字符串转出整数
@@ -212,6 +214,16 @@ int one_appear_count(int n);
给你5个球,每个球被抽到的可能性为30、50、20、40、10,设计一个随机算法,该算法的输出结果为本次执行的结果。输出A,B,C,D,E即可。
2.已知一随机发生器,产生0的概率是p,产生1的概率是1-p,现在要你构造一个发生器,
使得它构造0和1的概率均为1/2;构造一个发生器,使得它构造1、2、3的概率均为1/3;...,
构造一个发生器,使得它构造1、2、3、...n的概率均为1/n,要求复杂度最低。
+3
View File
@@ -572,6 +572,9 @@ b[0] *= a[1];
四对括号可以有多少种匹配排列方式?比如两对括号可以有两种:()()和(())