python中的pickle问题?
Pickle 每次序列化生成的字符串有独立头尾,pickle.load() 只会读取一个完整的结果,所以你只需要在 load 一次之后再 load 一次,就能读到第二次序列化的 。如果不知道文件里有多少 pickle 对象,可以在 while 循环中反复 load 文件对象,直到抛出异常为止。
■网友
特性有添加了索引文件, 就是说, 你想找哪个变量就找哪个变量, 跟shelve模块一样允许增量添加数据, 而无需全局重写, 只允许List, 比如说你已经有一个list了. 你想添加数据, 只要用内建的方法append就可以增量添加了.随机插入 删除List 而无需全局重写.磁盘空间回收, 你复写一个同名数据, 会把原来的数据的空间回收掉, 删除一个量同样也会回收.采用日志法,保证原子性与鲁棒性, 就是说, 在程序运行中途 断电, 出BUG, 下次启动会自动根据日志修复数据库生成器方式load数据, 只允许list 比如说, 你有一个巨大的list, 你只需要最新的100项 一般的方法是把整个list load进内存, 我改进的模块会一项一项生成.自动整理数据库, 一旦检测到数据库碎片化严重, 会自动整理一下数据库.使用方法:新建或者打开一个数据库exampleexample=Jpickle("example")添加一个值为value的,名为var的变量example.dump("var",value)取出一个名为var的变量example.load("var")以生成器的方式, 分段获得数据example.generator("var",load_len=-1)load_len为需要生成的单位数量, 默认为-1. 即整个list.增量写入名为"list"的空白listexample.append("list",value)往一个list加入值example.insert("list",position,value)删除x位example.remove("list",position)在程序结束前, 建议强制刷新一次索引, 清空日志.example.dbclear()import osimport picklefrom bisect import insort_left,bisect_rightfrom operator import itemgetterclass Jpickle(): def __init__(self,filename): logfile=filename+".log" self.datafile=filename+".dat" self.indexfile=filename+".index" self.emptydic={} self.emptykeys= self.cursorcurrent=0 self.init_status=False try: self.data=https://www.zhihu.com/api/v4/questions/22095333/open(self.datafile,"rb+") self.index=open(self.indexfile,"rb+") self.log=open(logfile,"rb+",buffering=0) self.dic=pickle.loads(self.index.read()) self.cursorfinish=os.path.getsize(self.datafile) if os.path.getsize(logfile): while True: try: i=pickle.load(self.log) except EOFError: break sign,argus=i if sign==4: self.attend(*argus) elif sign==3: self.dump(*argus) elif sign==0: self.insert(*argus) elif sign==1: self.remove(*argus) elif sign==2: self.dele(*argus) self.__writeindex() if self.dic: var_list= for i in self.dic.values(): if isinstance(i,tuple): var_list.append(i) else: var_list.extend(i) var_list.sort(key=itemgetter(0)) cursor_begin=0 for i in var_list: cursor_first,cursor_second=i self.__appendemptydic(cursor_first-cursor_begin,cursor_begin) cursor_begin=cursor_second self.__appendemptydic(self.cursorfinish-cursor_second,cursor_second) except FileNotFoundError: self.dic={} self.cursorfinish=0 data=https://www.zhihu.com/api/v4/questions/22095333/open(self.datafile,"wb").close() index=open(self.indexfile,"wb").close() log=open(logfile,"wb").close() self.data=https://www.zhihu.com/api/v4/questions/22095333/open(self.datafile,"rb+") self.index=open(self.indexfile,"rb+") self.log=open(logfile,"rb+",buffering=0) self.index.write(pickle.dumps(self.dic)) self.index.flush() self.init_status=True def __binwrite(self,value): value_len=len(value) if self.emptykeys: empty_len=0 group=value_len//10 plus_pos=bisect_right(self.emptykeys,group) if plus_pos!=len(self.emptykeys): group_plus=self.emptykeys empty_len,cursor_first=self.emptydic.pop(-1) if not self.emptydic: del self.emptydic,self.emptykeys else: if group in self.emptydic: empty_len,cursor_first=self.emptydic if empty_len\u0026lt;value_len: empty_len=0 else: del self.emptydic if not self.emptydic: del self.emptydic,self.emptykeys if empty_len: self.__dseek(cursor_first) self.__dwrite(value,value_len) self.__appendemptydic(empty_len-value_len,self.cursorcurrent) return cursor_first,self.cursorcurrent self.__dseek(self.cursorfinish) cursor_first=self.cursorfinish self.__dwrite(value,value_len) return cursor_first,self.cursorfinish def __appendemptydic(self,empty_len,cursor_first): if empty_len: group=empty_len//10 if group in self.emptydic: self.emptydic.append((empty_len,cursor_first)) else: self.emptydic= insort_left(self.emptykeys,group) def __dwrite(self,value,value_len): self.data.write(value) self.cursorcurrent+=value_len if self.cursorcurrent\u0026gt;self.cursorfinish: self.cursorfinish=self.cursorcurrent def __dseek(self,target): if self.cursorcurrent!=target: self.data.seek(target) self.cursorcurrent=target def __dread(self,size): self.cursorcurrent+=size return self.data.read(size) def dump(self,var,value): if var in self.dic: self.dele(var,record=False) if isinstance(value,tuple): cursor_first,cursor_second=self.__binwrite(pickle.dumps(value)) self.dic=(cursor_first,cursor_second) else: value_write=b"".join(pickle.dumps(i) for i in value) cursor_first,cursor_second=self.__binwrite(value_write) self.dic= if self.init_status: self.log.write(pickle.dumps((3,(var,value)))) def del(self,var,record=True): if isinstance(self.dic,tuple): cursor_first,cursor_second=self.dic self.__appendemptydic(cursor_second-cursor_first,cursor_first) else: for i in self.dic: cursor_first,cursor_second=i self.__appendemptydic(cursor_second-cursor_first,cursor_first) del self.dic if record: if self.init_status: self.log.write(pickle.dumps((2,(var,)))) def insert(self,var,position,value): cursor_first,cursor_second,suffix=self.__find(var,position) ori_first,ori_second=self.dic write_first,write_second=self.__binwrite(pickle.dumps(value)) if cursor_first==ori_first: self.dic.insert(suffix,(write_first,write_second)) change_len=1 else: self.dic=(ori_first,cursor_first) self.dic.insert(suffix+1,(cursor_first,ori_second)) self.dic.insert(suffix+1,(write_first,write_second)) self.dic+=1 if self.init_status: self.log.write(pickle.dumps((0,(var,position,value)))) def append(self,var,value): cursor_first,cursor_second=self.__binwrite(pickle.dumps(value)) if cursor_first==self.dic: self.dic=(self.dic,cursor_second) else: self.dic.append((cursor_first,cursor_second)) if self.init_status: self.log.write(pickle.dumps((4,(var,value)))) def __find(self,var,position): value_extent=self.dic if position\u0026gt;value_extent//3*2: for i in self.generator(var, load_len=value_extent-position, input_status=False): pass return i else: position_search=-1 suffix=0 for i in self.dic: suffix+=1 cursor_first,cursor_second=i self.__dseek(cursor_first) print(self.data.tell()) lock=0 while True: i=self.__dread(1) if lock==0 and i==b".": lock+=1 elif lock==1 and i==b"\\x80": lock+=1 elif lock==2 and i==b"\\x03": lock+=1 elif self.cursorcurrent==cursor_second: return cursor_first,self.cursorcurrent,suffix else: lock=0 if lock==3: position_search+=1 if position_search==position: return cursor_first,self.cursorcurrent-2,suffix else: lock=0 cursor_first=self.cursorcurrent-2 def remove(self,var,position): cursor_first,cursor_second,suffix=self.__find(var, position) ori_first,ori_second=self.dic if cursor_first==ori_first and cursor_second==ori_second: del self.dic elif cursor_first==ori_first: self.dic=(cursor_second,ori_second) elif cursor_second==ori_second: self.dic=(ori_first,cursor_first) else: self.dic=(ori_first,cursor_first) self.dic.insert(suffix+1,(cursor_second,ori_second)) if self.init_status: self.log.write(pickle.dumps((1,(var,position)))) def load(self,var): if isinstance(self.dic,tuple): cursor_first,cursor_second=self.dic self.__dseek(cursor_first) value=https://www.zhihu.com/api/v4/questions/22095333/pickle.loads(self.__dread(cursor_second-cursor_first)) else: value= for i in self.dic: cursor_first,cursor_second=i self.__dseek(cursor_first) while self.data.tell()!=cursor_second: value.append(pickle.load(self.data)) self.cursorcurrent=cursor_second return value def generator(self,var,load_len=-1,input_status=True): BLOCK=4096 load_number=0 suffix=len(self.dic) for i in reversed(self.dic): value_search=b"" suffix-=1 cursor_first,cursor_second=i read_times,lastread_size=divmod(cursor_second-cursor_first,BLOCK) self.__dseek(cursor_second) for m in range(read_times): self.__dseek(self.cursorcurrent-BLOCK) value_search=self.__dread(BLOCK)+value_search self.cursorcurrent-=BLOCK match_finish=len(value_search) while True: match_begin=value_search.rfind(b".\\x80\\x03",0,match_finish) if match_begin==-1: value_search=value_search break if input_status: value_yield=pickle.loads(value_search) else: value_yield=(self.cursorcurrent+match_begin+1, self.cursorcurrent+match_finish, suffix) match_finish=match_begin+1 yield value_yield load_number+=1 if load_number==load_len: self.cursorcurrent+=BLOCK return if lastread_size: self.__dseek(self.cursorcurrent-lastread_size) value_search=self.__dread(lastread_size)+value_search self.cursorcurrent-=lastread_size match_finish=len(value_search) while True: match_begin=value_search.rfind(b".\\x80\\x03",0,match_finish) if match_begin==-1: value_search=value_search break if input_status: value_yield=pickle.loads(value_search) else: value_yield=(self.cursorcurrent+match_begin+1, self.cursorcurrent+match_finish, suffix) match_finish=match_begin+1 yield value_yield load_number+=1 if load_number==load_len: self.cursorcurrent+=lastread_size return if input_status: yield pickle.loads(value_search) else: yield self.cursorcurrent,self.cursorcurrent+match_finish,suffix self.cursorcurrent+=lastread_size def dbclear(self): accept_len=len(self.dic)//3 extend_items=sum(len(x) for x in self.dic.values() if isinstance(x,list)) if accept_len\u0026gt;extend_items: self.__writeindex() else: data_temp=open("dtemp","wb") new_dic={} write_first=write_second=0 for i in self.dic.items(): var,value=https://www.zhihu.com/api/v4/questions/22095333/i if isinstance(value,tuple): cursor_first,cursor_second=value self.__dseek(cursor_first) data_size=cursor_second-cursor_first data_temp.write(self.__dread(data_size)) write_second=write_first+data_size new_dic=(write_first,write_second) else: value_extent=self.dic for m in value: cursor_first,cursor_second=m self.__dseek(cursor_first) data_size=cursor_second-cursor_first data_temp.write(self.__dread(data_size)) write_second+=data_size new_dic= write_first=write_second self.data.close() data_temp.close() self.dic=new_dic self.__writeindex() os.remove(self.datafile) os.rename("dtemp",self.datafile) self.data=https://www.zhihu.com/api/v4/questions/22095333/open(self.datafile,"rb+") self.cursorcurrent=0 self.cursorfinish=os.path.getsize(self.datafile) def __writeindex(self): index_temp=open("itemp","wb") index_temp.write(pickle.dumps(self.dic)) index_temp.close() self.index.close() os.remove(self.indexfile) os.rename("itemp",self.indexfile) self.index=open(self.indexfile,"rb+") self.log.seek(0) self.log.truncate()
推荐阅读
- 鄂温克冬季马赛-30℃极寒开赛:寒冬中的火热派对
- 大雪@大雪腌肉 适当进补 今日大雪
- |电商事业中的“闪光少年”
- 怎样成为一名合格的Python程序员?
- python 爬虫,咋获得输入验证码之后的搜索结果
- hadoop中的mapreduce链接(mapreduce chaining)怎样避免中间文件的产生
- python的html5lib这个库咋使用啊我在网上也没有找到相关文档
- 经观汽车|日系车企中的“异类”?东风日产将导入e-POWER技术大干增程式混动 | 经观汽车
- 中年|这些东西,比你想象中的还要大得多!
- 零基础入门学习啥语言好
