篇十一:去重简易算法
导语:筛选数据时,避免数据重复,需要去重,传统的逐一匹配,效率很低,先总结简单、快速去重方法。
一、方法的实现
1、add方法添加需要去重的数据;
2、contains用来判断是否包含,包含为true
package com.util; import java.util.ArrayList; import java.util.BitSet; import java.util.List; public class SimpleBloomFilter { private static final int DEFAULT_SIZE = 2 << 24 ; private static final int [] seeds = new int [] { 7 , 11 , 13 , 31 , 37 , 61 , }; private BitSet bits = new BitSet(DEFAULT_SIZE); private SimpleHash[] func = new SimpleHash[seeds.length]; public SimpleBloomFilter() { for ( int i = 0 ; i < seeds.length; i ++ ) { func[i] = new SimpleHash(DEFAULT_SIZE, seeds[i]); } } public void add(String value) { for (SimpleHash f : func) { bits.set(f.hash(value), true ); } } public boolean contains(String value) { if (value == null ) { return false ; } boolean ret = true ; for (SimpleHash f : func) { ret = ret && bits.get(f.hash(value)); } return ret; } public static class SimpleHash { private int cap; private int seed; public SimpleHash( int cap, int seed) { this .cap = cap; this .seed = seed; } public int hash(String value) { int result = 0 ; int len = value.length(); for ( int i = 0 ; i < len; i ++ ) { result = seed * result + value.charAt(i); } return (cap - 1 ) & result; } } }
二、使用
//存储需要去重的String值,最终用SimpleBloomFilter包含 SimpleBloomFilter filter=new SimpleBloomFilter(); for(String url:urlDistinct){ filter.add(url); } if(!filter.contains(url)) { //不重复时的操作 }

浙公网安备 33010602011771号