Wee! Pointless optimization Saturday!
Looks like my gen_stats_eesmith is even faster. Radim's is about 400x faster, iClaudiusX is 450x faster, and eesmith is 580x faster.
Under pypy2-v5.8.0 those are 65x, 16x, and 91x, respectively. The original algorithm is 10x faster under pypy.
def gen_stats_python(dataset_python):
start = time.time()
product_stats = []
unique_products = set([x[0] for x in dataset_python])
for product_id in unique_products:
product_items = [x for x in dataset_python if x[0]==product_id ]
num_orders = len(product_items)
total_quantity = 0
total_price = 0
for row in product_items:
total_quantity += row[2]
total_price += row[3]
avg_price = float(total_price/num_orders)
product_stats.append([int(product_id),int(num_orders),int(total_quantity),round(avg_price,2)])
end = time.time()
working_time = end-start
return product_stats,working_time
def gen_stats_radim(dataset_python):
start = time.time()
aggr = {}
for product_id, product_order_num, quantity, price in dataset_python:
item = aggr.get(product_id, [0, []])
item[0] += quantity
item[1].append(price)
aggr[product_id] = item
return [
[int(product_id), len(prices), int(quantity), round(float(sum(prices)) / len(prices), 2)]
for product_id, (quantity, prices) in aggr.items()
], time.time()-start
def gen_stats_iclaudiusx(data):
start = time.time()
stats = []
data.sort(key=itemgetter(0))
for p_id, p_items in itertools.groupby(data, key=itemgetter(0)):
num_orders, _, total_quantity, total_price = map(sum, zip(*p_items))
p_id, num_orders, total_quantity = map(int, [p_id, num_orders/p_id, total_quantity])
stats.append([p_id, num_orders, total_quantity, round(float(total_price/num_orders),2)])
end = time.time()
working_time = end-start
return stats, working_time
def gen_stats_eesmith(dataset_python):
start = time.time()
product_stats = []
total_num_orders = defaultdict(int)
total_quantity_by_product_id = defaultdict(int)
total_price_by_product_id = defaultdict(float)
for product_id, order_id, quantity, price in dataset_python:
total_num_orders[product_id] += 1
total_quantity_by_product_id[product_id] += quantity
total_price_by_product_id[product_id] += price
for product_id in total_price_by_product_id:
num_orders = total_num_orders[product_id]
total_quantity = total_quantity_by_product_id[product_id]
total_price = total_price_by_product_id[product_id]
avg_price = float(total_price)/num_orders
product_stats.append([int(product_id),int(num_orders),int(total_quantity),round(avg_price,2)])
end = time.time()
working_time = end-start
return product_stats,working_time
reference_result = reference_dt = None
gen_stats_functions = (gen_stats_python, gen_stats_radim,
gen_stats_iclaudiusx, gen_stats_eesmith)
for gen_stats_function in gen_stats_functions:
result, dt = gen_stats_function(dataset_python)
if reference_result is None:
reference_result = result
reference_dt = dt
assert result == reference_result, (gen_stats_function.__name__,
result[:2], reference_result[:2])
print(gen_stats_function.__name__, dt, reference_dt/dt)