[can't edit previous post]
In fact replacing the nopw that is used for alignment by a prefetch instruction gives me something slightly even faster:
00000000004004e0 <prefetchit>:
4004e0: 31 c0 xor %eax,%eax
4004e2: 0f 18 0d 87 04 20 00 prefetcht0 0x200487(%rip) # 600970 <counter>
4004e9: 48 8b 15 80 04 20 00 mov 0x200480(%rip),%rdx # 600970 <counter>
4004f0: 48 01 c2 add %rax,%rdx
4004f3: 48 83 c0 01 add $0x1,%rax
4004f7: 48 3d 00 84 d7 17 cmp $0x17d78400,%rax
4004fd: 48 89 15 6c 04 20 00 mov %rdx,0x20046c(%rip) # 600970 <counter>
400504: 75 dc jne 4004e2 <prefetchit+0x2>
400506: f3 c3 repz retq
400508: 0f 1f 84 00 00 00 00 nopl 0x0(%rax,%rax,1)
40050f: 00
2,445,505,116 cycles # 0.000 GHz ( +- 0.52% )
2,800,857,971 instructions # 1.15 insns per cycle ( +- 0.00% )
0.643010038 seconds time elapsed