forked from vmprof/vmprof-python
-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathreader.py
More file actions
288 lines (252 loc) · 9.1 KB
/
Copy pathreader.py
File metadata and controls
288 lines (252 loc) · 9.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
from __future__ import print_function
import re
import os
import struct
import subprocess
import sys
from six.moves import xrange
import io
import gzip
from vmprof.binary import read_word, read_string, read_words
PY3 = sys.version_info[0] >= 3
WORD_SIZE = struct.calcsize('L')
def read_trace(fileobj, depth, version, profile_lines=False):
if version == VERSION_TAG:
assert depth & 1 == 0
depth = depth // 2
kinds_and_pcs = read_words(fileobj, depth * 2)
# kinds_and_pcs is a list of [kind1, pc1, kind2, pc2, ...]
return [wrap_kind(kinds_and_pcs[i], kinds_and_pcs[i+1])
for i in xrange(len(kinds_and_pcs), None, 2)]
else:
trace = read_words(fileobj, depth)
if profile_lines:
for i in range(0, len(trace), 2):
# In the line profiling mode even items in the trace are line numbers.
# Every line number corresponds to the following frame, represented by an address.
trace[i] = -trace[i]
return trace
MARKER_STACKTRACE = b'\x01'
MARKER_VIRTUAL_IP = b'\x02'
MARKER_TRAILER = b'\x03'
MARKER_INTERP_NAME = b'\x04'
MARKER_HEADER = b'\x05'
VERSION_BASE = 0
VERSION_THREAD_ID = 1
VERSION_TAG = 2
VERSION_MEMORY = 3
VERSION_MODE_AWARE = 4
PROFILE_MEMORY = 1
PROFILE_LINES = 2
VMPROF_CODE_TAG = 1
VMPROF_BLACKHOLE_TAG = 2
VMPROF_JITTED_TAG = 3
VMPROF_JITTING_TAG = 4
VMPROF_GC_TAG = 5
VMPROF_ASSEMBLER_TAG = 6
class AssemblerCode(int):
pass
class JittedCode(int):
pass
def wrap_kind(kind, pc):
if kind == VMPROF_ASSEMBLER_TAG:
return AssemblerCode(pc)
elif kind == VMPROF_JITTED_TAG:
return JittedCode(pc)
assert kind == VMPROF_CODE_TAG
return pc
def gunzip(fileobj):
is_gzipped = fileobj.read(2) == b'\037\213'
fileobj.seek(-2, os.SEEK_CUR)
if is_gzipped:
fileobj = io.BufferedReader(gzip.GzipFile(fileobj=fileobj))
return fileobj
class BufferTooSmallError(Exception):
def get_buf(self):
return b"".join(self.args[0])
class FileObjWrapper(object):
def __init__(self, fileobj, buffer_so_far=None):
self._fileobj = fileobj
self._buf = []
self._buffer_so_far = buffer_so_far
self._buffer_pos = 0
def read(self, count):
if self._buffer_so_far is not None:
if self._buffer_pos + count >= len(self._buffer_so_far):
s = self._buffer_so_far[self._buffer_pos:]
s += self._fileobj.read(count - len(s))
self._buffer_so_far = None
else:
s = self._buffer_so_far[self._buffer_pos:self._buffer_pos + count]
self._buffer_pos += count
else:
s = self._fileobj.read(count)
self._buf.append(s)
if len(s) < count:
raise BufferTooSmallError(self._buf)
return s
class ReaderStatus(object):
def __init__(self, interp_name, period, version, previous_virtual_ips=None,
profile_memory=False, profile_lines=False):
if previous_virtual_ips is not None:
self.virtual_ips = previous_virtual_ips
else:
self.virtual_ips = {}
self.profiles = []
self.interp_name = interp_name
self.period = period
self.version = version
self.profile_memory = profile_memory
self.profile_lines = profile_lines
class FileReadError(Exception):
pass
def assert_error(condition, error="malformed file"):
if not condition:
raise FileReadError(error)
def read_header(fileobj, buffer_so_far=None):
fileobj = FileObjWrapper(fileobj, buffer_so_far)
assert_error(read_word(fileobj) == 0)
assert_error(read_word(fileobj) == 3)
assert_error(read_word(fileobj) == 0)
period = read_word(fileobj)
assert_error(read_word(fileobj) == 0)
marker = fileobj.read(1)
assert_error(marker == MARKER_HEADER, "expected header")
version, = struct.unpack("!h", fileobj.read(2))
if version >= VERSION_MODE_AWARE:
mode = ord(fileobj.read(1))
profile_memory = (mode & PROFILE_MEMORY) != 0
profile_lines = (mode & PROFILE_LINES) != 0
else:
profile_memory = version == VERSION_MEMORY
profile_lines = False
lgt = ord(fileobj.read(1))
interp_name = fileobj.read(lgt)
if PY3:
interp_name = interp_name.decode()
return ReaderStatus(interp_name, period, version, None, profile_memory, profile_lines)
def read_one_marker(fileobj, status, buffer_so_far=None):
fileobj = FileObjWrapper(fileobj, buffer_so_far)
marker = fileobj.read(1)
if marker == MARKER_STACKTRACE:
count = read_word(fileobj)
# for now
assert count == 1
depth = read_word(fileobj)
assert depth <= 2**16, 'stack strace depth too high'
trace = read_trace(fileobj, depth, status.version, status.profile_lines)
if status.version >= VERSION_THREAD_ID:
thread_id, = struct.unpack('l', fileobj.read(WORD_SIZE))
else:
thread_id = 0
if status.profile_memory:
mem_in_kb, = struct.unpack('l', fileobj.read(WORD_SIZE))
else:
mem_in_kb = 0
trace.reverse()
status.profiles.append((trace, 1, thread_id, mem_in_kb))
elif marker == MARKER_VIRTUAL_IP:
unique_id = read_word(fileobj)
name = read_string(fileobj)
if PY3:
name = name.decode()
status.virtual_ips[unique_id] = name
elif marker == MARKER_TRAILER:
return True # finished
else:
raise FileReadError("unexpected marker: %d" % ord(marker))
return False
def read_prof_bit_by_bit(fileobj):
fileobj = gunzip(fileobj)
# note that we don't want to use all of this on normal files, since it'll
# cost us quite a bit in memory and performance and parsing 200M files in
# CPython is slow (pypy does better, use pypy)
buf = None
while True:
try:
status = read_header(fileobj, buf)
break
except BufferTooSmallError as e:
buf = e.get_buf()
finished = False
buf = None
while not finished:
try:
finished = read_one_marker(fileobj, status, buf)
except BufferTooSmallError as e:
buf = e.get_buf()
return status.period, status.profiles, status.virtual_ips, status.interp_name
def read_prof(fileobj, virtual_ips_only=False):
fileobj = gunzip(fileobj)
assert read_word(fileobj) == 0 # header count
assert read_word(fileobj) == 3 # header size
assert read_word(fileobj) == 0
period = read_word(fileobj)
assert read_word(fileobj) == 0
virtual_ips = []
profiles = []
interp_name = None
version = 0
profile_memory = False
profile_lines = False
while True:
marker = fileobj.read(1)
if marker == MARKER_HEADER:
assert not version, "multiple headers"
version, = struct.unpack("!h", fileobj.read(2))
if version >= VERSION_MODE_AWARE:
mode = ord(fileobj.read(1))
profile_memory = (mode & PROFILE_MEMORY) != 0
profile_lines = (mode & PROFILE_LINES) != 0
else:
profile_memory = version == VERSION_MEMORY
profile_lines = False
lgt = ord(fileobj.read(1))
interp_name = fileobj.read(lgt)
if PY3:
interp_name = interp_name.decode()
elif marker == MARKER_STACKTRACE:
count = read_word(fileobj)
# for now
assert count == 1
depth = read_word(fileobj)
assert depth <= 2**16, 'stack strace depth too high'
if virtual_ips_only:
fileobj.read(WORD_SIZE * depth)
trace = []
else:
trace = read_trace(fileobj, depth, version, profile_lines)
if version >= VERSION_THREAD_ID:
thread_id, = struct.unpack('l', fileobj.read(WORD_SIZE))
else:
thread_id = 0
if profile_memory:
mem_in_kb, = struct.unpack('l', fileobj.read(WORD_SIZE))
else:
mem_in_kb = 0
trace.reverse()
profiles.append((trace, 1, thread_id, mem_in_kb))
elif marker == MARKER_INTERP_NAME:
assert not version, "multiple headers"
assert not interp_name, "Dual interpreter name header"
lgt = ord(fileobj.read(1))
interp_name = fileobj.read(lgt)
if PY3:
interp_name = interp_name.decode()
elif marker == MARKER_VIRTUAL_IP:
unique_id = read_word(fileobj)
name = read_string(fileobj)
if PY3:
name = name.decode()
virtual_ips.append((unique_id, name))
elif marker == MARKER_TRAILER:
#if not virtual_ips_only:
# symmap = read_ranges(fileobj.read())
break
else:
assert not marker, (fileobj.tell(), repr(marker))
break
virtual_ips.sort() # I think it's sorted, but who knows
if virtual_ips_only:
return virtual_ips
return period, profiles, virtual_ips, interp_name