Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions .github/workflows/tests.yml
Original file line number Diff line number Diff line change
Expand Up @@ -18,12 +18,12 @@ jobs:
matrix:
# Test all supported versions on Ubuntu:
os: [ubuntu-latest]
python: ["3.9", "3.10", "3.11", "3.12", "3.13", "3.14", "3.15.0-rc.2", "pypy-3.11"]
python: ["3.10", "3.11", "3.12", "3.13", "3.14", "3.15.0-rc.2", "pypy-3.11"]
experimental: [false]
include:
# Windows on the earliest and latest supported versions:
- os: windows-latest
python: "3.9"
python: "3.10"
experimental: false
- os: windows-latest
python: "3.15.0-rc.2"
Expand Down
2 changes: 1 addition & 1 deletion meson.build
Original file line number Diff line number Diff line change
@@ -1,5 +1,5 @@
project('vmprof', 'c',
version: '0.5.1',
version: '0.6.0',
license: 'MIT',
meson_version: '>=1.1.0',
)
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,7 @@ license-files = ["LICENSE"]
authors = [
{name = "vmprof team", email = "fijal@baroquesoftware.com"},
]
requires-python = ">=3.9,<3.16"
requires-python = ">=3.10,<3.16"
dependencies = [
"requests",
"colorama",
Expand Down
39 changes: 39 additions & 0 deletions src/compat.c
Original file line number Diff line number Diff line change
Expand Up @@ -96,9 +96,48 @@ int vmp_write_time_now(int marker) {
vmp_write_all(buffer, __SIZE);
return 0;
}

#if defined(VMPROF_APPLE) && !defined(CLOCK_MONOTONIC)
/* SDK older than macOS 10.12: no clock_gettime. mach_absolute_time is a
monotonic clock; there is no signal-safe process cpu clock, so both
modes use it (it is what setitimer(ITIMER_REAL) measures anyway). */
#include <mach/mach_time.h>
int64_t vmp_sample_time_ns(int cpu_time)
{
mach_timebase_info_data_t tb;
uint64_t t = mach_absolute_time();
(void)cpu_time;
if (mach_timebase_info(&tb) != KERN_SUCCESS || tb.denom == 0)
return (int64_t)t;
return (int64_t)(t / tb.denom * tb.numer + t % tb.denom * tb.numer / tb.denom);
}
#else
int64_t vmp_sample_time_ns(int cpu_time)
{
struct timespec ts;
clockid_t clk = cpu_time ? CLOCK_PROCESS_CPUTIME_ID : CLOCK_MONOTONIC;
if (clock_gettime(clk, &ts) != 0)
return 0;
return (int64_t)ts.tv_sec * 1000000000LL + ts.tv_nsec;
}
#endif
#endif

#ifdef VMPROF_WINDOWS
int64_t vmp_sample_time_ns(int cpu_time)
{
/* the frequency is fixed at boot, so it is safe to cache it */
static LARGE_INTEGER freq;
LARGE_INTEGER now;
(void)cpu_time;
if (freq.QuadPart == 0 && !QueryPerformanceFrequency(&freq))
return 0;
if (!QueryPerformanceCounter(&now))
return 0;
return (int64_t)(now.QuadPart / freq.QuadPart * 1000000000LL +
now.QuadPart % freq.QuadPart * 1000000000LL / freq.QuadPart);
}

int vmp_write_time_now(int marker) {
char buffer[__SIZE];
struct timezone_buf buf;
Expand Down
8 changes: 8 additions & 0 deletions src/compat.h
Original file line number Diff line number Diff line change
Expand Up @@ -23,5 +23,13 @@ int vmp_write_all(const char *buf, size_t bufsize);
int vmp_write_time_now(int marker);
int vmp_write_meta(const char * key, const char * value);

/* Nanosecond timestamp stored at the end of every stack sample. With
'cpu_time' set it reads the process cpu-time clock, which is the clock
that drives ITIMER_PROF; otherwise it reads a monotonic wall clock, which
drives ITIMER_REAL and the Windows sampler thread. The reader weights
each sample by the distance to the previous one on the same clock, so
timer signals that were dropped still count. Async-signal-safe. */
int64_t vmp_sample_time_ns(int cpu_time);

int vmp_profile_fileno(void);
void vmp_set_profile_fileno(int fileno);
3 changes: 3 additions & 0 deletions src/vmprof.h
Original file line number Diff line number Diff line change
Expand Up @@ -27,6 +27,9 @@
#define VERSION_MODE_AWARE '\x04'
#define VERSION_DURATION '\x05'
#define VERSION_TIMESTAMP '\x06'
/* every MARKER_STACKTRACE record ends with a 64-bit nanosecond timestamp,
see vmp_sample_time_ns() */
#define VERSION_SAMPLE_TIME '\x07'

#define PROFILE_MEMORY '\x01'
#define PROFILE_LINES '\x02'
Expand Down
2 changes: 1 addition & 1 deletion src/vmprof_common.c
Original file line number Diff line number Diff line change
Expand Up @@ -137,7 +137,7 @@ int opened_profile(const char *interp_name, int memory, int proflines, int nativ
}
header.interp_name[0] = MARKER_HEADER;
header.interp_name[1] = '\x00';
header.interp_name[2] = VERSION_TIMESTAMP;
header.interp_name[2] = VERSION_SAMPLE_TIME;
header.interp_name[3] = memory*PROFILE_MEMORY + proflines*PROFILE_LINES + \
native*PROFILE_NATIVE + real_time*PROFILE_REAL_TIME;
#ifdef RPYTHON_VMPROF
Expand Down
9 changes: 9 additions & 0 deletions src/vmprof_common.h
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,12 @@ long vmp_native_thread_id(void);
#define MAX_STACK_DEPTH \
((SINGLE_BUF_SIZE - sizeof(struct prof_stacktrace_s)) / sizeof(void *))

/* Slots at the end of a stack sample that are not stack entries: the thread
id, the (optional) rss and the 64-bit sample timestamp. Walk at most
MAX_STACK_DEPTH - STACK_TRAILER_SLOTS frames so that they always fit. */
#define STACK_TRAILER_SLOTS \
(2 + (sizeof(int64_t) + sizeof(void *) - 1) / sizeof(void *))

/*
* NOTE SHOULD NOT BE DONE THIS WAY. Here is an example why:
* assume the following struct content:
Expand Down Expand Up @@ -77,6 +83,9 @@ typedef struct prof_stacktrace_s {
char marker;
long count, depth;
void *stack[];
/* the file record continues after stack[depth] with the thread id,
the rss if memory profiling is on, and an int64_t timestamp
(VERSION_SAMPLE_TIME, see vmp_sample_time_ns) */
} prof_stacktrace_s;

#define SIZEOF_PROF_STACKTRACE sizeof(long)+sizeof(long)+sizeof(char)
Expand Down
14 changes: 11 additions & 3 deletions src/vmprof_unix.c
Original file line number Diff line number Diff line change
Expand Up @@ -95,13 +95,16 @@ void segfault_handler(int arg)
int _vmprof_sample_stack(struct profbuf_s *p, PY_THREAD_STATE_T * tstate, ucontext_t * uc)
{
int depth;
int64_t sample_time;
struct prof_stacktrace_s *st = (struct prof_stacktrace_s *)p->data;
st->marker = MARKER_STACKTRACE;
st->count = 1;
#ifdef RPYTHON_VMPROF
depth = get_stack_trace(get_vmprof_stack(), st->stack, MAX_STACK_DEPTH-1, (intptr_t)GetPC(uc));
depth = get_stack_trace(get_vmprof_stack(), st->stack,
MAX_STACK_DEPTH - STACK_TRAILER_SLOTS, (intptr_t)GetPC(uc));
#else
depth = get_stack_trace(tstate, st->stack, MAX_STACK_DEPTH-1, (intptr_t)NULL);
depth = get_stack_trace(tstate, st->stack,
MAX_STACK_DEPTH - STACK_TRAILER_SLOTS, (intptr_t)NULL);
#endif
// useful for tests (see test_stop_sampling)
#ifndef RPYTHON_LL2CTYPES
Expand All @@ -114,8 +117,13 @@ int _vmprof_sample_stack(struct profbuf_s *p, PY_THREAD_STATE_T * tstate, uconte
long rss = get_current_proc_rss();
if (rss >= 0)
st->stack[depth++] = (void*)rss;
/* ITIMER_PROF counts process cpu time, ITIMER_REAL wall time: stamp the
sample with the clock that drives the timer. memcpy: the slot is only
pointer-aligned, which is less than int64_t needs on 32 bit. */
sample_time = vmp_sample_time_ns(vmprof_get_itimer_type() == ITIMER_PROF);
memcpy(&st->stack[depth], &sample_time, sizeof(sample_time));
p->data_offset = offsetof(struct prof_stacktrace_s, marker);
p->data_size = (depth * sizeof(void *) +
p->data_size = (depth * sizeof(void *) + sizeof(sample_time) +
sizeof(struct prof_stacktrace_s) -
offsetof(struct prof_stacktrace_s, marker));
return 1;
Expand Down
22 changes: 17 additions & 5 deletions src/vmprof_win.c
Original file line number Diff line number Diff line change
Expand Up @@ -90,11 +90,17 @@ HANDLE write_mutex;

#include "vmprof_common.h"

/* Fill 'stack' with a sample of the given thread. Returns the number of
pointer-sized slots written after the header (frames plus the thread id),
or <= 0 if nothing was sampled. The record is followed by the int64_t
sample timestamp, so the caller must write
SIZEOF_PROF_STACKTRACE + depth * sizeof(void*) + sizeof(int64_t) bytes. */
int vmprof_snapshot_thread(DWORD thread_id, PY_WIN_THREAD_STATE *tstate, prof_stacktrace_s *stack)
{
HRESULT result;
HANDLE hThread;
int depth;
int64_t sample_time;
CONTEXT ctx;
#ifdef RPYTHON_LL2CTYPES
return 0; // not much we can do
Expand All @@ -115,9 +121,11 @@ int vmprof_snapshot_thread(DWORD thread_id, PY_WIN_THREAD_STATE *tstate, prof_st
if (!GetThreadContext(hThread, &ctx))
return -1;
depth = get_stack_trace(tstate->vmprof_tl_stack,
stack->stack, MAX_STACK_DEPTH-2, ctx.Eip);
stack->stack, MAX_STACK_DEPTH - STACK_TRAILER_SLOTS, ctx.Eip);
stack->depth = depth;
stack->stack[depth++] = thread_id;
sample_time = vmp_sample_time_ns(0);
memcpy(&stack->stack[depth], &sample_time, sizeof(sample_time));
stack->count = 1;
stack->marker = MARKER_STACKTRACE;
ResumeThread(hThread);
Expand All @@ -142,9 +150,9 @@ int vmprof_snapshot_thread(DWORD thread_id, PY_WIN_THREAD_STATE *tstate, prof_st
#else
frame = PyThreadState_GetFrame(tstate);
#endif
/* leave room for the thread id appended below */
/* leave room for the thread id and timestamp appended below */
depth = vmp_walk_and_record_stack(frame, stack->stack,
MAX_STACK_DEPTH - 1, 0, 0);
MAX_STACK_DEPTH - STACK_TRAILER_SLOTS, 0, 0);
#ifdef _MSC_VER
} __except (EXCEPTION_EXECUTE_HANDLER) {
depth = -1;
Expand All @@ -160,6 +168,9 @@ int vmprof_snapshot_thread(DWORD thread_id, PY_WIN_THREAD_STATE *tstate, prof_st
}
stack->depth = depth;
stack->stack[depth++] = (void*)((ULONG_PTR)thread_id);
/* the sampler thread sleeps in wall time, so use the wall clock */
sample_time = vmp_sample_time_ns(0);
memcpy(&stack->stack[depth], &sample_time, sizeof(sample_time));
stack->count = 1;
stack->marker = MARKER_STACKTRACE;
ResumeThread(hThread);
Expand Down Expand Up @@ -226,7 +237,8 @@ long __stdcall vmprof_mainloop(void *arg)
depth = vmprof_snapshot_thread(tstate->thread_id, tstate, stack);
if (depth > 0) {
vmp_write_all((char*)stack + offsetof(prof_stacktrace_s, marker),
SIZEOF_PROF_STACKTRACE + depth * sizeof(void*));
SIZEOF_PROF_STACKTRACE + depth * sizeof(void*) +
sizeof(int64_t));
}
sampler_busy = 0;
}
Expand All @@ -246,7 +258,7 @@ long __stdcall vmprof_mainloop(void *arg)
depth = vmprof_snapshot_thread(tstate->thread_ident, tstate, stack);
if (depth > 0) {
vmp_write_all((char*)stack + offsetof(prof_stacktrace_s, marker),
depth * sizeof(void *) +
depth * sizeof(void *) + sizeof(int64_t) +
sizeof(struct prof_stacktrace_s) -
offsetof(struct prof_stacktrace_s, marker));
}
Expand Down
12 changes: 9 additions & 3 deletions vmprof/profiler.py
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@
import tempfile

from vmprof.stats import Stats
from vmprof.reader import _read_prof
from vmprof.reader import _read_prof, DEFAULT_MAX_SAMPLE_GAP


class VMProfError(Exception):
Expand Down Expand Up @@ -32,12 +32,18 @@ def __exit__(self, type, value, traceback):
self.done = True


def read_profile(prof_file):
def read_profile(prof_file, max_sample_gap=DEFAULT_MAX_SAMPLE_GAP):
""" Read a profile file into a Stats object.

max_sample_gap caps, in seconds, how much time a single sample may
stand for when the timestamps show that timer signals were lost;
see vmprof.reader.LogReader.sample_weight.
"""
file_to_close = None
if not hasattr(prof_file, 'read'):
prof_file = file_to_close = open(str(prof_file), 'rb')

state = _read_prof(prof_file)
state = _read_prof(prof_file, max_sample_gap=max_sample_gap)

if file_to_close:
file_to_close.close()
Expand Down
Loading
Loading